Linux filesystem development
 help / color / mirror / Atom feed
From: Karl Mehltretter <kmehltretter@gmail.com>
To: selinux@vger.kernel.org
Cc: Karl Mehltretter <kmehltretter@gmail.com>,
	Paul Moore <paul@paul-moore.com>,
	Stephen Smalley <stephen.smalley.work@gmail.com>,
	Ondrej Mosnacek <omosnace@redhat.com>,
	Miklos Szeredi <miklos@szeredi.hu>,
	Amir Goldstein <amir73il@gmail.com>,
	Christian Brauner <brauner@kernel.org>,
	Baokun Li <libaokun@linux.alibaba.com>,
	linux-fsdevel@vger.kernel.org, linux-unionfs@vger.kernel.org,
	linux-kernel@vger.kernel.org, stable@vger.kernel.org
Subject: [PATCH v3 2/2] selinux: recheck intermediate backing files on mprotect
Date: Sat, 29 Aug 2026 23:32:56 +0200	[thread overview]
Message-ID: <20260829213256.51527-3-kmehltretter@gmail.com> (raw)
In-Reply-To: <20260829213256.51527-1-kmehltretter@gmail.com>

mprotect() can be used to bypass the SELinux checks that mmap() performs
against the intermediate layers of a stacked filesystem.

mmap() checks every backing layer as the request descends through the
stack.  mprotect() only has the lowest backing file in vma->vm_file, so it
rechecks the top-level user and the lowest mounter, but skips the mounters
of every layer in between.  With two nested overlayfs mounts and a policy
denying mounter_t -> middle_file_t:file { execute }, a direct
mmap(PROT_EXEC) is denied:

  avc:  denied  { execute } for  pid=71 comm="nested_exec"
    path="/payload" dev="overlay" ino=9
    scontext=user_u:base_r:mounter_t
    tcontext=user_u:object_r:middle_file_t tclass=file permissive=0

while mmap(PROT_NONE) followed by mprotect(PROT_EXEC) succeeds.

Preserve each intermediate path, mounter SID and file-description SID in
the backing-file security blob, copying the saved entries when another
backing layer is opened.  Allocate the array only for nested backing files,
and release it and the path references in the backing_file_free hook.

During mprotect(), recheck fd { use } and the requested inode permissions
for every saved mounter, and include the intermediate layers in the execmod
checks.  Policy for nested stacking may then need to grant intermediate
mounters what a direct mmap() already requires, and execmod on intermediate
labels for binaries using text relocations.

Tested on arm64 QEMU with a small BusyBox initramfs and a purpose-built
SELinux policy, on a mainline tree containing
commit f2381b546e7e ("fs: fix user path of nested backing files").

Fixes: 82544d36b172 ("selinux: fix overlayfs mmap() and mprotect() access checks")
Cc: <stable@vger.kernel.org>
Assisted-by: LLM
Signed-off-by: Karl Mehltretter <kmehltretter@gmail.com>
---
 security/selinux/hooks.c          | 141 ++++++++++++++++++++++++++----
 security/selinux/include/objsec.h |   8 ++
 2 files changed, 133 insertions(+), 16 deletions(-)

diff --git a/security/selinux/hooks.c b/security/selinux/hooks.c
index 232b7e7bfcafd..b6750edcc4783 100644
--- a/security/selinux/hooks.c
+++ b/security/selinux/hooks.c
@@ -1674,26 +1674,32 @@ static int cred_has_capability(const struct cred *cred,
 	return rc;
 }
 
-/* Check whether a task has a particular permission to an inode.
-   The 'adp' parameter is optional and allows other audit
-   data to be passed (e.g. the dentry). */
-static int inode_has_perm(const struct cred *cred,
-			  struct inode *inode,
-			  u32 perms,
-			  struct common_audit_data *adp)
+/*
+ * Check whether a SID has a particular permission to an inode.  The 'adp'
+ * parameter is optional and allows other audit data to be passed (e.g. the
+ * dentry).
+ */
+static int inode_sid_has_perm(u32 sid, struct inode *inode, u32 perms,
+			      struct common_audit_data *adp)
 {
 	struct inode_security_struct *isec;
-	u32 sid;
 
 	if (unlikely(IS_PRIVATE(inode)))
 		return 0;
 
-	sid = cred_sid(cred);
 	isec = selinux_inode(inode);
 
 	return avc_has_perm(sid, isec->sid, isec->sclass, perms, adp);
 }
 
+static int inode_has_perm(const struct cred *cred,
+			  struct inode *inode,
+			  u32 perms,
+			  struct common_audit_data *adp)
+{
+	return inode_sid_has_perm(cred_sid(cred), inode, perms, adp);
+}
+
 /* Same as inode_has_perm, but pass explicit audit data containing
    the dentry to help the auditing code to more easily generate the
    pathname if needed. */
@@ -3854,13 +3860,63 @@ static int selinux_backing_file_alloc(struct file *backing_file,
 				      const struct file *user_file)
 {
 	struct backing_file_security_struct *bfsec;
+	const struct backing_file_security_struct *ubfsec;
+	struct backing_file_security_layer *layer;
+	u32 i;
 
 	bfsec = selinux_backing_file(backing_file);
 	bfsec->uf_sid = selinux_file_user_sid(user_file);
+	if (!(user_file->f_mode & FMODE_BACKING))
+		return 0;
+
+	ubfsec = selinux_backing_file(user_file);
+	/* a wrapped count would make kmalloc_array() return ZERO_SIZE_PTR */
+	if (unlikely(ubfsec->layer_count == U32_MAX))
+		return -EOVERFLOW;
+
+	/*
+	 * The final VMA only retains the lowest backing file, so record the
+	 * whole chain here rather than in the mmap hook, where concurrent
+	 * mappings would have to be serialized.  Size it dynamically: erofs
+	 * inode sharing adds a backing file without bumping s_stack_depth.
+	 */
+	bfsec->layers = kmalloc_array(ubfsec->layer_count + 1,
+				      sizeof(*bfsec->layers), GFP_KERNEL);
+	if (!bfsec->layers)
+		return -ENOMEM;
+
+	for (i = 0; i < ubfsec->layer_count; i++) {
+		layer = &bfsec->layers[i];
+		*layer = ubfsec->layers[i];
+		path_get(&layer->path);
+	}
+
+	/* f_path, not file_user_path(): this layer, not the top-level file */
+	layer = &bfsec->layers[i];
+	layer->path = user_file->f_path;
+	layer->mounter_sid = cred_sid(user_file->f_cred);
+	layer->fd_sid = selinux_file(user_file)->sid;
+	path_get(&layer->path);
+	bfsec->layer_count = ubfsec->layer_count + 1;
 
 	return 0;
 }
 
+static void selinux_backing_file_free(struct file *backing_file)
+{
+	struct backing_file_security_struct *bfsec;
+
+	/* security_backing_file_free() may be called twice after an error */
+	if (!backing_file_security(backing_file))
+		return;
+
+	bfsec = selinux_backing_file(backing_file);
+	while (bfsec->layer_count)
+		path_put(&bfsec->layers[--bfsec->layer_count].path);
+	kfree(bfsec->layers);
+	bfsec->layers = NULL;
+}
+
 /*
  * Check whether a task has the ioctl permission and cmd
  * operation to an inode.
@@ -3978,6 +4034,53 @@ static int selinux_file_ioctl_compat(struct file *file, unsigned int cmd,
 
 static int default_noexec __ro_after_init;
 
+static u32 file_map_prot_to_av(unsigned long prot, bool shared)
+{
+	u32 av = FILE__READ;
+
+	if (shared && (prot & PROT_WRITE))
+		av |= FILE__WRITE;
+	if (prot & PROT_EXEC)
+		av |= FILE__EXECUTE;
+
+	return av;
+}
+
+static int backing_mounters_has_perm(const struct file *file, u32 av)
+{
+	const struct backing_file_security_struct *bfsec;
+	const struct backing_file_security_layer *layer;
+	struct common_audit_data ad;
+	struct inode *inode;
+	u32 i;
+	int rc;
+
+	if (WARN_ON_ONCE(!(file->f_mode & FMODE_BACKING)))
+		return -EIO;
+
+	bfsec = selinux_backing_file(file);
+	for (i = 0; i < bfsec->layer_count; i++) {
+		layer = &bfsec->layers[i];
+		inode = d_inode(layer->path.dentry);
+
+		ad.type = LSM_AUDIT_DATA_PATH;
+		ad.u.path = layer->path;
+
+		if (layer->mounter_sid != layer->fd_sid) {
+			rc = avc_has_perm(layer->mounter_sid, layer->fd_sid,
+					  SECCLASS_FD, FD__USE, &ad);
+			if (rc)
+				return rc;
+		}
+
+		rc = inode_sid_has_perm(layer->mounter_sid, inode, av, &ad);
+		if (rc)
+			return rc;
+	}
+
+	return 0;
+}
+
 static int __file_map_prot_check(const struct file *file, unsigned long prot,
 				 bool shared, bool mounter_check,
 				 bool bf_user_file)
@@ -4011,14 +4114,10 @@ static int __file_map_prot_check(const struct file *file, unsigned long prot,
 	if (file) {
 		const struct cred *cred = mounter_check ?
 				file->f_cred : current_cred();
-		/* "read" always possible, "write" only if shared */
-		u32 av = FILE__READ;
-		if (shared && prot_write)
-			av |= FILE__WRITE;
-		if (prot_exec)
-			av |= FILE__EXECUTE;
 
-		return __file_has_perm(cred, file, av, bf_user_file);
+		return __file_has_perm(cred, file,
+				       file_map_prot_to_av(prot, shared),
+				       bf_user_file);
 	}
 
 	return 0;
@@ -4113,6 +4212,7 @@ static int selinux_file_mprotect(struct vm_area_struct *vma,
 	int rc;
 	const struct cred *cred = current_cred();
 	u32 sid = cred_sid(cred);
+	u32 av;
 	const struct file *file = vma->vm_file;
 	bool backing_file;
 	bool shared = vma->vm_flags & VM_SHARED;
@@ -4156,6 +4256,10 @@ static int selinux_file_mprotect(struct vm_area_struct *vma,
 			if (rc)
 				return rc;
 			if (backing_file) {
+				rc = backing_mounters_has_perm(file,
+							       FILE__EXECMOD);
+				if (rc)
+					return rc;
 				rc = file_has_perm(file->f_cred, file,
 						   FILE__EXECMOD);
 				if (rc)
@@ -4168,6 +4272,10 @@ static int selinux_file_mprotect(struct vm_area_struct *vma,
 	if (rc)
 		return rc;
 	if (backing_file) {
+		av = file_map_prot_to_av(prot, shared);
+		rc = backing_mounters_has_perm(file, av);
+		if (rc)
+			return rc;
 		rc = file_map_prot_check(file, prot, shared, true);
 		if (rc)
 			return rc;
@@ -7642,6 +7750,7 @@ static struct security_hook_list selinux_hooks[] __ro_after_init = {
 	LSM_HOOK_INIT(file_permission, selinux_file_permission),
 	LSM_HOOK_INIT(file_alloc_security, selinux_file_alloc_security),
 	LSM_HOOK_INIT(backing_file_alloc, selinux_backing_file_alloc),
+	LSM_HOOK_INIT(backing_file_free, selinux_backing_file_free),
 	LSM_HOOK_INIT(file_ioctl, selinux_file_ioctl),
 	LSM_HOOK_INIT(file_ioctl_compat, selinux_file_ioctl_compat),
 	LSM_HOOK_INIT(mmap_file, selinux_mmap_file),
diff --git a/security/selinux/include/objsec.h b/security/selinux/include/objsec.h
index 853f7266ed189..2f21568251ffe 100644
--- a/security/selinux/include/objsec.h
+++ b/security/selinux/include/objsec.h
@@ -86,8 +86,16 @@ struct file_security_struct {
 	u32 pseqno; /* Policy seqno at the time of file open */
 };
 
+struct backing_file_security_layer {
+	struct path path; /* this layer's real path */
+	u32 mounter_sid; /* SID of the mounter that opened it */
+	u32 fd_sid; /* SID of its open file description */
+};
+
 struct backing_file_security_struct {
 	u32 uf_sid; /* top-level user file fsec->sid */
+	u32 layer_count; /* number of intermediate backing files */
+	struct backing_file_security_layer *layers;
 };
 
 struct superblock_security_struct {
-- 
2.53.0


  parent reply	other threads:[~2026-08-29 21:33 UTC|newest]

Thread overview: 5+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-08-29 21:32 [PATCH v3 0/2] selinux: fix nested backing-file mprotect checks Karl Mehltretter
2026-08-29 21:32 ` [PATCH v3 1/2] selinux: preserve user SID across nested backing files Karl Mehltretter
2026-08-31 13:47   ` Stephen Smalley
2026-08-29 21:32 ` Karl Mehltretter [this message]
2026-08-31 13:47   ` [PATCH v3 2/2] selinux: recheck intermediate backing files on mprotect Stephen Smalley

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260829213256.51527-3-kmehltretter@gmail.com \
    --to=kmehltretter@gmail.com \
    --cc=amir73il@gmail.com \
    --cc=brauner@kernel.org \
    --cc=libaokun@linux.alibaba.com \
    --cc=linux-fsdevel@vger.kernel.org \
    --cc=linux-kernel@vger.kernel.org \
    --cc=linux-unionfs@vger.kernel.org \
    --cc=miklos@szeredi.hu \
    --cc=omosnace@redhat.com \
    --cc=paul@paul-moore.com \
    --cc=selinux@vger.kernel.org \
    --cc=stable@vger.kernel.org \
    --cc=stephen.smalley.work@gmail.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox