Linux filesystem development
 help / color / mirror / Atom feed
From: Christian Brauner <brauner@kernel.org>
To: Jann Horn <jannh@google.com>,
	linux-fsdevel@vger.kernel.org,  Oleg Nesterov <oleg@redhat.com>
Cc: Alexander Viro <viro@zeniv.linux.org.uk>, Jan Kara <jack@suse.cz>,
	 Neil Brown <neil@brown.name>, Jeff Layton <jlayton@kernel.org>,
	 "Christian Brauner (Amutable)" <brauner@kernel.org>
Subject: [PATCH 09/10] file: add CLOSE_RANGE_CLOEXEC_ONLY
Date: Mon, 21 Sep 2026 16:15:37 +0200	[thread overview]
Message-ID: <20260921-work-file-close_range_except-v1-9-c20d0b49270d@kernel.org> (raw)
In-Reply-To: <20260921-work-file-close_range_except-v1-0-c20d0b49270d@kernel.org>

CLOSE_RANGE_CLOEXEC marks a range close-on-exec. Nothing happens to
these fds unless an exec happens. Add CLOSE_RANGE_CLOEXEC_ONLY which
closes the close-on-exec file descriptors in the range.

When combined with CLOSE_RANGE_EXCEPT it names the close-on-exec file
descriptors that are supposed to survive. Every close-on-exec fd outside
of the range is closed. File descriptors without that flag are left
alone. Whatever the caller deliberately passes down, stdio, LISTEN_FDS,
an inherited pipe, remains where it is.

The handful of close-on-exec descriptors the caller still needs for the
exec such as the executable, an error pipe, sit in a specific range.
That is what a task between clone(CLONE_FILES) and execve() actually
wants:

	clone(CLONE_FILES | CLONE_VM | CLONE_VFORK)
	child: close_range(lo, hi, CLOSE_RANGE_UNSHARE |
				   CLOSE_RANGE_CLOEXEC_ONLY |
				   CLOSE_RANGE_EXCEPT)
	child: rearrange descriptors in the now private table
	child: execve()

Between the clone and the close_range() the child holds no reference of
its own on any file because copy_files() only bumps the fdtable
refcount. So a close() in the parent takes effect immediately. The
unshare also never takes a reference on the fds it leaves behind either.

The range is expressed in the parent's numbering. So a caller that
cannot name the file descriptors it keeps contiguously picks a range
wide enough to cover them, unshares, and tidies up with a second
close_range() on the table it now owns alone. That one is cheap. To keep
nothing, name a range that cannot hold an open descriptor, e.g.
close_range(~0U, ~0U, ...).

CLOSE_RANGE_CLOEXEC and CLOSE_RANGE_CLOEXEC_ONLY are mutually exclusive.

Signed-off-by: Christian Brauner (Amutable) <brauner@kernel.org>
---
 fs/file.c                        | 30 +++++++++++++++++++++++++++---
 include/uapi/linux/close_range.h |  5 +++++
 2 files changed, 32 insertions(+), 3 deletions(-)

diff --git a/fs/file.c b/fs/file.c
index 8952efd2cf3a..37f0ba740c39 100644
--- a/fs/file.c
+++ b/fs/file.c
@@ -843,18 +843,29 @@ static inline void __range_cloexec(struct files_struct *cur_fds,
 	spin_unlock(&cur_fds->file_lock);
 }
 
+/* Next open descriptor in [fd, max_fd], or the next close-on-exec one. */
+static inline unsigned int next_open_fd(struct fdtable *fdt, unsigned int fd,
+					unsigned int max_fd,
+					struct fd_range *range)
+{
+	if (range->flags & FD_RANGE_CLOEXEC_ONLY)
+		return find_next_and_bit(fdt->open_fds, fdt->close_on_exec,
+					 max_fd + 1, fd);
+	return find_next_bit(fdt->open_fds, max_fd + 1, fd);
+}
+
 /* Next open descriptor in [fd, max_fd] that @range selects. */
 static inline unsigned int next_fd_to_close(struct fdtable *fdt,
 					    unsigned int fd, unsigned int max_fd,
 					    struct fd_range *range)
 {
-	fd = find_next_bit(fdt->open_fds, max_fd + 1, fd);
+	fd = next_open_fd(fdt, fd, max_fd, range);
 	/* Hop over the window the range keeps. */
 	if ((range->flags & FD_RANGE_EXCEPT) &&
 	    fd >= range->from && fd <= range->to) {
 		if (range->to >= max_fd)
 			return max_fd + 1;
-		fd = find_next_bit(fdt->open_fds, max_fd + 1, range->to + 1);
+		fd = next_open_fd(fdt, range->to + 1, max_fd, range);
 	}
 	return fd;
 }
@@ -911,6 +922,12 @@ static inline void __range_close(struct files_struct *files,
  * With CLOSE_RANGE_EXCEPT the range names what to leave alone instead:
  * every open file descriptor outside of [@fd, @max_fd] is closed, or
  * marked close-on-exec with CLOSE_RANGE_CLOEXEC.
+ *
+ * With CLOSE_RANGE_CLOEXEC_ONLY only file descriptors that have
+ * close-on-exec set are closed. Together with CLOSE_RANGE_EXCEPT the
+ * range names the close-on-exec file descriptors to keep. To keep none
+ * of them, name a range that cannot hold an open file descriptor, e.g.
+ * close_range(~0U, ~0U, ...).
  */
 SYSCALL_DEFINE3(close_range, unsigned int, fd, unsigned int, max_fd,
 		unsigned int, flags)
@@ -920,7 +937,12 @@ SYSCALL_DEFINE3(close_range, unsigned int, fd, unsigned int, max_fd,
 	struct fd_range range = {fd, max_fd};
 
 	if (flags & ~(CLOSE_RANGE_UNSHARE | CLOSE_RANGE_CLOEXEC |
-		      CLOSE_RANGE_EXCEPT))
+		      CLOSE_RANGE_EXCEPT | CLOSE_RANGE_CLOEXEC_ONLY))
+		return -EINVAL;
+
+	/* One marks close-on-exec, the other closes what is marked. */
+	if (hweight32(flags & (CLOSE_RANGE_CLOEXEC |
+			       CLOSE_RANGE_CLOEXEC_ONLY)) > 1)
 		return -EINVAL;
 
 	if (fd > max_fd)
@@ -928,6 +950,8 @@ SYSCALL_DEFINE3(close_range, unsigned int, fd, unsigned int, max_fd,
 
 	if (flags & CLOSE_RANGE_EXCEPT)
 		range.flags |= FD_RANGE_EXCEPT;
+	if (flags & CLOSE_RANGE_CLOEXEC_ONLY)
+		range.flags |= FD_RANGE_CLOEXEC_ONLY;
 
 	if ((flags & CLOSE_RANGE_UNSHARE) && atomic_read(&cur_fds->count) > 1) {
 		struct fd_range *drop = &range;
diff --git a/include/uapi/linux/close_range.h b/include/uapi/linux/close_range.h
index 39eddb3ab613..7da9ed95258a 100644
--- a/include/uapi/linux/close_range.h
+++ b/include/uapi/linux/close_range.h
@@ -9,6 +9,7 @@
 #undef CLOSE_RANGE_UNSHARE
 #undef CLOSE_RANGE_CLOEXEC
 #undef CLOSE_RANGE_EXCEPT
+#undef CLOSE_RANGE_CLOEXEC_ONLY
 
 enum close_range_flags {
 	/* Unshare the file descriptor table before closing file descriptors. */
@@ -19,12 +20,16 @@ enum close_range_flags {
 
 	/* Act on every file descriptor outside of the given range instead. */
 	CLOSE_RANGE_EXCEPT	= (1U << 3),
+
+	/* Only close file descriptors that have the FD_CLOEXEC bit set. */
+	CLOSE_RANGE_CLOEXEC_ONLY	= (1U << 4),
 };
 
 /* Keep #ifdef working and let glibc skip its own definitions. */
 #define CLOSE_RANGE_UNSHARE		CLOSE_RANGE_UNSHARE
 #define CLOSE_RANGE_CLOEXEC		CLOSE_RANGE_CLOEXEC
 #define CLOSE_RANGE_EXCEPT		CLOSE_RANGE_EXCEPT
+#define CLOSE_RANGE_CLOEXEC_ONLY	CLOSE_RANGE_CLOEXEC_ONLY
 
 #endif /* _UAPI_LINUX_CLOSE_RANGE_H */
 

-- 
2.53.0


  parent reply	other threads:[~2026-09-21 14:16 UTC|newest]

Thread overview: 14+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-21 14:15 [PATCH 00/10] files,close_range: add CLOSE_RANGE_{CLOEXEC_ONLY,EXCEPT} Christian Brauner
2026-09-21 14:15 ` [PATCH 01/10] file: let dup_fd() leave the punched hole behind Christian Brauner
2026-09-21 14:15 ` [PATCH 02/10] selftests/core: test the hole CLOSE_RANGE_UNSHARE leaves behind Christian Brauner
2026-09-21 14:15 ` [PATCH 03/10] file: rename dup_fd()'s punch_hole to range Christian Brauner
2026-09-21 14:15 ` [PATCH 04/10] file: let dup_fd() drop everything outside of the range Christian Brauner
2026-09-21 14:15 ` [PATCH 05/10] close_range: turn the flags into an enum Christian Brauner
2026-09-21 14:15 ` [PATCH 06/10] file: add CLOSE_RANGE_EXCEPT Christian Brauner
2026-09-21 14:15 ` [PATCH 07/10] selftests/core: test CLOSE_RANGE_EXCEPT Christian Brauner
2026-09-21 14:15 ` [PATCH 08/10] file: let dup_fd() drop only close-on-exec descriptors Christian Brauner
2026-09-21 16:24   ` Jann Horn
2026-09-25 14:57     ` Christian Brauner
2026-09-21 14:15 ` Christian Brauner [this message]
2026-09-21 14:15 ` [PATCH 10/10] selftests/core: test CLOSE_RANGE_CLOEXEC_ONLY Christian Brauner
2026-09-21 16:34 ` [PATCH 00/10] files,close_range: add CLOSE_RANGE_{CLOEXEC_ONLY,EXCEPT} Jann Horn

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260921-work-file-close_range_except-v1-9-c20d0b49270d@kernel.org \
    --to=brauner@kernel.org \
    --cc=jack@suse.cz \
    --cc=jannh@google.com \
    --cc=jlayton@kernel.org \
    --cc=linux-fsdevel@vger.kernel.org \
    --cc=neil@brown.name \
    --cc=oleg@redhat.com \
    --cc=viro@zeniv.linux.org.uk \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox