From: Christian Brauner <brauner@kernel.org>
To: Jann Horn <jannh@google.com>,
linux-fsdevel@vger.kernel.org, Oleg Nesterov <oleg@redhat.com>
Cc: Alexander Viro <viro@zeniv.linux.org.uk>, Jan Kara <jack@suse.cz>,
Neil Brown <neil@brown.name>, Jeff Layton <jlayton@kernel.org>,
"Christian Brauner (Amutable)" <brauner@kernel.org>
Subject: [PATCH 09/10] file: add CLOSE_RANGE_CLOEXEC_ONLY
Date: Mon, 21 Sep 2026 16:15:37 +0200 [thread overview]
Message-ID: <20260921-work-file-close_range_except-v1-9-c20d0b49270d@kernel.org> (raw)
In-Reply-To: <20260921-work-file-close_range_except-v1-0-c20d0b49270d@kernel.org>
CLOSE_RANGE_CLOEXEC marks a range close-on-exec. Nothing happens to
these fds unless an exec happens. Add CLOSE_RANGE_CLOEXEC_ONLY which
closes the close-on-exec file descriptors in the range.
When combined with CLOSE_RANGE_EXCEPT it names the close-on-exec file
descriptors that are supposed to survive. Every close-on-exec fd outside
of the range is closed. File descriptors without that flag are left
alone. Whatever the caller deliberately passes down, stdio, LISTEN_FDS,
an inherited pipe, remains where it is.
The handful of close-on-exec descriptors the caller still needs for the
exec such as the executable, an error pipe, sit in a specific range.
That is what a task between clone(CLONE_FILES) and execve() actually
wants:
clone(CLONE_FILES | CLONE_VM | CLONE_VFORK)
child: close_range(lo, hi, CLOSE_RANGE_UNSHARE |
CLOSE_RANGE_CLOEXEC_ONLY |
CLOSE_RANGE_EXCEPT)
child: rearrange descriptors in the now private table
child: execve()
Between the clone and the close_range() the child holds no reference of
its own on any file because copy_files() only bumps the fdtable
refcount. So a close() in the parent takes effect immediately. The
unshare also never takes a reference on the fds it leaves behind either.
The range is expressed in the parent's numbering. So a caller that
cannot name the file descriptors it keeps contiguously picks a range
wide enough to cover them, unshares, and tidies up with a second
close_range() on the table it now owns alone. That one is cheap. To keep
nothing, name a range that cannot hold an open descriptor, e.g.
close_range(~0U, ~0U, ...).
CLOSE_RANGE_CLOEXEC and CLOSE_RANGE_CLOEXEC_ONLY are mutually exclusive.
Signed-off-by: Christian Brauner (Amutable) <brauner@kernel.org>
---
fs/file.c | 30 +++++++++++++++++++++++++++---
include/uapi/linux/close_range.h | 5 +++++
2 files changed, 32 insertions(+), 3 deletions(-)
diff --git a/fs/file.c b/fs/file.c
index 8952efd2cf3a..37f0ba740c39 100644
--- a/fs/file.c
+++ b/fs/file.c
@@ -843,18 +843,29 @@ static inline void __range_cloexec(struct files_struct *cur_fds,
spin_unlock(&cur_fds->file_lock);
}
+/* Next open descriptor in [fd, max_fd], or the next close-on-exec one. */
+static inline unsigned int next_open_fd(struct fdtable *fdt, unsigned int fd,
+ unsigned int max_fd,
+ struct fd_range *range)
+{
+ if (range->flags & FD_RANGE_CLOEXEC_ONLY)
+ return find_next_and_bit(fdt->open_fds, fdt->close_on_exec,
+ max_fd + 1, fd);
+ return find_next_bit(fdt->open_fds, max_fd + 1, fd);
+}
+
/* Next open descriptor in [fd, max_fd] that @range selects. */
static inline unsigned int next_fd_to_close(struct fdtable *fdt,
unsigned int fd, unsigned int max_fd,
struct fd_range *range)
{
- fd = find_next_bit(fdt->open_fds, max_fd + 1, fd);
+ fd = next_open_fd(fdt, fd, max_fd, range);
/* Hop over the window the range keeps. */
if ((range->flags & FD_RANGE_EXCEPT) &&
fd >= range->from && fd <= range->to) {
if (range->to >= max_fd)
return max_fd + 1;
- fd = find_next_bit(fdt->open_fds, max_fd + 1, range->to + 1);
+ fd = next_open_fd(fdt, range->to + 1, max_fd, range);
}
return fd;
}
@@ -911,6 +922,12 @@ static inline void __range_close(struct files_struct *files,
* With CLOSE_RANGE_EXCEPT the range names what to leave alone instead:
* every open file descriptor outside of [@fd, @max_fd] is closed, or
* marked close-on-exec with CLOSE_RANGE_CLOEXEC.
+ *
+ * With CLOSE_RANGE_CLOEXEC_ONLY only file descriptors that have
+ * close-on-exec set are closed. Together with CLOSE_RANGE_EXCEPT the
+ * range names the close-on-exec file descriptors to keep. To keep none
+ * of them, name a range that cannot hold an open file descriptor, e.g.
+ * close_range(~0U, ~0U, ...).
*/
SYSCALL_DEFINE3(close_range, unsigned int, fd, unsigned int, max_fd,
unsigned int, flags)
@@ -920,7 +937,12 @@ SYSCALL_DEFINE3(close_range, unsigned int, fd, unsigned int, max_fd,
struct fd_range range = {fd, max_fd};
if (flags & ~(CLOSE_RANGE_UNSHARE | CLOSE_RANGE_CLOEXEC |
- CLOSE_RANGE_EXCEPT))
+ CLOSE_RANGE_EXCEPT | CLOSE_RANGE_CLOEXEC_ONLY))
+ return -EINVAL;
+
+ /* One marks close-on-exec, the other closes what is marked. */
+ if (hweight32(flags & (CLOSE_RANGE_CLOEXEC |
+ CLOSE_RANGE_CLOEXEC_ONLY)) > 1)
return -EINVAL;
if (fd > max_fd)
@@ -928,6 +950,8 @@ SYSCALL_DEFINE3(close_range, unsigned int, fd, unsigned int, max_fd,
if (flags & CLOSE_RANGE_EXCEPT)
range.flags |= FD_RANGE_EXCEPT;
+ if (flags & CLOSE_RANGE_CLOEXEC_ONLY)
+ range.flags |= FD_RANGE_CLOEXEC_ONLY;
if ((flags & CLOSE_RANGE_UNSHARE) && atomic_read(&cur_fds->count) > 1) {
struct fd_range *drop = ⦥
diff --git a/include/uapi/linux/close_range.h b/include/uapi/linux/close_range.h
index 39eddb3ab613..7da9ed95258a 100644
--- a/include/uapi/linux/close_range.h
+++ b/include/uapi/linux/close_range.h
@@ -9,6 +9,7 @@
#undef CLOSE_RANGE_UNSHARE
#undef CLOSE_RANGE_CLOEXEC
#undef CLOSE_RANGE_EXCEPT
+#undef CLOSE_RANGE_CLOEXEC_ONLY
enum close_range_flags {
/* Unshare the file descriptor table before closing file descriptors. */
@@ -19,12 +20,16 @@ enum close_range_flags {
/* Act on every file descriptor outside of the given range instead. */
CLOSE_RANGE_EXCEPT = (1U << 3),
+
+ /* Only close file descriptors that have the FD_CLOEXEC bit set. */
+ CLOSE_RANGE_CLOEXEC_ONLY = (1U << 4),
};
/* Keep #ifdef working and let glibc skip its own definitions. */
#define CLOSE_RANGE_UNSHARE CLOSE_RANGE_UNSHARE
#define CLOSE_RANGE_CLOEXEC CLOSE_RANGE_CLOEXEC
#define CLOSE_RANGE_EXCEPT CLOSE_RANGE_EXCEPT
+#define CLOSE_RANGE_CLOEXEC_ONLY CLOSE_RANGE_CLOEXEC_ONLY
#endif /* _UAPI_LINUX_CLOSE_RANGE_H */
--
2.53.0
next prev parent reply other threads:[~2026-09-21 14:16 UTC|newest]
Thread overview: 14+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-21 14:15 [PATCH 00/10] files,close_range: add CLOSE_RANGE_{CLOEXEC_ONLY,EXCEPT} Christian Brauner
2026-09-21 14:15 ` [PATCH 01/10] file: let dup_fd() leave the punched hole behind Christian Brauner
2026-09-21 14:15 ` [PATCH 02/10] selftests/core: test the hole CLOSE_RANGE_UNSHARE leaves behind Christian Brauner
2026-09-21 14:15 ` [PATCH 03/10] file: rename dup_fd()'s punch_hole to range Christian Brauner
2026-09-21 14:15 ` [PATCH 04/10] file: let dup_fd() drop everything outside of the range Christian Brauner
2026-09-21 14:15 ` [PATCH 05/10] close_range: turn the flags into an enum Christian Brauner
2026-09-21 14:15 ` [PATCH 06/10] file: add CLOSE_RANGE_EXCEPT Christian Brauner
2026-09-21 14:15 ` [PATCH 07/10] selftests/core: test CLOSE_RANGE_EXCEPT Christian Brauner
2026-09-21 14:15 ` [PATCH 08/10] file: let dup_fd() drop only close-on-exec descriptors Christian Brauner
2026-09-21 16:24 ` Jann Horn
2026-09-25 14:57 ` Christian Brauner
2026-09-21 14:15 ` Christian Brauner [this message]
2026-09-21 14:15 ` [PATCH 10/10] selftests/core: test CLOSE_RANGE_CLOEXEC_ONLY Christian Brauner
2026-09-21 16:34 ` [PATCH 00/10] files,close_range: add CLOSE_RANGE_{CLOEXEC_ONLY,EXCEPT} Jann Horn
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260921-work-file-close_range_except-v1-9-c20d0b49270d@kernel.org \
--to=brauner@kernel.org \
--cc=jack@suse.cz \
--cc=jannh@google.com \
--cc=jlayton@kernel.org \
--cc=linux-fsdevel@vger.kernel.org \
--cc=neil@brown.name \
--cc=oleg@redhat.com \
--cc=viro@zeniv.linux.org.uk \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox