From: Christian Brauner <brauner@kernel.org>
To: Jann Horn <jannh@google.com>,
linux-fsdevel@vger.kernel.org, Oleg Nesterov <oleg@redhat.com>
Cc: Alexander Viro <viro@zeniv.linux.org.uk>, Jan Kara <jack@suse.cz>,
Neil Brown <neil@brown.name>, Jeff Layton <jlayton@kernel.org>,
"Christian Brauner (Amutable)" <brauner@kernel.org>
Subject: [PATCH 06/10] file: add CLOSE_RANGE_EXCEPT
Date: Mon, 21 Sep 2026 16:15:34 +0200 [thread overview]
Message-ID: <20260921-work-file-close_range_except-v1-6-c20d0b49270d@kernel.org> (raw)
In-Reply-To: <20260921-work-file-close_range_except-v1-0-c20d0b49270d@kernel.org>
close_range() operates on [fd, max_fd]. Add CLOSE_RANGE_EXCEPT. This new
flag instructs it to operate on all file descriptors outside of the
range instead.
Without any other flags that means it closes everything except the file
descriptors in the specified range. Together with CLOSE_RANGE_CLOEXEC it
marks every file descriptor except the ones in the range as
close-on-exec. With CLOSE_RANGE_UNSHARE dup_fd() never takes a reference
on what is dropped.
A task between clone(CLONE_FILES | CLONE_VM | CLONE_VFORK) and execve()
that has the descriptors for the child in one window can shed the rest
of the shared table in one call:
close_range(lo, hi, CLOSE_RANGE_UNSHARE | CLOSE_RANGE_EXCEPT)
Nothing outside of the window is referenced by the child at any point.
Signed-off-by: Christian Brauner (Amutable) <brauner@kernel.org>
---
fs/file.c | 76 +++++++++++++++++++++++++++++++---------
include/uapi/linux/close_range.h | 5 +++
2 files changed, 64 insertions(+), 17 deletions(-)
diff --git a/fs/file.c b/fs/file.c
index ae7d021398dd..39651c7a992e 100644
--- a/fs/file.c
+++ b/fs/file.c
@@ -794,34 +794,67 @@ static inline unsigned last_fd(struct fdtable *fdt)
}
static inline void __range_cloexec(struct files_struct *cur_fds,
- unsigned int fd, unsigned int max_fd)
+ struct fd_range *range)
{
struct fdtable *fdt;
+ unsigned int last;
- /* make sure we're using the correct maximum value */
spin_lock(&cur_fds->file_lock);
fdt = files_fdtable(cur_fds);
- max_fd = min(last_fd(fdt), max_fd);
- if (fd <= max_fd)
- bitmap_set(fdt->close_on_exec, fd, max_fd - fd + 1);
+ /* make sure we're using the correct maximum value */
+ last = last_fd(fdt);
+ if (!(range->flags & FD_RANGE_EXCEPT)) {
+ if (range->from <= last)
+ bitmap_set(fdt->close_on_exec, range->from,
+ min(range->to, last) - range->from + 1);
+ } else {
+ if (range->from > 0)
+ bitmap_set(fdt->close_on_exec, 0,
+ min(range->from - 1, last) + 1);
+ if (range->to < last)
+ bitmap_set(fdt->close_on_exec, range->to + 1,
+ last - range->to);
+ }
spin_unlock(&cur_fds->file_lock);
}
-static inline void __range_close(struct files_struct *files, unsigned int fd,
- unsigned int max_fd)
+/* Next open descriptor in [fd, max_fd] that @range selects. */
+static inline unsigned int next_fd_to_close(struct fdtable *fdt,
+ unsigned int fd, unsigned int max_fd,
+ struct fd_range *range)
+{
+ fd = find_next_bit(fdt->open_fds, max_fd + 1, fd);
+ /* Hop over the window the range keeps. */
+ if ((range->flags & FD_RANGE_EXCEPT) &&
+ fd >= range->from && fd <= range->to) {
+ if (range->to >= max_fd)
+ return max_fd + 1;
+ fd = find_next_bit(fdt->open_fds, max_fd + 1, range->to + 1);
+ }
+ return fd;
+}
+
+static inline void __range_close(struct files_struct *files,
+ struct fd_range *range)
{
struct file *file;
struct fdtable *fdt;
- unsigned n;
+ unsigned int fd, max_fd;
spin_lock(&files->file_lock);
fdt = files_fdtable(files);
- n = last_fd(fdt);
- max_fd = min(max_fd, n);
+ if (range->flags & FD_RANGE_EXCEPT) {
+ /* Outside of the range means the whole table. */
+ fd = 0;
+ max_fd = last_fd(fdt);
+ } else {
+ fd = range->from;
+ max_fd = min(range->to, last_fd(fdt));
+ }
- for (fd = find_next_bit(fdt->open_fds, max_fd + 1, fd);
+ for (fd = next_fd_to_close(fdt, fd, max_fd, range);
fd <= max_fd;
- fd = find_next_bit(fdt->open_fds, max_fd + 1, fd + 1)) {
+ fd = next_fd_to_close(fdt, fd + 1, max_fd, range)) {
file = file_close_fd_locked(files, fd);
if (file) {
spin_unlock(&files->file_lock);
@@ -849,21 +882,30 @@ static inline void __range_close(struct files_struct *files, unsigned int fd,
* This closes a range of file descriptors. All file descriptors
* from @fd up to and including @max_fd are closed.
* Currently, errors to close a given file descriptor are ignored.
+ *
+ * With CLOSE_RANGE_EXCEPT the range names what to leave alone instead:
+ * every open file descriptor outside of [@fd, @max_fd] is closed, or
+ * marked close-on-exec with CLOSE_RANGE_CLOEXEC.
*/
SYSCALL_DEFINE3(close_range, unsigned int, fd, unsigned int, max_fd,
unsigned int, flags)
{
struct task_struct *me = current;
struct files_struct *cur_fds = me->files, *fds = NULL;
+ struct fd_range range = {fd, max_fd};
- if (flags & ~(CLOSE_RANGE_UNSHARE | CLOSE_RANGE_CLOEXEC))
+ if (flags & ~(CLOSE_RANGE_UNSHARE | CLOSE_RANGE_CLOEXEC |
+ CLOSE_RANGE_EXCEPT))
return -EINVAL;
if (fd > max_fd)
return -EINVAL;
+ if (flags & CLOSE_RANGE_EXCEPT)
+ range.flags |= FD_RANGE_EXCEPT;
+
if ((flags & CLOSE_RANGE_UNSHARE) && atomic_read(&cur_fds->count) > 1) {
- struct fd_range range = {fd, max_fd}, *drop = ⦥
+ struct fd_range *drop = ⦥
/*
* If the caller requested all fds to be made cloexec we always
@@ -884,10 +926,10 @@ SYSCALL_DEFINE3(close_range, unsigned int, fd, unsigned int, max_fd,
}
if (flags & CLOSE_RANGE_CLOEXEC) {
- __range_cloexec(cur_fds, fd, max_fd);
+ __range_cloexec(cur_fds, &range);
} else if (!fds) {
- /* If we unshared, dup_fd() left the range behind already. */
- __range_close(cur_fds, fd, max_fd);
+ /* If we unshared, dup_fd() already left behind what we'd close. */
+ __range_close(cur_fds, &range);
}
if (fds) {
diff --git a/include/uapi/linux/close_range.h b/include/uapi/linux/close_range.h
index b1b637d74deb..39eddb3ab613 100644
--- a/include/uapi/linux/close_range.h
+++ b/include/uapi/linux/close_range.h
@@ -8,6 +8,7 @@
*/
#undef CLOSE_RANGE_UNSHARE
#undef CLOSE_RANGE_CLOEXEC
+#undef CLOSE_RANGE_EXCEPT
enum close_range_flags {
/* Unshare the file descriptor table before closing file descriptors. */
@@ -15,11 +16,15 @@ enum close_range_flags {
/* Set the FD_CLOEXEC bit instead of closing the file descriptor. */
CLOSE_RANGE_CLOEXEC = (1U << 2),
+
+ /* Act on every file descriptor outside of the given range instead. */
+ CLOSE_RANGE_EXCEPT = (1U << 3),
};
/* Keep #ifdef working and let glibc skip its own definitions. */
#define CLOSE_RANGE_UNSHARE CLOSE_RANGE_UNSHARE
#define CLOSE_RANGE_CLOEXEC CLOSE_RANGE_CLOEXEC
+#define CLOSE_RANGE_EXCEPT CLOSE_RANGE_EXCEPT
#endif /* _UAPI_LINUX_CLOSE_RANGE_H */
--
2.53.0
next prev parent reply other threads:[~2026-09-21 14:16 UTC|newest]
Thread overview: 14+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-21 14:15 [PATCH 00/10] files,close_range: add CLOSE_RANGE_{CLOEXEC_ONLY,EXCEPT} Christian Brauner
2026-09-21 14:15 ` [PATCH 01/10] file: let dup_fd() leave the punched hole behind Christian Brauner
2026-09-21 14:15 ` [PATCH 02/10] selftests/core: test the hole CLOSE_RANGE_UNSHARE leaves behind Christian Brauner
2026-09-21 14:15 ` [PATCH 03/10] file: rename dup_fd()'s punch_hole to range Christian Brauner
2026-09-21 14:15 ` [PATCH 04/10] file: let dup_fd() drop everything outside of the range Christian Brauner
2026-09-21 14:15 ` [PATCH 05/10] close_range: turn the flags into an enum Christian Brauner
2026-09-21 14:15 ` Christian Brauner [this message]
2026-09-21 14:15 ` [PATCH 07/10] selftests/core: test CLOSE_RANGE_EXCEPT Christian Brauner
2026-09-21 14:15 ` [PATCH 08/10] file: let dup_fd() drop only close-on-exec descriptors Christian Brauner
2026-09-21 16:24 ` Jann Horn
2026-09-25 14:57 ` Christian Brauner
2026-09-21 14:15 ` [PATCH 09/10] file: add CLOSE_RANGE_CLOEXEC_ONLY Christian Brauner
2026-09-21 14:15 ` [PATCH 10/10] selftests/core: test CLOSE_RANGE_CLOEXEC_ONLY Christian Brauner
2026-09-21 16:34 ` [PATCH 00/10] files,close_range: add CLOSE_RANGE_{CLOEXEC_ONLY,EXCEPT} Jann Horn
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260921-work-file-close_range_except-v1-6-c20d0b49270d@kernel.org \
--to=brauner@kernel.org \
--cc=jack@suse.cz \
--cc=jannh@google.com \
--cc=jlayton@kernel.org \
--cc=linux-fsdevel@vger.kernel.org \
--cc=neil@brown.name \
--cc=oleg@redhat.com \
--cc=viro@zeniv.linux.org.uk \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox