From: Christian Brauner <brauner@kernel.org>
To: Jann Horn <jannh@google.com>,
linux-fsdevel@vger.kernel.org, Oleg Nesterov <oleg@redhat.com>
Cc: Alexander Viro <viro@zeniv.linux.org.uk>, Jan Kara <jack@suse.cz>,
Neil Brown <neil@brown.name>, Jeff Layton <jlayton@kernel.org>,
"Christian Brauner (Amutable)" <brauner@kernel.org>
Subject: [PATCH 01/10] file: let dup_fd() leave the punched hole behind
Date: Mon, 21 Sep 2026 16:15:29 +0200 [thread overview]
Message-ID: <20260921-work-file-close_range_except-v1-1-c20d0b49270d@kernel.org> (raw)
In-Reply-To: <20260921-work-file-close_range_except-v1-0-c20d0b49270d@kernel.org>
close_range(CLOSE_RANGE_UNSHARE) passed the range it is about to close
to dup_fd() so the clone is done without that range. This only works
when the last open descriptor falls into the range. A range in the
middle of the table is copied like everything else. That's wasteful.
Don't copy them. Descriptors in a skipped range stay open in the source
fdtable. They are never copied into the new table and no reference is
taken on them. Figuring out the size of the table follows the same rule.
That drops the special-case it has now.
This also means we stop calling ->flush() on fds from a table they
were never part of. With the range at the top of the table that was
already mostly the case.
Signed-off-by: Christian Brauner (Amutable) <brauner@kernel.org>
---
fs/file.c | 60 ++++++++++++++++++++++++++++++++++++++++++++++--------------
1 file changed, 46 insertions(+), 14 deletions(-)
diff --git a/fs/file.c b/fs/file.c
index 628ca07dc4b1..95dbdfee555a 100644
--- a/fs/file.c
+++ b/fs/file.c
@@ -352,27 +352,51 @@ static inline bool fd_is_open(unsigned int fd, const struct fdtable *fdt)
return test_bit(fd, fdt->open_fds);
}
+/* Bits of [range->from, range->to] that fall into word @i of a bitmap. */
+static unsigned long fd_range_word(struct fd_range *range, unsigned int i)
+{
+ unsigned int first = i * BITS_PER_LONG;
+ unsigned int last = first + BITS_PER_LONG - 1;
+
+ if (range->to < first || range->from > last)
+ return 0;
+ return GENMASK(min(range->to, last) - first,
+ max(range->from, first) - first);
+}
+
+/* Bits of word @i that dup_fd() leaves behind. */
+static unsigned long dup_fd_dropped_word(unsigned int i, struct fd_range *punch_hole)
+{
+ if (!punch_hole)
+ return 0;
+ return fd_range_word(punch_hole, i);
+}
+
/*
* Note that a sane fdtable size always has to be a multiple of
* BITS_PER_LONG, since we have bitmaps that are sized by this.
*
* punch_hole is optional - when close_range() is asked to unshare
- * and close, we don't need to copy descriptors in that range, so
- * a smaller cloned descriptor table might suffice if the last
- * currently opened descriptor falls into that range.
+ * and close, dup_fd() leaves the descriptors in that range behind,
+ * so the cloned table only has to reach the last open descriptor
+ * outside of it.
*/
static unsigned int sane_fdtable_size(struct fdtable *fdt, struct fd_range *punch_hole)
{
unsigned int last = find_last_bit(fdt->open_fds, fdt->max_fds);
+ unsigned int i;
if (last == fdt->max_fds)
return NR_OPEN_DEFAULT;
- if (punch_hole && punch_hole->to >= last && punch_hole->from <= last) {
- last = find_last_bit(fdt->open_fds, punch_hole->from);
- if (last == punch_hole->from)
- return NR_OPEN_DEFAULT;
+ /* Only words up to the last open descriptor can hold a kept one. */
+ i = last / BITS_PER_LONG + 1;
+ while (i--) {
+ unsigned long dropped = dup_fd_dropped_word(i, punch_hole);
+
+ if (fdt->open_fds[i] & ~dropped)
+ return (i + 1) * BITS_PER_LONG;
}
- return ALIGN(last + 1, BITS_PER_LONG);
+ return NR_OPEN_DEFAULT;
}
/*
@@ -384,7 +408,8 @@ struct files_struct *dup_fd(struct files_struct *oldf, struct fd_range *punch_ho
{
struct files_struct *newf;
struct file **old_fds, **new_fds;
- unsigned int open_files, i;
+ unsigned int open_files, fd;
+ unsigned long dropped = 0;
struct fdtable *old_fdt, *new_fdt;
newf = kmem_cache_alloc(files_cachep, GFP_KERNEL);
@@ -451,13 +476,18 @@ struct files_struct *dup_fd(struct files_struct *oldf, struct fd_range *punch_ho
*
* Instead of trying to placate userspace racing with itself, we
* ref the file if we see it and mark the fd slot as unused otherwise.
+ * Descriptors dup_fd() is asked to leave behind get the same treatment.
*/
- for (i = open_files; i != 0; i--) {
+ for (fd = 0; fd < open_files; fd++) {
struct file *f = rcu_dereference_raw(*old_fds++);
- if (f) {
+
+ if (!(fd % BITS_PER_LONG))
+ dropped = dup_fd_dropped_word(fd / BITS_PER_LONG, punch_hole);
+ if (f && !(dropped & BIT_MASK(fd))) {
get_file(f);
} else {
- __clear_open_fd(open_files - i, new_fdt);
+ f = NULL;
+ __clear_open_fd(fd, new_fdt);
}
rcu_assign_pointer(*new_fds++, f);
}
@@ -848,10 +878,12 @@ SYSCALL_DEFINE3(close_range, unsigned int, fd, unsigned int, max_fd,
swap(cur_fds, fds);
}
- if (flags & CLOSE_RANGE_CLOEXEC)
+ if (flags & CLOSE_RANGE_CLOEXEC) {
__range_cloexec(cur_fds, fd, max_fd);
- else
+ } else if (!fds) {
+ /* If we unshared, dup_fd() left the range behind already. */
__range_close(cur_fds, fd, max_fd);
+ }
if (fds) {
/*
--
2.53.0
next prev parent reply other threads:[~2026-09-21 14:15 UTC|newest]
Thread overview: 14+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-21 14:15 [PATCH 00/10] files,close_range: add CLOSE_RANGE_{CLOEXEC_ONLY,EXCEPT} Christian Brauner
2026-09-21 14:15 ` Christian Brauner [this message]
2026-09-21 14:15 ` [PATCH 02/10] selftests/core: test the hole CLOSE_RANGE_UNSHARE leaves behind Christian Brauner
2026-09-21 14:15 ` [PATCH 03/10] file: rename dup_fd()'s punch_hole to range Christian Brauner
2026-09-21 14:15 ` [PATCH 04/10] file: let dup_fd() drop everything outside of the range Christian Brauner
2026-09-21 14:15 ` [PATCH 05/10] close_range: turn the flags into an enum Christian Brauner
2026-09-21 14:15 ` [PATCH 06/10] file: add CLOSE_RANGE_EXCEPT Christian Brauner
2026-09-21 14:15 ` [PATCH 07/10] selftests/core: test CLOSE_RANGE_EXCEPT Christian Brauner
2026-09-21 14:15 ` [PATCH 08/10] file: let dup_fd() drop only close-on-exec descriptors Christian Brauner
2026-09-21 16:24 ` Jann Horn
2026-09-25 14:57 ` Christian Brauner
2026-09-21 14:15 ` [PATCH 09/10] file: add CLOSE_RANGE_CLOEXEC_ONLY Christian Brauner
2026-09-21 14:15 ` [PATCH 10/10] selftests/core: test CLOSE_RANGE_CLOEXEC_ONLY Christian Brauner
2026-09-21 16:34 ` [PATCH 00/10] files,close_range: add CLOSE_RANGE_{CLOEXEC_ONLY,EXCEPT} Jann Horn
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260921-work-file-close_range_except-v1-1-c20d0b49270d@kernel.org \
--to=brauner@kernel.org \
--cc=jack@suse.cz \
--cc=jannh@google.com \
--cc=jlayton@kernel.org \
--cc=linux-fsdevel@vger.kernel.org \
--cc=neil@brown.name \
--cc=oleg@redhat.com \
--cc=viro@zeniv.linux.org.uk \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox