From: Christian Brauner <brauner@kernel.org>
To: Jann Horn <jannh@google.com>,
linux-fsdevel@vger.kernel.org, Oleg Nesterov <oleg@redhat.com>
Cc: Alexander Viro <viro@zeniv.linux.org.uk>, Jan Kara <jack@suse.cz>,
Neil Brown <neil@brown.name>, Jeff Layton <jlayton@kernel.org>,
"Christian Brauner (Amutable)" <brauner@kernel.org>
Subject: [PATCH 07/10] selftests/core: test CLOSE_RANGE_EXCEPT
Date: Mon, 21 Sep 2026 16:15:35 +0200 [thread overview]
Message-ID: <20260921-work-file-close_range_except-v1-7-c20d0b49270d@kernel.org> (raw)
In-Reply-To: <20260921-work-file-close_range_except-v1-0-c20d0b49270d@kernel.org>
Cover the inverted range:
- with plain close the window stays and everything outside of it goes,
stdio included
- a window at the top, one that cannot hold a descriptor and one of a
single descriptor keep just what they name, and so does the unshare
form on a table that is not shared
- with CLOSE_RANGE_CLOEXEC everything outside of the window is marked
and the window is not, in place and in a clone, for a
window in the middle, at the bottom, at the top and above the table
- with CLOSE_RANGE_UNSHARE the table the child cloned from is untouched,
a window near the top of the table comes back whole, one at the bottom
keeps stdio, one that cannot hold a descriptor keeps nothing, and the
slots left behind are handed out again from the bottom
- the bounds are checked before the range is turned around
The extra cases came out of walking the window positions that
__range_close(), __range_cloexec() and dup_fd() tell apart: at the
bottom, in the middle, at the top, above the table, a single descriptor
and none, on a shared and on a private table.
Signed-off-by: Christian Brauner (Amutable) <brauner@kernel.org>
---
tools/testing/selftests/core/close_range_test.c | 426 ++++++++++++++++++++++++
1 file changed, 426 insertions(+)
diff --git a/tools/testing/selftests/core/close_range_test.c b/tools/testing/selftests/core/close_range_test.c
index afae3462502d..caeb2f1ea800 100644
--- a/tools/testing/selftests/core/close_range_test.c
+++ b/tools/testing/selftests/core/close_range_test.c
@@ -36,6 +36,14 @@ static inline int sys_close_range(unsigned int fd, unsigned int max_fd,
return syscall(__NR_close_range, fd, max_fd, flags);
}
+static void clear_cloexec(const int *fds, size_t n)
+{
+ size_t i;
+
+ for (i = 0; i < n; i++)
+ fcntl(fds[i], F_SETFD, 0);
+}
+
TEST(core_close_range)
{
int i, ret;
@@ -637,6 +645,424 @@ TEST(close_range_cloexec_unshare_syzbot)
EXPECT_EQ(close(fd3), 0);
}
+TEST(close_range_except)
+{
+ int i, ret, status;
+ pid_t pid;
+ int open_fds[101];
+ struct __clone_args args = {
+ .exit_signal = SIGCHLD,
+ };
+
+ for (i = 0; i < ARRAY_SIZE(open_fds); i++) {
+ int fd;
+
+ fd = open("/dev/null", O_RDONLY);
+ ASSERT_GE(fd, 0) {
+ if (errno == ENOENT)
+ SKIP(return, "Skipping test since /dev/null does not exist");
+ }
+
+ open_fds[i] = fd;
+ }
+
+ /* A range covering everything keeps everything. */
+ ret = sys_close_range(0, UINT_MAX, CLOSE_RANGE_EXCEPT);
+ if (ret < 0) {
+ if (errno == ENOSYS)
+ SKIP(return, "close_range() syscall not supported");
+ if (errno == EINVAL)
+ SKIP(return, "close_range() doesn't support CLOSE_RANGE_EXCEPT");
+ }
+ ASSERT_EQ(0, ret);
+
+ for (i = 0; i < ARRAY_SIZE(open_fds); i++)
+ EXPECT_NE(-1, fcntl(open_fds[i], F_GETFD));
+
+ /* The bounds are checked before the range is turned around. */
+ EXPECT_EQ(-1, sys_close_range(open_fds[20], open_fds[10],
+ CLOSE_RANGE_EXCEPT));
+ EXPECT_EQ(EINVAL, errno);
+
+ /* Everything above open_fds[50] goes. */
+ ASSERT_EQ(0, sys_close_range(0, open_fds[50], CLOSE_RANGE_EXCEPT));
+
+ for (i = 0; i < ARRAY_SIZE(open_fds); i++)
+ EXPECT_EQ(i <= 50, fcntl(open_fds[i], F_GETFD) != -1);
+
+ /* A window in the middle takes stdio with it, so do that in a fork. */
+ pid = sys_clone3(&args, sizeof(args));
+ ASSERT_GE(pid, 0);
+
+ if (pid == 0) {
+ ret = sys_close_range(open_fds[10], open_fds[20],
+ CLOSE_RANGE_EXCEPT);
+ if (ret)
+ exit(EXIT_FAILURE);
+
+ for (i = 0; i <= 50; i++) {
+ bool kept = i >= 10 && i <= 20;
+
+ if (kept != (fcntl(open_fds[i], F_GETFD) != -1))
+ exit(EXIT_FAILURE);
+ }
+
+ if (fcntl(STDERR_FILENO, F_GETFD) != -1)
+ exit(EXIT_FAILURE);
+
+ exit(EXIT_SUCCESS);
+ }
+
+ EXPECT_EQ(waitpid(pid, &status, 0), pid);
+ EXPECT_EQ(true, WIFEXITED(status));
+ EXPECT_EQ(0, WEXITSTATUS(status));
+
+ /* The fork had a table of its own. */
+ for (i = 0; i <= 50; i++)
+ EXPECT_NE(-1, fcntl(open_fds[i], F_GETFD));
+}
+
+TEST(close_range_except_cloexec)
+{
+ int i, ret;
+ int open_fds[101];
+
+ for (i = 0; i < ARRAY_SIZE(open_fds); i++) {
+ int fd;
+
+ fd = open("/dev/null", O_RDONLY);
+ ASSERT_GE(fd, 0) {
+ if (errno == ENOENT)
+ SKIP(return, "Skipping test since /dev/null does not exist");
+ }
+
+ open_fds[i] = fd;
+ }
+
+ ret = sys_close_range(open_fds[10], open_fds[20],
+ CLOSE_RANGE_CLOEXEC | CLOSE_RANGE_EXCEPT);
+ if (ret < 0) {
+ if (errno == ENOSYS)
+ SKIP(return, "close_range() syscall not supported");
+ if (errno == EINVAL)
+ SKIP(return, "close_range() doesn't support CLOSE_RANGE_EXCEPT");
+ }
+ ASSERT_EQ(0, ret);
+
+ for (i = 0; i < ARRAY_SIZE(open_fds); i++) {
+ bool inside = i >= 10 && i <= 20;
+ int flags = fcntl(open_fds[i], F_GETFD);
+
+ EXPECT_NE(-1, flags);
+ EXPECT_EQ(inside ? 0 : FD_CLOEXEC, flags & FD_CLOEXEC);
+ }
+
+ /* stdio sits outside of the window too. */
+ EXPECT_EQ(FD_CLOEXEC, fcntl(STDERR_FILENO, F_GETFD) & FD_CLOEXEC);
+
+ /* A window that starts at 0 marks only what lies above it. */
+ clear_cloexec(open_fds, ARRAY_SIZE(open_fds));
+ ASSERT_EQ(0, fcntl(STDERR_FILENO, F_SETFD, 0));
+ ASSERT_EQ(0, sys_close_range(0, open_fds[20],
+ CLOSE_RANGE_CLOEXEC | CLOSE_RANGE_EXCEPT));
+
+ for (i = 0; i < ARRAY_SIZE(open_fds); i++)
+ EXPECT_EQ(i <= 20 ? 0 : FD_CLOEXEC,
+ fcntl(open_fds[i], F_GETFD) & FD_CLOEXEC);
+ EXPECT_EQ(0, fcntl(STDERR_FILENO, F_GETFD) & FD_CLOEXEC);
+
+ /* One at the top marks only what lies below it, stdio included. */
+ clear_cloexec(open_fds, ARRAY_SIZE(open_fds));
+ ASSERT_EQ(0, sys_close_range(open_fds[80], UINT_MAX,
+ CLOSE_RANGE_CLOEXEC | CLOSE_RANGE_EXCEPT));
+
+ for (i = 0; i < ARRAY_SIZE(open_fds); i++)
+ EXPECT_EQ(i < 80 ? FD_CLOEXEC : 0,
+ fcntl(open_fds[i], F_GETFD) & FD_CLOEXEC);
+ EXPECT_EQ(FD_CLOEXEC, fcntl(STDERR_FILENO, F_GETFD) & FD_CLOEXEC);
+
+ /* One that cannot hold a descriptor marks everything. */
+ clear_cloexec(open_fds, ARRAY_SIZE(open_fds));
+ ASSERT_EQ(0, fcntl(STDERR_FILENO, F_SETFD, 0));
+ ASSERT_EQ(0, sys_close_range(UINT_MAX, UINT_MAX,
+ CLOSE_RANGE_CLOEXEC | CLOSE_RANGE_EXCEPT));
+
+ for (i = 0; i < ARRAY_SIZE(open_fds); i++)
+ EXPECT_EQ(FD_CLOEXEC, fcntl(open_fds[i], F_GETFD) & FD_CLOEXEC);
+ EXPECT_EQ(FD_CLOEXEC, fcntl(STDERR_FILENO, F_GETFD) & FD_CLOEXEC);
+}
+
+TEST(close_range_except_cloexec_unshare)
+{
+ int i, ret, status;
+ pid_t pid;
+ int open_fds[101];
+ struct __clone_args args = {
+ .flags = CLONE_FILES,
+ .exit_signal = SIGCHLD,
+ };
+
+ for (i = 0; i < ARRAY_SIZE(open_fds); i++) {
+ int fd;
+
+ fd = open("/dev/null", O_RDONLY);
+ ASSERT_GE(fd, 0) {
+ if (errno == ENOENT)
+ SKIP(return, "Skipping test since /dev/null does not exist");
+ }
+
+ open_fds[i] = fd;
+ }
+
+ /* A range covering everything marks nothing. */
+ ret = sys_close_range(0, UINT_MAX,
+ CLOSE_RANGE_CLOEXEC | CLOSE_RANGE_EXCEPT);
+ if (ret < 0) {
+ if (errno == ENOSYS)
+ SKIP(return, "close_range() syscall not supported");
+ if (errno == EINVAL)
+ SKIP(return, "close_range() doesn't support CLOSE_RANGE_EXCEPT");
+ }
+ ASSERT_EQ(0, ret);
+
+ for (i = 0; i < ARRAY_SIZE(open_fds); i++)
+ EXPECT_EQ(0, fcntl(open_fds[i], F_GETFD) & FD_CLOEXEC);
+
+ pid = sys_clone3(&args, sizeof(args));
+ ASSERT_GE(pid, 0);
+
+ if (pid == 0) {
+ ret = sys_close_range(open_fds[10], open_fds[20],
+ CLOSE_RANGE_UNSHARE | CLOSE_RANGE_CLOEXEC |
+ CLOSE_RANGE_EXCEPT);
+ if (ret)
+ exit(EXIT_FAILURE);
+
+ for (i = 0; i < ARRAY_SIZE(open_fds); i++) {
+ bool inside = i >= 10 && i <= 20;
+ int flags = fcntl(open_fds[i], F_GETFD);
+
+ if (flags == -1)
+ exit(EXIT_FAILURE);
+ if ((flags & FD_CLOEXEC) != (inside ? 0 : FD_CLOEXEC))
+ exit(EXIT_FAILURE);
+ }
+
+ exit(EXIT_SUCCESS);
+ }
+
+ EXPECT_EQ(waitpid(pid, &status, 0), pid);
+ EXPECT_EQ(true, WIFEXITED(status));
+ EXPECT_EQ(0, WEXITSTATUS(status));
+
+ /* The shared table the child unshared from is untouched. */
+ for (i = 0; i < ARRAY_SIZE(open_fds); i++)
+ EXPECT_EQ(0, fcntl(open_fds[i], F_GETFD) & FD_CLOEXEC);
+}
+
+TEST(close_range_except_bounds)
+{
+ int i, c, ret, status;
+ pid_t pid;
+ int open_fds[101];
+ struct __clone_args args = {
+ .exit_signal = SIGCHLD,
+ };
+
+ for (i = 0; i < ARRAY_SIZE(open_fds); i++) {
+ int fd;
+
+ fd = open("/dev/null", O_RDONLY);
+ ASSERT_GE(fd, 0) {
+ if (errno == ENOENT)
+ SKIP(return, "Skipping test since /dev/null does not exist");
+ }
+
+ open_fds[i] = fd;
+ }
+
+ /* A range covering everything keeps everything. */
+ ret = sys_close_range(0, UINT_MAX, CLOSE_RANGE_EXCEPT);
+ if (ret < 0) {
+ if (errno == ENOSYS)
+ SKIP(return, "close_range() syscall not supported");
+ if (errno == EINVAL)
+ SKIP(return, "close_range() doesn't support CLOSE_RANGE_EXCEPT");
+ }
+ ASSERT_EQ(0, ret);
+
+ struct {
+ unsigned int fd, max_fd, flags;
+ } cases[] = {
+ /* A window at the top drops everything below it. */
+ { open_fds[50], UINT_MAX, CLOSE_RANGE_EXCEPT },
+ /* One that cannot hold a descriptor keeps nothing. */
+ { UINT_MAX, UINT_MAX, CLOSE_RANGE_EXCEPT },
+ /* One of a single descriptor keeps just that. */
+ { open_fds[30], open_fds[30], CLOSE_RANGE_EXCEPT },
+ /* The unshare form on a table that is not shared acts in place. */
+ { open_fds[10], open_fds[20],
+ CLOSE_RANGE_UNSHARE | CLOSE_RANGE_EXCEPT },
+ };
+
+ /* Each of them takes stdio with it, so do that in a fork. */
+ for (c = 0; c < ARRAY_SIZE(cases); c++) {
+ pid = sys_clone3(&args, sizeof(args));
+ ASSERT_GE(pid, 0);
+
+ if (pid == 0) {
+ ret = sys_close_range(cases[c].fd, cases[c].max_fd,
+ cases[c].flags);
+ if (ret)
+ exit(EXIT_FAILURE);
+
+ for (i = 0; i < ARRAY_SIZE(open_fds); i++) {
+ unsigned int fd = open_fds[i];
+ bool kept = fd >= cases[c].fd &&
+ fd <= cases[c].max_fd;
+
+ if (kept != (fcntl(fd, F_GETFD) != -1))
+ exit(EXIT_FAILURE);
+ }
+
+ if (fcntl(STDERR_FILENO, F_GETFD) != -1)
+ exit(EXIT_FAILURE);
+
+ exit(EXIT_SUCCESS);
+ }
+
+ EXPECT_EQ(waitpid(pid, &status, 0), pid);
+ EXPECT_EQ(true, WIFEXITED(status));
+ EXPECT_EQ(0, WEXITSTATUS(status));
+ }
+
+ /* Each fork had a table of its own. */
+ for (i = 0; i < ARRAY_SIZE(open_fds); i++)
+ EXPECT_NE(-1, fcntl(open_fds[i], F_GETFD));
+}
+
+TEST(close_range_except_unshare)
+{
+ int i, ret, status;
+ pid_t pid;
+ int open_fds[200];
+ struct __clone_args args = {
+ .flags = CLONE_FILES,
+ .exit_signal = SIGCHLD,
+ };
+
+ for (i = 0; i < ARRAY_SIZE(open_fds); i++) {
+ int fd;
+
+ /* Odd slots are close-on-exec, which makes no difference here. */
+ fd = open("/dev/null", O_RDONLY | (i % 2 ? O_CLOEXEC : 0));
+ ASSERT_GE(fd, 0) {
+ if (errno == ENOENT)
+ SKIP(return, "Skipping test since /dev/null does not exist");
+ }
+
+ open_fds[i] = fd;
+ }
+
+ /* A range covering everything keeps everything. */
+ ret = sys_close_range(0, UINT_MAX, CLOSE_RANGE_EXCEPT);
+ if (ret < 0) {
+ if (errno == ENOSYS)
+ SKIP(return, "close_range() syscall not supported");
+ if (errno == EINVAL)
+ SKIP(return, "close_range() doesn't support CLOSE_RANGE_EXCEPT");
+ }
+ ASSERT_EQ(0, ret);
+
+ pid = sys_clone3(&args, sizeof(args));
+ ASSERT_GE(pid, 0);
+
+ if (pid == 0) {
+ /* The window sits near the top, so the clone is sized off its end. */
+ ret = sys_close_range(open_fds[150], open_fds[160],
+ CLOSE_RANGE_UNSHARE | CLOSE_RANGE_EXCEPT);
+ if (ret)
+ exit(EXIT_FAILURE);
+
+ for (i = 0; i < ARRAY_SIZE(open_fds); i++) {
+ bool kept = i >= 150 && i <= 160;
+
+ if (kept != (fcntl(open_fds[i], F_GETFD) != -1))
+ exit(EXIT_FAILURE);
+ }
+
+ if (fcntl(STDERR_FILENO, F_GETFD) != -1)
+ exit(EXIT_FAILURE);
+
+ /* What was left behind is handed out again, from the bottom. */
+ if (dup(open_fds[150]) != 0)
+ exit(EXIT_FAILURE);
+
+ exit(EXIT_SUCCESS);
+ }
+
+ EXPECT_EQ(waitpid(pid, &status, 0), pid);
+ EXPECT_EQ(true, WIFEXITED(status));
+ EXPECT_EQ(0, WEXITSTATUS(status));
+
+ /* A window at the bottom keeps just that, stdio included. */
+ pid = sys_clone3(&args, sizeof(args));
+ ASSERT_GE(pid, 0);
+
+ if (pid == 0) {
+ ret = sys_close_range(0, open_fds[10],
+ CLOSE_RANGE_UNSHARE | CLOSE_RANGE_EXCEPT);
+ if (ret)
+ exit(EXIT_FAILURE);
+
+ for (i = 0; i < ARRAY_SIZE(open_fds); i++) {
+ if ((i <= 10) != (fcntl(open_fds[i], F_GETFD) != -1))
+ exit(EXIT_FAILURE);
+ }
+
+ if (fcntl(STDERR_FILENO, F_GETFD) == -1)
+ exit(EXIT_FAILURE);
+
+ /* The first slot left behind is the next one handed out. */
+ if (dup(0) != open_fds[10] + 1)
+ exit(EXIT_FAILURE);
+
+ exit(EXIT_SUCCESS);
+ }
+
+ EXPECT_EQ(waitpid(pid, &status, 0), pid);
+ EXPECT_EQ(true, WIFEXITED(status));
+ EXPECT_EQ(0, WEXITSTATUS(status));
+
+ /* A window that cannot hold a descriptor keeps nothing. */
+ pid = sys_clone3(&args, sizeof(args));
+ ASSERT_GE(pid, 0);
+
+ if (pid == 0) {
+ ret = sys_close_range(UINT_MAX, UINT_MAX,
+ CLOSE_RANGE_UNSHARE | CLOSE_RANGE_EXCEPT);
+ if (ret)
+ exit(EXIT_FAILURE);
+
+ for (i = 0; i < ARRAY_SIZE(open_fds); i++)
+ if (fcntl(open_fds[i], F_GETFD) != -1)
+ exit(EXIT_FAILURE);
+
+ if (fcntl(STDERR_FILENO, F_GETFD) != -1)
+ exit(EXIT_FAILURE);
+
+ exit(EXIT_SUCCESS);
+ }
+
+ EXPECT_EQ(waitpid(pid, &status, 0), pid);
+ EXPECT_EQ(true, WIFEXITED(status));
+ EXPECT_EQ(0, WEXITSTATUS(status));
+
+ /* The shared table the child unshared from is untouched. */
+ for (i = 0; i < ARRAY_SIZE(open_fds); i++)
+ EXPECT_NE(-1, fcntl(open_fds[i], F_GETFD));
+}
+
TEST(close_range_bitmap_corruption)
{
pid_t pid;
--
2.53.0
next prev parent reply other threads:[~2026-09-21 14:16 UTC|newest]
Thread overview: 14+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-21 14:15 [PATCH 00/10] files,close_range: add CLOSE_RANGE_{CLOEXEC_ONLY,EXCEPT} Christian Brauner
2026-09-21 14:15 ` [PATCH 01/10] file: let dup_fd() leave the punched hole behind Christian Brauner
2026-09-21 14:15 ` [PATCH 02/10] selftests/core: test the hole CLOSE_RANGE_UNSHARE leaves behind Christian Brauner
2026-09-21 14:15 ` [PATCH 03/10] file: rename dup_fd()'s punch_hole to range Christian Brauner
2026-09-21 14:15 ` [PATCH 04/10] file: let dup_fd() drop everything outside of the range Christian Brauner
2026-09-21 14:15 ` [PATCH 05/10] close_range: turn the flags into an enum Christian Brauner
2026-09-21 14:15 ` [PATCH 06/10] file: add CLOSE_RANGE_EXCEPT Christian Brauner
2026-09-21 14:15 ` Christian Brauner [this message]
2026-09-21 14:15 ` [PATCH 08/10] file: let dup_fd() drop only close-on-exec descriptors Christian Brauner
2026-09-21 16:24 ` Jann Horn
2026-09-25 14:57 ` Christian Brauner
2026-09-21 14:15 ` [PATCH 09/10] file: add CLOSE_RANGE_CLOEXEC_ONLY Christian Brauner
2026-09-21 14:15 ` [PATCH 10/10] selftests/core: test CLOSE_RANGE_CLOEXEC_ONLY Christian Brauner
2026-09-21 16:34 ` [PATCH 00/10] files,close_range: add CLOSE_RANGE_{CLOEXEC_ONLY,EXCEPT} Jann Horn
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260921-work-file-close_range_except-v1-7-c20d0b49270d@kernel.org \
--to=brauner@kernel.org \
--cc=jack@suse.cz \
--cc=jannh@google.com \
--cc=jlayton@kernel.org \
--cc=linux-fsdevel@vger.kernel.org \
--cc=neil@brown.name \
--cc=oleg@redhat.com \
--cc=viro@zeniv.linux.org.uk \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox