Linux filesystem development
 help / color / mirror / Atom feed
From: Christian Brauner <brauner@kernel.org>
To: linux-fsdevel@vger.kernel.org
Cc: Linus Torvalds <torvalds@linux-foundation.org>,
	 Jann Horn <jannh@google.com>, Jan Kara <jack@suse.cz>,
	 Amir Goldstein <amir73il@gmail.com>,
	 Alexander Viro <viro@zeniv.linux.org.uk>,
	 "Christian Brauner (Amutable)" <brauner@kernel.org>
Subject: [PATCH 3/3] selftests/filesystems: test covered mounts
Date: Fri, 02 Oct 2026 16:14:29 +0200	[thread overview]
Message-ID: <20261002-work-mount-cover-v1-3-232a8f52b43c@kernel.org> (raw)
In-Reply-To: <20261002-work-mount-cover-v1-0-232a8f52b43c@kernel.org>

Test that mount cycles are resolved and test that mount covers behave as
expected.

Signed-off-by: Christian Brauner (Amutable) <brauner@kernel.org>
---
 .../selftests/filesystems/mount_cycle/.gitignore   |    3 +
 .../selftests/filesystems/mount_cycle/Makefile     |    7 +-
 .../selftests/filesystems/mount_cycle/config       |   39 +
 .../filesystems/mount_cycle/locked_handle_test.c   |  377 +++++
 .../filesystems/mount_cycle/loop_cycle_test.c      | 1542 ++++++++++++++++++++
 .../filesystems/mount_cycle/mount_cover_test.c     |  565 +++++++
 .../selftests/filesystems/mount_cycle/settings     |    1 +
 7 files changed, 2533 insertions(+), 1 deletion(-)

diff --git a/tools/testing/selftests/filesystems/mount_cycle/.gitignore b/tools/testing/selftests/filesystems/mount_cycle/.gitignore
index d11f5b720d5b..30f7dd071bcb 100644
--- a/tools/testing/selftests/filesystems/mount_cycle/.gitignore
+++ b/tools/testing/selftests/filesystems/mount_cycle/.gitignore
@@ -1,5 +1,8 @@
 # SPDX-License-Identifier: GPL-2.0-only
+loop_cycle_test
 unmounted_tree_test
 overmount_reparent_test
+mount_cover_test
+locked_handle_test
 nsfs_rbind_loop_test
 overmount_ns_file_test
diff --git a/tools/testing/selftests/filesystems/mount_cycle/Makefile b/tools/testing/selftests/filesystems/mount_cycle/Makefile
index 49a8402ca858..fbe8c5e19d42 100644
--- a/tools/testing/selftests/filesystems/mount_cycle/Makefile
+++ b/tools/testing/selftests/filesystems/mount_cycle/Makefile
@@ -1,7 +1,12 @@
 # SPDX-License-Identifier: GPL-2.0
-TEST_GEN_PROGS := unmounted_tree_test overmount_reparent_test
+TEST_GEN_PROGS := loop_cycle_test unmounted_tree_test overmount_reparent_test mount_cover_test locked_handle_test
 TEST_GEN_PROGS += nsfs_rbind_loop_test overmount_ns_file_test
 
 CFLAGS += -Wall -O2 -g $(KHDR_INCLUDES)
 
+LOCAL_HDRS += ../readdir_hold.h
+
 include ../../lib.mk
+
+$(OUTPUT)/locked_handle_test: LDLIBS += -pthread
+$(OUTPUT)/mount_cover_test: LDLIBS += -pthread
diff --git a/tools/testing/selftests/filesystems/mount_cycle/config b/tools/testing/selftests/filesystems/mount_cycle/config
new file mode 100644
index 000000000000..3bd5ce46af74
--- /dev/null
+++ b/tools/testing/selftests/filesystems/mount_cycle/config
@@ -0,0 +1,39 @@
+CONFIG_USER_NS=y
+CONFIG_TMPFS=y
+CONFIG_BLK_DEV_LOOP=y
+CONFIG_VFAT_FS=y
+CONFIG_MSDOS_FS=y
+CONFIG_NLS_CODEPAGE_437=y
+CONFIG_NLS_ISO8859_1=y
+CONFIG_MINIX_FS=y
+CONFIG_AUTOFS_FS=y
+CONFIG_ZRAM=y
+CONFIG_ZRAM_WRITEBACK=y
+CONFIG_KEYS=y
+CONFIG_ECRYPT_FS=y
+CONFIG_BINFMT_MISC=y
+CONFIG_FUSE_FS=y
+CONFIG_FUSE_PASSTHROUGH=y
+CONFIG_BLK_DEV_ZONED=y
+CONFIG_BLK_DEV_ZONED_LOOP=y
+CONFIG_CONFIGFS_FS=y
+CONFIG_USB_SUPPORT=y
+CONFIG_USB=y
+CONFIG_USB_STORAGE=y
+CONFIG_SCSI=y
+CONFIG_BLK_DEV_SD=y
+CONFIG_USB_GADGET=y
+CONFIG_USB_DUMMY_HCD=y
+CONFIG_USB_CONFIGFS=y
+CONFIG_USB_CONFIGFS_MASS_STORAGE=y
+CONFIG_MD=y
+CONFIG_BLK_DEV_MD=y
+CONFIG_MD_RAID1=y
+CONFIG_MD_BITMAP=y
+CONFIG_MD_BITMAP_FILE=y
+CONFIG_INOTIFY_USER=y
+CONFIG_DNOTIFY=y
+CONFIG_FANOTIFY=y
+CONFIG_FILE_LOCKING=y
+CONFIG_CRYPTO_AES=y
+CONFIG_USERFAULTFD=y
diff --git a/tools/testing/selftests/filesystems/mount_cycle/locked_handle_test.c b/tools/testing/selftests/filesystems/mount_cycle/locked_handle_test.c
new file mode 100644
index 000000000000..b21f43471e49
--- /dev/null
+++ b/tools/testing/selftests/filesystems/mount_cycle/locked_handle_test.c
@@ -0,0 +1,377 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * may_decode_fh() refuses a file handle below a mount with locked children
+ * to root in a user namespace. That has to hold while the mount is lazily
+ * unmounted and its children, the locked ones too, are taken off it.
+ */
+#define _GNU_SOURCE
+#include <errno.h>
+#include <fcntl.h>
+#include <pthread.h>
+#include <sched.h>
+#include <stdatomic.h>
+#include <stdbool.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <unistd.h>
+#include <sys/mount.h>
+#include <sys/prctl.h>
+#include <sys/stat.h>
+#include <sys/syscall.h>
+#include <sys/wait.h>
+
+#include "../../kselftest_harness.h"
+
+#ifndef OPEN_TREE_CLONE
+#define OPEN_TREE_CLONE		1
+#endif
+#ifndef OPEN_TREE_CLOEXEC
+#define OPEN_TREE_CLOEXEC	O_CLOEXEC
+#endif
+#ifndef AT_RECURSIVE
+#define AT_RECURSIVE		0x8000
+#endif
+
+#define DIR_LEN		64
+#define PATH_LEN	128
+
+#define ROUNDS		100	/* lazy umounts raced per test */
+#define RACERS		3	/* threads in open_by_handle_at() */
+#define EXTRA_MOUNTS	64	/* make umount_tree() hold mount_lock longer */
+#define MAX_DELAY_US	3000	/* before the umount */
+#define TAIL_US		2000	/* after it */
+
+struct handle {
+	struct file_handle fh;
+	unsigned char buf[MAX_HANDLE_SZ];
+};
+
+/* exit codes of the child */
+enum {
+	CHILD_OK,
+	CHILD_SETUP,
+	CHILD_DECODED,		/* decoded past a locked child */
+	CHILD_MOUNTED,		/* not refused while mounted */
+	CHILD_UNMOUNTED,	/* not refused once unmounted */
+	CHILD_ALLOWED,		/* refused where nothing is locked */
+	CHILD_NOUSERNS,		/* no user namespace to be had */
+};
+
+static int write_file(const char *path, const char *s)
+{
+	ssize_t n = -1;
+	int fd;
+
+	fd = open(path, O_WRONLY | O_CLOEXEC);
+	if (fd >= 0) {
+		n = write(fd, s, strlen(s));
+		close(fd);
+	}
+	return n == (ssize_t)strlen(s) ? 0 : -1;
+}
+
+/* Become root in a new user namespace with a private mount namespace. */
+static int enter_userns(void)
+{
+	uid_t uid = getuid();
+	gid_t gid = getgid();
+	char map[32];
+
+	prctl(PR_SET_DUMPABLE, 1);
+	/* EINVAL: no USER_NS, ENOSPC: user.max_user_namespaces is 0, EPERM: an LSM */
+	if (unshare(CLONE_NEWUSER | CLONE_NEWNS))
+		return errno == EINVAL || errno == ENOSPC || errno == EPERM ?
+		       CHILD_NOUSERNS : CHILD_SETUP;
+	if (write_file("/proc/self/setgroups", "deny") && errno != ENOENT)
+		return CHILD_SETUP;
+	snprintf(map, sizeof(map), "0 %d 1", uid);
+	if (write_file("/proc/self/uid_map", map))
+		return CHILD_SETUP;
+	snprintf(map, sizeof(map), "0 %d 1", gid);
+	if (write_file("/proc/self/gid_map", map))
+		return CHILD_SETUP;
+	if (setgid(0) || setuid(0))
+		return CHILD_SETUP;
+	if (mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL))
+		return CHILD_SETUP;
+	return CHILD_OK;
+}
+
+static int get_handle(const char *path, struct handle *h)
+{
+	int mntid;
+
+	h->fh.handle_bytes = MAX_HANDLE_SZ;
+	return name_to_handle_at(AT_FDCWD, path, &h->fh, &mntid, 0);
+}
+
+static int decode(int dfd, struct handle *h)
+{
+	return open_by_handle_at(dfd, &h->fh, O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+}
+
+struct race {
+	int dfd;
+	struct handle *h;
+	atomic_int go;
+	atomic_int stop;
+	atomic_int won;		/* the first decoded descriptor */
+};
+
+static void *racer(void *arg)
+{
+	struct race *r = arg;
+
+	while (!atomic_load(&r->go))
+		;
+	while (!atomic_load(&r->stop)) {
+		int none = -1;
+		int fd;
+
+		fd = decode(r->dfd, r->h);
+		if (fd < 0)
+			continue;
+		if (!atomic_compare_exchange_strong(&r->won, &none, fd))
+			close(fd);
+	}
+	return NULL;
+}
+
+/* stop the @n racers started so far and wait for them, they spin on @r */
+static void stop_racers(struct race *r, pthread_t *th, int n)
+{
+	int i;
+
+	atomic_store(&r->stop, 1);
+	atomic_store(&r->go, 1);	/* one still waiting for the start sees stop next */
+	for (i = 0; i < n; i++)
+		pthread_join(th[i], NULL);
+}
+
+/*
+ * Bind @h recursively at @w, or take a detached copy of it, and race
+ * open_by_handle_at() of @inner against the lazy umount. The decoded
+ * descriptor, if there was one, is left in @won.
+ */
+static int race_round(const char *h, const char *w, struct handle *inner,
+		      bool dissolve, int *won)
+{
+	struct race r = { .h = inner, .won = -1 };
+	pthread_t th[RACERS];
+	int treefd = -1, fd, i;
+
+	if (dissolve) {
+		treefd = syscall(__NR_open_tree, AT_FDCWD, h,
+				 OPEN_TREE_CLONE | AT_RECURSIVE | OPEN_TREE_CLOEXEC);
+		if (treefd < 0)
+			return CHILD_SETUP;
+		/* an ordinary descriptor on the copy, treefd's close dissolves it */
+		r.dfd = openat(treefd, ".", O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+	} else {
+		if (mount(h, w, NULL, MS_BIND | MS_REC, NULL))
+			return CHILD_SETUP;
+		r.dfd = open(w, O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+	}
+	if (r.dfd < 0)
+		return CHILD_SETUP;
+
+	fd = decode(r.dfd, inner);
+	if (fd >= 0 || errno != EPERM) {
+		if (fd >= 0)
+			close(fd);
+		return CHILD_MOUNTED;
+	}
+
+	for (i = 0; i < RACERS; i++) {
+		if (pthread_create(&th[i], NULL, racer, &r)) {
+			stop_racers(&r, th, i);
+			return CHILD_SETUP;
+		}
+	}
+	atomic_store(&r.go, 1);
+	usleep(rand() % MAX_DELAY_US);
+	if (dissolve)
+		close(treefd);
+	else if (umount2(w, MNT_DETACH)) {
+		stop_racers(&r, th, RACERS);
+		return CHILD_SETUP;
+	}
+	usleep(TAIL_US);
+	stop_racers(&r, th, RACERS);
+
+	fd = decode(r.dfd, inner);
+	if (fd >= 0) {
+		close(fd);
+		return CHILD_UNMOUNTED;
+	}
+	close(r.dfd);
+	*won = atomic_load(&r.won);
+	return CHILD_OK;
+}
+
+static int race_child(const char *h, const char *w, struct handle *inner,
+		      bool dissolve)
+{
+	int ret, won, i;
+
+	srand(getpid());
+	ret = enter_userns();
+	if (ret)
+		return ret;
+	for (i = 0; i < ROUNDS; i++) {
+		ret = race_round(h, w, inner, dissolve, &won);
+		if (ret)
+			return ret;
+		if (won >= 0)
+			return CHILD_DECODED;
+	}
+	return CHILD_OK;
+}
+
+/* a bind without locked children decodes while mounted and not after */
+static int allowed_child(const char *plain, const char *w, struct handle *h)
+{
+	int dfd, fd, ret;
+
+	ret = enter_userns();
+	if (ret)
+		return ret;
+	if (mount(plain, w, NULL, MS_BIND | MS_REC, NULL))
+		return CHILD_SETUP;
+	dfd = open(w, O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+	if (dfd < 0)
+		return CHILD_SETUP;
+	fd = decode(dfd, h);
+	if (fd < 0)
+		return CHILD_ALLOWED;
+	close(fd);
+	if (umount2(w, MNT_DETACH))
+		return CHILD_SETUP;
+	fd = decode(dfd, h);
+	if (fd >= 0) {
+		close(fd);
+		return CHILD_UNMOUNTED;
+	}
+	return errno == EPERM ? CHILD_OK : CHILD_UNMOUNTED;
+}
+
+FIXTURE(locked_handle) {
+	char base[DIR_LEN];
+	char h[PATH_LEN];		/* the tree with the covered directory */
+	char plain[PATH_LEN];		/* a tree with no mount in it */
+	char w[PATH_LEN];		/* where the child binds either */
+	struct handle inner;		/* h/top/secret/inner, covered */
+	struct handle uncovered;	/* plain/inner */
+};
+
+FIXTURE_SETUP(locked_handle)
+{
+	char p[PATH_LEN], e[PATH_LEN];
+	int i;
+
+	if (geteuid())
+		SKIP(return, "test requires root");
+
+	snprintf(self->base, sizeof(self->base), "/tmp/locked_handle.XXXXXX");
+	ASSERT_NE(mkdtemp(self->base), NULL);
+	ASSERT_EQ(unshare(CLONE_NEWNS), 0);
+	ASSERT_EQ(mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL), 0);
+	ASSERT_EQ(mount("tmpfs", self->base, "tmpfs", 0, NULL), 0);
+
+	snprintf(self->h, sizeof(self->h), "%s/h", self->base);
+	snprintf(self->plain, sizeof(self->plain), "%s/plain", self->base);
+	snprintf(self->w, sizeof(self->w), "%s/w", self->base);
+	ASSERT_EQ(mkdir(self->h, 0755), 0);
+	ASSERT_EQ(mkdir(self->plain, 0755), 0);
+	ASSERT_EQ(mkdir(self->w, 0755), 0);
+	snprintf(p, sizeof(p), "%s/plain/inner", self->base);
+	ASSERT_EQ(mkdir(p, 0755), 0);
+	ASSERT_EQ(get_handle(p, &self->uncovered), 0);
+
+	/* the handle is for this filesystem */
+	ASSERT_EQ(mount("tmpfs", self->h, "tmpfs", 0, NULL), 0);
+	snprintf(p, sizeof(p), "%s/h/top", self->base);
+	ASSERT_EQ(mkdir(p, 0755), 0);
+	snprintf(p, sizeof(p), "%s/h/top/secret", self->base);
+	ASSERT_EQ(mkdir(p, 0755), 0);
+	snprintf(p, sizeof(p), "%s/h/top/secret/inner", self->base);
+	ASSERT_EQ(mkdir(p, 0755), 0);
+	ASSERT_EQ(get_handle(p, &self->inner), 0);
+	snprintf(e, sizeof(e), "%s/h/empty", self->base);
+	ASSERT_EQ(mkdir(e, 0755), 0);
+
+	/* cover it, and some more so the umount takes longer */
+	snprintf(p, sizeof(p), "%s/h/top/secret", self->base);
+	ASSERT_EQ(mount("tmpfs", p, "tmpfs", 0, NULL), 0);
+	for (i = 0; i < EXTRA_MOUNTS; i++) {
+		snprintf(p, sizeof(p), "%s/h/c%d", self->base, i);
+		ASSERT_EQ(mkdir(p, 0755), 0);
+		ASSERT_EQ(mount(e, p, NULL, MS_BIND, NULL), 0);
+	}
+}
+
+FIXTURE_TEARDOWN(locked_handle)
+{
+	umount2(self->base, MNT_DETACH);
+	rmdir(self->base);
+}
+
+static int wait_child(pid_t pid)
+{
+	int status;
+
+	if (waitpid(pid, &status, 0) != pid || !WIFEXITED(status))
+		return -1;
+	return WEXITSTATUS(status);
+}
+
+TEST_F(locked_handle, lazy_umount)
+{
+	pid_t pid;
+	int ret;
+
+	pid = fork();
+	ASSERT_GE(pid, 0);
+	if (pid == 0)
+		_exit(race_child(self->h, self->w, &self->inner, false));
+	ret = wait_child(pid);
+	TH_LOG("child exit code %d", ret);
+	if (ret == CHILD_NOUSERNS)
+		SKIP(return, "no user namespaces");
+	EXPECT_EQ(ret, CHILD_OK);
+}
+
+TEST_F(locked_handle, dissolve)
+{
+	pid_t pid;
+	int ret;
+
+	pid = fork();
+	ASSERT_GE(pid, 0);
+	if (pid == 0)
+		_exit(race_child(self->h, self->w, &self->inner, true));
+	ret = wait_child(pid);
+	TH_LOG("child exit code %d", ret);
+	if (ret == CHILD_NOUSERNS)
+		SKIP(return, "no user namespaces");
+	EXPECT_EQ(ret, CHILD_OK);
+}
+
+TEST_F(locked_handle, allowed_use)
+{
+	pid_t pid;
+	int ret;
+
+	pid = fork();
+	ASSERT_GE(pid, 0);
+	if (pid == 0)
+		_exit(allowed_child(self->plain, self->w, &self->uncovered));
+	ret = wait_child(pid);
+	TH_LOG("child exit code %d", ret);
+	if (ret == CHILD_NOUSERNS)
+		SKIP(return, "no user namespaces");
+	EXPECT_EQ(ret, CHILD_OK);
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/filesystems/mount_cycle/loop_cycle_test.c b/tools/testing/selftests/filesystems/mount_cycle/loop_cycle_test.c
new file mode 100644
index 000000000000..6b4f5304c2e4
--- /dev/null
+++ b/tools/testing/selftests/filesystems/mount_cycle/loop_cycle_test.c
@@ -0,0 +1,1542 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * A mount that another namespace's rmdir detached, or that went down with
+ * the detached tree it was in when the tree's last fd was closed, keeps
+ * its submounts connected, and a connected submount is put by its
+ * parent's final mntput(). A submount whose filesystem keeps a file open
+ * on the parent then holds the parent's count above zero for good:
+ * nothing in userspace refers to either mount any more and nothing can
+ * release them. A loop device is the simplest such filesystem, its
+ * backing file sits on the parent.
+ */
+#define _GNU_SOURCE
+#include <dirent.h>
+#include <errno.h>
+#include <fcntl.h>
+#include <sched.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <sys/ioctl.h>
+#include <sys/mount.h>
+#include <sys/stat.h>
+#include <stdbool.h>
+#include <sys/wait.h>
+#include <unistd.h>
+#include <linux/fuse.h>
+#include <linux/keyctl.h>
+#include <linux/major.h>
+#include <linux/raid/md_u.h>
+#include <linux/raid/md_p.h>
+#include <linux/loop.h>
+#include <linux/magic.h>
+#include <sys/syscall.h>
+#include <sys/sysmacros.h>
+#include <sys/uio.h>
+
+#include "../wrappers.h"
+#include "../../kselftest_harness.h"
+
+#define IMAGE_SIZE	(1440 * 1024)
+#define SECTOR		512
+
+/*
+ * The tmpfs of the fixture. A directory of its own from mkdtemp(), so that
+ * the test needs no writable root directory and two instances don't take
+ * each other's mounts down.
+ */
+static char base[64];
+
+/* The configfs directory of the gadget, named after the test process. */
+static char gadget[96];
+
+/* A path below the base directory. Eight of them can be in use at a time. */
+static const char *at(const char *rel)
+{
+	static char buf[8][PATH_MAX];
+	static unsigned int next;
+	char *p = buf[next++ % 8];
+
+	snprintf(p, PATH_MAX, "%s/%s", base, rel);
+	return p;
+}
+
+/* A path below the gadget's configfs directory. */
+static const char *gat(const char *rel)
+{
+	static char buf[4][PATH_MAX];
+	static unsigned int next;
+	char *p = buf[next++ % 4];
+
+	snprintf(p, PATH_MAX, "%s%s", gadget, rel);
+	return p;
+}
+
+/* The loop devices this process has bound, to release them on every path. */
+#define MAX_LOOPS	4
+static int bound[MAX_LOOPS];
+static int nr_bound;
+
+/* What a child tells its parent: the loop devices it has bound, -1 for none. */
+struct report {
+	int n[2];
+};
+
+/* A blank FAT12 floppy image: boot sector, two FATs, an empty root directory. */
+static int write_fat12(int fd)
+{
+	struct stat st;
+	unsigned char sector[SECTOR] = {
+		0xeb, 0x3c, 0x90, 'M', 'S', 'W', 'I', 'N', '4', '.', '1',
+		[11] = 0x00, 0x02,	/* bytes per sector: 512 */
+		[13] = 1,		/* sectors per cluster */
+		[14] = 1, 0,		/* reserved sectors */
+		[16] = 2,		/* FATs */
+		[17] = 0xe0, 0x00,	/* root directory entries: 224 */
+		[19] = 0x40, 0x0b,	/* total sectors: 2880 */
+		[21] = 0xf0,		/* media descriptor */
+		[22] = 9, 0,		/* sectors per FAT */
+		[24] = 18, 0,		/* sectors per track */
+		[26] = 2, 0,		/* heads */
+		[38] = 0x29,		/* extended boot signature */
+		[39] = 0x12, 0x34, 0x56, 0x78,
+		[43] = 'N', 'O', ' ', 'N', 'A', 'M', 'E', ' ', ' ', ' ', ' ',
+		[54] = 'F', 'A', 'T', '1', '2', ' ', ' ', ' ',
+		[510] = 0x55, 0xaa,
+	};
+	unsigned char fat[SECTOR] = { 0xf0, 0xff, 0xff };
+
+	if (pwrite(fd, sector, SECTOR, 0) != SECTOR)
+		return -1;
+	/* the first FAT and the second one, one sector each is enough */
+	if (pwrite(fd, fat, SECTOR, 1 * SECTOR) != SECTOR ||
+	    pwrite(fd, fat, SECTOR, 10 * SECTOR) != SECTOR)
+		return -1;
+	if (fstat(fd, &st) || S_ISBLK(st.st_mode))
+		return 0;
+	return ftruncate(fd, IMAGE_SIZE);
+}
+
+/*
+ * A blank FAT12 image with 4 KiB sectors and @sectors of them, for a device
+ * with 4 KiB logical blocks (zram) or for a bigger image than a floppy.
+ */
+static int write_fat12_4k(int fd, unsigned int sectors)
+{
+	unsigned int fat_sectors = (sectors * 3 / 2 + 4095) / 4096;
+	unsigned char sector[4096] = {
+		0xeb, 0x3c, 0x90, 'M', 'S', 'W', 'I', 'N', '4', '.', '1',
+		[11] = 0x00, 0x10,	/* bytes per sector: 4096 */
+		[13] = 1,		/* sectors per cluster */
+		[14] = 1, 0,		/* reserved sectors */
+		[16] = 2,		/* FATs */
+		[17] = 128, 0,		/* root directory entries: one sector */
+		[19] = sectors & 0xff, sectors >> 8,
+		[21] = 0xf8,		/* media descriptor */
+		[22] = fat_sectors, 0,
+		[24] = 63, 0,		/* sectors per track */
+		[26] = 255, 0,		/* heads */
+		[38] = 0x29,		/* extended boot signature */
+		[39] = 0x12, 0x34, 0x56, 0x78,
+		[43] = 'N', 'O', ' ', 'N', 'A', 'M', 'E', ' ', ' ', ' ', ' ',
+		[54] = 'F', 'A', 'T', '1', '2', ' ', ' ', ' ',
+		[510] = 0x55, 0xaa,
+	};
+	unsigned char fat[4096] = { 0xf8, 0xff, 0xff };
+	struct stat st;
+
+	if (pwrite(fd, sector, sizeof(sector), 0) != sizeof(sector))
+		return -1;
+	if (pwrite(fd, fat, sizeof(fat), 1 * 4096) != sizeof(fat) ||
+	    pwrite(fd, fat, sizeof(fat), (1 + fat_sectors) * 4096) != sizeof(fat))
+		return -1;
+	if (fstat(fd, &st) || S_ISBLK(st.st_mode))
+		return 0;
+	return ftruncate(fd, (off_t)sectors * 4096);
+}
+
+#define MINIX_BLOCK	1024
+#define MINIX_BLOCKS	4096			/* a 4 MiB image */
+#define MINIX_INODES	512
+#define MINIX_ITABLE	(MINIX_INODES * 32 / MINIX_BLOCK)
+#define MINIX_FIRSTDATA	(2 + 1 + 1 + MINIX_ITABLE)	/* boot, super, imap, zmap, inodes */
+
+/*
+ * A blank minix v1 image, for the holders that need a FIFO or a device
+ * node on the dying mount, which vfat can't hold. Superblock in block 1,
+ * one block each for the inode and zone bitmaps, the inode table, and
+ * the root directory in the first data zone.
+ */
+static int write_minix(int fd)
+{
+	struct {
+		__u16 s_ninodes, s_nzones, s_imap_blocks, s_zmap_blocks;
+		__u16 s_firstdatazone, s_log_zone_size;
+		__u32 s_max_size;
+		__u16 s_magic, s_state;
+	} sb = {
+		.s_ninodes = MINIX_INODES,
+		.s_nzones = MINIX_BLOCKS,
+		.s_imap_blocks = 1,
+		.s_zmap_blocks = 1,
+		.s_firstdatazone = MINIX_FIRSTDATA,
+		.s_max_size = (7 + 512 + 512 * 512) * MINIX_BLOCK,
+		.s_magic = MINIX_SUPER_MAGIC,
+		.s_state = 1,		/* MINIX_VALID_FS */
+	};
+	struct {
+		__u16 i_mode, i_uid;
+		__u32 i_size, i_time;
+		__u8 i_gid, i_nlinks;
+		__u16 i_zone[9];
+	} root = {
+		.i_mode = S_IFDIR | 0755,
+		.i_size = 2 * 16,
+		.i_nlinks = 2,
+		.i_zone = { MINIX_FIRSTDATA },
+	};
+	unsigned char imap[MINIX_BLOCK], zmap[MINIX_BLOCK], dir[MINIX_BLOCK] = {};
+	int i;
+
+	/* bit 0 is reserved in both maps, the root inode and its zone are in use */
+	memset(imap, 0xff, sizeof(imap));
+	for (i = 2; i <= MINIX_INODES; i++)
+		imap[i / 8] &= ~(1 << (i % 8));
+	memset(zmap, 0xff, sizeof(zmap));
+	for (i = 2; i <= MINIX_BLOCKS - MINIX_FIRSTDATA; i++)
+		zmap[i / 8] &= ~(1 << (i % 8));
+	dir[0] = 1;
+	dir[2] = '.';
+	dir[16] = 1;
+	dir[18] = '.';
+	dir[19] = '.';
+
+	if (pwrite(fd, &sb, sizeof(sb), 1 * MINIX_BLOCK) != sizeof(sb) ||
+	    pwrite(fd, imap, sizeof(imap), 2 * MINIX_BLOCK) != sizeof(imap) ||
+	    pwrite(fd, zmap, sizeof(zmap), 3 * MINIX_BLOCK) != sizeof(zmap) ||
+	    pwrite(fd, &root, sizeof(root), 4 * MINIX_BLOCK) != sizeof(root) ||
+	    pwrite(fd, dir, sizeof(dir), MINIX_FIRSTDATA * MINIX_BLOCK) != sizeof(dir))
+		return -1;
+	return ftruncate(fd, (off_t)MINIX_BLOCKS * MINIX_BLOCK);
+}
+
+static int read_sysfs(const char *path, char *buf, size_t size)
+{
+	ssize_t n;
+	int fd;
+
+	fd = open(path, O_RDONLY);
+	if (fd < 0)
+		return -1;
+	n = read(fd, buf, size - 1);
+	close(fd);
+	if (n < 0)
+		return -1;
+	buf[n] = '\0';
+	return 0;
+}
+
+/* Is @dev the mount source of this line of mountinfo? loop1 is not loop10. */
+static bool line_has_source(const char *line, const char *dev)
+{
+	const char *sep, *src, *end;
+
+	sep = strstr(line, " - ");
+	if (!sep)
+		return false;
+	src = strchr(sep + 3, ' ');	/* skip the filesystem type */
+	if (!src)
+		return false;
+	src++;
+	end = strchr(src, ' ');
+	if (!end)
+		return false;
+	return (size_t)(end - src) == strlen(dev) && !strncmp(src, dev, end - src);
+}
+
+/*
+ * Does a mount namespace of a process that /proc shows have a mount of @dev?
+ * The mounts under test are in no namespace at this point on any kernel, so
+ * this only checks that the test got as far as it thinks.
+ */
+static bool mounted_anywhere(const char *dev)
+{
+	char path[PATH_MAX], line[4096];
+	struct dirent *de;
+	bool found = false;
+	DIR *proc;
+	FILE *f;
+
+	proc = opendir("/proc");
+	if (!proc)
+		return false;
+	while (!found && (de = readdir(proc))) {
+		if (de->d_name[0] < '0' || de->d_name[0] > '9')
+			continue;
+		snprintf(path, sizeof(path), "/proc/%s/mountinfo", de->d_name);
+		f = fopen(path, "re");
+		if (!f)
+			continue;
+		while (fgets(line, sizeof(line), f)) {
+			if (line_has_source(line, dev)) {
+				found = true;
+				break;
+			}
+		}
+		fclose(f);
+	}
+	closedir(proc);
+	return found;
+}
+
+/* Wait up to @ms milliseconds for the loop device to give up its backing file. */
+static bool loop_released(const char *sysfs, int ms)
+{
+	char buf[PATH_MAX];
+
+	for (; ms > 0; ms -= 100) {
+		if (read_sysfs(sysfs, buf, sizeof(buf)) < 0)
+			return errno == ENOENT;
+		usleep(100000);
+	}
+	return read_sysfs(sysfs, buf, sizeof(buf)) < 0 && errno == ENOENT;
+}
+
+/* Tell the loop device to give its file up, now or when its last user is gone. */
+static void loop_clear(int n)
+{
+	char dev[32];
+	int lfd;
+
+	if (n < 0)
+		return;
+	snprintf(dev, sizeof(dev), "/dev/loop%d", n);
+	lfd = open(dev, O_RDWR);
+	if (lfd < 0)
+		return;
+	ioctl(lfd, LOOP_CLR_FD);
+	close(lfd);
+}
+
+FIXTURE(loop_cycle) {
+	char dev[32];		/* the loop device the child set up */
+	char sysfs[64];		/* its backing_file attribute */
+	int loops[MAX_LOOPS];	/* every loop device the case has bound */
+	int nr_loops;
+	bool gadget;		/* the gadget's configfs directory exists */
+};
+
+static void remember_loop(FIXTURE_DATA(loop_cycle) *self, int n)
+{
+	if (n >= 0 && self->nr_loops < MAX_LOOPS)
+		self->loops[self->nr_loops++] = n;
+}
+
+FIXTURE_SETUP(loop_cycle)
+{
+	if (geteuid() != 0)
+		SKIP(return, "test requires CAP_SYS_ADMIN");
+	if (access("/dev/loop-control", R_OK | W_OK))
+		SKIP(return, "test requires loop devices");
+
+	ASSERT_EQ(unshare(CLONE_NEWNS), 0);
+	ASSERT_EQ(mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL), 0);
+
+	snprintf(base, sizeof(base), "/tmp/loop_cycle.XXXXXX");
+	ASSERT_NE(mkdtemp(base), NULL);
+	/* a failed assertion in the setup does not run the teardown */
+	ASSERT_EQ(mount("tmpfs", base, "tmpfs", 0, NULL), 0)
+		rmdir(base);
+	snprintf(gadget, sizeof(gadget),
+		 "/sys/kernel/config/usb_gadget/kselftest_cycle_%d", getpid());
+	self->dev[0] = '\0';
+	self->nr_loops = 0;
+	self->gadget = false;
+}
+
+static int mount_configfs(void);
+static void gadget_remove(void);
+
+FIXTURE_TEARDOWN(loop_cycle)
+{
+	if (self->gadget)
+		gadget_remove();
+	/* whatever a failed assertion has left bound */
+	for (int i = 0; i < self->nr_loops; i++)
+		loop_clear(self->loops[i]);
+	umount2(base, MNT_DETACH);
+	rmdir(base);
+}
+
+/* Child exit codes. */
+enum {
+	CHILD_OK,
+	CHILD_NS,		/* could not set up the namespace or the tmpfs */
+	CHILD_IMAGE,		/* could not write the image */
+	CHILD_LOOP,		/* could not set up the loop device */
+	CHILD_MOUNT,		/* could not mount it (vfat and msdos both refused) */
+	CHILD_PIPE,		/* the parent went away */
+	CHILD_HOLDER,		/* could not set the holder up below the mount */
+	CHILD_SKIP,		/* the kernel lacks what the holder needs */
+	CHILD_NOFS,		/* the kernel lacks the filesystem of the image */
+};
+
+/* Bind the loop device that LOOP_CTL_GET_FREE names to @ifd; the device number. */
+static int loop_bind_free(int ifd)
+{
+	int cfd, lfd, n;
+	char dev[32];
+
+	cfd = open("/dev/loop-control", O_RDWR);
+	if (cfd < 0)
+		return -1;
+	n = ioctl(cfd, LOOP_CTL_GET_FREE);
+	close(cfd);
+	if (n < 0)
+		return -1;
+	snprintf(dev, sizeof(dev), "/dev/loop%d", n);
+	lfd = open(dev, O_RDWR);
+	if (lfd < 0)
+		return -1;
+	if (ioctl(lfd, LOOP_SET_FD, ifd))
+		n = -1;
+	close(lfd);
+	return n;
+}
+
+/*
+ * Bind a free loop device to the open image @ifd; the device number. The
+ * device is free when LOOP_CTL_GET_FREE names it and may be somebody else's
+ * a moment later, so try again when it is busy.
+ */
+static int loop_bind(int ifd)
+{
+	int n = -1;
+
+	for (int i = 0; i < 64 && n < 0; i++) {
+		n = loop_bind_free(ifd);
+		if (n < 0 && errno != EBUSY)
+			return -1;
+	}
+	if (n >= 0 && nr_bound < MAX_LOOPS)
+		bound[nr_bound++] = n;
+	return n;
+}
+
+/* Mount a FAT image, as vfat or as msdos; -1 with ENODEV if the kernel has neither. */
+static int mount_fat(const char *dev, const char *mp)
+{
+	int err;
+
+	if (!mount(dev, mp, "vfat", 0, NULL))
+		return 0;
+	err = errno;
+	if (!mount(dev, mp, "msdos", 0, NULL))
+		return 0;
+	if (err != ENODEV)
+		errno = err;
+	return -1;
+}
+
+/*
+ * A child that gives up has to take down what it has set up: nobody else
+ * knows about it. Unmount, then tell the loop devices to let go. The mounts
+ * are released after this process is gone and the devices follow them.
+ */
+static int child_fails(int ret)
+{
+	umount2(at("vol"), MNT_DETACH);
+	umount2(at("vol2"), MNT_DETACH);
+	umount2(at("p"), MNT_DETACH);
+	umount2(at("img"), MNT_DETACH);
+	for (int i = 0; i < nr_bound; i++)
+		loop_clear(bound[i]);
+	return ret;
+}
+
+/* Write an image to @img, bind a loop device to it and mount that at @mp; the device number. */
+static int loop_mount(const char *img, const char *mp)
+{
+	char dev[32];
+	int ifd, n;
+
+	ifd = open(img, O_RDWR | O_CREAT | O_EXCL, 0600);
+	if (ifd < 0 || write_fat12(ifd))
+		return -CHILD_IMAGE;
+	n = loop_bind(ifd);
+	close(ifd);	/* the loop device holds the file from now on */
+	if (n < 0)
+		return -CHILD_LOOP;
+	snprintf(dev, sizeof(dev), "/dev/loop%d", n);
+	if (mkdir(mp, 0755))
+		return -CHILD_MOUNT;
+	if (mount_fat(dev, mp))
+		return errno == ENODEV ? -CHILD_NOFS : -CHILD_MOUNT;
+	return n;
+}
+
+/*
+ * Two tmpfs mounts, each carrying the image of the loop mount below the
+ * other: the loop mount below vol has its image on vol2 and the other way
+ * round.
+ */
+static int crossed_child(int to_parent, int from_parent)
+{
+	struct report r;
+	char c;
+
+	if (unshare(CLONE_NEWNS) || mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL))
+		return CHILD_NS;
+	if (mkdir(at("vol"), 0755) || mount("tmpfs", at("vol"), "tmpfs", 0, NULL) ||
+	    mkdir(at("vol2"), 0755) || mount("tmpfs", at("vol2"), "tmpfs", 0, NULL))
+		return child_fails(CHILD_NS);
+	r.n[0] = loop_mount(at("vol2/img"), at("vol/mnt"));
+	if (r.n[0] < 0)
+		return child_fails(-r.n[0]);
+	r.n[1] = loop_mount(at("vol/img"), at("vol2/mnt"));
+	if (r.n[1] < 0)
+		return child_fails(-r.n[1]);
+	if (write(to_parent, &r, sizeof(r)) != sizeof(r))
+		return child_fails(CHILD_PIPE);
+	if (read(from_parent, &c, 1) != 1)
+		return child_fails(CHILD_PIPE);
+	return CHILD_OK;
+}
+
+/*
+ * In its own mount namespace the child mounts a tmpfs on vol,
+ * puts a filesystem image on it, binds a loop device to the image and
+ * mounts that loop device below. The loop device's backing file is a
+ * reference on the mount the image is on, held by the loop device, held
+ * by the mounted filesystem, held by the mount below.
+ */
+static int loop_child(int to_parent, int from_parent)
+{
+	struct report r = { { -1, -1 } };
+	char c;
+
+	if (unshare(CLONE_NEWNS) || mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL))
+		return CHILD_NS;
+	if (mkdir(at("vol"), 0755) || mount("tmpfs", at("vol"), "tmpfs", 0, NULL))
+		return child_fails(CHILD_NS);
+	r.n[0] = loop_mount(at("vol/img"), at("vol/mnt"));
+	if (r.n[0] < 0)
+		return child_fails(-r.n[0]);
+
+	if (write(to_parent, &r, sizeof(r)) != sizeof(r))
+		return child_fails(CHILD_PIPE);
+	/* keep the namespace alive while the parent removes the directory */
+	if (read(from_parent, &c, 1) != 1)
+		return child_fails(CHILD_PIPE);
+	return CHILD_OK;
+}
+
+/*
+ * The child did not report. Skip if the kernel lacks something, fail with
+ * what the child said otherwise. A child that a signal killed has failed.
+ */
+#define CHILD_GAVE_UP(pid) do {						\
+	int __status;							\
+									\
+	ASSERT_EQ(waitpid(pid, &__status, 0), pid);			\
+	if (WIFEXITED(__status) && WEXITSTATUS(__status) == CHILD_NOFS)	\
+		SKIP(return, "test requires the filesystem of the image (FAT or minix)"); \
+	if (WIFEXITED(__status) && WEXITSTATUS(__status) == CHILD_SKIP)	\
+		SKIP(return, "the kernel lacks what this holder needs"); \
+	ASSERT_TRUE(false)						\
+		TH_LOG("child failed to set up: %s %d",			\
+		       WIFEXITED(__status) ? "exit status" : "signal",	\
+		       WIFEXITED(__status) ? WEXITSTATUS(__status) : WTERMSIG(__status)); \
+} while (0)
+
+/* The child has to leave by itself and with nothing to complain about. */
+#define CHILD_LEFT(pid) do {						\
+	int __status;							\
+									\
+	ASSERT_EQ(waitpid(pid, &__status, 0), pid);			\
+	ASSERT_TRUE(WIFEXITED(__status));				\
+	ASSERT_EQ(WEXITSTATUS(__status), CHILD_OK);			\
+} while (0)
+
+/*
+ * An exclusive open of the device fails while a filesystem holds it, and
+ * the only filesystem that ever did is the one mounted below the dead
+ * mount. Give a release in flight a moment. Then the device must clear
+ * right away rather than only be marked for autoclear.
+ */
+static void assert_loop_released(struct __test_metadata *_metadata,
+				 FIXTURE_DATA(loop_cycle) *self)
+{
+	int lfd;
+
+	for (int i = 0; i < 20; i++) {
+		lfd = open(self->dev, O_RDONLY | O_EXCL);
+		if (lfd >= 0)
+			break;
+		usleep(100000);
+	}
+	EXPECT_GE(lfd, 0)
+		TH_LOG("%s is still held by the loop mount below the dead mount: nothing refers to either mount and nothing can release them",
+		       self->dev);
+	if (lfd >= 0)
+		close(lfd);
+
+	lfd = open(self->dev, O_RDWR);
+	ASSERT_GE(lfd, 0);
+	ASSERT_EQ(ioctl(lfd, LOOP_CLR_FD), 0);
+	close(lfd);
+	ASSERT_TRUE(loop_released(self->sysfs, 5000))
+		TH_LOG("%s kept its backing file after LOOP_CLR_FD: the filesystem on it is still mounted somewhere nobody can reach",
+		       self->dev);
+	/* somebody else may bind it from now on, so the teardown leaves it alone */
+	for (int i = 0; i < self->nr_loops; i++) {
+		char dev[32];
+
+		snprintf(dev, sizeof(dev), "/dev/loop%d", self->loops[i]);
+		if (!strcmp(dev, self->dev))
+			self->loops[i] = -1;
+	}
+}
+
+/*
+ * rmdir of vol from here, where it is not a mountpoint, detaches
+ * the child's tmpfs with the loop mount connected below it. Once the child
+ * is gone nothing refers to either mount. The loop device must then be
+ * free to give up its backing file, which only happens when the mounted
+ * filesystem below the detached tmpfs has been released.
+ */
+TEST_F(loop_cycle, detached_loop_mount_released)
+{
+	int to_parent[2], to_child[2];
+	struct report r = { { -1, -1 } };
+	char buf[PATH_MAX];
+	pid_t pid;
+
+	ASSERT_EQ(pipe(to_parent), 0);
+	ASSERT_EQ(pipe(to_child), 0);
+
+	pid = fork();
+	ASSERT_GE(pid, 0);
+	if (pid == 0) {
+		close(to_parent[0]);
+		close(to_child[1]);
+		_exit(loop_child(to_parent[1], to_child[0]));
+	}
+	close(to_parent[1]);
+	close(to_child[0]);
+
+	if (read(to_parent[0], &r, sizeof(r)) != sizeof(r))
+		CHILD_GAVE_UP(pid);
+	remember_loop(self, r.n[0]);
+	snprintf(self->dev, sizeof(self->dev), "/dev/loop%d", r.n[0]);
+	snprintf(self->sysfs, sizeof(self->sysfs), "/sys/block/loop%d/loop/backing_file", r.n[0]);
+	ASSERT_EQ(read_sysfs(self->sysfs, buf, sizeof(buf)), 0);
+	ASSERT_NE(strstr(buf, "/vol/img"), NULL);
+
+	/* not a mountpoint in this namespace, so the directory can go */
+	ASSERT_EQ(rmdir(at("vol")), 0);
+
+	/* the child leaves: its namespace and every reference it held are gone */
+	ASSERT_EQ(write(to_child[1], "", 1), 1);
+	CHILD_LEFT(pid);
+	close(to_parent[0]);
+	close(to_child[1]);
+
+	/* nothing can reach the two mounts any more */
+	ASSERT_EQ(access(at("vol"), F_OK), -1);
+	ASSERT_FALSE(mounted_anywhere(self->dev));
+
+	assert_loop_released(_metadata, self);
+}
+
+/*
+ * The same two mounts in a detached tree: a clone of vol from
+ * open_tree(), the image opened through the clone so that the loop device
+ * holds the clone, and the loop mount moved below the clone. The last
+ * close of the tree's fd dissolves the tree with the loop mount left
+ * connected below the dead clone, and nothing refers to either afterwards.
+ */
+TEST_F(loop_cycle, dissolved_tree_loop_mount_released)
+{
+	int tfd, ifd, fsfd, mfd, n;
+	char buf[PATH_MAX];
+
+	fsfd = sys_fsopen("vfat", 0);
+	if (fsfd < 0)
+		fsfd = sys_fsopen("msdos", 0);
+	if (fsfd < 0)
+		SKIP(return, "test requires a FAT filesystem");
+
+	ASSERT_EQ(mkdir(at("vol"), 0755), 0);
+	ASSERT_EQ(mount("tmpfs", at("vol"), "tmpfs", 0, NULL), 0);
+	tfd = sys_open_tree(AT_FDCWD, at("vol"), OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
+	ASSERT_GE(tfd, 0);
+
+	/* the image, opened through the clone: the loop device holds the clone */
+	ifd = openat(tfd, "img", O_RDWR | O_CREAT | O_EXCL, 0600);
+	ASSERT_GE(ifd, 0);
+	ASSERT_EQ(write_fat12(ifd), 0);
+	n = loop_bind(ifd);
+	close(ifd);
+	ASSERT_GE(n, 0);
+	remember_loop(self, n);
+	snprintf(self->dev, sizeof(self->dev), "/dev/loop%d", n);
+	snprintf(self->sysfs, sizeof(self->sysfs), "/sys/block/loop%d/loop/backing_file", n);
+
+	/* the loop mount, moved below the clone */
+	ASSERT_EQ(mkdirat(tfd, "mnt", 0755), 0);
+	ASSERT_EQ(sys_fsconfig(fsfd, FSCONFIG_SET_STRING, "source", self->dev, 0), 0);
+	ASSERT_EQ(sys_fsconfig(fsfd, FSCONFIG_CMD_CREATE, NULL, NULL, 0), 0);
+	mfd = sys_fsmount(fsfd, 0, 0);
+	ASSERT_GE(mfd, 0);
+	close(fsfd);
+	ASSERT_EQ(sys_move_mount(mfd, "", tfd, "mnt", MOVE_MOUNT_F_EMPTY_PATH), 0);
+	close(mfd);
+
+	ASSERT_EQ(read_sysfs(self->sysfs, buf, sizeof(buf)), 0);
+	ASSERT_NE(strstr(buf, "img"), NULL);
+
+	/* the last fd of the tree: both mounts die, the loop mount connected */
+	close(tfd);
+
+	/* nothing can reach the two mounts any more */
+	ASSERT_FALSE(mounted_anywhere(self->dev));
+
+	assert_loop_released(_metadata, self);
+}
+
+/*
+ * The cycle in two steps: rmdir of vol leaves the loop mount below it
+ * connected while its image's mount, vol2, is alive; then rmdir of vol2
+ * takes that one with the loop mount whose image is on the dead vol. Each
+ * dead mount now owns a loop mount whose filesystem pins the other.
+ */
+TEST_F(loop_cycle, crossed_images_released)
+{
+	int to_parent[2], to_child[2];
+	struct report r = { { -1, -1 } };
+	char sysfs[2][64], dev[2][32];
+	pid_t pid;
+
+	ASSERT_EQ(pipe(to_parent), 0);
+	ASSERT_EQ(pipe(to_child), 0);
+
+	pid = fork();
+	ASSERT_GE(pid, 0);
+	if (pid == 0) {
+		close(to_parent[0]);
+		close(to_child[1]);
+		_exit(crossed_child(to_parent[1], to_child[0]));
+	}
+	close(to_parent[1]);
+	close(to_child[0]);
+
+	if (read(to_parent[0], &r, sizeof(r)) != sizeof(r))
+		CHILD_GAVE_UP(pid);
+	for (int i = 0; i < 2; i++) {
+		remember_loop(self, r.n[i]);
+		snprintf(dev[i], sizeof(dev[i]), "/dev/loop%d", r.n[i]);
+		snprintf(sysfs[i], sizeof(sysfs[i]), "/sys/block/loop%d/loop/backing_file", r.n[i]);
+	}
+
+	/* step one: the mount with the first loop mount below it goes */
+	ASSERT_EQ(rmdir(at("vol")), 0);
+
+	/* step two: the other one, with the loop mount whose image is on the first */
+	ASSERT_EQ(rmdir(at("vol2")), 0);
+
+	/* the child leaves: its namespace and every reference it held are gone */
+	ASSERT_EQ(write(to_child[1], "", 1), 1);
+	CHILD_LEFT(pid);
+	close(to_parent[0]);
+	close(to_child[1]);
+
+	/* nothing can reach the four mounts any more */
+	for (int i = 0; i < 2; i++) {
+		ASSERT_FALSE(mounted_anywhere(dev[i]));
+		strcpy(self->dev, dev[i]);
+		strcpy(self->sysfs, sysfs[i]);
+		assert_loop_released(_metadata, self);
+	}
+}
+
+/*
+ * The holders. Each keeps a file or a path on a mount P for as long as
+ * its own filesystem or device lives, and each has that filesystem or
+ * device mounted at C below P. Once another namespace's rmdir has
+ * detached P with C connected below it and the child is gone, P is owned
+ * by nobody, C by P, and P is kept by whatever the holder still holds.
+ *
+ * P is a loop mount so that its death can be observed: the loop device
+ * gives its backing file up when P's superblock goes. The image sits on
+ * a tmpfs next to P, not above it, so the loop device's own reference is
+ * not part of the picture.
+ */
+enum holder {
+	HOLDER_AUTOFS,		/* a FIFO on P as the daemon's pipe */
+	HOLDER_ZRAM,		/* a device node on P as the writeback device */
+	HOLDER_ECRYPTFS,	/* a directory on P as the lower directory */
+	HOLDER_BINFMT_MISC,	/* an executable on P as an 'F' interpreter */
+	HOLDER_FUSE,		/* a file on P as a passthrough backing file */
+	HOLDER_ZLOOP,		/* a directory on P for the zone files */
+	HOLDER_GADGET,		/* a file on P as a mass storage LUN, over dummy_hcd */
+	HOLDER_MD,		/* a file on P as an array's bitmap file */
+};
+
+#define HOLDER_IMG	at("img/p.img")
+#define HOLDER_MNT	at("p")
+#define HOLDER_BELOW	at("p/c")
+
+static int write_file(const char *path, const char *s)
+{
+	int fd = open(path, O_WRONLY);
+	ssize_t n;
+
+	if (fd < 0)
+		return -1;
+	n = write(fd, s, strlen(s));
+	close(fd);
+	return n == (ssize_t)strlen(s) ? 0 : -1;
+}
+
+/* Bind a free loop device to @img; the device number. */
+static int loop_attach(const char *img)
+{
+	int ifd, n;
+
+	ifd = open(img, O_RDWR);
+	if (ifd < 0)
+		return -1;
+	n = loop_bind(ifd);
+	close(ifd);
+	return n;
+}
+
+static int holder_autofs(void)
+{
+	char opts[64];
+	int pfd;
+
+	if (mkfifo(at("p/pipe"), 0600))
+		return CHILD_HOLDER;
+	pfd = open(at("p/pipe"), O_RDWR);
+	if (pfd < 0)
+		return CHILD_HOLDER;
+	snprintf(opts, sizeof(opts), "fd=%d,minproto=5,maxproto=5", pfd);
+	if (mount("autofs", HOLDER_BELOW, "autofs", 0, opts))
+		return errno == ENODEV ? CHILD_SKIP : CHILD_HOLDER;
+	close(pfd);	/* the mount keeps its own */
+	return CHILD_OK;
+}
+
+/*
+ * zram's writeback device has to be a block device node, and that node has
+ * to be on P. Point it at a second loop device.
+ */
+static void zram_reset(void);
+
+static int holder_zram(void)
+{
+	char dev[32], buf[64];
+	struct stat st;
+	int fd, n;
+
+	if (access("/sys/block/zram0/backing_dev", W_OK))
+		return CHILD_SKIP;
+	/* somebody else's device, leave it alone */
+	if (read_sysfs("/sys/block/zram0/initstate", buf, sizeof(buf)) || buf[0] != '0')
+		return CHILD_SKIP;
+	fd = open(at("img/wb.img"), O_RDWR | O_CREAT | O_EXCL, 0600);
+	if (fd < 0 || ftruncate(fd, IMAGE_SIZE))
+		return CHILD_HOLDER;
+	close(fd);
+	n = loop_attach(at("img/wb.img"));
+	if (n < 0)
+		return CHILD_HOLDER;
+	/* the minor is not the number of the device when loop has partitions */
+	snprintf(dev, sizeof(dev), "/dev/loop%d", n);
+	if (stat(dev, &st) || mknod(at("p/wbdev"), S_IFBLK | 0600, st.st_rdev))
+		return CHILD_HOLDER;
+	if (write_file("/sys/block/zram0/backing_dev", at("p/wbdev")))
+		return CHILD_HOLDER;
+	snprintf(buf, sizeof(buf), "%d", 4 * 1024 * 1024);
+	if (write_file("/sys/block/zram0/disksize", buf))
+		goto undo;
+	fd = open("/dev/zram0", O_RDWR);
+	if (fd < 0)
+		goto undo;
+	n = write_fat12_4k(fd, 1024);	/* zram has 4 KiB blocks */
+	close(fd);	/* the reset in undo is refused while the device is open */
+	if (n)
+		goto undo;
+	if (mount_fat("/dev/zram0", HOLDER_BELOW)) {
+		n = errno == ENODEV ? CHILD_NOFS : CHILD_HOLDER;
+		zram_reset();
+		return n;
+	}
+	return CHILD_OK;
+undo:
+	zram_reset();
+	return CHILD_HOLDER;
+}
+
+/* Is @alg in /proc/crypto? A module is listed once a request has loaded it. */
+static bool crypto_has(const char *alg)
+{
+	char line[256], name[64];
+	bool found = false;
+	FILE *f;
+
+	f = fopen("/proc/crypto", "r");
+	if (!f)
+		return true;	/* no way to tell, assume it is */
+	while (!found && fgets(line, sizeof(line), f))
+		found = sscanf(line, "name : %63s", name) == 1 && !strcmp(name, alg);
+	fclose(f);
+	return found;
+}
+
+/* The kernel's auth token layout, which userspace has to match byte for byte. */
+struct ecryptfs_auth_tok {
+	__u16 version;
+	__u16 token_type;
+	__u32 flags;
+	struct {
+		__u32 flags, encrypted_key_size, decrypted_key_size;
+		__u8 encrypted_key[512], decrypted_key[64];
+	} session_key;
+	__u8 reserved[32];
+	struct {
+		__u32 password_bytes;
+		__s32 hash_algo;
+		__u32 hash_iterations, session_key_encryption_key_bytes, flags;
+		__u8 session_key_encryption_key[64];
+		__u8 signature[17];
+		__u8 salt[8];
+	} password;
+} __attribute__((packed));
+
+#define ECRYPTFS_SIG	"0123456789abcdef"
+
+static int holder_ecryptfs(void)
+{
+	struct ecryptfs_auth_tok tok = {
+		.version = 0x0004,
+		.token_type = 0,		/* ECRYPTFS_PASSWORD */
+		.password.session_key_encryption_key_bytes = 16,
+		.password.flags = 0x02,		/* ECRYPTFS_SESSION_KEY_ENCRYPTION_KEY_SET */
+		.password.signature = ECRYPTFS_SIG,
+	};
+
+	/* a session keyring of this child's own, so that the key goes with it */
+	if (syscall(__NR_keyctl, KEYCTL_JOIN_SESSION_KEYRING, NULL) < 0)
+		return errno == ENOSYS ? CHILD_SKIP : CHILD_HOLDER;
+	if (syscall(__NR_add_key, "user", ECRYPTFS_SIG, &tok, sizeof(tok),
+		    KEY_SPEC_SESSION_KEYRING) < 0)
+		return CHILD_HOLDER;
+	if (mkdir(at("p/lower"), 0755))
+		return CHILD_HOLDER;
+	if (mount(at("p/lower"), HOLDER_BELOW, "ecryptfs", 0,
+		  "ecryptfs_sig=" ECRYPTFS_SIG ",ecryptfs_cipher=aes,ecryptfs_key_bytes=16")) {
+		/* EINVAL without the cipher; a module is loaded by the attempt */
+		if (errno == ENODEV || (errno == EINVAL && !crypto_has("aes")))
+			return CHILD_SKIP;
+		return CHILD_HOLDER;
+	}
+	return CHILD_OK;
+}
+
+/*
+ * binfmt_misc instances are per user namespace, so the mount below P is
+ * made from a new one, which gets a copy of P.
+ */
+static int holder_binfmt_misc(void)
+{
+	static const char interp[] = "#!/bin/true\n";
+	char reg[PATH_MAX];
+	int out;
+
+	/* 'F' opens the interpreter at registration, nothing runs it here */
+	out = open(at("p/interp"), O_WRONLY | O_CREAT | O_EXCL, 0755);
+	if (out < 0 || write(out, interp, sizeof(interp) - 1) != sizeof(interp) - 1)
+		return CHILD_HOLDER;
+	close(out);
+
+	/* ENOSPC: user.max_user_namespaces is 0, EPERM: an LSM says no */
+	if (unshare(CLONE_NEWUSER | CLONE_NEWNS)) {
+		if (errno == EINVAL || errno == ENOSPC || errno == EPERM)
+			return CHILD_SKIP;
+		return CHILD_HOLDER;
+	}
+	if (write_file("/proc/self/setgroups", "deny") ||
+	    write_file("/proc/self/uid_map", "0 0 1") ||
+	    write_file("/proc/self/gid_map", "0 0 1"))
+		return CHILD_HOLDER;
+	if (mount("binfmt_misc", HOLDER_BELOW, "binfmt_misc", 0, NULL))
+		return errno == ENODEV ? CHILD_SKIP : CHILD_HOLDER;
+	snprintf(reg, sizeof(reg), ":cycle:E::cyc::%s:F", at("p/interp"));
+	if (write_file(at("p/c/register"), reg))
+		return CHILD_HOLDER;
+	return CHILD_OK;
+}
+
+/*
+ * A fuse server that only ever answers FUSE_INIT, with passthrough on, and
+ * then registers a file on P as a backing file. The registration alone
+ * makes the fuse superblock hold the file.
+ */
+static int holder_fuse(void)
+{
+	struct fuse_backing_map map = {};
+	struct fuse_in_header *ih;
+	struct fuse_init_out init = {
+		.major = FUSE_KERNEL_VERSION,
+		.minor = FUSE_KERNEL_MINOR_VERSION,
+		.flags = FUSE_INIT_EXT,
+		.flags2 = FUSE_PASSTHROUGH >> 32,
+		.max_write = 4096,
+		.max_stack_depth = 1,
+	};
+	struct fuse_out_header oh = { .len = sizeof(oh) + sizeof(init) };
+	struct iovec iov[2] = { { &oh, sizeof(oh) }, { &init, sizeof(init) } };
+	char opts[64], buf[8192];
+	int ffd, bfd;
+	ssize_t n;
+
+	bfd = open(at("p/backing"), O_RDWR | O_CREAT | O_EXCL, 0600);
+	if (bfd < 0)
+		return CHILD_HOLDER;
+	ffd = open("/dev/fuse", O_RDWR);
+	if (ffd < 0)
+		return CHILD_SKIP;
+	snprintf(opts, sizeof(opts), "fd=%d,rootmode=40000,user_id=0,group_id=0", ffd);
+	if (mount("fuse", HOLDER_BELOW, "fuse", 0, opts))
+		return errno == ENODEV ? CHILD_SKIP : CHILD_HOLDER;
+
+	n = read(ffd, buf, sizeof(buf));
+	ih = (void *)buf;
+	if (n < (ssize_t)sizeof(*ih) || ih->opcode != FUSE_INIT)
+		return CHILD_HOLDER;
+	oh.unique = ih->unique;
+	if (writev(ffd, iov, 2) != (ssize_t)oh.len)
+		return CHILD_HOLDER;
+
+	map.fd = bfd;
+	if (ioctl(ffd, FUSE_DEV_IOC_BACKING_OPEN, &map) < 0) {
+		/* EOPNOTSUPP: no FUSE_PASSTHROUGH, ENOTTY: no such ioctl */
+		if (errno == EPERM || errno == EOPNOTSUPP || errno == ENOTTY)
+			return CHILD_SKIP;
+		return CHILD_HOLDER;
+	}
+	close(bfd);	/* the connection keeps its own */
+	return CHILD_OK;	/* ffd stays open until the child exits */
+}
+
+/* Wait for a device node the kernel is about to create. */
+static int open_when_there(const char *dev, int flags, int ms)
+{
+	int fd;
+
+	for (; ms > 0; ms -= 100) {
+		fd = open(dev, flags);
+		if (fd >= 0)
+			return fd;
+		usleep(100000);
+	}
+	return -1;
+}
+
+/*
+ * zloop keeps every zone file open. One conventional zone is enough for a
+ * FAT image, the sequential one stays empty.
+ */
+static int holder_zloop(void)
+{
+	char cmd[PATH_MAX];
+	int fd, ret = CHILD_HOLDER;
+
+	if (access("/dev/zloop-control", W_OK))
+		return CHILD_SKIP;
+	/* somebody else's device, leave it alone */
+	if (!access("/sys/block/zloop0", F_OK))
+		return CHILD_SKIP;
+	if (mkdir(at("p/zl"), 0755) || mkdir(at("p/zl/0"), 0755))
+		return CHILD_HOLDER;
+	snprintf(cmd, sizeof(cmd),
+		 "add id=0,capacity_mb=8,zone_size_mb=4,conv_zones=1,base_dir=%s",
+		 at("p/zl"));
+	if (write_file("/dev/zloop-control", cmd))
+		return CHILD_HOLDER;
+	fd = open_when_there("/dev/zloop0", O_RDWR, 5000);
+	if (fd < 0 || write_fat12_4k(fd, 1024))	/* 4 KiB blocks, like P */
+		goto undo;
+	close(fd);
+	fd = -1;
+	if (mount_fat("/dev/zloop0", HOLDER_BELOW)) {
+		ret = errno == ENODEV ? CHILD_NOFS : CHILD_HOLDER;
+		goto undo;
+	}
+	return CHILD_OK;
+undo:
+	if (fd >= 0)
+		close(fd);
+	write_file("/dev/zloop-control", "remove id=0");
+	return ret;
+}
+
+static int mount_configfs(void)
+{
+	if (access("/sys/kernel/config", F_OK))
+		return -1;
+	if (mount("configfs", "/sys/kernel/config", "configfs", 0, NULL) && errno != EBUSY)
+		return -1;
+	return 0;
+}
+
+/* Take the gadget out of configfs again, in the reverse order of its creation. */
+static void gadget_remove(void)
+{
+	if (mount_configfs())
+		return;
+	write_file(gat("/functions/mass_storage.0/lun.0/file"), "\n");
+	write_file(gat("/UDC"), "\n");
+	unlink(gat("/configs/c.1/mass_storage.0"));
+	rmdir(gat("/configs/c.1/strings/0x409"));
+	rmdir(gat("/configs/c.1"));
+	rmdir(gat("/functions/mass_storage.0"));
+	rmdir(gat("/strings/0x409"));
+	rmdir(gat(""));
+}
+
+/* The disk usb-storage created for the gadget, by the LUN's inquiry string. */
+static int find_gadget_disk(char *dev, size_t len, int ms)
+{
+	char path[PATH_MAX], model[64];
+	struct dirent *de;
+	DIR *d;
+
+	for (; ms > 0; ms -= 100, usleep(100000)) {
+		d = opendir("/sys/block");
+		if (!d)
+			return -1;
+		while ((de = readdir(d))) {
+			if (strncmp(de->d_name, "sd", 2))
+				continue;
+			snprintf(path, sizeof(path), "/sys/block/%s/device/model", de->d_name);
+			if (read_sysfs(path, model, sizeof(model)) ||
+			    strncmp(model, "File-Stor Gadget", 16))
+				continue;
+			snprintf(dev, len, "/dev/%s", de->d_name);
+			closedir(d);
+			return 0;
+		}
+		closedir(d);
+	}
+	return -1;
+}
+
+/*
+ * A mass storage gadget bound to the dummy UDC, so that this kernel is
+ * also the USB host that sees the LUN as a SCSI disk. sd locks the
+ * medium on open, which is what keeps the LUN's file from being ejected.
+ */
+#define GADGET_STEP(x) do {						\
+	if (x) {							\
+		fprintf(stderr, "gadget: %s failed: %s\n", #x, strerror(errno)); \
+		gadget_remove();					\
+		return CHILD_HOLDER;					\
+	}								\
+} while (0)
+
+static int holder_gadget(void)
+{
+	char dev[PATH_MAX];
+	int fd;
+
+	if (mount_configfs())
+		return CHILD_SKIP;
+	if (access("/sys/kernel/config/usb_gadget", F_OK) ||
+	    access("/sys/class/udc/dummy_udc.0", F_OK))
+		return CHILD_SKIP;
+
+	fd = open(at("p/lun.img"), O_RDWR | O_CREAT | O_EXCL, 0600);
+	if (fd < 0 || write_fat12(fd))
+		return CHILD_HOLDER;
+	close(fd);
+
+	GADGET_STEP(mkdir(gat(""), 0755));
+	GADGET_STEP(write_file(gat("/idVendor"), "0x1d6b"));
+	GADGET_STEP(write_file(gat("/idProduct"), "0x0104"));
+	GADGET_STEP(mkdir(gat("/strings/0x409"), 0755));
+	GADGET_STEP(write_file(gat("/strings/0x409/serialnumber"), "1"));
+	GADGET_STEP(write_file(gat("/strings/0x409/manufacturer"), "kselftest"));
+	GADGET_STEP(write_file(gat("/strings/0x409/product"), "cycle"));
+	GADGET_STEP(mkdir(gat("/configs/c.1"), 0755));
+	GADGET_STEP(mkdir(gat("/configs/c.1/strings/0x409"), 0755));
+	GADGET_STEP(write_file(gat("/configs/c.1/strings/0x409/configuration"), "c"));
+	GADGET_STEP(mkdir(gat("/functions/mass_storage.0"), 0755));
+	GADGET_STEP(write_file(gat("/functions/mass_storage.0/lun.0/removable"), "1"));
+	GADGET_STEP(write_file(gat("/functions/mass_storage.0/lun.0/file"), at("p/lun.img")));
+	GADGET_STEP(symlink(gat("/functions/mass_storage.0"), gat("/configs/c.1/mass_storage.0")));
+	/* EBUSY: the controller is somebody else's, leave it alone */
+	if (write_file(gat("/UDC"), "dummy_udc.0")) {
+		fd = errno == EBUSY ? CHILD_SKIP : CHILD_HOLDER;
+		gadget_remove();
+		return fd;
+	}
+
+	/* usb-storage waits a second before it scans the device */
+	if (find_gadget_disk(dev, sizeof(dev), 15000)) {
+		/* without usb-storage and sd the LUN never shows up as a disk */
+		fd = (access("/sys/bus/usb/drivers/usb-storage", F_OK) ||
+		      access("/sys/bus/scsi/drivers/sd", F_OK)) ? CHILD_SKIP : CHILD_HOLDER;
+		gadget_remove();
+		return fd;
+	}
+	fd = open_when_there(dev, O_RDONLY, 5000);
+	GADGET_STEP(fd < 0);
+	close(fd);
+	if (mount_fat(dev, HOLDER_BELOW)) {
+		fd = errno == ENODEV ? CHILD_NOFS : CHILD_HOLDER;
+		gadget_remove();
+		return fd;
+	}
+	return CHILD_OK;
+}
+
+#define BITMAP_MAGIC	0x6d746962
+
+/*
+ * A RAID1 of one loop device, not persistent, with its bitmap in a file
+ * on P. The bitmap file needs a superblock the kernel accepts; sync_size,
+ * uuid and events are not looked at for a non-persistent array.
+ */
+static int holder_md(void)
+{
+	struct {
+		__u32 magic, version;
+		__u8 uuid[16];
+		__u64 events, events_cleared, sync_size;
+		__u32 state, chunksize, daemon_sleep, write_behind;
+	} bsb = {
+		.magic = BITMAP_MAGIC,
+		.version = 4,
+		.chunksize = 64 * 1024,
+		.daemon_sleep = 5,
+	};
+	mdu_array_info_t info = {
+		.level = 1,
+		.raid_disks = 1,
+		.size = 8 * 1024,		/* KiB */
+		.not_persistent = 1,
+	};
+	mdu_disk_info_t disk = {
+		.major = 7,
+		.state = (1 << MD_DISK_ACTIVE) | (1 << MD_DISK_SYNC),
+	};
+	int fd, mdfd, bfd, n, ret = CHILD_HOLDER;
+	char dev[32], buf[4096];
+	struct stat st;
+
+	fd = open(at("img/md.img"), O_RDWR | O_CREAT | O_EXCL, 0600);
+	if (fd < 0 || ftruncate(fd, 8 * 1024 * 1024))
+		return CHILD_HOLDER;
+	close(fd);
+	n = loop_attach(at("img/md.img"));
+	if (n < 0)
+		return CHILD_HOLDER;
+	/* the minor is not the number of the device when loop has partitions */
+	snprintf(dev, sizeof(dev), "/dev/loop%d", n);
+	if (stat(dev, &st))
+		return CHILD_HOLDER;
+	disk.major = major(st.st_rdev);
+	disk.minor = minor(st.st_rdev);
+
+	bfd = open(at("p/bitmap"), O_RDWR | O_CREAT | O_EXCL, 0600);
+	if (bfd < 0 || write(bfd, &bsb, sizeof(bsb)) != sizeof(bsb) || ftruncate(bfd, 4096))
+		return CHILD_HOLDER;
+
+	if (read_sysfs("/proc/mdstat", buf, sizeof(buf))) {
+		fprintf(stderr, "md: no /proc/mdstat\n");
+		return CHILD_SKIP;
+	}
+	if (!strstr(buf, "[raid1]")) {
+		fprintf(stderr, "md: no raid1 personality: %s\n", buf);
+		return CHILD_SKIP;
+	}
+	/* a node of our own: /dev may not have one and is not ours to change */
+	if (mknod(at("img/md0"), S_IFBLK | 0600, makedev(MD_MAJOR, 0)))
+		return CHILD_HOLDER;
+	mdfd = open(at("img/md0"), O_RDWR);
+	if (mdfd < 0)
+		return CHILD_SKIP;
+	/* somebody else's array: SET_ARRAY_INFO would say EINVAL, not EBUSY */
+	if (!read_sysfs("/sys/block/md0/md/array_state", buf, sizeof(buf)) &&
+	    strncmp(buf, "clear", 5)) {
+		fprintf(stderr, "md: md0 is in use: %s", buf);
+		close(mdfd);
+		return CHILD_SKIP;
+	}
+	/* the bitmap ops are only installed once a bitmap type is chosen */
+	write_file("/sys/block/md0/md/bitmap_type", "bitmap");
+	/* EBUSY: somebody else's array, leave it alone */
+	if (ioctl(mdfd, SET_ARRAY_INFO, &info))
+		return errno == EBUSY ? CHILD_SKIP : CHILD_HOLDER;
+	/* the array is ours from here on and has to be stopped on every path */
+	if (ioctl(mdfd, ADD_NEW_DISK, &disk))
+		goto undo;
+	/* attach the bitmap file before the array runs, the way mdadm does */
+	if (ioctl(mdfd, SET_BITMAP_FILE, bfd)) {
+		fprintf(stderr, "md: SET_BITMAP_FILE: %s\n", strerror(errno));
+		if (errno == EINVAL)
+			ret = CHILD_SKIP;
+		goto undo;
+	}
+	close(bfd);	/* the array keeps its own */
+	if (ioctl(mdfd, RUN_ARRAY, NULL)) {
+		fprintf(stderr, "md: RUN_ARRAY: %s\n", strerror(errno));
+		goto undo;
+	}
+	if (write_fat12(mdfd))
+		goto undo;
+	close(mdfd);
+	if (mount_fat(at("img/md0"), HOLDER_BELOW)) {
+		ret = errno == ENODEV ? CHILD_NOFS : CHILD_HOLDER;
+		mdfd = open(at("img/md0"), O_RDWR);
+		goto undo;
+	}
+	return CHILD_OK;
+undo:
+	if (mdfd >= 0) {
+		ioctl(mdfd, STOP_ARRAY);
+		close(mdfd);
+	}
+	return ret;
+}
+
+/*
+ * A device keeps its file for as long as it is configured, so once nothing
+ * below the dead mount is left the device has to be told to let go. With
+ * the cycle unbroken C's filesystem still holds the device and every one
+ * of these refuses.
+ */
+static int holder_let_go_once(enum holder holder)
+{
+	int fd, ret;
+
+	switch (holder) {
+	case HOLDER_ZRAM:
+		return write_file("/sys/block/zram0/reset", "1");
+	case HOLDER_ZLOOP:
+		return write_file("/dev/zloop-control", "remove id=0");
+	case HOLDER_GADGET:
+		if (mount_configfs())
+			return -1;
+		/* a zero-length write is a no-op for configfs; a newline ejects */
+		return write_file(gat("/functions/mass_storage.0/lun.0/file"), "\n");
+	case HOLDER_MD:
+		fd = open(at("img/md0"), O_RDONLY);
+		if (fd < 0) {
+			/* the child's node went with its tmpfs */
+			mknod(at("md0"), S_IFBLK | 0600, makedev(MD_MAJOR, 0));
+			fd = open(at("md0"), O_RDONLY);
+		}
+		if (fd < 0)
+			return -1;
+		ret = ioctl(fd, STOP_ARRAY);
+		close(fd);
+		return ret;
+	default:
+		return 0;
+	}
+}
+
+static void zram_reset(void)
+{
+	holder_let_go_once(HOLDER_ZRAM);
+}
+
+/*
+ * In its own mount namespace the child puts the image on a tmpfs next to
+ * P, mounts P from a loop device, and sets the holder up with its own
+ * filesystem or device mounted at C below P.
+ */
+static int holder_child(int to_parent, int from_parent, enum holder holder)
+{
+	bool fifo = holder == HOLDER_AUTOFS || holder == HOLDER_ZRAM;
+	/* room for a 4 MiB zone file or a floppy image on P */
+	bool big = holder == HOLDER_ZLOOP || holder == HOLDER_GADGET;
+	struct report r = { { -1, -1 } };
+	char dev[32], c;
+	int ifd, n, ret;
+
+	if (unshare(CLONE_NEWNS) || mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL))
+		return CHILD_NS;
+	if (mkdir(at("img"), 0755) || mount("tmpfs", at("img"), "tmpfs", 0, NULL))
+		return child_fails(CHILD_NS);
+
+	ifd = open(HOLDER_IMG, O_RDWR | O_CREAT | O_EXCL, 0600);
+	if (ifd < 0 || (fifo ? write_minix(ifd) :
+			big ? write_fat12_4k(ifd, 3072) : write_fat12(ifd)))
+		return child_fails(CHILD_IMAGE);
+	close(ifd);
+	n = loop_attach(HOLDER_IMG);
+	if (n < 0)
+		return child_fails(CHILD_LOOP);
+	snprintf(dev, sizeof(dev), "/dev/loop%d", n);
+	if (mkdir(HOLDER_MNT, 0755))
+		return child_fails(CHILD_MOUNT);
+	if (fifo ? mount(dev, HOLDER_MNT, "minix", 0, NULL) : mount_fat(dev, HOLDER_MNT))
+		return child_fails(errno == ENODEV ? CHILD_NOFS : CHILD_MOUNT);
+	if (mkdir(HOLDER_BELOW, 0755))
+		return child_fails(CHILD_MOUNT);
+
+	switch (holder) {
+	case HOLDER_AUTOFS:
+		ret = holder_autofs();
+		break;
+	case HOLDER_ZRAM:
+		ret = holder_zram();
+		break;
+	case HOLDER_ECRYPTFS:
+		ret = holder_ecryptfs();
+		break;
+	case HOLDER_BINFMT_MISC:
+		ret = holder_binfmt_misc();
+		break;
+	case HOLDER_FUSE:
+		ret = holder_fuse();
+		break;
+	case HOLDER_ZLOOP:
+		ret = holder_zloop();
+		break;
+	case HOLDER_GADGET:
+		ret = holder_gadget();
+		break;
+	case HOLDER_MD:
+		ret = holder_md();
+		break;
+	default:
+		ret = CHILD_HOLDER;
+	}
+	if (ret != CHILD_OK)
+		return child_fails(ret);
+
+	/* P's device first, then the one the holder has bound for itself */
+	r.n[0] = n;
+	if (nr_bound > 1)
+		r.n[1] = bound[1];
+	if (write(to_parent, &r, sizeof(r)) != sizeof(r) ||
+	    read(from_parent, &c, 1) != 1) {
+		/* the parent went away and will not tell the devices to let go */
+		umount2(HOLDER_BELOW, MNT_DETACH);
+		for (int i = 0; i < 50 && holder_let_go_once(holder); i++)
+			usleep(100000);
+		if (holder == HOLDER_GADGET)
+			gadget_remove();
+		return child_fails(CHILD_PIPE);
+	}
+	return CHILD_OK;
+}
+
+/*
+ * The release of C's filesystem may still be in flight when the child is
+ * gone, and a device that is still held refuses to let go, so try for a
+ * while. With the cycle unbroken it refuses for good.
+ */
+static void holder_let_go(enum holder holder)
+{
+	for (int i = 0; i < 50; i++) {
+		if (!holder_let_go_once(holder))
+			return;
+		usleep(100000);
+	}
+}
+
+/*
+ * rmdir of p from here, where it is a plain directory, detaches
+ * P in the child's namespace with C connected below it. Once the child is
+ * gone the holder's file on P is the only thing left that refers to P,
+ * and it is dropped only when C's filesystem dies, which waits for P.
+ * The loop device backing P tells whether that resolved.
+ */
+static void holder_cycle(struct __test_metadata *_metadata,
+			 FIXTURE_DATA(loop_cycle) *self, enum holder holder)
+{
+	struct report r = { { -1, -1 } };
+	int to_parent[2], from_parent[2];
+	pid_t pid;
+
+	ASSERT_EQ(pipe(to_parent), 0);
+	ASSERT_EQ(pipe(from_parent), 0);
+	pid = fork();
+	ASSERT_GE(pid, 0);
+	if (pid == 0) {
+		close(to_parent[0]);
+		close(from_parent[1]);
+		_exit(holder_child(to_parent[1], from_parent[0], holder));
+	}
+	close(to_parent[1]);
+	close(from_parent[0]);
+
+	if (read(to_parent[0], &r, sizeof(r)) != sizeof(r))
+		CHILD_GAVE_UP(pid);
+	self->gadget = holder == HOLDER_GADGET;
+	remember_loop(self, r.n[0]);
+	remember_loop(self, r.n[1]);
+	snprintf(self->dev, sizeof(self->dev), "/dev/loop%d", r.n[0]);
+	snprintf(self->sysfs, sizeof(self->sysfs), "/sys/block/loop%d/loop/backing_file", r.n[0]);
+
+	ASSERT_EQ(rmdir(HOLDER_MNT), 0);
+	ASSERT_EQ(write(from_parent[1], "x", 1), 1);
+	CHILD_LEFT(pid);
+	close(to_parent[0]);
+	close(from_parent[1]);
+
+	holder_let_go(holder);
+	assert_loop_released(_metadata, self);
+	/* the teardown takes the gadget and the holder's own loop device away */
+}
+
+TEST_F(loop_cycle, autofs_pipe_on_dead_mount_released)
+{
+	holder_cycle(_metadata, self, HOLDER_AUTOFS);
+}
+
+TEST_F(loop_cycle, zram_writeback_node_on_dead_mount_released)
+{
+	holder_cycle(_metadata, self, HOLDER_ZRAM);
+}
+
+TEST_F(loop_cycle, ecryptfs_lower_on_dead_mount_released)
+{
+	holder_cycle(_metadata, self, HOLDER_ECRYPTFS);
+}
+
+TEST_F(loop_cycle, binfmt_misc_interpreter_on_dead_mount_released)
+{
+	holder_cycle(_metadata, self, HOLDER_BINFMT_MISC);
+}
+
+TEST_F(loop_cycle, fuse_backing_file_on_dead_mount_released)
+{
+	holder_cycle(_metadata, self, HOLDER_FUSE);
+}
+
+TEST_F(loop_cycle, zloop_zone_files_on_dead_mount_released)
+{
+	holder_cycle(_metadata, self, HOLDER_ZLOOP);
+}
+
+/* up to 15 s for the disk to show up and 12 s for a device that is held */
+TEST_F_TIMEOUT(loop_cycle, mass_storage_lun_on_dead_mount_released, 120)
+{
+	holder_cycle(_metadata, self, HOLDER_GADGET);
+}
+
+TEST_F(loop_cycle, md_bitmap_file_on_dead_mount_released)
+{
+	holder_cycle(_metadata, self, HOLDER_MD);
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/filesystems/mount_cycle/mount_cover_test.c b/tools/testing/selftests/filesystems/mount_cycle/mount_cover_test.c
new file mode 100644
index 000000000000..e3092454ae68
--- /dev/null
+++ b/tools/testing/selftests/filesystems/mount_cycle/mount_cover_test.c
@@ -0,0 +1,565 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * An unmounted mount that would have stayed attached to its unmounted parent
+ * leaves a cover behind instead. A lookup on the parent at the mountpoint
+ * finds knullfs: an empty read-only directory that is shared by every cover
+ * and every kernel thread, so it can't be watched or locked and nothing can
+ * be mounted on it. The mount itself is a root from then on. Where a file
+ * was mounted, the stand-in is an empty regular file of the same instance.
+ */
+#define _GNU_SOURCE
+#include <errno.h>
+#include <fcntl.h>
+#include <sched.h>
+#include <stdint.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <unistd.h>
+#include <sys/fanotify.h>
+#include <sys/file.h>
+#include <sys/inotify.h>
+#include <sys/mount.h>
+#include <sys/prctl.h>
+#include <sys/stat.h>
+#include <sys/statvfs.h>
+#include <sys/syscall.h>
+#include <sys/vfs.h>
+#include <sys/wait.h>
+
+#include "../../kselftest_harness.h"
+#include "../readdir_hold.h"
+
+#ifndef NULL_FS_MAGIC
+#define NULL_FS_MAGIC 0x4E554C4C
+#endif
+
+#ifndef TMPFS_MAGIC
+#define TMPFS_MAGIC 0x01021994
+#endif
+
+#ifndef F_SETDELEG
+#define F_SETDELEG	(F_SETLEASE + 16)
+#endif
+
+/* struct delegation of <linux/fcntl.h>, which doesn't mix with <fcntl.h> */
+struct delegation_req {
+	uint32_t d_flags;
+	uint16_t d_type;
+	uint16_t __pad;
+};
+
+#define DIR_LEN 64
+#define PATH_LEN 128
+
+/* what the parent asks the child to do */
+#define CMD_RMDIR	'r'
+#define CMD_CLOSE_T	't'
+#define CMD_QUIT	'q'
+
+static int write_file(const char *path, const char *s)
+{
+	ssize_t n = -1;
+	int fd;
+
+	fd = open(path, O_WRONLY | O_CLOEXEC);
+	if (fd >= 0) {
+		n = write(fd, s, strlen(s));
+		close(fd);
+	}
+	return n == (ssize_t)strlen(s) ? 0 : -1;
+}
+
+static int touch(const char *path)
+{
+	int fd;
+
+	fd = open(path, O_WRONLY | O_CREAT | O_EXCL | O_CLOEXEC, 0644);
+	if (fd < 0)
+		return -1;
+	close(fd);
+	return 0;
+}
+
+/* Become root in a new user namespace with a private mount namespace. */
+static int enter_userns(void)
+{
+	uid_t uid = getuid();
+	gid_t gid = getgid();
+	char map[32];
+
+	prctl(PR_SET_DUMPABLE, 1);
+	if (unshare(CLONE_NEWUSER | CLONE_NEWNS))
+		return -1;
+	if (write_file("/proc/self/setgroups", "deny") && errno != ENOENT)
+		return -1;
+	snprintf(map, sizeof(map), "0 %d 1", uid);
+	if (write_file("/proc/self/uid_map", map))
+		return -1;
+	snprintf(map, sizeof(map), "0 %d 1", gid);
+	if (write_file("/proc/self/gid_map", map))
+		return -1;
+	if (setgid(0) || setuid(0))
+		return -1;
+	return mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL);
+}
+
+/*
+ * The child mounts P on @base/p, C on P/covered and the file P/src on P/file
+ * in a mount namespace of its own, binds P a second time at @base/q and hands
+ * out descriptors on P and on C. rmdir() of @base/p from here unmounts P
+ * together with C and the file bind. P and C are held by the descriptors and
+ * the unmounted children leave their covers behind. On request the child
+ * removes C's mountpoint through the bind.
+ *
+ * It also mounts T on @base/t with a child on T/covered, binds T at @base/u
+ * with a second child on the same dentry and hands out descriptors on T and
+ * U. rmdir() of both from here leaves two covers on one mountpoint. On request
+ * the child lets go of T.
+ */
+static int cover_child(const char *base, int to_parent, int from_parent)
+{
+	char p[PATH_LEN], c[PATH_LEN], q[PATH_LEN], qc[PATH_LEN];
+	char f[PATH_LEN], src[PATH_LEN], t[PATH_LEN], u[PATH_LEN], tc[PATH_LEN];
+	int fds[4], ret;
+	char cmd;
+
+	if (unshare(CLONE_NEWNS) || mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL))
+		return 1;
+	snprintf(p, sizeof(p), "%s/p", base);
+	if (mkdir(p, 0755) || mount("tmpfs", p, "tmpfs", 0, NULL))
+		return 2;
+	snprintf(c, sizeof(c), "%s/p/covered", base);
+	if (mkdir(c, 0755) || mount("tmpfs", c, "tmpfs", 0, NULL))
+		return 3;
+	snprintf(f, sizeof(f), "%s/p/file", base);
+	snprintf(src, sizeof(src), "%s/p/src", base);
+	if (touch(src) || touch(f) || mount(src, f, NULL, MS_BIND, NULL))
+		return 4;
+	snprintf(q, sizeof(q), "%s/q", base);
+	if (mkdir(q, 0755) || mount(p, q, NULL, MS_BIND, NULL))
+		return 5;
+	snprintf(t, sizeof(t), "%s/t", base);
+	if (mkdir(t, 0755) || mount("tmpfs", t, "tmpfs", 0, NULL))
+		return 6;
+	snprintf(tc, sizeof(tc), "%s/t/covered", base);
+	if (mkdir(tc, 0755) || mount("tmpfs", tc, "tmpfs", 0, NULL))
+		return 7;
+	snprintf(u, sizeof(u), "%s/u", base);
+	if (mkdir(u, 0755) || mount(t, u, NULL, MS_BIND, NULL))
+		return 8;
+	snprintf(tc, sizeof(tc), "%s/u/covered", base);
+	if (mount("tmpfs", tc, "tmpfs", 0, NULL))
+		return 9;
+	fds[0] = open(p, O_PATH | O_DIRECTORY | O_CLOEXEC);
+	fds[1] = open(c, O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+	fds[2] = open(t, O_PATH | O_DIRECTORY | O_CLOEXEC);
+	fds[3] = open(u, O_PATH | O_DIRECTORY | O_CLOEXEC);
+	if (fds[0] < 0 || fds[1] < 0 || fds[2] < 0 || fds[3] < 0)
+		return 10;
+	if (write(to_parent, fds, sizeof(fds)) != sizeof(fds))
+		return 11;
+
+	snprintf(qc, sizeof(qc), "%s/q/covered", base);
+	for (;;) {
+		if (read(from_parent, &cmd, 1) != 1)
+			return 12;
+		switch (cmd) {
+		case CMD_RMDIR:
+			ret = rmdir(qc) ? errno : 0;
+			if (write(to_parent, &ret, sizeof(ret)) != sizeof(ret))
+				return 13;
+			break;
+		case CMD_CLOSE_T:
+			ret = close(fds[2]) ? errno : 0;
+			if (write(to_parent, &ret, sizeof(ret)) != sizeof(ret))
+				return 13;
+			break;
+		case CMD_QUIT:
+			return 0;
+		default:
+			return 14;
+		}
+	}
+}
+
+FIXTURE(mount_cover) {
+	char base[DIR_LEN];
+	pid_t child;
+	int to_child;
+	int from_child;
+	int dfd;	/* P, unmounted, held */
+	int cfd;	/* C, unmounted, held */
+	int fd;		/* what a lookup on P finds at C's mountpoint */
+	int tfd;	/* T, unmounted, held */
+	int ufd;	/* U, a bind of T, unmounted, held */
+	int fan;	/* fanotify group from before the user namespace, or -1 */
+	struct readdir_hold hold;	/* likewise from before, uffd -1 without */
+};
+
+/*
+ * Everything the setup has made. The harness does not run the teardown
+ * when an assertion of the setup fails, so the setup calls this itself.
+ */
+static void cover_cleanup(FIXTURE_DATA(mount_cover) *self)
+{
+	char cmd = CMD_QUIT;
+	int status;
+
+	if (self->fd >= 0)
+		close(self->fd);
+	if (self->cfd >= 0)
+		close(self->cfd);
+	if (self->dfd >= 0)
+		close(self->dfd);
+	if (self->tfd >= 0)
+		close(self->tfd);
+	if (self->ufd >= 0)
+		close(self->ufd);
+	if (self->fan >= 0)
+		close(self->fan);
+	readdir_hold_destroy(&self->hold);
+	if (self->child > 0) {
+		if (write(self->to_child, &cmd, 1) != 1)
+			kill(self->child, SIGKILL);
+		waitpid(self->child, &status, 0);
+	}
+	if (self->to_child >= 0)
+		close(self->to_child);
+	if (self->from_child >= 0)
+		close(self->from_child);
+	umount2(self->base, MNT_DETACH);
+	rmdir(self->base);
+}
+
+static int same_file(int fd1, int fd2)
+{
+	struct stat st1, st2;
+
+	if (fstat(fd1, &st1) || fstat(fd2, &st2))
+		return 0;
+	return st1.st_dev == st2.st_dev && st1.st_ino == st2.st_ino;
+}
+
+FIXTURE_SETUP(mount_cover)
+{
+	int to_parent[2], to_child[2], fds[4], pidfd, status;
+	char dir[PATH_LEN];
+	struct statfs sf;
+
+	self->child = 0;
+	self->to_child = self->from_child = -1;
+	self->dfd = self->cfd = self->fd = self->tfd = self->ufd = -1;
+
+	/*
+	 * A group for plain events takes CAP_SYS_ADMIN in the initial user
+	 * namespace, so get one before that is gone. An inode mark can be
+	 * added to it from anywhere.
+	 */
+	self->fan = fanotify_init(FAN_CLASS_NOTIF | FAN_CLOEXEC, O_RDONLY);
+	/* same for the userfaultfd that holds a readdir in its fault */
+	readdir_hold_init(&self->hold);
+
+	snprintf(self->base, sizeof(self->base), "/tmp/mount_cover.XXXXXX");
+	ASSERT_NE(mkdtemp(self->base), NULL);
+	if (enter_userns()) {
+		cover_cleanup(self);
+		SKIP(return, "test requires user namespaces");
+	}
+	ASSERT_EQ(mount("tmpfs", self->base, "tmpfs", 0, NULL), 0)
+		cover_cleanup(self);
+
+	snprintf(dir, sizeof(dir), "%s/p", self->base);
+	ASSERT_EQ(pipe(to_parent), 0)
+		cover_cleanup(self);
+	ASSERT_EQ(pipe(to_child), 0)
+		cover_cleanup(self);
+	self->child = fork();
+	ASSERT_GE(self->child, 0)
+		cover_cleanup(self);
+	if (self->child == 0) {
+		close(to_parent[0]);
+		close(to_child[1]);
+		_exit(cover_child(self->base, to_parent[1], to_child[0]));
+	}
+	close(to_parent[1]);
+	close(to_child[0]);
+	self->to_child = to_child[1];
+	self->from_child = to_parent[0];
+	if (read(self->from_child, fds, sizeof(fds)) != sizeof(fds)) {
+		pid_t pid = self->child;
+
+		waitpid(pid, &status, 0);
+		self->child = 0;
+		cover_cleanup(self);
+		ASSERT_TRUE(false)
+			TH_LOG("child failed to set up: %s %d",
+			       WIFEXITED(status) ? "step" : "signal",
+			       WIFEXITED(status) ? WEXITSTATUS(status) : WTERMSIG(status));
+	}
+
+	pidfd = syscall(__NR_pidfd_open, self->child, 0);
+	ASSERT_GE(pidfd, 0)
+		cover_cleanup(self);
+	self->dfd = syscall(__NR_pidfd_getfd, pidfd, fds[0], 0);
+	self->cfd = syscall(__NR_pidfd_getfd, pidfd, fds[1], 0);
+	self->tfd = syscall(__NR_pidfd_getfd, pidfd, fds[2], 0);
+	self->ufd = syscall(__NR_pidfd_getfd, pidfd, fds[3], 0);
+	close(pidfd);
+	ASSERT_GE(self->dfd, 0)
+		cover_cleanup(self);
+	ASSERT_GE(self->cfd, 0)
+		cover_cleanup(self);
+	ASSERT_GE(self->tfd, 0)
+		cover_cleanup(self);
+	ASSERT_GE(self->ufd, 0)
+		cover_cleanup(self);
+
+	/* unmounts P and C, C leaves its cover behind */
+	ASSERT_EQ(rmdir(dir), 0)
+		cover_cleanup(self);
+	/* unmounts T and U with their children, two covers on one mountpoint */
+	snprintf(dir, sizeof(dir), "%s/t", self->base);
+	ASSERT_EQ(rmdir(dir), 0)
+		cover_cleanup(self);
+	snprintf(dir, sizeof(dir), "%s/u", self->base);
+	ASSERT_EQ(rmdir(dir), 0)
+		cover_cleanup(self);
+
+	self->fd = openat(self->dfd, "covered", O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+	ASSERT_GE(self->fd, 0)
+		cover_cleanup(self);
+	ASSERT_EQ(fstatfs(self->fd, &sf), 0)
+		cover_cleanup(self);
+	ASSERT_EQ(sf.f_type, NULL_FS_MAGIC)
+		cover_cleanup(self);
+}
+
+FIXTURE_TEARDOWN(mount_cover)
+{
+	cover_cleanup(self);
+}
+
+TEST_F(mount_cover, not_watchable)
+{
+	char p[PATH_LEN];
+	int ifd;
+
+	snprintf(p, sizeof(p), "/proc/self/fd/%d", self->fd);
+
+	ifd = inotify_init1(IN_CLOEXEC);
+	if (ifd < 0 && errno == ENOSYS) {
+		TH_LOG("no inotify in this kernel, skipping that part");
+	} else {
+		ASSERT_GE(ifd, 0);
+		EXPECT_EQ(inotify_add_watch(ifd, p, IN_OPEN), -1);
+		EXPECT_EQ(errno, EINVAL);
+		close(ifd);
+	}
+
+	/* dnotify ends up at the same place */
+	EXPECT_EQ(fcntl(self->fd, F_NOTIFY, DN_ACCESS), -1);
+	EXPECT_EQ(errno, EINVAL);
+
+	if (self->fan < 0) {
+		TH_LOG("no fanotify group without CAP_SYS_ADMIN in the initial user namespace, skipping the fanotify part");
+		return;
+	}
+	EXPECT_EQ(fanotify_mark(self->fan, FAN_MARK_ADD, FAN_OPEN, self->fd, NULL), -1);
+	EXPECT_EQ(errno, EINVAL);
+}
+
+TEST_F(mount_cover, not_lockable)
+{
+	struct flock fl = {
+		.l_type = F_RDLCK,
+		.l_whence = SEEK_SET,
+	};
+
+	EXPECT_EQ(flock(self->fd, LOCK_EX | LOCK_NB), -1);
+	EXPECT_EQ(errno, ENOLCK);
+	EXPECT_EQ(fcntl(self->fd, F_SETLK, &fl), -1);
+	EXPECT_EQ(errno, ENOLCK);
+	EXPECT_EQ(fcntl(self->fd, F_GETLK, &fl), -1);
+	EXPECT_EQ(errno, ENOLCK);
+}
+
+/* a lease is refused for the owner and for everybody else alike */
+TEST_F(mount_cover, not_leasable)
+{
+	struct delegation_req deleg = {
+		.d_type = F_RDLCK,
+	};
+
+	EXPECT_EQ(fcntl(self->fd, F_SETLEASE, F_RDLCK), -1);
+	EXPECT_TRUE(errno == EINVAL || errno == EACCES);
+	EXPECT_EQ(fcntl(self->fd, F_SETDELEG, &deleg), -1);
+	EXPECT_TRUE(errno == EINVAL || errno == EACCES);
+}
+
+TEST_F(mount_cover, not_mountable)
+{
+	char p[PATH_LEN];
+
+	snprintf(p, sizeof(p), "/proc/self/fd/%d", self->fd);
+	EXPECT_EQ(mount("tmpfs", p, "tmpfs", 0, NULL), -1);
+	EXPECT_EQ(errno, ENOENT);
+}
+
+TEST_F(mount_cover, read_only)
+{
+	struct statvfs sv;
+
+	EXPECT_EQ(mkdirat(self->fd, "x", 0755), -1);
+	EXPECT_EQ(errno, ENOENT);
+	EXPECT_EQ(fchmod(self->fd, 0777), -1);
+	EXPECT_EQ(errno, EROFS);
+	/* the immutable inode is checked before the read-only mount */
+	EXPECT_EQ(faccessat(self->fd, ".", W_OK, 0), -1);
+	EXPECT_EQ(errno, EPERM);
+	ASSERT_EQ(fstatvfs(self->fd, &sv), 0);
+	EXPECT_TRUE(sv.f_flag & ST_RDONLY);
+}
+
+/* the stand-in is a root of its own, ".." stays put */
+TEST_F(mount_cover, island)
+{
+	int fd;
+
+	fd = openat(self->fd, "..", O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+	ASSERT_GE(fd, 0);
+	EXPECT_TRUE(same_file(fd, self->fd));
+	close(fd);
+}
+
+/*
+ * C is alive for as long as the child holds it, but it can't be reached
+ * through P anymore and it's a root of its own as well.
+ */
+TEST_F(mount_cover, held_child_detached)
+{
+	struct statfs sf;
+	int fd;
+
+	ASSERT_EQ(fstatfs(self->cfd, &sf), 0);
+	EXPECT_EQ(sf.f_type, TMPFS_MAGIC);
+
+	fd = openat(self->cfd, "..", O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+	ASSERT_GE(fd, 0);
+	EXPECT_TRUE(same_file(fd, self->cfd));
+	close(fd);
+
+	fd = openat(self->dfd, "covered", O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+	ASSERT_GE(fd, 0);
+	ASSERT_EQ(fstatfs(fd, &sf), 0);
+	EXPECT_EQ(sf.f_type, NULL_FS_MAGIC);
+	close(fd);
+}
+
+/*
+ * The cover goes with its mountpoint: once the child has removed C's
+ * mountpoint through the bind of P, the name is gone from P as well.
+ * What was opened through the cover before stays open.
+ */
+TEST_F(mount_cover, cover_goes_with_mountpoint)
+{
+	char cmd = CMD_RMDIR;
+	struct statfs sf;
+	int ret;
+
+	ASSERT_EQ(write(self->to_child, &cmd, 1), 1);
+	ASSERT_EQ(read(self->from_child, &ret, sizeof(ret)), sizeof(ret));
+	ASSERT_EQ(ret, 0);
+
+	EXPECT_EQ(openat(self->dfd, "covered", O_RDONLY | O_DIRECTORY | O_CLOEXEC), -1);
+	EXPECT_EQ(errno, ENOENT);
+	ASSERT_EQ(fstatfs(self->fd, &sf), 0);
+	EXPECT_EQ(sf.f_type, NULL_FS_MAGIC);
+}
+
+/*
+ * A readdir of the stand-in holds nothing that others wait for: one holder
+ * sticks in the page fault of its buffer, and a create and a lookup of
+ * another come back meanwhile.
+ */
+TEST_F(mount_cover, readdir_blocks_nobody)
+{
+	bool stalled;
+
+	if (self->hold.uffd < 0)
+		SKIP(return, "test requires userfaultfd");
+	ASSERT_EQ(readdir_hold_check(&self->hold, self->fd, &stalled), 0);
+	EXPECT_FALSE(stalled);
+}
+
+/* where a file was mounted, the stand-in is an empty regular file */
+TEST_F(mount_cover, file_stand_in)
+{
+	char src[PATH_LEN], p[PATH_LEN];
+	struct statfs sf;
+	struct stat st;
+	char c;
+	int fd;
+
+	fd = openat(self->dfd, "file", O_RDONLY | O_CLOEXEC);
+	ASSERT_GE(fd, 0);
+	ASSERT_EQ(fstat(fd, &st), 0);
+	EXPECT_TRUE(S_ISREG(st.st_mode));
+	ASSERT_EQ(fstatfs(fd, &sf), 0);
+	EXPECT_EQ(sf.f_type, NULL_FS_MAGIC);
+	/* the two stand-ins are two inodes */
+	EXPECT_FALSE(same_file(fd, self->fd));
+	EXPECT_EQ(read(fd, &c, 1), 0);
+	EXPECT_EQ(flock(fd, LOCK_EX | LOCK_NB), -1);
+	EXPECT_EQ(errno, ENOLCK);
+
+	snprintf(src, sizeof(src), "%s/src", self->base);
+	ASSERT_EQ(touch(src), 0);
+	snprintf(p, sizeof(p), "/proc/self/fd/%d", fd);
+	EXPECT_EQ(mount(src, p, NULL, MS_BIND, NULL), -1);
+	EXPECT_EQ(errno, ENOENT);
+	close(fd);
+
+	EXPECT_EQ(openat(self->dfd, "file", O_WRONLY | O_CLOEXEC), -1);
+	EXPECT_EQ(errno, EPERM);
+	EXPECT_EQ(openat(self->dfd, "file", O_RDONLY | O_DIRECTORY | O_CLOEXEC), -1);
+	EXPECT_EQ(errno, ENOTDIR);
+}
+
+/*
+ * Two unmounted parents left covers on the same dentry. A lookup on
+ * either finds a stand-in, and U's cover stays when T goes.
+ */
+TEST_F(mount_cover, shared_mountpoint)
+{
+	char cmd = CMD_CLOSE_T;
+	struct statfs sf;
+	int fd, ret;
+
+	fd = openat(self->tfd, "covered", O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+	ASSERT_GE(fd, 0);
+	ASSERT_EQ(fstatfs(fd, &sf), 0);
+	EXPECT_EQ(sf.f_type, NULL_FS_MAGIC);
+	close(fd);
+
+	fd = openat(self->ufd, "covered", O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+	ASSERT_GE(fd, 0);
+	ASSERT_EQ(fstatfs(fd, &sf), 0);
+	EXPECT_EQ(sf.f_type, NULL_FS_MAGIC);
+	close(fd);
+
+	/* the last references to T go, with T its cover */
+	ASSERT_EQ(write(self->to_child, &cmd, 1), 1);
+	ASSERT_EQ(read(self->from_child, &ret, sizeof(ret)), sizeof(ret));
+	ASSERT_EQ(ret, 0);
+	close(self->tfd);
+	self->tfd = -1;
+
+	fd = openat(self->ufd, "covered", O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+	ASSERT_GE(fd, 0);
+	ASSERT_EQ(fstatfs(fd, &sf), 0);
+	EXPECT_EQ(sf.f_type, NULL_FS_MAGIC);
+	close(fd);
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/filesystems/mount_cycle/settings b/tools/testing/selftests/filesystems/mount_cycle/settings
new file mode 100644
index 000000000000..694d70710ff0
--- /dev/null
+++ b/tools/testing/selftests/filesystems/mount_cycle/settings
@@ -0,0 +1 @@
+timeout=300

-- 
2.53.0


      parent reply	other threads:[~2026-10-02 14:14 UTC|newest]

Thread overview: 6+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-10-02 14:14 [PATCH 0/3] namespace: rework connected mounts Christian Brauner
2026-10-02 14:14 ` [PATCH 1/3] nullfs: add an empty immutable regular file Christian Brauner
2026-10-02 14:31   ` Jann Horn
2026-10-05 10:33     ` Christian Brauner
2026-10-02 14:14 ` [PATCH 2/3] namespace: rework connected mounts Christian Brauner
2026-10-02 14:14 ` Christian Brauner [this message]

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20261002-work-mount-cover-v1-3-232a8f52b43c@kernel.org \
    --to=brauner@kernel.org \
    --cc=amir73il@gmail.com \
    --cc=jack@suse.cz \
    --cc=jannh@google.com \
    --cc=linux-fsdevel@vger.kernel.org \
    --cc=torvalds@linux-foundation.org \
    --cc=viro@zeniv.linux.org.uk \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox