Rust for Linux List
 help / color / mirror / Atom feed
* [PATCH v5 1/2] block: Add post_release() operation
@ 2026-09-23  5:09 Tetsuo Handa
  2026-09-23  5:10 ` [PATCH v5 2/2] loop: Perform __loop_clr_fd() from post_release callback Tetsuo Handa
                   ` (3 more replies)
  0 siblings, 4 replies; 10+ messages in thread
From: Tetsuo Handa @ 2026-09-23  5:09 UTC (permalink / raw)
  To: Bart Van Assche, Jens Axboe, Christoph Hellwig, Jan Kara,
	linux-block
  Cc: Al Viro, Andrew Morton, Brian Foster, Damien Le Moal,
	Hillf Danton, Markus Elfring, Ming Lei, Qu Wenruo, Tao Cui,
	kernel test robot, rust-for-linux, Linus Torvalds, Nilay Shroff

Add post_release() block device operation which provides a hook for
performing synchronous cleanup without disk->open_mutex held, which is
needed by the loop devices.

Real-world container engines, test suites, and system utilities rely on
fput() from __loop_clr_fd() being completed when lo_release() returns.
But changes which went to the v7.1 merge window broke an assumption that
there is no outstanding I/O when __loop_clr_fd() is called, causing NULL
pointer dereference problem in lo_rw_aio().

In order to fix this regression, we want to allow __loop_clr_fd() to flush
outstanding I/O. But calling drain_workqueue() from __loop_clr_fd() with
disk->open_mutex held causes lockdep warnings. We need a mechanism which
can flush outstanding I/O without disk->open_mutex held.

But deferring __loop_clr_fd() to WQ context has a problem that there is no
way to wait for completion of __loop_clr_fd() before the calling thread
returns to the userspace, for there is no hook for calling flush_work().
Despite what LO_FLAGS_AUTOCLEAR can guarantee is to clear backing device
"eventually" after the last thread called lo_release(), abovementioned
programs are expecting "synchronously" when a thread who is going to call
umount() or open() as soon as returning from close() called close().
That is an unsatisfiable expectation because the former is an objective
behavior and the latter is a subjective dependency. Nonetheless, we need
to try to wait for completion of __loop_clr_fd() at best-effort basis.

Also, since __loop_clr_fd() calls module_put(THIS_MODULE) and there is no
API for waiting for completion of remote thread's task work context,
deferring __loop_clr_fd() to task work context has a problem (aside from
task_work_add() being not exported to loadable modules) that module unload
operation can unmap code/data segment before __loop_clr_fd() completes.

Therefore, allow the loop driver to safely know completion of
__loop_clr_fd(), by adding a hook which is called after disk->open_mutex
is released.

Signed-off-by: Tetsuo Handa <penguin-kernel@I-love.SAKURA.ne.jp>
---
Changes in v5:
  Since Bart Van Assche commented that post_release() is more clear
  and more descriptive than run_todo(), renamed run_todo() back to
  post_release(), and make it be called only if release() was called.

 block/bdev.c                     | 28 +++++++++++++++++++---------
 include/linux/blkdev.h           | 10 ++++++++++
 rust/kernel/block/mq/gen_disk.rs |  1 +
 3 files changed, 30 insertions(+), 9 deletions(-)

diff --git a/block/bdev.c b/block/bdev.c
index cd8323083740..50cd82719236 100644
--- a/block/bdev.c
+++ b/block/bdev.c
@@ -766,7 +766,8 @@ static void blkdev_put_whole(struct block_device *bdev)
 		bdev->bd_disk->fops->release(bdev->bd_disk);
 }
 
-static int blkdev_get_whole(struct block_device *bdev, blk_mode_t mode)
+static int blkdev_get_whole(struct block_device *bdev, blk_mode_t mode,
+			    bool *called_release)
 {
 	struct gendisk *disk = bdev->bd_disk;
 	int ret;
@@ -793,18 +794,20 @@ static int blkdev_get_whole(struct block_device *bdev, blk_mode_t mode)
 		ret = bdev_disk_changed(disk, false);
 		if (ret && (mode & BLK_OPEN_STRICT_SCAN)) {
 			blkdev_put_whole(bdev);
+			*called_release = true;
 			return ret;
 		}
 	}
 	return 0;
 }
 
-static int blkdev_get_part(struct block_device *part, blk_mode_t mode)
+static int blkdev_get_part(struct block_device *part, blk_mode_t mode,
+			   bool *called_release)
 {
 	struct gendisk *disk = part->bd_disk;
 	int ret;
 
-	ret = blkdev_get_whole(bdev_whole(part), mode);
+	ret = blkdev_get_whole(bdev_whole(part), mode, called_release);
 	if (ret)
 		return ret;
 
@@ -821,6 +824,7 @@ static int blkdev_get_part(struct block_device *part, blk_mode_t mode)
 
 out_blkdev_put:
 	blkdev_put_whole(bdev_whole(part));
+	*called_release = true;
 	return ret;
 }
 
@@ -974,6 +978,8 @@ int bdev_open(struct block_device *bdev, blk_mode_t mode, void *holder,
 	bool unblock_events = true;
 	struct gendisk *disk = bdev->bd_disk;
 	int ret;
+	bool called_release = false;
+	struct module *fops_owner = NULL;
 
 	if (holder) {
 		mode |= BLK_OPEN_EXCL;
@@ -993,15 +999,16 @@ int bdev_open(struct block_device *bdev, blk_mode_t mode, void *holder,
 		goto abort_claiming;
 	if (!try_module_get(disk->fops->owner))
 		goto abort_claiming;
+	fops_owner = disk->fops->owner;
 	ret = -EBUSY;
 	if (!bdev_may_open(bdev, mode))
-		goto put_module;
+		goto abort_claiming;
 	if (bdev_is_partition(bdev))
-		ret = blkdev_get_part(bdev, mode);
+		ret = blkdev_get_part(bdev, mode, &called_release);
 	else
-		ret = blkdev_get_whole(bdev, mode);
+		ret = blkdev_get_whole(bdev, mode, &called_release);
 	if (ret)
-		goto put_module;
+		goto abort_claiming;
 	bdev_claim_write_access(bdev, mode);
 	if (holder) {
 		bd_finish_claiming(bdev, holder, hops);
@@ -1036,13 +1043,14 @@ int bdev_open(struct block_device *bdev, blk_mode_t mode, void *holder,
 	bdev_file->private_data = holder;
 
 	return 0;
-put_module:
-	module_put(disk->fops->owner);
 abort_claiming:
 	if (holder)
 		bd_abort_claiming(bdev, holder);
 	mutex_unlock(&disk->open_mutex);
 	disk_unblock_events(disk);
+	if (called_release && disk->fops->post_release)
+		disk->fops->post_release(disk);
+	module_put(fops_owner);
 	return ret;
 }
 
@@ -1188,6 +1196,8 @@ void bdev_release(struct file *bdev_file)
 	else
 		blkdev_put_whole(bdev);
 	mutex_unlock(&disk->open_mutex);
+	if (disk->fops->post_release)
+		disk->fops->post_release(disk);
 
 	module_put(disk->fops->owner);
 put_no_open:
diff --git a/include/linux/blkdev.h b/include/linux/blkdev.h
index 4f7905c3412b..5354eb006c02 100644
--- a/include/linux/blkdev.h
+++ b/include/linux/blkdev.h
@@ -1577,6 +1577,16 @@ struct block_device_operations {
 			unsigned int flags);
 	int (*open)(struct gendisk *disk, blk_mode_t mode);
 	void (*release)(struct gendisk *disk);
+	/*
+	 * This operation is for performing synchronous cleanup after
+	 * release() was called. Since this operation is called without
+	 * disk->open_mutex held, users of this operation must implement
+	 * appropriate serialization. For example, even if thread-A called
+	 * release() operation before thread-B calls release() operation,
+	 * it is possible that thread-B calls post_release() operation
+	 * before thread-A calls post_release() operation.
+	 */
+	void (*post_release)(struct gendisk *disk);
 	int (*ioctl)(struct block_device *bdev, blk_mode_t mode,
 			unsigned cmd, unsigned long arg);
 	int (*compat_ioctl)(struct block_device *bdev, blk_mode_t mode,
diff --git a/rust/kernel/block/mq/gen_disk.rs b/rust/kernel/block/mq/gen_disk.rs
index fc97dd873974..2ff77ef49781 100644
--- a/rust/kernel/block/mq/gen_disk.rs
+++ b/rust/kernel/block/mq/gen_disk.rs
@@ -129,6 +129,7 @@ pub fn build<T: Operations>(
             submit_bio: None,
             open: None,
             release: None,
+            post_release: None,
             ioctl: None,
             compat_ioctl: None,
             check_events: None,
-- 
2.52.0


^ permalink raw reply related	[flat|nested] 10+ messages in thread

end of thread, other threads:[~2026-10-05 21:07 UTC | newest]

Thread overview: 10+ messages (download: mbox.gz follow: Atom feed
-- links below jump to the message on this page --
2026-09-23  5:09 [PATCH v5 1/2] block: Add post_release() operation Tetsuo Handa
2026-09-23  5:10 ` [PATCH v5 2/2] loop: Perform __loop_clr_fd() from post_release callback Tetsuo Handa
2026-09-23 17:16   ` Bart Van Assche
2026-09-23 17:14 ` [PATCH v5 1/2] block: Add post_release() operation Bart Van Assche
2026-09-23 18:38 ` Gary Guo
2026-09-24 13:46   ` Tetsuo Handa
2026-09-28 17:29 ` Bart Van Assche
2026-10-05  8:45   ` Christoph Hellwig
2026-10-05 14:23     ` Tetsuo Handa
2026-10-05 21:07     ` Bart Van Assche

This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox