Linux block layer
 help / color / mirror / Atom feed
From: Bart Van Assche <bvanassche@acm.org>
To: Jens Axboe <axboe@kernel.dk>
Cc: linux-block@vger.kernel.org, Christoph Hellwig <hch@lst.de>,
	Tetsuo Handa <penguin-kernel@I-love.SAKURA.ne.jp>,
	Nilay Shroff <nilay@linux.ibm.com>,
	Bart Van Assche <bvanassche@acm.org>
Subject: [PATCH v2 2/2] loop: Perform __loop_clr_fd() after disk->open_mutex is dropped
Date: Thu, 17 Sep 2026 14:20:26 -0700	[thread overview]
Message-ID: <cdaebcf71c5de3e994f535dfdc9daa15a77bf59b.1789680010.git.bvanassche@acm.org> (raw)
In-Reply-To: <72661d0de27da8c843faf18f9b79442b0e1bb6b5.1789680010.git.bvanassche@acm.org>

In order to prevent NULL pointer dereferences in lo_rw_aio() when tearing
down a loop device, outstanding I/O must be flushed before clearing the
backing file and device state. However, calling blk_mq_wait_quiesce_done(),
drain_workqueue(), or blk_mq_freeze_queue() with disk->open_mutex held
causes lockdep warnings and potential deadlocks.

Use the .post_release() block device operation to execute __loop_clr_fd()
synchronously after disk->open_mutex has been released by the block layer.
Inside __loop_clr_fd(), outstanding I/O is flushed and the request queue
is frozen before acquiring disk->open_mutex to perform the remaining
device teardown and partition rescans.

Introduce a new loop device state to ensure that __loop_clr_fd() clears
a loop device once even if it is called multiple times concurrently.

Signed-off-by: Bart Van Assche <bvanassche@acm.org>
---
 drivers/block/loop.c | 60 ++++++++++++++++++++++++++++++++------------
 1 file changed, 44 insertions(+), 16 deletions(-)

diff --git a/drivers/block/loop.c b/drivers/block/loop.c
index 758c20678bf6..0ebee9a8816a 100644
--- a/drivers/block/loop.c
+++ b/drivers/block/loop.c
@@ -42,6 +42,7 @@ enum {
 	Lo_unbound,
 	Lo_bound,
 	Lo_rundown,
+	Lo_clearing,
 	Lo_deleting,
 };
 
@@ -1138,11 +1139,38 @@ static int loop_configure(struct loop_device *lo, blk_mode_t mode,
 
 static void __loop_clr_fd(struct loop_device *lo)
 {
+	struct gendisk *disk = lo->lo_disk;
 	struct queue_limits lim;
 	struct file *filp;
 	gfp_t gfp = lo->old_gfp_mask;
+	unsigned int memflags;
 	int err;
 
+	scoped_guard(mutex, &lo->lo_mutex) {
+		if (READ_ONCE(lo->lo_state) != Lo_rundown)
+			return;
+		WRITE_ONCE(lo->lo_state, Lo_clearing);
+	}
+
+	/*
+	 * Wait for ongoing loop_queue_rq() calls. Subsequent loop_queue_rq()
+	 * calls which are made after this call returned will see lo->lo_state
+	 * != Lo_bound and return with BLK_STS_IOERR.
+	 */
+	blk_mq_quiesce_queue(lo->lo_queue);
+	blk_mq_unquiesce_queue(lo->lo_queue);
+
+	/* loop_queue_rq() queues work on lo->workqueue, hence drain it. */
+	drain_workqueue(lo->workqueue);
+
+	lim = queue_limits_start_update(lo->lo_queue);
+
+	/*
+	 * Freeze the request queue while updating parameters used while
+	 * processing requests.
+	 */
+	memflags = blk_mq_freeze_queue(lo->lo_queue);
+
 	spin_lock_irq(&lo->lo_lock);
 	filp = lo->lo_backing_file;
 	lo->lo_backing_file = NULL;
@@ -1153,18 +1181,17 @@ static void __loop_clr_fd(struct loop_device *lo)
 	lo->lo_sizelimit = 0;
 	memset(lo->lo_file_name, 0, LO_NAME_SIZE);
 
-	/*
-	 * Reset the block size to the default.
-	 *
-	 * No queue freezing needed because this is called from the final
-	 * ->release call only, so there can't be any outstanding I/O.
-	 */
-	lim = queue_limits_start_update(lo->lo_queue);
+	/* Reset the block size to the default. */
 	lim.logical_block_size = SECTOR_SIZE;
 	lim.physical_block_size = SECTOR_SIZE;
 	lim.io_min = SECTOR_SIZE;
 	queue_limits_commit_update(lo->lo_queue, &lim);
 
+	blk_mq_unfreeze_queue(lo->lo_queue, memflags);
+
+	/* Serialize against concurrent bdev_open() calls. */
+	mutex_lock(&disk->open_mutex);
+
 	invalidate_disk(lo->lo_disk);
 	loop_sysfs_exit(lo);
 	/* let user-space know about this change */
@@ -1178,9 +1205,6 @@ static void __loop_clr_fd(struct loop_device *lo)
 	/*
 	 * Remove all partitions, including partitions added manually with
 	 * BLKPG, which may exist even if LO_FLAGS_PARTSCAN is not set.
-	 *
-	 * open_mutex has been held already in release path, so don't acquire
-	 * it here.
 	 */
 	err = bdev_disk_changed(lo->lo_disk, false);
 	if (err)
@@ -1197,6 +1221,8 @@ static void __loop_clr_fd(struct loop_device *lo)
 	lo->lo_flags = 0;
 	if (!part_shift)
 		set_bit(GD_SUPPRESS_PART_SCAN, &lo->lo_disk->state);
+	mutex_unlock(&disk->open_mutex);
+
 	mutex_lock(&lo->lo_mutex);
 	WRITE_ONCE(lo->lo_state, Lo_unbound);
 	mutex_unlock(&lo->lo_mutex);
@@ -1745,7 +1771,7 @@ static int lo_open(struct gendisk *disk, blk_mode_t mode)
 	if (err)
 		return err;
 
-	if (lo->lo_state == Lo_deleting || lo->lo_state == Lo_rundown)
+	if (lo->lo_state != Lo_bound && lo->lo_state != Lo_unbound)
 		err = -ENXIO;
 	mutex_unlock(&lo->lo_mutex);
 	return err;
@@ -1754,7 +1780,6 @@ static int lo_open(struct gendisk *disk, blk_mode_t mode)
 static void lo_release(struct gendisk *disk)
 {
 	struct loop_device *lo = disk->private_data;
-	bool need_clear = false;
 
 	if (disk_openers(disk) > 0)
 		return;
@@ -1767,12 +1792,14 @@ static void lo_release(struct gendisk *disk)
 	mutex_lock(&lo->lo_mutex);
 	if (lo->lo_state == Lo_bound && (lo->lo_flags & LO_FLAGS_AUTOCLEAR))
 		WRITE_ONCE(lo->lo_state, Lo_rundown);
-
-	need_clear = (lo->lo_state == Lo_rundown);
 	mutex_unlock(&lo->lo_mutex);
+}
+
+static void lo_post_release(struct gendisk *disk)
+{
+	struct loop_device *lo = disk->private_data;
 
-	if (need_clear)
-		__loop_clr_fd(lo);
+	__loop_clr_fd(lo);
 }
 
 static void lo_free_disk(struct gendisk *disk)
@@ -1791,6 +1818,7 @@ static const struct block_device_operations lo_fops = {
 	.owner =	THIS_MODULE,
 	.open =         lo_open,
 	.release =	lo_release,
+	.post_release = lo_post_release,
 	.ioctl =	lo_ioctl,
 #ifdef CONFIG_COMPAT
 	.compat_ioctl =	lo_compat_ioctl,

      reply	other threads:[~2026-09-17 21:20 UTC|newest]

Thread overview: 3+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-17 21:20 [PATCH v2 0/2] loop: Fix teardown Bart Van Assche
2026-09-17 21:20 ` [PATCH v2 1/2] block: Add post_release() operation Bart Van Assche
2026-09-17 21:20   ` Bart Van Assche [this message]

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=cdaebcf71c5de3e994f535dfdc9daa15a77bf59b.1789680010.git.bvanassche@acm.org \
    --to=bvanassche@acm.org \
    --cc=axboe@kernel.dk \
    --cc=hch@lst.de \
    --cc=linux-block@vger.kernel.org \
    --cc=nilay@linux.ibm.com \
    --cc=penguin-kernel@I-love.SAKURA.ne.jp \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox