* [PATCH 1/2] sbitmap: add __sbitmap_add_wait_queue()
2026-09-20 10:11 [PATCH 0/2] blk-mq: fix dispatch waiter accounting for reserved tags huhai
@ 2026-09-20 10:11 ` huhai
2026-09-20 10:11 ` [PATCH 2/2] blk-mq: fix dispatch waiter accounting for reserved tags huhai
1 sibling, 0 replies; 3+ messages in thread
From: huhai @ 2026-09-20 10:11 UTC (permalink / raw)
To: axboe; +Cc: linux-block, shikemeng, Henry Hu
From: Henry Hu <huhai@kylinos.cn>
sbitmap_add_wait_queue() takes the waitqueue lock, so a caller that
already holds it cannot use that helper.
Add __sbitmap_add_wait_queue() for that case. It updates the sbitmap
accounting and inserts the wait entry as non-exclusive, and asserts
with lockdep_assert_held() that the caller holds the waitqueue lock.
sbitmap_add_wait_queue() now takes the lock and calls it, so the
accounting and insertion stay under the same lock.
blk-mq can then use the sbitmap accounting helpers without changing
its waitqueue and dispatch lock ordering.
Assisted-by: LLM
Signed-off-by: Henry Hu <huhai@kylinos.cn>
---
include/linux/sbitmap.h | 9 ++++++++-
lib/sbitmap.c | 23 +++++++++++++++++++----
2 files changed, 27 insertions(+), 5 deletions(-)
diff --git a/include/linux/sbitmap.h b/include/linux/sbitmap.h
index cc7ad189caa5..ab8f34a97aa0 100644
--- a/include/linux/sbitmap.h
+++ b/include/linux/sbitmap.h
@@ -614,6 +614,13 @@ void sbitmap_prepare_to_wait(struct sbitmap_queue *sbq,
void sbitmap_finish_wait(struct sbitmap_queue *sbq, struct sbq_wait_state *ws,
struct sbq_wait *sbq_wait);
+/*
+ * Like sbitmap_add_wait_queue(), but the caller must hold ws->wait.lock.
+ */
+void __sbitmap_add_wait_queue(struct sbitmap_queue *sbq,
+ struct sbq_wait_state *ws,
+ struct sbq_wait *sbq_wait);
+
/*
* Wrapper around add_wait_queue(), which maintains some extra internal state
*/
@@ -622,7 +629,7 @@ void sbitmap_add_wait_queue(struct sbitmap_queue *sbq,
struct sbq_wait *sbq_wait);
/*
- * Must be paired with sbitmap_add_wait_queue()
+ * Must be paired with sbitmap_add_wait_queue() or __sbitmap_add_wait_queue().
*/
void sbitmap_del_wait_queue(struct sbq_wait *sbq_wait);
diff --git a/lib/sbitmap.c b/lib/sbitmap.c
index 4d188d05db15..eb450c6bdaba 100644
--- a/lib/sbitmap.c
+++ b/lib/sbitmap.c
@@ -756,16 +756,31 @@ void sbitmap_queue_show(struct sbitmap_queue *sbq, struct seq_file *m)
}
EXPORT_SYMBOL_GPL(sbitmap_queue_show);
-void sbitmap_add_wait_queue(struct sbitmap_queue *sbq,
- struct sbq_wait_state *ws,
- struct sbq_wait *sbq_wait)
+void __sbitmap_add_wait_queue(struct sbitmap_queue *sbq,
+ struct sbq_wait_state *ws,
+ struct sbq_wait *sbq_wait)
{
+ lockdep_assert_held(&ws->wait.lock);
+
if (!sbq_wait->sbq) {
sbq_wait->sbq = sbq;
atomic_inc(&sbq->ws_active);
- add_wait_queue(&ws->wait, &sbq_wait->wait);
+ sbq_wait->wait.flags &= ~WQ_FLAG_EXCLUSIVE;
+ __add_wait_queue(&ws->wait, &sbq_wait->wait);
}
}
+EXPORT_SYMBOL_GPL(__sbitmap_add_wait_queue);
+
+void sbitmap_add_wait_queue(struct sbitmap_queue *sbq,
+ struct sbq_wait_state *ws,
+ struct sbq_wait *sbq_wait)
+{
+ unsigned long flags;
+
+ spin_lock_irqsave(&ws->wait.lock, flags);
+ __sbitmap_add_wait_queue(sbq, ws, sbq_wait);
+ spin_unlock_irqrestore(&ws->wait.lock, flags);
+}
EXPORT_SYMBOL_GPL(sbitmap_add_wait_queue);
void sbitmap_del_wait_queue(struct sbq_wait *sbq_wait)
--
2.40.1
^ permalink raw reply related [flat|nested] 3+ messages in thread* [PATCH 2/2] blk-mq: fix dispatch waiter accounting for reserved tags
2026-09-20 10:11 [PATCH 0/2] blk-mq: fix dispatch waiter accounting for reserved tags huhai
2026-09-20 10:11 ` [PATCH 1/2] sbitmap: add __sbitmap_add_wait_queue() huhai
@ 2026-09-20 10:11 ` huhai
1 sibling, 0 replies; 3+ messages in thread
From: huhai @ 2026-09-20 10:11 UTC (permalink / raw)
To: axboe; +Cc: linux-block, shikemeng, Henry Hu
From: Henry Hu <huhai@kylinos.cn>
blk_mq_mark_tag_wait() can queue dispatch_wait on breserved_tags and
increment its ws_active. But blk_mq_dispatch_wake() always decrements
tags->bitmap_tags.ws_active. The reserved counter stays elevated and
bitmap_tags.ws_active gets decremented without a matching increment.
If bitmap_tags.ws_active reaches zero while waiters are still queued,
sbitmap_queue_wake_up() skips their wakeups and dispatch can stall.
Make dispatch_wait a struct sbq_wait that records the sbitmap_queue it
was accounted against. Add it with __sbitmap_add_wait_queue() while
holding the waitqueue lock, and remove it with sbitmap_del_wait_queue()
on both the wakeup and successful allocation retry paths, so the
matching ws_active counter is decremented and the association is
cleared.
Fixes: 98b99e9412d0 ("blk-mq: wait on correct sbitmap_queue in blk_mq_mark_tag_wait")
Assisted-by: LLM
Signed-off-by: Henry Hu <huhai@kylinos.cn>
---
block/blk-mq.c | 33 +++++++++++++--------------------
include/linux/blk-mq.h | 2 +-
2 files changed, 14 insertions(+), 21 deletions(-)
diff --git a/block/blk-mq.c b/block/blk-mq.c
index a26a11c73ee3..2e07aad3853a 100644
--- a/block/blk-mq.c
+++ b/block/blk-mq.c
@@ -1863,16 +1863,10 @@ static int blk_mq_dispatch_wake(wait_queue_entry_t *wait, unsigned mode,
{
struct blk_mq_hw_ctx *hctx;
- hctx = container_of(wait, struct blk_mq_hw_ctx, dispatch_wait);
+ hctx = container_of(wait, struct blk_mq_hw_ctx, dispatch_wait.wait);
spin_lock(&hctx->dispatch_wait_lock);
- if (!list_empty(&wait->entry)) {
- struct sbitmap_queue *sbq;
-
- list_del_init(&wait->entry);
- sbq = &hctx->tags->bitmap_tags;
- atomic_dec(&sbq->ws_active);
- }
+ sbitmap_del_wait_queue(&hctx->dispatch_wait);
spin_unlock(&hctx->dispatch_wait_lock);
blk_mq_run_hw_queue(hctx, true);
@@ -1889,8 +1883,9 @@ static bool blk_mq_mark_tag_wait(struct blk_mq_hw_ctx *hctx,
struct request *rq)
{
struct sbitmap_queue *sbq;
+ struct sbq_wait_state *ws;
struct wait_queue_head *wq;
- wait_queue_entry_t *wait;
+ struct sbq_wait *wait;
bool ret;
if (!(hctx->flags & BLK_MQ_F_TAG_QUEUE_SHARED) &&
@@ -1909,26 +1904,25 @@ static bool blk_mq_mark_tag_wait(struct blk_mq_hw_ctx *hctx,
}
wait = &hctx->dispatch_wait;
- if (!list_empty_careful(&wait->entry))
+ if (!list_empty_careful(&wait->wait.entry))
return false;
if (blk_mq_tag_is_reserved(rq->mq_hctx->sched_tags, rq->internal_tag))
sbq = &hctx->tags->breserved_tags;
else
sbq = &hctx->tags->bitmap_tags;
- wq = &bt_wait_ptr(sbq, hctx)->wait;
+ ws = bt_wait_ptr(sbq, hctx);
+ wq = &ws->wait;
spin_lock_irq(&wq->lock);
spin_lock(&hctx->dispatch_wait_lock);
- if (!list_empty(&wait->entry)) {
+ if (!list_empty(&wait->wait.entry)) {
spin_unlock(&hctx->dispatch_wait_lock);
spin_unlock_irq(&wq->lock);
return false;
}
- atomic_inc(&sbq->ws_active);
- wait->flags &= ~WQ_FLAG_EXCLUSIVE;
- __add_wait_queue(wq, wait);
+ __sbitmap_add_wait_queue(sbq, ws, wait);
/*
* Add one explicit barrier since blk_mq_get_driver_tag() may
@@ -1962,8 +1956,7 @@ static bool blk_mq_mark_tag_wait(struct blk_mq_hw_ctx *hctx,
* We got a tag, remove ourselves from the wait queue to ensure
* someone else gets the wakeup.
*/
- list_del_init(&wait->entry);
- atomic_dec(&sbq->ws_active);
+ sbitmap_del_wait_queue(wait);
spin_unlock(&hctx->dispatch_wait_lock);
spin_unlock_irq(&wq->lock);
@@ -2197,7 +2190,7 @@ bool blk_mq_dispatch_rq_list(struct blk_mq_hw_ctx *hctx, struct list_head *list,
if (prep == PREP_DISPATCH_NO_BUDGET)
needs_resource = true;
if (!needs_restart ||
- (no_tag && list_empty_careful(&hctx->dispatch_wait.entry)))
+ (no_tag && list_empty_careful(&hctx->dispatch_wait.wait.entry)))
blk_mq_run_hw_queue(hctx, true);
else if (needs_resource)
blk_mq_delay_run_hw_queue(hctx, BLK_MQ_RESOURCE_DELAY);
@@ -4058,8 +4051,8 @@ blk_mq_alloc_hctx(struct request_queue *q, struct blk_mq_tag_set *set,
hctx->nr_ctx = 0;
spin_lock_init(&hctx->dispatch_wait_lock);
- init_waitqueue_func_entry(&hctx->dispatch_wait, blk_mq_dispatch_wake);
- INIT_LIST_HEAD(&hctx->dispatch_wait.entry);
+ init_waitqueue_func_entry(&hctx->dispatch_wait.wait, blk_mq_dispatch_wake);
+ INIT_LIST_HEAD(&hctx->dispatch_wait.wait.entry);
blk_mq_hctx_kobj_init(hctx);
diff --git a/include/linux/blk-mq.h b/include/linux/blk-mq.h
index af878597afb8..7912139bca9f 100644
--- a/include/linux/blk-mq.h
+++ b/include/linux/blk-mq.h
@@ -407,7 +407,7 @@ struct blk_mq_hw_ctx {
* @dispatch_wait: Waitqueue to put requests when there is no tag
* available at the moment, to wait for another try in the future.
*/
- wait_queue_entry_t dispatch_wait;
+ struct sbq_wait dispatch_wait;
/**
* @wait_index: Index of next available dispatch_wait queue to insert
--
2.40.1
^ permalink raw reply related [flat|nested] 3+ messages in thread