* [PATCHv2 1/2] blk-mq: set RQF_USE_SCHED when the operation is known
@ 2026-09-22 17:27 Keith Busch
2026-09-22 17:27 ` [PATCHv2 2/2] blk-mq: allow cached requests to be used for flush operations Keith Busch
` (2 more replies)
0 siblings, 3 replies; 5+ messages in thread
From: Keith Busch @ 2026-09-22 17:27 UTC (permalink / raw)
To: axboe, linux-block; +Cc: hare, hch, Keith Busch, Henry Hu
From: Keith Busch <kbusch@kernel.org>
The cached requests are allocated for one operation but can be handed
out for another. A passthrough command has RQF_USE_SCHED cleared, so
using those flags for a subsequent read/write bio will insert it into
the scheduler without ->prepare_request() and frees it without
->finish_request(). For kyber, this leaks the domain token acquired at
dispatch and stalls the queue.
Don't set RQF_USE_SCHED based on the first operation the batch happened
to be allocated for. Instead, set it after the request is claimed by an
operation. Introduce a helper function, blk_mq_rq_late_init() for this,
and absorb blk_mq_rq_time_init() since the two run together.
Fixes: 4b6a5d9cea91 ("block: enable batched allocation for blk_mq_alloc_request()")
Link: https://lore.kernel.org/linux-block/20260916151655.2588-1-15815827059@163.com/
Reported-by: Henry Hu <huhai@kylinos.cn>
Signed-off-by: Keith Busch <kbusch@kernel.org>
---
v1->v2:
Introduced a new helper function for the late request init
Split off the flush handling into a separate patch.
Updated mq-deadline comment
block/blk-mq.c | 59 ++++++++++++++++++++++++++++-----------------
block/mq-deadline.c | 2 +-
2 files changed, 38 insertions(+), 23 deletions(-)
diff --git a/block/blk-mq.c b/block/blk-mq.c
index a26a11c73ee3e..5df6d2244db82 100644
--- a/block/blk-mq.c
+++ b/block/blk-mq.c
@@ -447,16 +447,6 @@ static struct request *blk_mq_rq_ctx_init(struct blk_mq_alloc_data *data,
WRITE_ONCE(rq->deadline, 0);
req_ref_set(rq, 1);
- if (rq->rq_flags & RQF_USE_SCHED) {
- struct elevator_queue *e = data->q->elevator;
-
- INIT_HLIST_NODE(&rq->hash);
- RB_CLEAR_NODE(&rq->rb_node);
-
- if (e->type->ops.prepare_request)
- e->type->ops.prepare_request(rq);
- }
-
return rq;
}
@@ -498,6 +488,12 @@ __blk_mq_alloc_requests_batch(struct blk_mq_alloc_data *data)
return rq_list_pop(data->cached_rqs);
}
+static bool blk_op_bypass_sched(blk_opf_t opf)
+{
+ return (opf & REQ_OP_MASK) == REQ_OP_FLUSH ||
+ blk_op_is_passthrough(opf);
+}
+
static void blk_mq_limit_depth(struct blk_mq_alloc_data *data)
{
struct elevator_mq_ops *ops;
@@ -518,12 +514,10 @@ static void blk_mq_limit_depth(struct blk_mq_alloc_data *data)
* Flush/passthrough requests are special and go directly to the
* dispatch list, they are not subject to the async_depth limit.
*/
- if ((data->cmd_flags & REQ_OP_MASK) == REQ_OP_FLUSH ||
- blk_op_is_passthrough(data->cmd_flags))
+ if (blk_op_bypass_sched(data->cmd_flags))
return;
WARN_ON_ONCE(data->flags & BLK_MQ_REQ_RESERVED);
- data->rq_flags |= RQF_USE_SCHED;
/*
* By default, sync requests have no limit, and async requests are
@@ -534,6 +528,29 @@ static void blk_mq_limit_depth(struct blk_mq_alloc_data *data)
ops->limit_depth(data->cmd_flags, data);
}
+/*
+ * Finish initializing a request once it has been claimed for an operation.
+ * Cached requests are allocated before that operation is known.
+ */
+static void blk_mq_rq_late_init(struct request *rq, u64 alloc_time_ns)
+{
+ struct elevator_queue *e;
+
+ blk_mq_rq_time_init(rq, alloc_time_ns);
+
+ if (!(rq->rq_flags & RQF_SCHED_TAGS) || (rq->rq_flags & RQF_RESV) ||
+ blk_op_bypass_sched(rq->cmd_flags))
+ return;
+
+ rq->rq_flags |= RQF_USE_SCHED;
+ INIT_HLIST_NODE(&rq->hash);
+ RB_CLEAR_NODE(&rq->rb_node);
+
+ e = rq->q->elevator;
+ if (e->type->ops.prepare_request)
+ e->type->ops.prepare_request(rq);
+}
+
static struct request *__blk_mq_alloc_requests(struct blk_mq_alloc_data *data)
{
struct request_queue *q = data->q;
@@ -562,7 +579,7 @@ static struct request *__blk_mq_alloc_requests(struct blk_mq_alloc_data *data)
if (data->nr_tags > 1) {
rq = __blk_mq_alloc_requests_batch(data);
if (rq) {
- blk_mq_rq_time_init(rq, alloc_time_ns);
+ blk_mq_rq_late_init(rq, alloc_time_ns);
return rq;
}
data->nr_tags = 1;
@@ -590,7 +607,7 @@ static struct request *__blk_mq_alloc_requests(struct blk_mq_alloc_data *data)
if (!(data->rq_flags & RQF_SCHED_TAGS))
blk_mq_inc_active_requests(data->hctx);
rq = blk_mq_rq_ctx_init(data, blk_mq_tags_from_data(data), tag);
- blk_mq_rq_time_init(rq, alloc_time_ns);
+ blk_mq_rq_late_init(rq, alloc_time_ns);
return rq;
}
@@ -637,8 +654,6 @@ static struct request *blk_mq_alloc_cached_request(struct request_queue *q,
if (plug->nr_ios == 1)
return NULL;
rq = blk_mq_rq_cache_fill(q, plug, opf, flags);
- if (!rq)
- return NULL;
} else {
rq = rq_list_peek(&plug->cached_rqs);
if (!rq || rq->q != q)
@@ -650,11 +665,11 @@ static struct request *blk_mq_alloc_cached_request(struct request_queue *q,
return NULL;
rq_list_pop(&plug->cached_rqs);
- blk_mq_rq_time_init(rq, blk_time_get_ns());
+ rq->cmd_flags = opf;
+ INIT_LIST_HEAD(&rq->queuelist);
+ blk_mq_rq_late_init(rq, blk_time_get_ns());
}
- rq->cmd_flags = opf;
- INIT_LIST_HEAD(&rq->queuelist);
return rq;
}
@@ -766,7 +781,7 @@ struct request *blk_mq_alloc_request_hctx(struct request_queue *q,
if (!(data.rq_flags & RQF_SCHED_TAGS))
blk_mq_inc_active_requests(data.hctx);
rq = blk_mq_rq_ctx_init(&data, blk_mq_tags_from_data(&data), tag);
- blk_mq_rq_time_init(rq, alloc_time_ns);
+ blk_mq_rq_late_init(rq, alloc_time_ns);
rq->__data_len = 0;
rq->phys_gap_bit = 0;
rq->__sector = (sector_t) -1;
@@ -3163,9 +3178,9 @@ void blk_mq_submit_bio(struct bio *bio)
new_request:
if (rq) {
rq_qos_throttle(rq->q, bio);
- blk_mq_rq_time_init(rq, blk_time_get_ns());
rq->cmd_flags = bio->bi_opf;
INIT_LIST_HEAD(&rq->queuelist);
+ blk_mq_rq_late_init(rq, blk_time_get_ns());
} else {
rq = blk_mq_get_new_requests(q, plug, bio);
if (unlikely(!rq)) {
diff --git a/block/mq-deadline.c b/block/mq-deadline.c
index 5f643c0ce2a86..e5db1ee097c35 100644
--- a/block/mq-deadline.c
+++ b/block/mq-deadline.c
@@ -685,7 +685,7 @@ static void dd_insert_requests(struct blk_mq_hw_ctx *hctx,
blk_mq_free_requests(&free);
}
-/* Callback from inside blk_mq_rq_ctx_init(). */
+/* Callback from inside blk_mq_rq_late_init(). */
static void dd_prepare_request(struct request *rq)
{
rq->elv.priv[0] = NULL;
--
2.52.0
^ permalink raw reply related [flat|nested] 5+ messages in thread* [PATCHv2 2/2] blk-mq: allow cached requests to be used for flush operations
2026-09-22 17:27 [PATCHv2 1/2] blk-mq: set RQF_USE_SCHED when the operation is known Keith Busch
@ 2026-09-22 17:27 ` Keith Busch
2026-09-23 4:41 ` Christoph Hellwig
2026-09-23 4:41 ` [PATCHv2 1/2] blk-mq: set RQF_USE_SCHED when the operation is known Christoph Hellwig
2026-09-23 11:26 ` Jens Axboe
2 siblings, 1 reply; 5+ messages in thread
From: Keith Busch @ 2026-09-22 17:27 UTC (permalink / raw)
To: axboe, linux-block; +Cc: hare, hch, Keith Busch
From: Keith Busch <kbusch@kernel.org>
The cached request lookups reject a request whose op_is_flush() state does
not match the incoming operation. That check is from when rq->elv and
rq->flush shared a union, so a request could not be both a scheduler
request and carry flush state. Commit be4c427809b0 ("blk-mq: use the I/O
scheduler for writes from the flush state machine") made them separate
members of struct request.
A cached request now gets its scheduler state when it is claimed by an
operation, so one claimed for a flush is set up exactly as a freshly
allocated request. Drop the check so a cached batch can mix flush and
non-flush operations.
Signed-off-by: Keith Busch <kbusch@kernel.org>
---
block/blk-mq.c | 4 ----
1 file changed, 4 deletions(-)
diff --git a/block/blk-mq.c b/block/blk-mq.c
index 5df6d2244db82..025a799f35b92 100644
--- a/block/blk-mq.c
+++ b/block/blk-mq.c
@@ -661,8 +661,6 @@ static struct request *blk_mq_alloc_cached_request(struct request_queue *q,
if (blk_mq_get_hctx_type(opf) != rq->mq_hctx->type)
return NULL;
- if (op_is_flush(rq->cmd_flags) != op_is_flush(opf))
- return NULL;
rq_list_pop(&plug->cached_rqs);
rq->cmd_flags = opf;
@@ -3075,8 +3073,6 @@ static struct request *blk_mq_get_cached_request(struct blk_plug *plug,
if (type != rq->mq_hctx->type &&
(type != HCTX_TYPE_READ || rq->mq_hctx->type != HCTX_TYPE_DEFAULT))
return NULL;
- if (op_is_flush(rq->cmd_flags) != op_is_flush(opf))
- return NULL;
rq_list_pop(&plug->cached_rqs);
return rq;
}
--
2.52.0
^ permalink raw reply related [flat|nested] 5+ messages in thread* Re: [PATCHv2 1/2] blk-mq: set RQF_USE_SCHED when the operation is known
2026-09-22 17:27 [PATCHv2 1/2] blk-mq: set RQF_USE_SCHED when the operation is known Keith Busch
2026-09-22 17:27 ` [PATCHv2 2/2] blk-mq: allow cached requests to be used for flush operations Keith Busch
@ 2026-09-23 4:41 ` Christoph Hellwig
2026-09-23 11:26 ` Jens Axboe
2 siblings, 0 replies; 5+ messages in thread
From: Christoph Hellwig @ 2026-09-23 4:41 UTC (permalink / raw)
To: Keith Busch; +Cc: axboe, linux-block, hare, Keith Busch, Henry Hu
Looks good:
Reviewed-by: Christoph Hellwig <hch@lst.de>
^ permalink raw reply [flat|nested] 5+ messages in thread
* Re: [PATCHv2 1/2] blk-mq: set RQF_USE_SCHED when the operation is known
2026-09-22 17:27 [PATCHv2 1/2] blk-mq: set RQF_USE_SCHED when the operation is known Keith Busch
2026-09-22 17:27 ` [PATCHv2 2/2] blk-mq: allow cached requests to be used for flush operations Keith Busch
2026-09-23 4:41 ` [PATCHv2 1/2] blk-mq: set RQF_USE_SCHED when the operation is known Christoph Hellwig
@ 2026-09-23 11:26 ` Jens Axboe
2 siblings, 0 replies; 5+ messages in thread
From: Jens Axboe @ 2026-09-23 11:26 UTC (permalink / raw)
To: linux-block, Keith Busch; +Cc: hare, hch, Keith Busch, Henry Hu
On Tue, 22 Sep 2026 10:27:10 -0700, Keith Busch wrote:
> The cached requests are allocated for one operation but can be handed
> out for another. A passthrough command has RQF_USE_SCHED cleared, so
> using those flags for a subsequent read/write bio will insert it into
> the scheduler without ->prepare_request() and frees it without
> ->finish_request(). For kyber, this leaks the domain token acquired at
> dispatch and stalls the queue.
>
> [...]
Applied, thanks!
[1/2] blk-mq: set RQF_USE_SCHED when the operation is known
commit: ab6c756f28c741d704a7c3e04814bfc3adf6f818
[2/2] blk-mq: allow cached requests to be used for flush operations
commit: 9d2c70986bb7838c1c441a93bda415eecb52f3dd
Best regards,
--
Jens Axboe
^ permalink raw reply [flat|nested] 5+ messages in thread
end of thread, other threads:[~2026-09-23 11:26 UTC | newest]
Thread overview: 5+ messages (download: mbox.gz follow: Atom feed
-- links below jump to the message on this page --
2026-09-22 17:27 [PATCHv2 1/2] blk-mq: set RQF_USE_SCHED when the operation is known Keith Busch
2026-09-22 17:27 ` [PATCHv2 2/2] blk-mq: allow cached requests to be used for flush operations Keith Busch
2026-09-23 4:41 ` Christoph Hellwig
2026-09-23 4:41 ` [PATCHv2 1/2] blk-mq: set RQF_USE_SCHED when the operation is known Christoph Hellwig
2026-09-23 11:26 ` Jens Axboe
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox