Linux block layer
 help / color / mirror / Atom feed
* [PATCH] blk-mq: check passthrough state before reusing cached requests
@ 2026-09-16 15:16 huhai
  2026-09-16 19:19 ` John Garry
  2026-09-17 14:21 ` Keith Busch
  0 siblings, 2 replies; 5+ messages in thread
From: huhai @ 2026-09-16 15:16 UTC (permalink / raw)
  To: axboe; +Cc: linux-block, Henry Hu

From: Henry Hu <huhai@kylinos.cn>

blk_mq_alloc_request() and blk_mq_submit_bio() share plug->cached_rqs,
but neither lookup checks the passthrough state. This allows an io_uring
batch to reuse a cached NVMe passthrough request for an ordinary bio.

When an I/O scheduler is enabled, passthrough requests have RQF_SCHED_TAGS
set but RQF_USE_SCHED clear, and skip ->prepare_request(). Reuse updates
rq->cmd_flags without reinitializing the scheduler state, so the request
is inserted into the scheduler on plug flush, while completion skips
->finish_request(). With kyber, this leaks the domain token acquired at
dispatch and can stall further I/O in that domain.

This was reproduced with kyber and request merging disabled on a QEMU
NVMe device by submitting a passthrough read followed by O_DIRECT writes
in one io_uring batch. Tokens leak until the kyber write-domain token pool
is exhausted, blocking ext4 journal I/O and fsync():

  INFO: task jbd2/nvme0n1-8:97 blocked in I/O wait for more than 241 seconds.
    Call Trace:
      schedule+0xe9/0x300
      io_schedule+0xca/0x150
      bit_wait_io+0x1b/0x140
      __wait_on_bit+0x63/0x170
      jbd2_write_superblock+0x3d7/0x600
      jbd2_journal_update_sb_log_tail+0x1e3/0x2d0
      jbd2_journal_commit_transaction+0x11d1/0x5b50

  INFO: task fsynctest:107 blocked in I/O wait for more than 241 seconds.
    Call Trace:
      schedule+0xe9/0x300
      io_schedule+0xca/0x150
      folio_wait_bit_common+0x2ed/0x780
      folio_wait_bit+0x18/0x30
      folio_wait_writeback+0x4b/0x1e0
      __filemap_fdatawait_range+0x120/0x1e0
      file_write_and_wait_range+0xeb/0x130
      ext4_sync_file+0x368/0xa80
      do_fsync+0xa2/0x200

Require the passthrough state to match before reusing a cached request.
With the fix, the workload completes normally.

Assisted-by: LLM
Signed-off-by: Henry Hu <huhai@kylinos.cn>
---
 block/blk-mq.c | 4 ++++
 1 file changed, 4 insertions(+)

diff --git a/block/blk-mq.c b/block/blk-mq.c
index a26a11c73ee3..3a9d57f6a569 100644
--- a/block/blk-mq.c
+++ b/block/blk-mq.c
@@ -648,6 +648,8 @@ static struct request *blk_mq_alloc_cached_request(struct request_queue *q,
 			return NULL;
 		if (op_is_flush(rq->cmd_flags) != op_is_flush(opf))
 			return NULL;
+		if (blk_rq_is_passthrough(rq) != blk_op_is_passthrough(opf))
+			return NULL;
 
 		rq_list_pop(&plug->cached_rqs);
 		blk_mq_rq_time_init(rq, blk_time_get_ns());
@@ -3062,6 +3064,8 @@ static struct request *blk_mq_get_cached_request(struct blk_plug *plug,
 		return NULL;
 	if (op_is_flush(rq->cmd_flags) != op_is_flush(opf))
 		return NULL;
+	if (blk_rq_is_passthrough(rq) != blk_op_is_passthrough(opf))
+		return NULL;
 	rq_list_pop(&plug->cached_rqs);
 	return rq;
 }
-- 
2.25.1


^ permalink raw reply related	[flat|nested] 5+ messages in thread

* Re: [PATCH] blk-mq: check passthrough state before reusing cached requests
  2026-09-16 15:16 [PATCH] blk-mq: check passthrough state before reusing cached requests huhai
@ 2026-09-16 19:19 ` John Garry
  2026-09-17  5:52   ` huhai
  2026-09-17 14:21 ` Keith Busch
  1 sibling, 1 reply; 5+ messages in thread
From: John Garry @ 2026-09-16 19:19 UTC (permalink / raw)
  To: huhai, axboe; +Cc: linux-block, Henry Hu

On 9/16/26 16:16, huhai wrote:
> From: Henry Hu <huhai@kylinos.cn>
> 
> blk_mq_alloc_request() and blk_mq_submit_bio() share plug->cached_rqs,
> but neither lookup checks the passthrough state. This allows an io_uring
> batch to reuse a cached NVMe passthrough request for an ordinary bio.
> 
> When an I/O scheduler is enabled, passthrough requests have RQF_SCHED_TAGS
> set but RQF_USE_SCHED clear, and skip ->prepare_request(). Reuse updates
> rq->cmd_flags without reinitializing the scheduler state, so the request
> is inserted into the scheduler on plug flush, while completion skips
> ->finish_request(). With kyber, this leaks the domain token acquired at
> dispatch and can stall further I/O in that domain.
> 
> This was reproduced with kyber and request merging disabled on a QEMU
> NVMe device by submitting a passthrough read followed by O_DIRECT writes
> in one io_uring batch. Tokens leak until the kyber write-domain token pool
> is exhausted, blocking ext4 journal I/O and fsync():
> 
>    INFO: task jbd2/nvme0n1-8:97 blocked in I/O wait for more than 241 seconds.
>      Call Trace:
>        schedule+0xe9/0x300
>        io_schedule+0xca/0x150
>        bit_wait_io+0x1b/0x140
>        __wait_on_bit+0x63/0x170
>        jbd2_write_superblock+0x3d7/0x600
>        jbd2_journal_update_sb_log_tail+0x1e3/0x2d0
>        jbd2_journal_commit_transaction+0x11d1/0x5b50
> 
>    INFO: task fsynctest:107 blocked in I/O wait for more than 241 seconds.
>      Call Trace:
>        schedule+0xe9/0x300
>        io_schedule+0xca/0x150
>        folio_wait_bit_common+0x2ed/0x780
>        folio_wait_bit+0x18/0x30
>        folio_wait_writeback+0x4b/0x1e0
>        __filemap_fdatawait_range+0x120/0x1e0
>        file_write_and_wait_range+0xeb/0x130
>        ext4_sync_file+0x368/0xa80
>        do_fsync+0xa2/0x200
> 
> Require the passthrough state to match before reusing a cached request.
> With the fix, the workload completes normally.
> 
> Assisted-by: LLM
> Signed-off-by: Henry Hu <huhai@kylinos.cn>
> ---
>   block/blk-mq.c | 4 ++++
>   1 file changed, 4 insertions(+)
> 
> diff --git a/block/blk-mq.c b/block/blk-mq.c
> index a26a11c73ee3..3a9d57f6a569 100644
> --- a/block/blk-mq.c
> +++ b/block/blk-mq.c
> @@ -648,6 +648,8 @@ static struct request *blk_mq_alloc_cached_request(struct request_queue *q,
>   			return NULL;
>   		if (op_is_flush(rq->cmd_flags) != op_is_flush(opf))
>   			return NULL;
> +		if (blk_rq_is_passthrough(rq) != blk_op_is_passthrough(opf))
> +			return NULL;
>   
>   		rq_list_pop(&plug->cached_rqs);
>   		blk_mq_rq_time_init(rq, blk_time_get_ns());
> @@ -3062,6 +3064,8 @@ static struct request *blk_mq_get_cached_request(struct blk_plug *plug,
>   		return NULL;
>   	if (op_is_flush(rq->cmd_flags) != op_is_flush(opf))
>   		return NULL;
> +	if (blk_rq_is_passthrough(rq) != blk_op_is_passthrough(opf))
> +		return NULL;

It seems like some code which code be factored out (between 
blk_mq_alloc_cached_request() and blk_mq_get_cached_request()).

And maybe 7746564793978fe2f43b18a302b22dca0ad3a0e8 could to be repeated 
for blk_mq_alloc_cached_request(), which could mean even more factoring out.

>   	rq_list_pop(&plug->cached_rqs);
>   	return rq;
>   }


^ permalink raw reply	[flat|nested] 5+ messages in thread

* Re:Re: [PATCH] blk-mq: check passthrough state before reusing cached requests
  2026-09-16 19:19 ` John Garry
@ 2026-09-17  5:52   ` huhai
  0 siblings, 0 replies; 5+ messages in thread
From: huhai @ 2026-09-17  5:52 UTC (permalink / raw)
  To: John Garry; +Cc: axboe, linux-block, Henry Hu

At 2026-09-17 03:19:58, "John Garry" <john.garry@linux.dev> wrote:
>On 9/16/26 16:16, huhai wrote:
>> From: Henry Hu <huhai@kylinos.cn>
>> 
>> blk_mq_alloc_request() and blk_mq_submit_bio() share plug->cached_rqs,
>> but neither lookup checks the passthrough state. This allows an io_uring
>> batch to reuse a cached NVMe passthrough request for an ordinary bio.
>> 
>> When an I/O scheduler is enabled, passthrough requests have RQF_SCHED_TAGS
>> set but RQF_USE_SCHED clear, and skip ->prepare_request(). Reuse updates
>> rq->cmd_flags without reinitializing the scheduler state, so the request
>> is inserted into the scheduler on plug flush, while completion skips
>> ->finish_request(). With kyber, this leaks the domain token acquired at
>> dispatch and can stall further I/O in that domain.
>> 
>> This was reproduced with kyber and request merging disabled on a QEMU
>> NVMe device by submitting a passthrough read followed by O_DIRECT writes
>> in one io_uring batch. Tokens leak until the kyber write-domain token pool
>> is exhausted, blocking ext4 journal I/O and fsync():
>> 
>>    INFO: task jbd2/nvme0n1-8:97 blocked in I/O wait for more than 241 seconds.
>>      Call Trace:
>>        schedule+0xe9/0x300
>>        io_schedule+0xca/0x150
>>        bit_wait_io+0x1b/0x140
>>        __wait_on_bit+0x63/0x170
>>        jbd2_write_superblock+0x3d7/0x600
>>        jbd2_journal_update_sb_log_tail+0x1e3/0x2d0
>>        jbd2_journal_commit_transaction+0x11d1/0x5b50
>> 
>>    INFO: task fsynctest:107 blocked in I/O wait for more than 241 seconds.
>>      Call Trace:
>>        schedule+0xe9/0x300
>>        io_schedule+0xca/0x150
>>        folio_wait_bit_common+0x2ed/0x780
>>        folio_wait_bit+0x18/0x30
>>        folio_wait_writeback+0x4b/0x1e0
>>        __filemap_fdatawait_range+0x120/0x1e0
>>        file_write_and_wait_range+0xeb/0x130
>>        ext4_sync_file+0x368/0xa80
>>        do_fsync+0xa2/0x200
>> 
>> Require the passthrough state to match before reusing a cached request.
>> With the fix, the workload completes normally.
>> 
>> Assisted-by: LLM
>> Signed-off-by: Henry Hu <huhai@kylinos.cn>
>> ---
>>   block/blk-mq.c | 4 ++++
>>   1 file changed, 4 insertions(+)
>> 
>> diff --git a/block/blk-mq.c b/block/blk-mq.c
>> index a26a11c73ee3..3a9d57f6a569 100644
>> --- a/block/blk-mq.c
>> +++ b/block/blk-mq.c
>> @@ -648,6 +648,8 @@ static struct request *blk_mq_alloc_cached_request(struct request_queue *q,
>>   			return NULL;
>>   		if (op_is_flush(rq->cmd_flags) != op_is_flush(opf))
>>   			return NULL;
>> +		if (blk_rq_is_passthrough(rq) != blk_op_is_passthrough(opf))
>> +			return NULL;
>>   
>>   		rq_list_pop(&plug->cached_rqs);
>>   		blk_mq_rq_time_init(rq, blk_time_get_ns());
>> @@ -3062,6 +3064,8 @@ static struct request *blk_mq_get_cached_request(struct blk_plug *plug,
>>   		return NULL;
>>   	if (op_is_flush(rq->cmd_flags) != op_is_flush(opf))
>>   		return NULL;
>> +	if (blk_rq_is_passthrough(rq) != blk_op_is_passthrough(opf))
>> +		return NULL;
>
>It seems like some code which code be factored out (between 
>blk_mq_alloc_cached_request() and blk_mq_get_cached_request()).
>
>And maybe 7746564793978fe2f43b18a302b22dca0ad3a0e8 could to be repeated 
>for blk_mq_alloc_cached_request(), which could mean even more factoring out.

Thanks for the review and suggestions.

I've confirmed dd6216bb16e8 is the patch that introduced the issue. I'll therefore
add the missing tag in v2:

Fixes: dd6216bb16e8 ("blk-mq: make sure elevator callbacks aren't called for passthrough request")

Before dd6216bb16e8 the reproducer is clean. At dd6216bb16e8 it triggers a
KASAN slab-out-of-bounds write in nvme_queue_rqs() during the io_uring plug
flush:

  BUG: KASAN: slab-out-of-bounds in nvme_queue_rqs+0x8f3/0xa50
  Write of size 8 by task repro/94
  Call Trace:
    nvme_queue_rqs+0x8f3/0xa50
    __blk_mq_flush_plug_list+0x93/0xc0
    blk_mq_flush_plug_list+0x152d/0x1d20
    blk_finish_plug+0x5c/0xb0
    io_submit_sqes+0x1142/0x1db0
    __do_sys_io_uring_enter+0x7aa/0x1dc0

With this fix applied, the reproducer completes without the KASAN report.

I will continue to evaluate the suggested refactoring and applying 774656479397
to blk_mq_alloc_cached_request(), and confirm in v2..

Thanks,
Henry Hu.

>
>>   	rq_list_pop(&plug->cached_rqs);
>>   	return rq;
>>   }

^ permalink raw reply	[flat|nested] 5+ messages in thread

* Re: [PATCH] blk-mq: check passthrough state before reusing cached requests
  2026-09-16 15:16 [PATCH] blk-mq: check passthrough state before reusing cached requests huhai
  2026-09-16 19:19 ` John Garry
@ 2026-09-17 14:21 ` Keith Busch
  2026-09-18  5:16   ` huhai
  1 sibling, 1 reply; 5+ messages in thread
From: Keith Busch @ 2026-09-17 14:21 UTC (permalink / raw)
  To: huhai; +Cc: axboe, linux-block, Henry Hu

On Wed, Sep 16, 2026 at 11:16:55PM +0800, huhai wrote:
> @@ -648,6 +648,8 @@ static struct request *blk_mq_alloc_cached_request(struct request_queue *q,
>  			return NULL;
>  		if (op_is_flush(rq->cmd_flags) != op_is_flush(opf))
>  			return NULL;
> +		if (blk_rq_is_passthrough(rq) != blk_op_is_passthrough(opf))
> +			return NULL;

I see how this addresses your observation, but I don't think this is the
right way to fix it. The batch of requests allocated to prime the cache
shouldn't be locked into a specific type of command that can use it,
IMO. This is a bit more involved of a change, but it should allow more
flexible use of the cached requests. Here's a quick idea to attempt
that:

---
diff --git a/block/blk-mq.c b/block/blk-mq.c
index a26a11c73ee3e..c1fd20807da30 100644
--- a/block/blk-mq.c
+++ b/block/blk-mq.c
@@ -447,17 +447,28 @@ static struct request *blk_mq_rq_ctx_init(struct blk_mq_alloc_data *data,
 	WRITE_ONCE(rq->deadline, 0);
 	req_ref_set(rq, 1);
 
-	if (rq->rq_flags & RQF_USE_SCHED) {
-		struct elevator_queue *e = data->q->elevator;
+	return rq;
+}
+
+static bool blk_op_bypass_sched(blk_opf_t opf)
+{
+	return (opf & REQ_OP_MASK) == REQ_OP_FLUSH ||
+	       blk_op_is_passthrough(opf);
+}
 
-		INIT_HLIST_NODE(&rq->hash);
-		RB_CLEAR_NODE(&rq->rb_node);
+static void blk_mq_set_rq_sched(struct request *rq, blk_opf_t opf)
+{
+	struct elevator_queue *e = rq->q->elevator;
 
-		if (e->type->ops.prepare_request)
-			e->type->ops.prepare_request(rq);
-	}
+	if (!e || blk_op_bypass_sched(opf))
+		return;
 
-	return rq;
+	rq->rq_flags |= RQF_USE_SCHED;
+	INIT_HLIST_NODE(&rq->hash);
+	RB_CLEAR_NODE(&rq->rb_node);
+
+	if (e->type->ops.prepare_request)
+		e->type->ops.prepare_request(rq);
 }
 
 static inline struct request *
@@ -518,12 +529,10 @@ static void blk_mq_limit_depth(struct blk_mq_alloc_data *data)
 	 * Flush/passthrough requests are special and go directly to the
 	 * dispatch list, they are not subject to the async_depth limit.
 	 */
-	if ((data->cmd_flags & REQ_OP_MASK) == REQ_OP_FLUSH ||
-	    blk_op_is_passthrough(data->cmd_flags))
+	if (blk_op_bypass_sched(data->cmd_flags))
 		return;
 
 	WARN_ON_ONCE(data->flags & BLK_MQ_REQ_RESERVED);
-	data->rq_flags |= RQF_USE_SCHED;
 
 	/*
 	 * By default, sync requests have no limit, and async requests are
@@ -563,6 +572,7 @@ static struct request *__blk_mq_alloc_requests(struct blk_mq_alloc_data *data)
 		rq = __blk_mq_alloc_requests_batch(data);
 		if (rq) {
 			blk_mq_rq_time_init(rq, alloc_time_ns);
+			blk_mq_set_rq_sched(rq, data->cmd_flags);
 			return rq;
 		}
 		data->nr_tags = 1;
@@ -591,6 +601,7 @@ static struct request *__blk_mq_alloc_requests(struct blk_mq_alloc_data *data)
 		blk_mq_inc_active_requests(data->hctx);
 	rq = blk_mq_rq_ctx_init(data, blk_mq_tags_from_data(data), tag);
 	blk_mq_rq_time_init(rq, alloc_time_ns);
+	blk_mq_set_rq_sched(rq, data->cmd_flags);
 	return rq;
 }
 
@@ -637,8 +648,6 @@ static struct request *blk_mq_alloc_cached_request(struct request_queue *q,
 		if (plug->nr_ios == 1)
 			return NULL;
 		rq = blk_mq_rq_cache_fill(q, plug, opf, flags);
-		if (!rq)
-			return NULL;
 	} else {
 		rq = rq_list_peek(&plug->cached_rqs);
 		if (!rq || rq->q != q)
@@ -646,15 +655,14 @@ static struct request *blk_mq_alloc_cached_request(struct request_queue *q,
 
 		if (blk_mq_get_hctx_type(opf) != rq->mq_hctx->type)
 			return NULL;
-		if (op_is_flush(rq->cmd_flags) != op_is_flush(opf))
-			return NULL;
 
 		rq_list_pop(&plug->cached_rqs);
 		blk_mq_rq_time_init(rq, blk_time_get_ns());
+		rq->cmd_flags = opf;
+		INIT_LIST_HEAD(&rq->queuelist);
+		blk_mq_set_rq_sched(rq, opf);
 	}
 
-	rq->cmd_flags = opf;
-	INIT_LIST_HEAD(&rq->queuelist);
 	return rq;
 }
 
@@ -3060,8 +3068,6 @@ static struct request *blk_mq_get_cached_request(struct blk_plug *plug,
 	if (type != rq->mq_hctx->type &&
 	    (type != HCTX_TYPE_READ || rq->mq_hctx->type != HCTX_TYPE_DEFAULT))
 		return NULL;
-	if (op_is_flush(rq->cmd_flags) != op_is_flush(opf))
-		return NULL;
 	rq_list_pop(&plug->cached_rqs);
 	return rq;
 }
@@ -3166,6 +3172,7 @@ void blk_mq_submit_bio(struct bio *bio)
 		blk_mq_rq_time_init(rq, blk_time_get_ns());
 		rq->cmd_flags = bio->bi_opf;
 		INIT_LIST_HEAD(&rq->queuelist);
+		blk_mq_set_rq_sched(rq, bio->bi_opf);
 	} else {
 		rq = blk_mq_get_new_requests(q, plug, bio);
 		if (unlikely(!rq)) {
--

^ permalink raw reply related	[flat|nested] 5+ messages in thread

* Re:Re: [PATCH] blk-mq: check passthrough state before reusing cached requests
  2026-09-17 14:21 ` Keith Busch
@ 2026-09-18  5:16   ` huhai
  0 siblings, 0 replies; 5+ messages in thread
From: huhai @ 2026-09-18  5:16 UTC (permalink / raw)
  To: Keith Busch; +Cc: axboe, linux-block, Henry Hu

At 2026-09-17 22:21:58, "Keith Busch" <kbusch@kernel.org> wrote:
>On Wed, Sep 16, 2026 at 11:16:55PM +0800, huhai wrote:
>> @@ -648,6 +648,8 @@ static struct request *blk_mq_alloc_cached_request(struct request_queue *q,
>>  			return NULL;
>>  		if (op_is_flush(rq->cmd_flags) != op_is_flush(opf))
>>  			return NULL;
>> +		if (blk_rq_is_passthrough(rq) != blk_op_is_passthrough(opf))
>> +			return NULL;
>
>I see how this addresses your observation, but I don't think this is the
>right way to fix it. The batch of requests allocated to prime the cache
>shouldn't be locked into a specific type of command that can use it,
>IMO. This is a bit more involved of a change, but it should allow more
>flexible use of the cached requests. Here's a quick idea to attempt
>that:
>

Thanks, I think this is the better direction. cached requests shouldn't be
locked into a specific type of command. Initializing the scheduler state
when a request is actually handed out is cleaner than rejecting a mismatch
during cache lookup.

I tested your diff with my reproducer: the kyber token count returns to zero
after IO completes, and the workload no longer stalls and the hung task is
no longer reported.

I'll drop my original fix. If a stable backport is needed, my original minimal
fix is small and self-contained; happy to respin it for stable if useful. 

Feel free to turn this into a formal patch; I can test it further once it's posted. 

Thanks,
Henry Hu

>---
>diff --git a/block/blk-mq.c b/block/blk-mq.c
>index a26a11c73ee3e..c1fd20807da30 100644
>--- a/block/blk-mq.c
>+++ b/block/blk-mq.c
>@@ -447,17 +447,28 @@ static struct request *blk_mq_rq_ctx_init(struct blk_mq_alloc_data *data,
> 	WRITE_ONCE(rq->deadline, 0);
> 	req_ref_set(rq, 1);
> 
>-	if (rq->rq_flags & RQF_USE_SCHED) {
>-		struct elevator_queue *e = data->q->elevator;
>+	return rq;
>+}
>+
>+static bool blk_op_bypass_sched(blk_opf_t opf)
>+{
>+	return (opf & REQ_OP_MASK) == REQ_OP_FLUSH ||
>+	       blk_op_is_passthrough(opf);
>+}
> 
>-		INIT_HLIST_NODE(&rq->hash);
>-		RB_CLEAR_NODE(&rq->rb_node);
>+static void blk_mq_set_rq_sched(struct request *rq, blk_opf_t opf)
>+{
>+	struct elevator_queue *e = rq->q->elevator;
> 
>-		if (e->type->ops.prepare_request)
>-			e->type->ops.prepare_request(rq);
>-	}
>+	if (!e || blk_op_bypass_sched(opf))
>+		return;
> 
>-	return rq;
>+	rq->rq_flags |= RQF_USE_SCHED;
>+	INIT_HLIST_NODE(&rq->hash);
>+	RB_CLEAR_NODE(&rq->rb_node);
>+
>+	if (e->type->ops.prepare_request)
>+		e->type->ops.prepare_request(rq);
> }
> 
> static inline struct request *
>@@ -518,12 +529,10 @@ static void blk_mq_limit_depth(struct blk_mq_alloc_data *data)
> 	 * Flush/passthrough requests are special and go directly to the
> 	 * dispatch list, they are not subject to the async_depth limit.
> 	 */
>-	if ((data->cmd_flags & REQ_OP_MASK) == REQ_OP_FLUSH ||
>-	    blk_op_is_passthrough(data->cmd_flags))
>+	if (blk_op_bypass_sched(data->cmd_flags))
> 		return;
> 
> 	WARN_ON_ONCE(data->flags & BLK_MQ_REQ_RESERVED);
>-	data->rq_flags |= RQF_USE_SCHED;
> 
> 	/*
> 	 * By default, sync requests have no limit, and async requests are
>@@ -563,6 +572,7 @@ static struct request *__blk_mq_alloc_requests(struct blk_mq_alloc_data *data)
> 		rq = __blk_mq_alloc_requests_batch(data);
> 		if (rq) {
> 			blk_mq_rq_time_init(rq, alloc_time_ns);
>+			blk_mq_set_rq_sched(rq, data->cmd_flags);
> 			return rq;
> 		}
> 		data->nr_tags = 1;
>@@ -591,6 +601,7 @@ static struct request *__blk_mq_alloc_requests(struct blk_mq_alloc_data *data)
> 		blk_mq_inc_active_requests(data->hctx);
> 	rq = blk_mq_rq_ctx_init(data, blk_mq_tags_from_data(data), tag);
> 	blk_mq_rq_time_init(rq, alloc_time_ns);
>+	blk_mq_set_rq_sched(rq, data->cmd_flags);
> 	return rq;
> }
> 
>@@ -637,8 +648,6 @@ static struct request *blk_mq_alloc_cached_request(struct request_queue *q,
> 		if (plug->nr_ios == 1)
> 			return NULL;
> 		rq = blk_mq_rq_cache_fill(q, plug, opf, flags);
>-		if (!rq)
>-			return NULL;
> 	} else {
> 		rq = rq_list_peek(&plug->cached_rqs);
> 		if (!rq || rq->q != q)
>@@ -646,15 +655,14 @@ static struct request *blk_mq_alloc_cached_request(struct request_queue *q,
> 
> 		if (blk_mq_get_hctx_type(opf) != rq->mq_hctx->type)
> 			return NULL;
>-		if (op_is_flush(rq->cmd_flags) != op_is_flush(opf))
>-			return NULL;
> 
> 		rq_list_pop(&plug->cached_rqs);
> 		blk_mq_rq_time_init(rq, blk_time_get_ns());
>+		rq->cmd_flags = opf;
>+		INIT_LIST_HEAD(&rq->queuelist);
>+		blk_mq_set_rq_sched(rq, opf);
> 	}
> 
>-	rq->cmd_flags = opf;
>-	INIT_LIST_HEAD(&rq->queuelist);
> 	return rq;
> }
> 
>@@ -3060,8 +3068,6 @@ static struct request *blk_mq_get_cached_request(struct blk_plug *plug,
> 	if (type != rq->mq_hctx->type &&
> 	    (type != HCTX_TYPE_READ || rq->mq_hctx->type != HCTX_TYPE_DEFAULT))
> 		return NULL;
>-	if (op_is_flush(rq->cmd_flags) != op_is_flush(opf))
>-		return NULL;
> 	rq_list_pop(&plug->cached_rqs);
> 	return rq;
> }
>@@ -3166,6 +3172,7 @@ void blk_mq_submit_bio(struct bio *bio)
> 		blk_mq_rq_time_init(rq, blk_time_get_ns());
> 		rq->cmd_flags = bio->bi_opf;
> 		INIT_LIST_HEAD(&rq->queuelist);
>+		blk_mq_set_rq_sched(rq, bio->bi_opf);
> 	} else {
> 		rq = blk_mq_get_new_requests(q, plug, bio);
> 		if (unlikely(!rq)) {
>--

^ permalink raw reply	[flat|nested] 5+ messages in thread

end of thread, other threads:[~2026-09-18  5:17 UTC | newest]

Thread overview: 5+ messages (download: mbox.gz follow: Atom feed
-- links below jump to the message on this page --
2026-09-16 15:16 [PATCH] blk-mq: check passthrough state before reusing cached requests huhai
2026-09-16 19:19 ` John Garry
2026-09-17  5:52   ` huhai
2026-09-17 14:21 ` Keith Busch
2026-09-18  5:16   ` huhai

This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox