Linux block layer
 help / color / mirror / Atom feed
From: huhai  <15815827059@163.com>
To: "Keith Busch" <kbusch@kernel.org>
Cc: axboe@kernel.dk, linux-block@vger.kernel.org,
	"Henry Hu" <huhai@kylinos.cn>
Subject: Re:Re: [PATCH] blk-mq: check passthrough state before reusing cached requests
Date: Fri, 18 Sep 2026 13:16:58 +0800 (CST)	[thread overview]
Message-ID: <24313658.46fe.1a0b2f24bcc.Coremail.15815827059@163.com> (raw)
In-Reply-To: <aqv3hrg_fCZZCojW@kbusch-mbp>

At 2026-09-17 22:21:58, "Keith Busch" <kbusch@kernel.org> wrote:
>On Wed, Sep 16, 2026 at 11:16:55PM +0800, huhai wrote:
>> @@ -648,6 +648,8 @@ static struct request *blk_mq_alloc_cached_request(struct request_queue *q,
>>  			return NULL;
>>  		if (op_is_flush(rq->cmd_flags) != op_is_flush(opf))
>>  			return NULL;
>> +		if (blk_rq_is_passthrough(rq) != blk_op_is_passthrough(opf))
>> +			return NULL;
>
>I see how this addresses your observation, but I don't think this is the
>right way to fix it. The batch of requests allocated to prime the cache
>shouldn't be locked into a specific type of command that can use it,
>IMO. This is a bit more involved of a change, but it should allow more
>flexible use of the cached requests. Here's a quick idea to attempt
>that:
>

Thanks, I think this is the better direction. cached requests shouldn't be
locked into a specific type of command. Initializing the scheduler state
when a request is actually handed out is cleaner than rejecting a mismatch
during cache lookup.

I tested your diff with my reproducer: the kyber token count returns to zero
after IO completes, and the workload no longer stalls and the hung task is
no longer reported.

I'll drop my original fix. If a stable backport is needed, my original minimal
fix is small and self-contained; happy to respin it for stable if useful. 

Feel free to turn this into a formal patch; I can test it further once it's posted. 

Thanks,
Henry Hu

>---
>diff --git a/block/blk-mq.c b/block/blk-mq.c
>index a26a11c73ee3e..c1fd20807da30 100644
>--- a/block/blk-mq.c
>+++ b/block/blk-mq.c
>@@ -447,17 +447,28 @@ static struct request *blk_mq_rq_ctx_init(struct blk_mq_alloc_data *data,
> 	WRITE_ONCE(rq->deadline, 0);
> 	req_ref_set(rq, 1);
> 
>-	if (rq->rq_flags & RQF_USE_SCHED) {
>-		struct elevator_queue *e = data->q->elevator;
>+	return rq;
>+}
>+
>+static bool blk_op_bypass_sched(blk_opf_t opf)
>+{
>+	return (opf & REQ_OP_MASK) == REQ_OP_FLUSH ||
>+	       blk_op_is_passthrough(opf);
>+}
> 
>-		INIT_HLIST_NODE(&rq->hash);
>-		RB_CLEAR_NODE(&rq->rb_node);
>+static void blk_mq_set_rq_sched(struct request *rq, blk_opf_t opf)
>+{
>+	struct elevator_queue *e = rq->q->elevator;
> 
>-		if (e->type->ops.prepare_request)
>-			e->type->ops.prepare_request(rq);
>-	}
>+	if (!e || blk_op_bypass_sched(opf))
>+		return;
> 
>-	return rq;
>+	rq->rq_flags |= RQF_USE_SCHED;
>+	INIT_HLIST_NODE(&rq->hash);
>+	RB_CLEAR_NODE(&rq->rb_node);
>+
>+	if (e->type->ops.prepare_request)
>+		e->type->ops.prepare_request(rq);
> }
> 
> static inline struct request *
>@@ -518,12 +529,10 @@ static void blk_mq_limit_depth(struct blk_mq_alloc_data *data)
> 	 * Flush/passthrough requests are special and go directly to the
> 	 * dispatch list, they are not subject to the async_depth limit.
> 	 */
>-	if ((data->cmd_flags & REQ_OP_MASK) == REQ_OP_FLUSH ||
>-	    blk_op_is_passthrough(data->cmd_flags))
>+	if (blk_op_bypass_sched(data->cmd_flags))
> 		return;
> 
> 	WARN_ON_ONCE(data->flags & BLK_MQ_REQ_RESERVED);
>-	data->rq_flags |= RQF_USE_SCHED;
> 
> 	/*
> 	 * By default, sync requests have no limit, and async requests are
>@@ -563,6 +572,7 @@ static struct request *__blk_mq_alloc_requests(struct blk_mq_alloc_data *data)
> 		rq = __blk_mq_alloc_requests_batch(data);
> 		if (rq) {
> 			blk_mq_rq_time_init(rq, alloc_time_ns);
>+			blk_mq_set_rq_sched(rq, data->cmd_flags);
> 			return rq;
> 		}
> 		data->nr_tags = 1;
>@@ -591,6 +601,7 @@ static struct request *__blk_mq_alloc_requests(struct blk_mq_alloc_data *data)
> 		blk_mq_inc_active_requests(data->hctx);
> 	rq = blk_mq_rq_ctx_init(data, blk_mq_tags_from_data(data), tag);
> 	blk_mq_rq_time_init(rq, alloc_time_ns);
>+	blk_mq_set_rq_sched(rq, data->cmd_flags);
> 	return rq;
> }
> 
>@@ -637,8 +648,6 @@ static struct request *blk_mq_alloc_cached_request(struct request_queue *q,
> 		if (plug->nr_ios == 1)
> 			return NULL;
> 		rq = blk_mq_rq_cache_fill(q, plug, opf, flags);
>-		if (!rq)
>-			return NULL;
> 	} else {
> 		rq = rq_list_peek(&plug->cached_rqs);
> 		if (!rq || rq->q != q)
>@@ -646,15 +655,14 @@ static struct request *blk_mq_alloc_cached_request(struct request_queue *q,
> 
> 		if (blk_mq_get_hctx_type(opf) != rq->mq_hctx->type)
> 			return NULL;
>-		if (op_is_flush(rq->cmd_flags) != op_is_flush(opf))
>-			return NULL;
> 
> 		rq_list_pop(&plug->cached_rqs);
> 		blk_mq_rq_time_init(rq, blk_time_get_ns());
>+		rq->cmd_flags = opf;
>+		INIT_LIST_HEAD(&rq->queuelist);
>+		blk_mq_set_rq_sched(rq, opf);
> 	}
> 
>-	rq->cmd_flags = opf;
>-	INIT_LIST_HEAD(&rq->queuelist);
> 	return rq;
> }
> 
>@@ -3060,8 +3068,6 @@ static struct request *blk_mq_get_cached_request(struct blk_plug *plug,
> 	if (type != rq->mq_hctx->type &&
> 	    (type != HCTX_TYPE_READ || rq->mq_hctx->type != HCTX_TYPE_DEFAULT))
> 		return NULL;
>-	if (op_is_flush(rq->cmd_flags) != op_is_flush(opf))
>-		return NULL;
> 	rq_list_pop(&plug->cached_rqs);
> 	return rq;
> }
>@@ -3166,6 +3172,7 @@ void blk_mq_submit_bio(struct bio *bio)
> 		blk_mq_rq_time_init(rq, blk_time_get_ns());
> 		rq->cmd_flags = bio->bi_opf;
> 		INIT_LIST_HEAD(&rq->queuelist);
>+		blk_mq_set_rq_sched(rq, bio->bi_opf);
> 	} else {
> 		rq = blk_mq_get_new_requests(q, plug, bio);
> 		if (unlikely(!rq)) {
>--

      reply	other threads:[~2026-09-18  5:17 UTC|newest]

Thread overview: 5+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-16 15:16 [PATCH] blk-mq: check passthrough state before reusing cached requests huhai
2026-09-16 19:19 ` John Garry
2026-09-17  5:52   ` huhai
2026-09-17 14:21 ` Keith Busch
2026-09-18  5:16   ` huhai [this message]

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=24313658.46fe.1a0b2f24bcc.Coremail.15815827059@163.com \
    --to=15815827059@163.com \
    --cc=axboe@kernel.dk \
    --cc=huhai@kylinos.cn \
    --cc=kbusch@kernel.org \
    --cc=linux-block@vger.kernel.org \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox