The Linux Kernel Mailing List
 help / color / mirror / Atom feed
* [PATCH v2] nvme-tcp: pin io_cpu to submitter cpu
@ 2026-08-20  8:36 Saravanan D
  2026-08-20 15:39 ` Daniel Wagner
                   ` (2 more replies)
  0 siblings, 3 replies; 9+ messages in thread
From: Saravanan D @ 2026-08-20  8:36 UTC (permalink / raw)
  To: linux-nvme; +Cc: kbusch, hch, sagi, axboe, linux-kernel, Saravanan D

nvme_tcp_set_queue_io_cpu() picks each queue's io_cpu at connect time,
before any I/O exists, as the least loaded CPU in the queue's blk-mq
map group, and all socket work then runs there for the connection's
lifetime. On hosts that partition CPUs between pinned workloads a map
group can straddle a partition boundary, so the pick can land one
workload's socket processing on CPUs owned by another. On a 384 cpu
multi tenant host with one VM driving ~1.1 GB/s of writes, 9% of
nvme_tcp_io_work executions ran outside the submitting VM's cpuset,
all on io_cpus of boundary straddling map groups, observed by the
neighbor as steal time it did not cause.

Adopt the submitting CPU as io_cpu for every command except the
fabrics Connect. The submitter is a member of the map group by
construction, and the nvme_tcp_cpu_queues accounting moves with each
adoption. Connect is the only command on an I/O queue that does not
represent the data path, since it is injected on an arbitrary CPU by
blk_mq_alloc_request_hctx(), so it is skipped and the first real read
or write decides. User passthrough is submitted from a real task on
the submitting CPU and adopts like any other command.

Queues outlive the workloads that submit through them, so adoption
re-arms after 30 seconds of queue quiet. An idle queue is reclaimed by
its next submitter, while a busy queue keeps a stable io_cpu and
cannot ping pong between two live submitters. Concurrent writers on
different CPUs serialize on a cmpxchg on io_cpu.

The behavior is opt in per controller via the io_cpu_adopt fabrics
option at connect time. wq_unbound takes precedence when set.

Signed-off-by: Saravanan D <saravanand@crusoe.ai>
---
Changes since v1 [1]:
- Special case the fabrics Connect command instead of skipping all
  passthrough commands, so user passthrough I/O adopts too.
- Make it a per-controller io_cpu_adopt fabrics option instead of a
  global wq_adopt module parameter, set once at connect time rather
  than flipped under a live connection.

Both per Christoph Hellwig's review.

[1] https://lore.kernel.org/linux-nvme/20260806023947.94680-2-saravanand@crusoe.ai/

 drivers/nvme/host/fabrics.c |  4 ++
 drivers/nvme/host/fabrics.h |  2 +
 drivers/nvme/host/tcp.c     | 75 ++++++++++++++++++++++++++++++++++++-
 3 files changed, 80 insertions(+), 1 deletion(-)

diff --git a/drivers/nvme/host/fabrics.c b/drivers/nvme/host/fabrics.c
index fd5abd04e080..26f8703744de 100644
--- a/drivers/nvme/host/fabrics.c
+++ b/drivers/nvme/host/fabrics.c
@@ -695,6 +695,7 @@ static const match_table_t opt_tokens = {
 	{ NVMF_OPT_NR_WRITE_QUEUES,	"nr_write_queues=%d"	},
 	{ NVMF_OPT_NR_POLL_QUEUES,	"nr_poll_queues=%d"	},
 	{ NVMF_OPT_TOS,			"tos=%d"		},
+	{ NVMF_OPT_IO_CPU_ADOPT,	"io_cpu_adopt"		},
 #ifdef CONFIG_NVME_TCP_TLS
 	{ NVMF_OPT_KEYRING,		"keyring=%d"		},
 	{ NVMF_OPT_TLS_KEY,		"tls_key=%d"		},
@@ -951,6 +952,9 @@ static int nvmf_parse_options(struct nvmf_ctrl_options *opts,
 		case NVMF_OPT_DATA_DIGEST:
 			opts->data_digest = true;
 			break;
+		case NVMF_OPT_IO_CPU_ADOPT:
+			opts->io_cpu_adopt = true;
+			break;
 		case NVMF_OPT_NR_WRITE_QUEUES:
 			if (match_int(args, &token)) {
 				ret = -EINVAL;
diff --git a/drivers/nvme/host/fabrics.h b/drivers/nvme/host/fabrics.h
index caf5503d0833..3ecac041a628 100644
--- a/drivers/nvme/host/fabrics.h
+++ b/drivers/nvme/host/fabrics.h
@@ -67,6 +67,7 @@ enum {
 	NVMF_OPT_KEYRING	= 1 << 26,
 	NVMF_OPT_TLS_KEY	= 1 << 27,
 	NVMF_OPT_CONCAT		= 1 << 28,
+	NVMF_OPT_IO_CPU_ADOPT	= 1 << 29,
 };
 
 /**
@@ -140,6 +141,7 @@ struct nvmf_ctrl_options {
 	unsigned int		nr_poll_queues;
 	int			tos;
 	int			fast_io_fail_tmo;
+	bool			io_cpu_adopt;
 };
 
 /*
diff --git a/drivers/nvme/host/tcp.c b/drivers/nvme/host/tcp.c
index 87d8067f3283..530e38695257 100644
--- a/drivers/nvme/host/tcp.c
+++ b/drivers/nvme/host/tcp.c
@@ -92,6 +92,7 @@ enum nvme_tcp_queue_flags {
 	NVME_TCP_Q_LIVE		= 1,
 	NVME_TCP_Q_POLLING	= 2,
 	NVME_TCP_Q_IO_CPU_SET	= 3,
+	NVME_TCP_Q_IO_CPU_ADOPTED = 4,
 };
 
 enum nvme_tcp_recv_state {
@@ -105,6 +106,7 @@ struct nvme_tcp_queue {
 	struct socket		*sock;
 	struct work_struct	io_work;
 	int			io_cpu;
+	unsigned long		last_data;
 
 	struct mutex		queue_lock;
 	struct mutex		send_mutex;
@@ -2783,6 +2785,74 @@ static void nvme_tcp_commit_rqs(struct blk_mq_hw_ctx *hctx)
 		queue_work_on(queue->io_cpu, nvme_tcp_wq, &queue->io_work);
 }
 
+/* Re-adopt io_cpu on the first data request after this much queue idle time */
+#define NVME_TCP_READOPT_IDLE	(30 * HZ)
+
+/*
+ * Adopt the CPU of the current data submission as the queue's io_cpu.
+ *
+ * The connect time choice in nvme_tcp_set_queue_io_cpu() picks the least
+ * loaded CPU in the queue's mq_map group before any I/O exists, so it
+ * cannot know which side of the group the actual submitters live on. On
+ * hosts that partition CPUs between pinned workloads a group that
+ * straddles a partition boundary can get an io_cpu on CPUs the submitting
+ * workload does not own, and its network processing then preempts an
+ * unrelated workload. The submitting CPU is in the queue's mq_map group
+ * by construction, so adopting it preserves the spreading property while
+ * landing the work on the side that generates it.
+ *
+ * Queues belong to the controller connection and outlive the workloads
+ * that submit through them, so adoption re-arms after NVME_TCP_READOPT_IDLE
+ * of queue quiet. A successor workload reclaims an idle queue with its
+ * first data request, while a continuously busy queue keeps a stable
+ * io_cpu and cannot ping pong between two live submitters.
+ *
+ * The fabrics Connect command targets a specific queue via
+ * blk_mq_alloc_request_hctx() and so runs on an arbitrary CPU that does
+ * not represent the data path, so it is skipped and the first real read
+ * or write decides. All other commands, including user passthrough,
+ * carry a real submitting CPU and adopt.
+ *
+ * Adoption is opt in per controller via the io_cpu_adopt connect option
+ * and is bypassed when wq_unbound is set.
+ */
+static void nvme_tcp_adopt_io_cpu(struct nvme_tcp_queue *queue,
+		struct request *rq)
+{
+	struct nvme_command *cmd = nvme_req(rq)->cmd;
+	int old, new;
+
+	if (!queue->ctrl->ctrl.opts->io_cpu_adopt || wq_unbound)
+		return;
+	if (!nvme_tcp_queue_id(queue))
+		return;
+	if (nvme_is_fabrics(cmd) &&
+	    cmd->fabrics.fctype == nvme_fabrics_type_connect)
+		return;
+
+	if (test_bit(NVME_TCP_Q_IO_CPU_ADOPTED, &queue->flags) &&
+	    time_before(jiffies, READ_ONCE(queue->last_data) +
+				NVME_TCP_READOPT_IDLE)) {
+		WRITE_ONCE(queue->last_data, jiffies);
+		return;
+	}
+
+	WRITE_ONCE(queue->last_data, jiffies);
+	set_bit(NVME_TCP_Q_IO_CPU_ADOPTED, &queue->flags);
+
+	old = READ_ONCE(queue->io_cpu);
+	new = raw_smp_processor_id();
+	if (old == new || !try_cmpxchg(&queue->io_cpu, &old, new))
+		return;
+
+	if (test_bit(NVME_TCP_Q_IO_CPU_SET, &queue->flags)) {
+		atomic_dec(&nvme_tcp_cpu_queues[old]);
+		atomic_inc(&nvme_tcp_cpu_queues[new]);
+	}
+	dev_dbg(queue->ctrl->ctrl.device, "queue %d: adopted io_cpu %d\n",
+		nvme_tcp_queue_id(queue), new);
+}
+
 static blk_status_t nvme_tcp_queue_rq(struct blk_mq_hw_ctx *hctx,
 		const struct blk_mq_queue_data *bd)
 {
@@ -2802,6 +2872,8 @@ static blk_status_t nvme_tcp_queue_rq(struct blk_mq_hw_ctx *hctx,
 
 	nvme_start_request(rq);
 
+	nvme_tcp_adopt_io_cpu(queue, rq);
+
 	nvme_tcp_queue_request(req, bd->last);
 
 	return BLK_STS_OK;
@@ -3047,7 +3119,8 @@ static struct nvmf_transport_ops nvme_tcp_transport = {
 			  NVMF_OPT_HDR_DIGEST | NVMF_OPT_DATA_DIGEST |
 			  NVMF_OPT_NR_WRITE_QUEUES | NVMF_OPT_NR_POLL_QUEUES |
 			  NVMF_OPT_TOS | NVMF_OPT_HOST_IFACE | NVMF_OPT_TLS |
-			  NVMF_OPT_KEYRING | NVMF_OPT_TLS_KEY | NVMF_OPT_CONCAT,
+			  NVMF_OPT_KEYRING | NVMF_OPT_TLS_KEY | NVMF_OPT_CONCAT |
+			  NVMF_OPT_IO_CPU_ADOPT,
 	.create_ctrl	= nvme_tcp_create_ctrl,
 };
 

base-commit: bf881dd20062db5e951a0d0703cb476df8c9fdee
-- 
2.53.0


^ permalink raw reply related	[flat|nested] 9+ messages in thread

end of thread, other threads:[~2026-08-23 18:20 UTC | newest]

Thread overview: 9+ messages (download: mbox.gz follow: Atom feed
-- links below jump to the message on this page --
2026-08-20  8:36 [PATCH v2] nvme-tcp: pin io_cpu to submitter cpu Saravanan D
2026-08-20 15:39 ` Daniel Wagner
2026-08-22  0:08   ` Saravanan D
2026-08-21 15:47 ` Nilay Shroff
2026-08-22  0:49   ` Saravanan D
2026-08-22 12:18     ` Nilay Shroff
2026-08-23 18:20       ` Saravanan D
2026-08-22 21:20 ` Sagi Grimberg
2026-08-23 17:43   ` Saravanan D

This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox