From: Chuck Lever <cel@kernel.org>
To: Jeff Layton <jlayton@kernel.org>, NeilBrown <neil@brown.name>,
Olga Kornievskaia <okorniev@redhat.com>,
Dai Ngo <Dai.Ngo@oracle.com>, Tom Talpey <tom@talpey.com>
Cc: Rick Macklem <rmacklem@uoguelph.ca>,
linux-nfs@vger.kernel.org, Chuck Lever <cel@kernel.org>
Subject: [PATCH v6 10/12] svcrdma: Publish RDMA reply positions
Date: Mon, 21 Sep 2026 09:22:36 -0400 [thread overview]
Message-ID: <20260921-duplicate-reply-cache-v6-10-db5e13fd9944@kernel.org> (raw)
In-Reply-To: <20260921-duplicate-reply-cache-v6-0-db5e13fd9944@kernel.org>
Add the svcrdma side of reply-position reporting. A reply's position
is the sequence number of its Send on the transport, and the Send
completion for that reply publishes the same number as the
transport's acknowledged position. A successful Send completion means
the peer's HCA has acknowledged the message, so a reply whose Send
has completed is in the client's receive buffer.
Completions on one queue pair arrive in posting order. Several nfsd
threads post on one queue pair with no serialization above the
provider, so the order in which Sends are queued is decided inside
the provider. Assign the sequence number and post the Send under one
lock so positions match queuing order. ib_post_send() does not
sleep; the Send Queue wait stays outside the lock.
Only a posted RPC Reply reports a position. The RDMA_ERROR message
and backchannel calls post with no position pointer, and a Send that
fails to post leaves rq_reply_pos at zero.
Assisted-by: LLM
Signed-off-by: Chuck Lever <cel@kernel.org>
---
include/linux/sunrpc/svc_rdma.h | 5 ++++-
net/sunrpc/xprtrdma/svc_rdma_backchannel.c | 2 +-
net/sunrpc/xprtrdma/svc_rdma_sendto.c | 27 ++++++++++++++++++++++++---
net/sunrpc/xprtrdma/svc_rdma_transport.c | 1 +
4 files changed, 30 insertions(+), 5 deletions(-)
diff --git a/include/linux/sunrpc/svc_rdma.h b/include/linux/sunrpc/svc_rdma.h
index 76aa5ec4ab40..7627afed16bb 100644
--- a/include/linux/sunrpc/svc_rdma.h
+++ b/include/linux/sunrpc/svc_rdma.h
@@ -96,6 +96,8 @@ struct svcxprt_rdma {
spinlock_t sc_send_lock;
struct llist_head sc_send_ctxts;
+ spinlock_t sc_post_lock; /* orders Send posting */
+ u64 sc_post_seq; /* Sends posted so far */
spinlock_t sc_rw_ctxt_lock;
struct llist_head sc_rw_ctxts;
@@ -242,6 +244,7 @@ struct svc_rdma_send_ctxt {
struct ib_send_wr *sc_wr_chain;
int sc_sqecount;
struct ib_cqe sc_cqe;
+ u64 sc_pos;
struct xdr_buf sc_hdrbuf;
struct xdr_stream sc_stream;
@@ -305,7 +308,7 @@ extern struct svc_rdma_send_ctxt *
extern void svc_rdma_send_ctxt_put(struct svcxprt_rdma *rdma,
struct svc_rdma_send_ctxt *ctxt);
extern int svc_rdma_post_send(struct svcxprt_rdma *rdma,
- struct svc_rdma_send_ctxt *ctxt);
+ struct svc_rdma_send_ctxt *ctxt, u64 *pos);
extern int svc_rdma_map_reply_msg(struct svcxprt_rdma *rdma,
struct svc_rdma_send_ctxt *sctxt,
const struct svc_rdma_pcl *write_pcl,
diff --git a/net/sunrpc/xprtrdma/svc_rdma_backchannel.c b/net/sunrpc/xprtrdma/svc_rdma_backchannel.c
index e5a78b761012..549c0c39a97a 100644
--- a/net/sunrpc/xprtrdma/svc_rdma_backchannel.c
+++ b/net/sunrpc/xprtrdma/svc_rdma_backchannel.c
@@ -90,7 +90,7 @@ static int svc_rdma_bc_sendto(struct svcxprt_rdma *rdma,
*/
get_page(virt_to_page(rqst->rq_buffer));
sctxt->sc_send_wr.opcode = IB_WR_SEND;
- return svc_rdma_post_send(rdma, sctxt);
+ return svc_rdma_post_send(rdma, sctxt, NULL);
}
/* Server-side transport endpoint wants a whole page for its send
diff --git a/net/sunrpc/xprtrdma/svc_rdma_sendto.c b/net/sunrpc/xprtrdma/svc_rdma_sendto.c
index c09659b17351..47391cc9d750 100644
--- a/net/sunrpc/xprtrdma/svc_rdma_sendto.c
+++ b/net/sunrpc/xprtrdma/svc_rdma_sendto.c
@@ -222,6 +222,7 @@ struct svc_rdma_send_ctxt *svc_rdma_send_ctxt_get(struct svcxprt_rdma *rdma)
ctxt->sc_page_count = 0;
ctxt->sc_wr_chain = &ctxt->sc_send_wr;
ctxt->sc_sqecount = 1;
+ ctxt->sc_pos = 0;
return ctxt;
@@ -471,6 +472,8 @@ static void svc_rdma_wc_send(struct ib_cq *cq, struct ib_wc *wc)
goto flushed;
trace_svcrdma_wc_send(&ctxt->sc_cid);
+ if (ctxt->sc_pos > atomic64_read(&rdma->sc_xprt.xpt_acked_pos))
+ atomic64_set(&rdma->sc_xprt.xpt_acked_pos, ctxt->sc_pos);
svc_rdma_send_ctxt_put(rdma, ctxt);
return;
@@ -487,23 +490,29 @@ static void svc_rdma_wc_send(struct ib_cq *cq, struct ib_wc *wc)
* svc_rdma_post_send - Post a WR chain to the Send Queue
* @rdma: transport context
* @ctxt: WR chain to post
+ * @pos: OUT: position of this Send on @rdma, or NULL
*
* Copy fields in @ctxt to stack variables in order to guarantee
* that these values remain available after the ib_post_send() call.
* In some error flow cases, svc_rdma_wc_send() releases @ctxt.
*
+ * @pos is written only when the chain was posted. Positions on one
+ * transport increase in posting order, and the Send completion of
+ * @ctxt publishes its position in xpt_acked_pos.
+ *
* Return values:
* %0: @ctxt's WR chain was posted successfully
* %-ENOTCONN: The connection was lost
*/
int svc_rdma_post_send(struct svcxprt_rdma *rdma,
- struct svc_rdma_send_ctxt *ctxt)
+ struct svc_rdma_send_ctxt *ctxt, u64 *pos)
{
struct ib_send_wr *first_wr = ctxt->sc_wr_chain;
struct ib_send_wr *send_wr = &ctxt->sc_send_wr;
const struct ib_send_wr *bad_wr = first_wr;
struct rpc_rdma_cid cid = ctxt->sc_cid;
int ret, sqecount = ctxt->sc_sqecount;
+ u64 seq;
might_sleep();
@@ -518,10 +527,22 @@ int svc_rdma_post_send(struct svcxprt_rdma *rdma,
return ret;
trace_svcrdma_post_send(ctxt);
+
+ /*
+ * Assign the position and post under one lock so positions
+ * match the order the provider queues the Sends, and thus
+ * completion order.
+ */
+ spin_lock(&rdma->sc_post_lock);
+ seq = ++rdma->sc_post_seq;
+ ctxt->sc_pos = seq;
ret = ib_post_send(rdma->sc_qp, first_wr, &bad_wr);
+ spin_unlock(&rdma->sc_post_lock);
if (ret)
return svc_rdma_post_send_err(rdma, &cid, bad_wr,
first_wr, sqecount, ret);
+ if (pos)
+ *pos = seq;
return 0;
}
@@ -1052,7 +1073,7 @@ static int svc_rdma_send_reply_msg(struct svcxprt_rdma *rdma,
send_wr->opcode = IB_WR_SEND;
}
- return svc_rdma_post_send(rdma, sctxt);
+ return svc_rdma_post_send(rdma, sctxt, &rqstp->rq_reply_pos);
}
/**
@@ -1122,7 +1143,7 @@ void svc_rdma_send_error_msg(struct svcxprt_rdma *rdma,
*/
sctxt->sc_wr_chain = &sctxt->sc_send_wr;
sctxt->sc_sqecount = 1;
- if (svc_rdma_post_send(rdma, sctxt))
+ if (svc_rdma_post_send(rdma, sctxt, NULL))
goto put_ctxt;
return;
diff --git a/net/sunrpc/xprtrdma/svc_rdma_transport.c b/net/sunrpc/xprtrdma/svc_rdma_transport.c
index f949601b2144..84042688bb3d 100644
--- a/net/sunrpc/xprtrdma/svc_rdma_transport.c
+++ b/net/sunrpc/xprtrdma/svc_rdma_transport.c
@@ -207,6 +207,7 @@ static struct svcxprt_rdma *svc_rdma_create_xprt(struct svc_serv *serv,
lockdep_set_class(&cma_xprt->sc_send_lock, &svcrdma_sctx_lock);
spin_lock_init(&cma_xprt->sc_rw_ctxt_lock);
lockdep_set_class(&cma_xprt->sc_rw_ctxt_lock, &svcrdma_rwctx_lock);
+ spin_lock_init(&cma_xprt->sc_post_lock);
/*
* Note that this implies that the underlying transport support
--
2.55.0
next prev parent reply other threads:[~2026-09-21 13:22 UTC|newest]
Thread overview: 16+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-21 13:22 [PATCH v6 00/12] Improve the scalability of NFSD's classic DRC Chuck Lever
2026-09-21 13:22 ` [PATCH v6 01/12] NFSD: Make the DRC size limit independent of page size Chuck Lever
2026-09-21 13:22 ` [PATCH v6 02/12] NFSD: Remove hard cap on duplicate reply cache size Chuck Lever
2026-09-21 13:22 ` [PATCH v6 03/12] SUNRPC: Assign a unique identifier to each svc_xprt Chuck Lever
2026-09-21 13:22 ` [PATCH v6 04/12] NFSD: Track transport in DRC entries Chuck Lever
2026-09-21 13:22 ` [PATCH v6 05/12] NFSD: Prepare bucket pruning for additional eviction reasons Chuck Lever
2026-09-21 13:22 ` [PATCH v6 06/12] NFSD: Add tracepoints for DRC entry eviction Chuck Lever
2026-09-21 13:22 ` [PATCH v6 07/12] NFSD: Record DRC population in lookup tracepoints Chuck Lever
2026-09-21 13:22 ` [PATCH v6 08/12] SUNRPC: Publish reply positions for upper-layer consumers Chuck Lever
2026-09-21 13:22 ` [PATCH v6 09/12] SUNRPC: Publish TCP reply positions Chuck Lever
2026-09-21 13:22 ` Chuck Lever [this message]
2026-09-21 13:22 ` [PATCH v6 11/12] NFSD: Evict acknowledged DRC entries Chuck Lever
2026-09-21 13:22 ` [PATCH v6 12/12] NFSD: Remove DRC checksum and payload_misses stat Chuck Lever
2026-09-23 14:56 ` Jeff Layton
2026-09-23 16:36 ` Chuck Lever
2026-09-22 6:55 ` [PATCH v6 00/12] Improve the scalability of NFSD's classic DRC NeilBrown
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260921-duplicate-reply-cache-v6-10-db5e13fd9944@kernel.org \
--to=cel@kernel.org \
--cc=Dai.Ngo@oracle.com \
--cc=jlayton@kernel.org \
--cc=linux-nfs@vger.kernel.org \
--cc=neil@brown.name \
--cc=okorniev@redhat.com \
--cc=rmacklem@uoguelph.ca \
--cc=tom@talpey.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox