From: Chuck Lever <cel@kernel.org>
To: Jeff Layton <jlayton@kernel.org>, NeilBrown <neil@brown.name>,
Olga Kornievskaia <okorniev@redhat.com>,
Dai Ngo <Dai.Ngo@oracle.com>, Tom Talpey <tom@talpey.com>
Cc: Rick Macklem <rmacklem@uoguelph.ca>,
linux-nfs@vger.kernel.org, Chuck Lever <cel@kernel.org>
Subject: [PATCH v5 09/11] svcrdma: Publish RDMA reply positions
Date: Fri, 18 Sep 2026 13:21:21 -0400 [thread overview]
Message-ID: <20260918-duplicate-reply-cache-v5-9-b6aba9ebf2f4@kernel.org> (raw)
In-Reply-To: <20260918-duplicate-reply-cache-v5-0-b6aba9ebf2f4@kernel.org>
Add the svcrdma side of reply-position reporting. A reply's position
is the sequence number of its Send on the transport, and the Send
completion for that reply publishes the same number as the
transport's acknowledged position. A successful Send completion means
the peer's HCA has acknowledged the message, so a reply whose Send
has completed is in the client's receive buffer.
Completions on one queue pair arrive in posting order. Several nfsd
threads post on one queue pair with no serialization above the
provider, so the order in which Sends are queued is decided inside
the provider. Assign the sequence number and post the Send under one
lock so positions match queuing order. ib_post_send() does not
sleep; the Send Queue wait stays outside the lock.
Only a posted RPC Reply reports a position. The RDMA_ERROR message
and backchannel calls post with no position pointer, and a Send that
fails to post leaves rq_reply_pos at zero.
Assisted-by: LLM
Signed-off-by: Chuck Lever <cel@kernel.org>
---
include/linux/sunrpc/svc_rdma.h | 5 ++++-
net/sunrpc/xprtrdma/svc_rdma_backchannel.c | 2 +-
net/sunrpc/xprtrdma/svc_rdma_sendto.c | 27 ++++++++++++++++++++++++---
net/sunrpc/xprtrdma/svc_rdma_transport.c | 1 +
4 files changed, 30 insertions(+), 5 deletions(-)
diff --git a/include/linux/sunrpc/svc_rdma.h b/include/linux/sunrpc/svc_rdma.h
index 76aa5ec4ab40..7627afed16bb 100644
--- a/include/linux/sunrpc/svc_rdma.h
+++ b/include/linux/sunrpc/svc_rdma.h
@@ -96,6 +96,8 @@ struct svcxprt_rdma {
spinlock_t sc_send_lock;
struct llist_head sc_send_ctxts;
+ spinlock_t sc_post_lock; /* orders Send posting */
+ u64 sc_post_seq; /* Sends posted so far */
spinlock_t sc_rw_ctxt_lock;
struct llist_head sc_rw_ctxts;
@@ -242,6 +244,7 @@ struct svc_rdma_send_ctxt {
struct ib_send_wr *sc_wr_chain;
int sc_sqecount;
struct ib_cqe sc_cqe;
+ u64 sc_pos;
struct xdr_buf sc_hdrbuf;
struct xdr_stream sc_stream;
@@ -305,7 +308,7 @@ extern struct svc_rdma_send_ctxt *
extern void svc_rdma_send_ctxt_put(struct svcxprt_rdma *rdma,
struct svc_rdma_send_ctxt *ctxt);
extern int svc_rdma_post_send(struct svcxprt_rdma *rdma,
- struct svc_rdma_send_ctxt *ctxt);
+ struct svc_rdma_send_ctxt *ctxt, u64 *pos);
extern int svc_rdma_map_reply_msg(struct svcxprt_rdma *rdma,
struct svc_rdma_send_ctxt *sctxt,
const struct svc_rdma_pcl *write_pcl,
diff --git a/net/sunrpc/xprtrdma/svc_rdma_backchannel.c b/net/sunrpc/xprtrdma/svc_rdma_backchannel.c
index e5a78b761012..549c0c39a97a 100644
--- a/net/sunrpc/xprtrdma/svc_rdma_backchannel.c
+++ b/net/sunrpc/xprtrdma/svc_rdma_backchannel.c
@@ -90,7 +90,7 @@ static int svc_rdma_bc_sendto(struct svcxprt_rdma *rdma,
*/
get_page(virt_to_page(rqst->rq_buffer));
sctxt->sc_send_wr.opcode = IB_WR_SEND;
- return svc_rdma_post_send(rdma, sctxt);
+ return svc_rdma_post_send(rdma, sctxt, NULL);
}
/* Server-side transport endpoint wants a whole page for its send
diff --git a/net/sunrpc/xprtrdma/svc_rdma_sendto.c b/net/sunrpc/xprtrdma/svc_rdma_sendto.c
index c09659b17351..47391cc9d750 100644
--- a/net/sunrpc/xprtrdma/svc_rdma_sendto.c
+++ b/net/sunrpc/xprtrdma/svc_rdma_sendto.c
@@ -222,6 +222,7 @@ struct svc_rdma_send_ctxt *svc_rdma_send_ctxt_get(struct svcxprt_rdma *rdma)
ctxt->sc_page_count = 0;
ctxt->sc_wr_chain = &ctxt->sc_send_wr;
ctxt->sc_sqecount = 1;
+ ctxt->sc_pos = 0;
return ctxt;
@@ -471,6 +472,8 @@ static void svc_rdma_wc_send(struct ib_cq *cq, struct ib_wc *wc)
goto flushed;
trace_svcrdma_wc_send(&ctxt->sc_cid);
+ if (ctxt->sc_pos > atomic64_read(&rdma->sc_xprt.xpt_acked_pos))
+ atomic64_set(&rdma->sc_xprt.xpt_acked_pos, ctxt->sc_pos);
svc_rdma_send_ctxt_put(rdma, ctxt);
return;
@@ -487,23 +490,29 @@ static void svc_rdma_wc_send(struct ib_cq *cq, struct ib_wc *wc)
* svc_rdma_post_send - Post a WR chain to the Send Queue
* @rdma: transport context
* @ctxt: WR chain to post
+ * @pos: OUT: position of this Send on @rdma, or NULL
*
* Copy fields in @ctxt to stack variables in order to guarantee
* that these values remain available after the ib_post_send() call.
* In some error flow cases, svc_rdma_wc_send() releases @ctxt.
*
+ * @pos is written only when the chain was posted. Positions on one
+ * transport increase in posting order, and the Send completion of
+ * @ctxt publishes its position in xpt_acked_pos.
+ *
* Return values:
* %0: @ctxt's WR chain was posted successfully
* %-ENOTCONN: The connection was lost
*/
int svc_rdma_post_send(struct svcxprt_rdma *rdma,
- struct svc_rdma_send_ctxt *ctxt)
+ struct svc_rdma_send_ctxt *ctxt, u64 *pos)
{
struct ib_send_wr *first_wr = ctxt->sc_wr_chain;
struct ib_send_wr *send_wr = &ctxt->sc_send_wr;
const struct ib_send_wr *bad_wr = first_wr;
struct rpc_rdma_cid cid = ctxt->sc_cid;
int ret, sqecount = ctxt->sc_sqecount;
+ u64 seq;
might_sleep();
@@ -518,10 +527,22 @@ int svc_rdma_post_send(struct svcxprt_rdma *rdma,
return ret;
trace_svcrdma_post_send(ctxt);
+
+ /*
+ * Assign the position and post under one lock so positions
+ * match the order the provider queues the Sends, and thus
+ * completion order.
+ */
+ spin_lock(&rdma->sc_post_lock);
+ seq = ++rdma->sc_post_seq;
+ ctxt->sc_pos = seq;
ret = ib_post_send(rdma->sc_qp, first_wr, &bad_wr);
+ spin_unlock(&rdma->sc_post_lock);
if (ret)
return svc_rdma_post_send_err(rdma, &cid, bad_wr,
first_wr, sqecount, ret);
+ if (pos)
+ *pos = seq;
return 0;
}
@@ -1052,7 +1073,7 @@ static int svc_rdma_send_reply_msg(struct svcxprt_rdma *rdma,
send_wr->opcode = IB_WR_SEND;
}
- return svc_rdma_post_send(rdma, sctxt);
+ return svc_rdma_post_send(rdma, sctxt, &rqstp->rq_reply_pos);
}
/**
@@ -1122,7 +1143,7 @@ void svc_rdma_send_error_msg(struct svcxprt_rdma *rdma,
*/
sctxt->sc_wr_chain = &sctxt->sc_send_wr;
sctxt->sc_sqecount = 1;
- if (svc_rdma_post_send(rdma, sctxt))
+ if (svc_rdma_post_send(rdma, sctxt, NULL))
goto put_ctxt;
return;
diff --git a/net/sunrpc/xprtrdma/svc_rdma_transport.c b/net/sunrpc/xprtrdma/svc_rdma_transport.c
index f949601b2144..84042688bb3d 100644
--- a/net/sunrpc/xprtrdma/svc_rdma_transport.c
+++ b/net/sunrpc/xprtrdma/svc_rdma_transport.c
@@ -207,6 +207,7 @@ static struct svcxprt_rdma *svc_rdma_create_xprt(struct svc_serv *serv,
lockdep_set_class(&cma_xprt->sc_send_lock, &svcrdma_sctx_lock);
spin_lock_init(&cma_xprt->sc_rw_ctxt_lock);
lockdep_set_class(&cma_xprt->sc_rw_ctxt_lock, &svcrdma_rwctx_lock);
+ spin_lock_init(&cma_xprt->sc_post_lock);
/*
* Note that this implies that the underlying transport support
--
2.55.0
next prev parent reply other threads:[~2026-09-18 17:21 UTC|newest]
Thread overview: 19+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-18 17:21 [PATCH v5 00/11] Improve the scalability of NFSD's classic DRC Chuck Lever
2026-09-18 17:21 ` [PATCH v5 01/11] NFSD: Remove hard cap on duplicate reply cache size Chuck Lever
2026-09-18 23:00 ` NeilBrown
2026-09-20 17:47 ` Chuck Lever
2026-09-20 22:10 ` NeilBrown
2026-09-18 17:21 ` [PATCH v5 02/11] SUNRPC: Assign a unique identifier to each svc_xprt Chuck Lever
2026-09-18 17:21 ` [PATCH v5 03/11] NFSD: Track transport in DRC entries Chuck Lever
2026-09-18 17:21 ` [PATCH v5 04/11] NFSD: Prepare bucket pruning for additional eviction reasons Chuck Lever
2026-09-18 17:21 ` [PATCH v5 05/11] NFSD: Add tracepoints for DRC entry eviction Chuck Lever
2026-09-18 17:21 ` [PATCH v5 06/11] NFSD: Record DRC population in lookup tracepoints Chuck Lever
2026-09-18 17:21 ` [PATCH v5 07/11] SUNRPC: Publish reply positions for upper-layer consumers Chuck Lever
2026-09-18 17:21 ` [PATCH v5 08/11] SUNRPC: Publish TCP reply positions Chuck Lever
2026-09-18 17:21 ` Chuck Lever [this message]
2026-09-18 17:21 ` [PATCH v5 10/11] NFSD: Evict acknowledged DRC entries Chuck Lever
2026-09-18 17:21 ` [PATCH v5 11/11] NFSD: Remove DRC checksum and payload_misses stat Chuck Lever
2026-09-18 23:39 ` NeilBrown
2026-09-19 16:11 ` Chuck Lever
2026-09-20 10:37 ` NeilBrown
2026-09-20 17:49 ` Chuck Lever
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260918-duplicate-reply-cache-v5-9-b6aba9ebf2f4@kernel.org \
--to=cel@kernel.org \
--cc=Dai.Ngo@oracle.com \
--cc=jlayton@kernel.org \
--cc=linux-nfs@vger.kernel.org \
--cc=neil@brown.name \
--cc=okorniev@redhat.com \
--cc=rmacklem@uoguelph.ca \
--cc=tom@talpey.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.