From: J Louis Kaplan <Louis.Kaplan@arm.com>
To: cel@kernel.org, dai.ngo@oracle.com, jlayton@kernel.org,
neil@brown.name, okorniev@redhat.com, tom@talpey.com
Cc: linux-nfs@vger.kernel.org, linux-rdma@vger.kernel.org,
anna@kernel.org, jgg@ziepe.ca, leon@kernel.org,
trondmy@kernel.org, J Louis Kaplan <Louis.Kaplan@arm.com>
Subject: [RFC PATCH 4/6] svcrdma: Coalesce contiguous pages in RDMA Write chunks
Date: Tue, 6 Oct 2026 10:00:26 +0100 [thread overview]
Message-ID: <20261006090028.3412544-5-Louis.Kaplan@arm.com> (raw)
In-Reply-To: <20261006090028.3412544-1-Louis.Kaplan@arm.com>
When sending replies with remote direct memory access (RDMA)
Writes, the pagelist is represented by one bvec per base page.
This prevents contiguous pages from sharing a buffer descriptor.
Coalesce contiguous runs within each remote segment, subject to
device length limits. Count the resulting bvecs before acquiring
the context so its allocation reflects the coalesced entry count.
The counting pass leaves the payload cursor untouched.
Signed-off-by: J Louis Kaplan <Louis.Kaplan@arm.com>
Assisted-by: LLM
---
net/sunrpc/xprtrdma/svc_rdma_rw.c | 77 +++++++++++++++++--------------
1 file changed, 43 insertions(+), 34 deletions(-)
diff --git a/net/sunrpc/xprtrdma/svc_rdma_rw.c b/net/sunrpc/xprtrdma/svc_rdma_rw.c
index 63d00f0cc1dbd..a2e1c693b1bc1 100644
--- a/net/sunrpc/xprtrdma/svc_rdma_rw.c
+++ b/net/sunrpc/xprtrdma/svc_rdma_rw.c
@@ -445,46 +445,48 @@ static int svc_rdma_post_chunk_ctxt(struct svcxprt_rdma *rdma,
return 0;
}
-/* Build a bvec that covers one kvec in an xdr_buf.
+/*
+ * With a NULL @ctxt, bvec constructors count entries without advancing
+ * @info. Otherwise they populate the context and advance @info.
*/
-static void svc_rdma_vec_to_bvec(struct svc_rdma_write_info *info,
- unsigned int len,
- struct svc_rdma_rw_ctxt *ctxt)
+static unsigned int
+svc_rdma_vec_to_bvec(struct svc_rdma_write_info *info, unsigned int len,
+ struct svc_rdma_rw_ctxt *ctxt)
{
+ if (!ctxt)
+ return 1;
+
bvec_set_virt(&ctxt->rw_bvec[0], info->wi_base, len);
info->wi_base += len;
ctxt->rw_nents = 1;
+ return 1;
}
/* Build a bvec array that covers part of an xdr_buf's pagelist.
+ * If @ctxt is NULL, only count the required entries.
*/
-static void svc_rdma_pagelist_to_bvec(struct svc_rdma_write_info *info,
- unsigned int remaining,
- struct svc_rdma_rw_ctxt *ctxt)
+static unsigned int
+svc_rdma_pagelist_to_bvec(struct svc_rdma_write_info *info,
+ unsigned int remaining,
+ struct svc_rdma_rw_ctxt *ctxt)
{
- unsigned int bvec_idx, bvec_len, page_off, page_no;
const struct xdr_buf *xdr = info->wi_xdr;
- struct page **page;
-
- page_off = info->wi_next_off + xdr->page_base;
- page_no = page_off >> PAGE_SHIFT;
- page_off = offset_in_page(page_off);
- page = xdr->pages + page_no;
- info->wi_next_off += remaining;
- bvec_idx = 0;
- do {
- bvec_len = min_t(unsigned int, remaining,
- PAGE_SIZE - page_off);
- bvec_set_page(&ctxt->rw_bvec[bvec_idx], *page, bvec_len,
- page_off);
- remaining -= bvec_len;
- page_off = 0;
- bvec_idx++;
- page++;
- } while (remaining);
-
- ctxt->rw_nents = bvec_idx;
+ unsigned int page_pos = info->wi_next_off + xdr->page_base;
+ unsigned int page_no = page_pos >> PAGE_SHIFT;
+ unsigned int max_bvec_len = svc_rdma_max_bvec_len(info->wi_rdma);
+ unsigned int page_off = offset_in_page(page_pos);
+ unsigned int nr_pages = DIV_ROUND_UP(page_off + remaining, PAGE_SIZE);
+ struct bio_vec *bvecs = ctxt ? ctxt->rw_bvec : NULL;
+ unsigned int nents;
+
+ nents = svc_pages_to_bvecs(bvecs, xdr->pages + page_no, nr_pages,
+ page_off, remaining, max_bvec_len);
+ if (ctxt) {
+ ctxt->rw_nents = nents;
+ info->wi_next_off += remaining;
+ }
+ return nents;
}
/* Construct RDMA Write WRs to send a portion of an xdr_buf containing
@@ -492,9 +494,9 @@ static void svc_rdma_pagelist_to_bvec(struct svc_rdma_write_info *info,
*/
static int
svc_rdma_build_writes(struct svc_rdma_write_info *info,
- void (*constructor)(struct svc_rdma_write_info *info,
- unsigned int len,
- struct svc_rdma_rw_ctxt *ctxt),
+ unsigned int (*constructor)(struct svc_rdma_write_info *,
+ unsigned int,
+ struct svc_rdma_rw_ctxt *),
unsigned int remaining)
{
struct svc_rdma_chunk_ctxt *cc = &info->wi_cc;
@@ -504,6 +506,7 @@ svc_rdma_build_writes(struct svc_rdma_write_info *info,
int ret;
do {
+ unsigned int nr_bvec;
unsigned int write_len;
u64 offset;
@@ -514,12 +517,18 @@ svc_rdma_build_writes(struct svc_rdma_write_info *info,
write_len = min(remaining, seg->rs_length - info->wi_seg_off);
if (!write_len)
goto out_overflow;
- ctxt = svc_rdma_get_rw_ctxt(rdma,
- (write_len >> PAGE_SHIFT) + 2);
+ nr_bvec = constructor(info, write_len, NULL);
+ if (!nr_bvec)
+ return -EIO;
+
+ ctxt = svc_rdma_get_rw_ctxt(rdma, nr_bvec);
if (!ctxt)
return -ENOMEM;
- constructor(info, write_len, ctxt);
+ if (WARN_ON_ONCE(constructor(info, write_len, ctxt) != nr_bvec)) {
+ svc_rdma_put_rw_ctxt(rdma, ctxt);
+ return -EIO;
+ }
offset = seg->rs_offset + info->wi_seg_off;
ret = svc_rdma_rw_ctx_init(rdma, ctxt, offset, seg->rs_handle,
write_len, DMA_TO_DEVICE);
--
2.43.0
next prev parent reply other threads:[~2026-10-06 9:01 UTC|newest]
Thread overview: 16+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-10-06 9:00 [RFC PATCH 0/6] Improve NFS server direct throughput with passthrough enabled J Louis Kaplan
2026-10-06 9:00 ` [RFC PATCH 1/6] sunrpc: Add helpers to build bvecs from contiguous pages J Louis Kaplan
2026-10-06 13:49 ` Chuck Lever
2026-10-08 22:44 ` J Louis Kaplan
2026-10-06 9:00 ` [RFC PATCH 2/6] nfsd: Coalesce contiguous pages for direct reads J Louis Kaplan
2026-10-06 9:00 ` [RFC PATCH 3/6] svcrdma: Coalesce contiguous pages when mapping replies J Louis Kaplan
2026-10-06 13:55 ` Chuck Lever
2026-10-08 22:54 ` J Louis Kaplan
2026-10-06 9:00 ` J Louis Kaplan [this message]
2026-10-06 9:00 ` [RFC PATCH 5/6] svcrdma: Coalesce contiguous pages in RDMA Read chunks J Louis Kaplan
2026-10-06 9:00 ` [RFC PATCH 6/6] sunrpc: Allocate svc request pages from large folios J Louis Kaplan
2026-10-06 14:00 ` Chuck Lever
2026-10-08 22:58 ` J Louis Kaplan
2026-10-09 15:28 ` Chuck Lever
2026-10-06 13:47 ` [RFC PATCH 0/6] Improve NFS server direct throughput with passthrough enabled Chuck Lever
2026-10-08 22:50 ` J Louis Kaplan
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20261006090028.3412544-5-Louis.Kaplan@arm.com \
--to=louis.kaplan@arm.com \
--cc=anna@kernel.org \
--cc=cel@kernel.org \
--cc=dai.ngo@oracle.com \
--cc=jgg@ziepe.ca \
--cc=jlayton@kernel.org \
--cc=leon@kernel.org \
--cc=linux-nfs@vger.kernel.org \
--cc=linux-rdma@vger.kernel.org \
--cc=neil@brown.name \
--cc=okorniev@redhat.com \
--cc=tom@talpey.com \
--cc=trondmy@kernel.org \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox