From: J Louis Kaplan <Louis.Kaplan@arm.com>
To: cel@kernel.org, dai.ngo@oracle.com, jlayton@kernel.org,
neil@brown.name, okorniev@redhat.com, tom@talpey.com
Cc: linux-nfs@vger.kernel.org, linux-rdma@vger.kernel.org,
anna@kernel.org, jgg@ziepe.ca, leon@kernel.org,
trondmy@kernel.org, J Louis Kaplan <Louis.Kaplan@arm.com>
Subject: [RFC PATCH 3/6] svcrdma: Coalesce contiguous pages when mapping replies
Date: Tue, 6 Oct 2026 10:00:25 +0100 [thread overview]
Message-ID: <20261006090028.3412544-4-Louis.Kaplan@arm.com> (raw)
In-Reply-To: <20261006090028.3412544-1-Louis.Kaplan@arm.com>
Mapping reply pagelists one page at a time uses a separate direct
memory access (DMA) mapping and scatter/gather entry (SGE) for
each base page, even when those pages are physically contiguous.
Instead, map contiguous runs as multipage bvecs, capped by the device's
mapping and segment length limits. Use the same coalescing rules
when counting SGEs for the pull-up decision, avoiding unnecessary
copying when the coalesced reply fits the Send SGE limit.
Virtual-mapping devices bypass device-limit check.
Signed-off-by: J Louis Kaplan <Louis.Kaplan@arm.com>
Assisted-by: LLM
---
include/linux/sunrpc/svc_rdma.h | 15 ++++
net/sunrpc/xprtrdma/svc_rdma_sendto.c | 112 +++++++++++++++-----------
2 files changed, 82 insertions(+), 45 deletions(-)
diff --git a/include/linux/sunrpc/svc_rdma.h b/include/linux/sunrpc/svc_rdma.h
index 71bd5bcc5ce2a..c296bf8b8b1e3 100644
--- a/include/linux/sunrpc/svc_rdma.h
+++ b/include/linux/sunrpc/svc_rdma.h
@@ -42,6 +42,7 @@
#ifndef SVC_RDMA_H
#define SVC_RDMA_H
+#include <linux/dma-mapping.h>
#include <linux/llist.h>
#include <linux/sunrpc/xdr.h>
#include <linux/sunrpc/svcsock.h>
@@ -134,6 +135,20 @@ static inline struct svcxprt_rdma *svc_rdma_rqst_rdma(struct svc_rqst *rqstp)
return container_of(xprt, struct svcxprt_rdma, sc_xprt);
}
+static inline unsigned int
+svc_rdma_max_bvec_len(const struct svcxprt_rdma *rdma)
+{
+ struct ib_device *device = rdma->sc_cm_id->device;
+
+ if (ib_uses_virt_dma(device))
+ return UINT_MAX;
+
+ /* Respect both DMA mapping and RDMA device segment limits. */
+ return min_t(size_t,
+ dma_max_mapping_size(device->dma_device),
+ ib_dma_max_seg_size(device));
+}
+
/*
* Default connection parameters
*/
diff --git a/net/sunrpc/xprtrdma/svc_rdma_sendto.c b/net/sunrpc/xprtrdma/svc_rdma_sendto.c
index efa352ddf9d71..9f2e3ee3c2936 100644
--- a/net/sunrpc/xprtrdma/svc_rdma_sendto.c
+++ b/net/sunrpc/xprtrdma/svc_rdma_sendto.c
@@ -260,10 +260,8 @@ static void svc_rdma_send_ctxt_unmap(struct svcxprt_rdma *rdma,
trace_svcrdma_dma_unmap_page(&ctxt->sc_cid,
ctxt->sc_sges[i].addr,
ctxt->sc_sges[i].length);
- ib_dma_unmap_page(device,
- ctxt->sc_sges[i].addr,
- ctxt->sc_sges[i].length,
- DMA_TO_DEVICE);
+ ib_dma_unmap_bvec(device, ctxt->sc_sges[i].addr,
+ ctxt->sc_sges[i].length, DMA_TO_DEVICE);
}
}
@@ -730,44 +728,45 @@ svc_rdma_encode_reply_chunk(struct svc_rdma_recv_ctxt *rctxt,
}
struct svc_rdma_map_data {
- struct svcxprt_rdma *md_rdma;
struct svc_rdma_send_ctxt *md_ctxt;
+ unsigned int md_max_bvec_len;
};
/**
- * svc_rdma_page_dma_map - DMA map one page
+ * svc_rdma_bvec_dma_map - DMA map one physically contiguous bvec
* @data: pointer to arguments
- * @page: struct page to DMA map
- * @offset: offset into the page
- * @len: number of bytes to map
+ * @bv: contiguous range to DMA map
*
* Returns:
* %0 if DMA mapping was successful
- * %-EIO if the page cannot be DMA mapped
+ * %-EIO if the range cannot be DMA mapped
*/
-static int svc_rdma_page_dma_map(void *data, struct page *page,
- unsigned long offset, unsigned int len)
+static int svc_rdma_bvec_dma_map(void *data, struct bio_vec *bv)
{
struct svc_rdma_map_data *args = data;
- struct svcxprt_rdma *rdma = args->md_rdma;
struct svc_rdma_send_ctxt *ctxt = args->md_ctxt;
+ struct svcxprt_rdma *rdma = ctxt->sc_rdma;
struct ib_device *dev = rdma->sc_cm_id->device;
dma_addr_t dma_addr;
+ if (bv->bv_len > args->md_max_bvec_len)
+ return -EIO;
+ if (WARN_ON_ONCE(ctxt->sc_send_wr.num_sge >= rdma->sc_max_send_sges))
+ return -EIO;
++ctxt->sc_cur_sge_no;
- dma_addr = ib_dma_map_page(dev, page, offset, len, DMA_TO_DEVICE);
+ dma_addr = ib_dma_map_bvec(dev, bv, DMA_TO_DEVICE);
if (ib_dma_mapping_error(dev, dma_addr))
goto out_maperr;
- trace_svcrdma_dma_map_page(&ctxt->sc_cid, dma_addr, len);
+ trace_svcrdma_dma_map_page(&ctxt->sc_cid, dma_addr, bv->bv_len);
ctxt->sc_sges[ctxt->sc_cur_sge_no].addr = dma_addr;
- ctxt->sc_sges[ctxt->sc_cur_sge_no].length = len;
+ ctxt->sc_sges[ctxt->sc_cur_sge_no].length = bv->bv_len;
ctxt->sc_send_wr.num_sge++;
return 0;
out_maperr:
- trace_svcrdma_dma_map_err(&ctxt->sc_cid, dma_addr, len);
+ trace_svcrdma_dma_map_err(&ctxt->sc_cid, dma_addr, bv->bv_len);
return -EIO;
}
@@ -776,20 +775,18 @@ static int svc_rdma_page_dma_map(void *data, struct page *page,
* @data: pointer to arguments
* @iov: kvec to DMA map
*
- * ib_dma_map_page() is used here because svc_rdma_dma_unmap()
- * handles DMA-unmap and it uses ib_dma_unmap_page() exclusively.
- *
* Returns:
* %0 if DMA mapping was successful
* %-EIO if the iovec cannot be DMA mapped
*/
static int svc_rdma_iov_dma_map(void *data, const struct kvec *iov)
{
+ struct bio_vec bv;
+
if (!iov->iov_len)
return 0;
- return svc_rdma_page_dma_map(data, virt_to_page(iov->iov_base),
- offset_in_page(iov->iov_base),
- iov->iov_len);
+ bvec_set_virt(&bv, iov->iov_base, iov->iov_len);
+ return svc_rdma_bvec_dma_map(data, &bv);
}
/**
@@ -798,7 +795,7 @@ static int svc_rdma_iov_dma_map(void *data, const struct kvec *iov)
* @data: pointer to arguments
*
* Returns:
- * %0 if DMA mapping was successful
+ * The number of mapped bytes if DMA mapping was successful
* %-EIO if DMA mapping failed
*
* On failure, any DMA mappings that have been already done must be
@@ -806,9 +803,11 @@ static int svc_rdma_iov_dma_map(void *data, const struct kvec *iov)
*/
static int svc_rdma_xb_dma_map(const struct xdr_buf *xdr, void *data)
{
- unsigned int len, remaining;
+ struct svc_rdma_map_data *args = data;
+ unsigned int remaining;
unsigned long pageoff;
struct page **ppages;
+ unsigned int nr_pages;
int ret;
ret = svc_rdma_iov_dma_map(data, &xdr->head[0]);
@@ -818,15 +817,25 @@ static int svc_rdma_xb_dma_map(const struct xdr_buf *xdr, void *data)
ppages = xdr->pages + (xdr->page_base >> PAGE_SHIFT);
pageoff = offset_in_page(xdr->page_base);
remaining = xdr->page_len;
+ nr_pages = DIV_ROUND_UP(pageoff + remaining, PAGE_SIZE);
while (remaining) {
- len = min_t(u32, PAGE_SIZE - pageoff, remaining);
-
- ret = svc_rdma_page_dma_map(data, *ppages++, pageoff, len);
+ struct bio_vec bv;
+ unsigned int bytes = svc_pages_to_bvec(&bv, ppages,
+ nr_pages, pageoff, remaining,
+ args->md_max_bvec_len);
+ unsigned int advanced =
+ (pageoff + bytes) >> PAGE_SHIFT;
+
+ if (!bytes)
+ return -EIO;
+ ret = svc_rdma_bvec_dma_map(data, &bv);
if (ret < 0)
return ret;
- remaining -= len;
- pageoff = 0;
+ pageoff = offset_in_page(pageoff + bytes);
+ ppages += advanced;
+ nr_pages -= advanced;
+ remaining -= bytes;
}
ret = svc_rdma_iov_dma_map(data, &xdr->tail[0]);
@@ -836,10 +845,27 @@ static int svc_rdma_xb_dma_map(const struct xdr_buf *xdr, void *data)
return xdr->len;
}
+static unsigned int
+svc_rdma_xb_page_sges(const struct xdr_buf *xdr,
+ unsigned int max_bvec_len)
+{
+ const unsigned long offset = offset_in_page(xdr->page_base);
+ const unsigned int nr_pages =
+ DIV_ROUND_UP(offset + xdr->page_len, PAGE_SIZE);
+ struct page *const *ppages;
+
+ if (!xdr->page_len)
+ return 0;
+ ppages = xdr->pages + (xdr->page_base >> PAGE_SHIFT);
+ return svc_pages_to_bvecs(NULL, ppages, nr_pages, offset,
+ xdr->page_len, max_bvec_len);
+}
+
struct svc_rdma_pullup_data {
u8 *pd_dest;
unsigned int pd_length;
unsigned int pd_num_sges;
+ unsigned int pd_max_bvec_len;
};
/**
@@ -848,25 +874,22 @@ struct svc_rdma_pullup_data {
* @data: pointer to arguments
*
* Returns:
- * Number of SGEs needed to Send the contents of @xdr inline
+ * %0 if the SGE count was updated
+ * %-EIO if the page list cannot be described
*/
static int svc_rdma_xb_count_sges(const struct xdr_buf *xdr,
void *data)
{
struct svc_rdma_pullup_data *args = data;
- unsigned int remaining;
- unsigned long offset;
+ unsigned int page_sges =
+ svc_rdma_xb_page_sges(xdr, args->pd_max_bvec_len);
if (xdr->head[0].iov_len)
++args->pd_num_sges;
- offset = offset_in_page(xdr->page_base);
- remaining = xdr->page_len;
- while (remaining) {
- ++args->pd_num_sges;
- remaining -= min_t(u32, PAGE_SIZE - offset, remaining);
- offset = 0;
- }
+ if (xdr->page_len && !page_sges)
+ return -EIO;
+ args->pd_num_sges += page_sges;
if (xdr->tail[0].iov_len)
++args->pd_num_sges;
@@ -896,6 +919,7 @@ static int svc_rdma_check_pull_up(const struct svcxprt_rdma *rdma,
struct svc_rdma_pullup_data args = {
.pd_length = sctxt->sc_hdrbuf.len,
.pd_num_sges = 1,
+ .pd_max_bvec_len = svc_rdma_max_bvec_len(rdma),
};
int ret;
@@ -964,7 +988,6 @@ static int svc_rdma_xb_linearize(const struct xdr_buf *xdr,
/**
* svc_rdma_pull_up_reply_msg - Copy Reply into a single buffer
- * @rdma: controlling transport
* @sctxt: send_ctxt for the Send WR; xprt hdr is already prepared
* @write_pcl: Write chunk list provided by client
* @xdr: prepared xdr_buf containing RPC message
@@ -979,8 +1002,7 @@ static int svc_rdma_xb_linearize(const struct xdr_buf *xdr,
* %0 if pull-up was successful
* %-EMSGSIZE if a buffer manipulation problem occurred
*/
-static int svc_rdma_pull_up_reply_msg(const struct svcxprt_rdma *rdma,
- struct svc_rdma_send_ctxt *sctxt,
+static int svc_rdma_pull_up_reply_msg(struct svc_rdma_send_ctxt *sctxt,
const struct svc_rdma_pcl *write_pcl,
const struct xdr_buf *xdr)
{
@@ -1021,8 +1043,8 @@ int svc_rdma_map_reply_msg(struct svcxprt_rdma *rdma,
const struct xdr_buf *xdr)
{
struct svc_rdma_map_data args = {
- .md_rdma = rdma,
.md_ctxt = sctxt,
+ .md_max_bvec_len = svc_rdma_max_bvec_len(rdma),
};
int ret;
@@ -1043,7 +1065,7 @@ int svc_rdma_map_reply_msg(struct svcxprt_rdma *rdma,
if (ret < 0)
return ret;
if (ret)
- return svc_rdma_pull_up_reply_msg(rdma, sctxt, write_pcl, xdr);
+ return svc_rdma_pull_up_reply_msg(sctxt, write_pcl, xdr);
return pcl_process_nonpayloads(write_pcl, xdr,
svc_rdma_xb_dma_map, &args);
--
2.43.0
next prev parent reply other threads:[~2026-10-06 9:01 UTC|newest]
Thread overview: 15+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-10-06 9:00 [RFC PATCH 0/6] Improve NFS server direct throughput with passthrough enabled J Louis Kaplan
2026-10-06 9:00 ` [RFC PATCH 1/6] sunrpc: Add helpers to build bvecs from contiguous pages J Louis Kaplan
2026-10-06 13:49 ` Chuck Lever
2026-10-08 22:44 ` J Louis Kaplan
2026-10-06 9:00 ` [RFC PATCH 2/6] nfsd: Coalesce contiguous pages for direct reads J Louis Kaplan
2026-10-06 9:00 ` J Louis Kaplan [this message]
2026-10-06 13:55 ` [RFC PATCH 3/6] svcrdma: Coalesce contiguous pages when mapping replies Chuck Lever
2026-10-08 22:54 ` J Louis Kaplan
2026-10-06 9:00 ` [RFC PATCH 4/6] svcrdma: Coalesce contiguous pages in RDMA Write chunks J Louis Kaplan
2026-10-06 9:00 ` [RFC PATCH 5/6] svcrdma: Coalesce contiguous pages in RDMA Read chunks J Louis Kaplan
2026-10-06 9:00 ` [RFC PATCH 6/6] sunrpc: Allocate svc request pages from large folios J Louis Kaplan
2026-10-06 14:00 ` Chuck Lever
2026-10-08 22:58 ` J Louis Kaplan
2026-10-06 13:47 ` [RFC PATCH 0/6] Improve NFS server direct throughput with passthrough enabled Chuck Lever
2026-10-08 22:50 ` J Louis Kaplan
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20261006090028.3412544-4-Louis.Kaplan@arm.com \
--to=louis.kaplan@arm.com \
--cc=anna@kernel.org \
--cc=cel@kernel.org \
--cc=dai.ngo@oracle.com \
--cc=jgg@ziepe.ca \
--cc=jlayton@kernel.org \
--cc=leon@kernel.org \
--cc=linux-nfs@vger.kernel.org \
--cc=linux-rdma@vger.kernel.org \
--cc=neil@brown.name \
--cc=okorniev@redhat.com \
--cc=tom@talpey.com \
--cc=trondmy@kernel.org \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox