From: "Chuck Lever" <cel@kernel.org>
To: "J Louis Kaplan" <Louis.Kaplan@arm.com>,
"Dai Ngo" <dai.ngo@oracle.com>,
"Jeff Layton" <jlayton@kernel.org>, NeilBrown <neil@brown.name>,
"Olga Kornievskaia" <okorniev@redhat.com>,
"Tom Talpey" <tom@talpey.com>
Cc: linux-nfs@vger.kernel.org, linux-rdma@vger.kernel.org,
"Anna Schumaker" <anna@kernel.org>,
"Jason Gunthorpe" <jgg@ziepe.ca>,
"Leon Romanovsky" <leon@kernel.org>,
"Trond Myklebust" <trondmy@kernel.org>
Subject: Re: [RFC PATCH 3/6] svcrdma: Coalesce contiguous pages when mapping replies
Date: Tue, 06 Oct 2026 09:55:23 -0400 [thread overview]
Message-ID: <59261d53-53fc-4195-a7f4-d9f117bb0150@app.fastmail.com> (raw)
In-Reply-To: <20261006090028.3412544-4-Louis.Kaplan@arm.com>
On Tue, Oct 6, 2026, at 5:00 AM, J Louis Kaplan wrote:
> Mapping reply pagelists one page at a time uses a separate direct
> memory access (DMA) mapping and scatter/gather entry (SGE) for
> each base page, even when those pages are physically contiguous.
>
> Instead, map contiguous runs as multipage bvecs, capped by the device's
> mapping and segment length limits. Use the same coalescing rules
> when counting SGEs for the pull-up decision, avoiding unnecessary
> copying when the coalesced reply fits the Send SGE limit.
>
> Virtual-mapping devices bypass device-limit check.
>
> Signed-off-by: J Louis Kaplan <Louis.Kaplan@arm.com>
> Assisted-by: LLM
> ---
> include/linux/sunrpc/svc_rdma.h | 15 ++++
> net/sunrpc/xprtrdma/svc_rdma_sendto.c | 112 +++++++++++++++-----------
> 2 files changed, 82 insertions(+), 45 deletions(-)
>
> diff --git a/include/linux/sunrpc/svc_rdma.h b/include/linux/sunrpc/svc_rdma.h
> index 71bd5bcc5ce2a..c296bf8b8b1e3 100644
> --- a/include/linux/sunrpc/svc_rdma.h
> +++ b/include/linux/sunrpc/svc_rdma.h
> @@ -42,6 +42,7 @@
>
> #ifndef SVC_RDMA_H
> #define SVC_RDMA_H
> +#include <linux/dma-mapping.h>
> #include <linux/llist.h>
> #include <linux/sunrpc/xdr.h>
> #include <linux/sunrpc/svcsock.h>
> @@ -134,6 +135,20 @@ static inline struct svcxprt_rdma
> *svc_rdma_rqst_rdma(struct svc_rqst *rqstp)
> return container_of(xprt, struct svcxprt_rdma, sc_xprt);
> }
>
> +static inline unsigned int
> +svc_rdma_max_bvec_len(const struct svcxprt_rdma *rdma)
There are some interesting infrastructural changes in this series.
Using bvecs through the svcrdma code has been on my to-do list
for quite some time.
> +{
> + struct ib_device *device = rdma->sc_cm_id->device;
> +
> + if (ib_uses_virt_dma(device))
> + return UINT_MAX;
> +
> + /* Respect both DMA mapping and RDMA device segment limits. */
> + return min_t(size_t,
> + dma_max_mapping_size(device->dma_device),
> + ib_dma_max_seg_size(device));
> +}
> +
> /*
> * Default connection parameters
> */
> diff --git a/net/sunrpc/xprtrdma/svc_rdma_sendto.c
> b/net/sunrpc/xprtrdma/svc_rdma_sendto.c
> index efa352ddf9d71..9f2e3ee3c2936 100644
> --- a/net/sunrpc/xprtrdma/svc_rdma_sendto.c
> +++ b/net/sunrpc/xprtrdma/svc_rdma_sendto.c
> @@ -260,10 +260,8 @@ static void svc_rdma_send_ctxt_unmap(struct
> svcxprt_rdma *rdma,
> trace_svcrdma_dma_unmap_page(&ctxt->sc_cid,
> ctxt->sc_sges[i].addr,
> ctxt->sc_sges[i].length);
> - ib_dma_unmap_page(device,
> - ctxt->sc_sges[i].addr,
> - ctxt->sc_sges[i].length,
> - DMA_TO_DEVICE);
> + ib_dma_unmap_bvec(device, ctxt->sc_sges[i].addr,
> + ctxt->sc_sges[i].length, DMA_TO_DEVICE);
> }
> }
>
> @@ -730,44 +728,45 @@ svc_rdma_encode_reply_chunk(struct
> svc_rdma_recv_ctxt *rctxt,
> }
>
> struct svc_rdma_map_data {
> - struct svcxprt_rdma *md_rdma;
> struct svc_rdma_send_ctxt *md_ctxt;
> + unsigned int md_max_bvec_len;
> };
>
> /**
> - * svc_rdma_page_dma_map - DMA map one page
> + * svc_rdma_bvec_dma_map - DMA map one physically contiguous bvec
> * @data: pointer to arguments
> - * @page: struct page to DMA map
> - * @offset: offset into the page
> - * @len: number of bytes to map
> + * @bv: contiguous range to DMA map
> *
> * Returns:
> * %0 if DMA mapping was successful
> - * %-EIO if the page cannot be DMA mapped
> + * %-EIO if the range cannot be DMA mapped
> */
> -static int svc_rdma_page_dma_map(void *data, struct page *page,
> - unsigned long offset, unsigned int len)
> +static int svc_rdma_bvec_dma_map(void *data, struct bio_vec *bv)
> {
> struct svc_rdma_map_data *args = data;
> - struct svcxprt_rdma *rdma = args->md_rdma;
> struct svc_rdma_send_ctxt *ctxt = args->md_ctxt;
> + struct svcxprt_rdma *rdma = ctxt->sc_rdma;
> struct ib_device *dev = rdma->sc_cm_id->device;
> dma_addr_t dma_addr;
>
> + if (bv->bv_len > args->md_max_bvec_len)
> + return -EIO;
> + if (WARN_ON_ONCE(ctxt->sc_send_wr.num_sge >= rdma->sc_max_send_sges))
> + return -EIO;
> ++ctxt->sc_cur_sge_no;
>
> - dma_addr = ib_dma_map_page(dev, page, offset, len, DMA_TO_DEVICE);
> + dma_addr = ib_dma_map_bvec(dev, bv, DMA_TO_DEVICE);
> if (ib_dma_mapping_error(dev, dma_addr))
> goto out_maperr;
>
> - trace_svcrdma_dma_map_page(&ctxt->sc_cid, dma_addr, len);
> + trace_svcrdma_dma_map_page(&ctxt->sc_cid, dma_addr, bv->bv_len);
> ctxt->sc_sges[ctxt->sc_cur_sge_no].addr = dma_addr;
> - ctxt->sc_sges[ctxt->sc_cur_sge_no].length = len;
> + ctxt->sc_sges[ctxt->sc_cur_sge_no].length = bv->bv_len;
> ctxt->sc_send_wr.num_sge++;
> return 0;
>
> out_maperr:
> - trace_svcrdma_dma_map_err(&ctxt->sc_cid, dma_addr, len);
> + trace_svcrdma_dma_map_err(&ctxt->sc_cid, dma_addr, bv->bv_len);
> return -EIO;
> }
>
> @@ -776,20 +775,18 @@ static int svc_rdma_page_dma_map(void *data,
> struct page *page,
> * @data: pointer to arguments
> * @iov: kvec to DMA map
> *
> - * ib_dma_map_page() is used here because svc_rdma_dma_unmap()
> - * handles DMA-unmap and it uses ib_dma_unmap_page() exclusively.
> - *
> * Returns:
> * %0 if DMA mapping was successful
> * %-EIO if the iovec cannot be DMA mapped
> */
> static int svc_rdma_iov_dma_map(void *data, const struct kvec *iov)
> {
> + struct bio_vec bv;
> +
> if (!iov->iov_len)
> return 0;
> - return svc_rdma_page_dma_map(data, virt_to_page(iov->iov_base),
> - offset_in_page(iov->iov_base),
> - iov->iov_len);
> + bvec_set_virt(&bv, iov->iov_base, iov->iov_len);
> + return svc_rdma_bvec_dma_map(data, &bv);
> }
>
> /**
> @@ -798,7 +795,7 @@ static int svc_rdma_iov_dma_map(void *data, const
> struct kvec *iov)
> * @data: pointer to arguments
> *
> * Returns:
> - * %0 if DMA mapping was successful
> + * The number of mapped bytes if DMA mapping was successful
> * %-EIO if DMA mapping failed
> *
> * On failure, any DMA mappings that have been already done must be
> @@ -806,9 +803,11 @@ static int svc_rdma_iov_dma_map(void *data, const
> struct kvec *iov)
> */
> static int svc_rdma_xb_dma_map(const struct xdr_buf *xdr, void *data)
> {
> - unsigned int len, remaining;
> + struct svc_rdma_map_data *args = data;
> + unsigned int remaining;
> unsigned long pageoff;
> struct page **ppages;
> + unsigned int nr_pages;
> int ret;
>
> ret = svc_rdma_iov_dma_map(data, &xdr->head[0]);
> @@ -818,15 +817,25 @@ static int svc_rdma_xb_dma_map(const struct
> xdr_buf *xdr, void *data)
> ppages = xdr->pages + (xdr->page_base >> PAGE_SHIFT);
> pageoff = offset_in_page(xdr->page_base);
> remaining = xdr->page_len;
> + nr_pages = DIV_ROUND_UP(pageoff + remaining, PAGE_SIZE);
> while (remaining) {
> - len = min_t(u32, PAGE_SIZE - pageoff, remaining);
> -
> - ret = svc_rdma_page_dma_map(data, *ppages++, pageoff, len);
> + struct bio_vec bv;
> + unsigned int bytes = svc_pages_to_bvec(&bv, ppages,
> + nr_pages, pageoff, remaining,
> + args->md_max_bvec_len);
> + unsigned int advanced =
> + (pageoff + bytes) >> PAGE_SHIFT;
> +
> + if (!bytes)
> + return -EIO;
> + ret = svc_rdma_bvec_dma_map(data, &bv);
> if (ret < 0)
> return ret;
>
> - remaining -= len;
> - pageoff = 0;
> + pageoff = offset_in_page(pageoff + bytes);
> + ppages += advanced;
> + nr_pages -= advanced;
> + remaining -= bytes;
> }
>
> ret = svc_rdma_iov_dma_map(data, &xdr->tail[0]);
> @@ -836,10 +845,27 @@ static int svc_rdma_xb_dma_map(const struct
> xdr_buf *xdr, void *data)
> return xdr->len;
> }
>
> +static unsigned int
> +svc_rdma_xb_page_sges(const struct xdr_buf *xdr,
> + unsigned int max_bvec_len)
> +{
> + const unsigned long offset = offset_in_page(xdr->page_base);
> + const unsigned int nr_pages =
> + DIV_ROUND_UP(offset + xdr->page_len, PAGE_SIZE);
> + struct page *const *ppages;
> +
> + if (!xdr->page_len)
> + return 0;
> + ppages = xdr->pages + (xdr->page_base >> PAGE_SHIFT);
> + return svc_pages_to_bvecs(NULL, ppages, nr_pages, offset,
> + xdr->page_len, max_bvec_len);
> +}
> +
> struct svc_rdma_pullup_data {
> u8 *pd_dest;
> unsigned int pd_length;
> unsigned int pd_num_sges;
> + unsigned int pd_max_bvec_len;
> };
>
> /**
> @@ -848,25 +874,22 @@ struct svc_rdma_pullup_data {
> * @data: pointer to arguments
> *
> * Returns:
> - * Number of SGEs needed to Send the contents of @xdr inline
> + * %0 if the SGE count was updated
> + * %-EIO if the page list cannot be described
> */
> static int svc_rdma_xb_count_sges(const struct xdr_buf *xdr,
> void *data)
> {
> struct svc_rdma_pullup_data *args = data;
> - unsigned int remaining;
> - unsigned long offset;
> + unsigned int page_sges =
> + svc_rdma_xb_page_sges(xdr, args->pd_max_bvec_len);
>
> if (xdr->head[0].iov_len)
> ++args->pd_num_sges;
>
> - offset = offset_in_page(xdr->page_base);
> - remaining = xdr->page_len;
> - while (remaining) {
> - ++args->pd_num_sges;
> - remaining -= min_t(u32, PAGE_SIZE - offset, remaining);
> - offset = 0;
> - }
> + if (xdr->page_len && !page_sges)
> + return -EIO;
> + args->pd_num_sges += page_sges;
>
> if (xdr->tail[0].iov_len)
> ++args->pd_num_sges;
> @@ -896,6 +919,7 @@ static int svc_rdma_check_pull_up(const struct
> svcxprt_rdma *rdma,
> struct svc_rdma_pullup_data args = {
> .pd_length = sctxt->sc_hdrbuf.len,
> .pd_num_sges = 1,
> + .pd_max_bvec_len = svc_rdma_max_bvec_len(rdma),
> };
> int ret;
>
> @@ -964,7 +988,6 @@ static int svc_rdma_xb_linearize(const struct xdr_buf *xdr,
>
> /**
> * svc_rdma_pull_up_reply_msg - Copy Reply into a single buffer
> - * @rdma: controlling transport
> * @sctxt: send_ctxt for the Send WR; xprt hdr is already prepared
> * @write_pcl: Write chunk list provided by client
> * @xdr: prepared xdr_buf containing RPC message
> @@ -979,8 +1002,7 @@ static int svc_rdma_xb_linearize(const struct xdr_buf *xdr,
> * %0 if pull-up was successful
> * %-EMSGSIZE if a buffer manipulation problem occurred
> */
> -static int svc_rdma_pull_up_reply_msg(const struct svcxprt_rdma *rdma,
> - struct svc_rdma_send_ctxt *sctxt,
> +static int svc_rdma_pull_up_reply_msg(struct svc_rdma_send_ctxt *sctxt,
> const struct svc_rdma_pcl *write_pcl,
> const struct xdr_buf *xdr)
> {
> @@ -1021,8 +1043,8 @@ int svc_rdma_map_reply_msg(struct svcxprt_rdma *rdma,
> const struct xdr_buf *xdr)
> {
> struct svc_rdma_map_data args = {
> - .md_rdma = rdma,
> .md_ctxt = sctxt,
> + .md_max_bvec_len = svc_rdma_max_bvec_len(rdma),
> };
> int ret;
>
> @@ -1043,7 +1065,7 @@ int svc_rdma_map_reply_msg(struct svcxprt_rdma *rdma,
> if (ret < 0)
> return ret;
> if (ret)
> - return svc_rdma_pull_up_reply_msg(rdma, sctxt, write_pcl, xdr);
> + return svc_rdma_pull_up_reply_msg(sctxt, write_pcl, xdr);
>
> return pcl_process_nonpayloads(write_pcl, xdr,
> svc_rdma_xb_dma_map, &args);
> --
> 2.43.0
--
Chuck Lever (Come to NFS bake-a-thon! https://nfsv4bat.org)
next prev parent reply other threads:[~2026-10-06 13:55 UTC|newest]
Thread overview: 16+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-10-06 9:00 [RFC PATCH 0/6] Improve NFS server direct throughput with passthrough enabled J Louis Kaplan
2026-10-06 9:00 ` [RFC PATCH 1/6] sunrpc: Add helpers to build bvecs from contiguous pages J Louis Kaplan
2026-10-06 13:49 ` Chuck Lever
2026-10-08 22:44 ` J Louis Kaplan
2026-10-06 9:00 ` [RFC PATCH 2/6] nfsd: Coalesce contiguous pages for direct reads J Louis Kaplan
2026-10-06 9:00 ` [RFC PATCH 3/6] svcrdma: Coalesce contiguous pages when mapping replies J Louis Kaplan
2026-10-06 13:55 ` Chuck Lever [this message]
2026-10-08 22:54 ` J Louis Kaplan
2026-10-06 9:00 ` [RFC PATCH 4/6] svcrdma: Coalesce contiguous pages in RDMA Write chunks J Louis Kaplan
2026-10-06 9:00 ` [RFC PATCH 5/6] svcrdma: Coalesce contiguous pages in RDMA Read chunks J Louis Kaplan
2026-10-06 9:00 ` [RFC PATCH 6/6] sunrpc: Allocate svc request pages from large folios J Louis Kaplan
2026-10-06 14:00 ` Chuck Lever
2026-10-08 22:58 ` J Louis Kaplan
2026-10-09 15:28 ` Chuck Lever
2026-10-06 13:47 ` [RFC PATCH 0/6] Improve NFS server direct throughput with passthrough enabled Chuck Lever
2026-10-08 22:50 ` J Louis Kaplan
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=59261d53-53fc-4195-a7f4-d9f117bb0150@app.fastmail.com \
--to=cel@kernel.org \
--cc=Louis.Kaplan@arm.com \
--cc=anna@kernel.org \
--cc=dai.ngo@oracle.com \
--cc=jgg@ziepe.ca \
--cc=jlayton@kernel.org \
--cc=leon@kernel.org \
--cc=linux-nfs@vger.kernel.org \
--cc=linux-rdma@vger.kernel.org \
--cc=neil@brown.name \
--cc=okorniev@redhat.com \
--cc=tom@talpey.com \
--cc=trondmy@kernel.org \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox