From: Lorenzo Bianconi <lorenzo.bianconi@oss.qualcomm.com>
To: Maxime Chevallier <maxime.chevallier@bootlin.com>,
Andrew Lunn <andrew+netdev@lunn.ch>,
"David S. Miller" <davem@davemloft.net>,
Eric Dumazet <edumazet@google.com>,
Jakub Kicinski <kuba@kernel.org>, Paolo Abeni <pabeni@redhat.com>,
Maxime Coquelin <mcoquelin.stm32@gmail.com>,
Alexandre Torgue <alexandre.torgue@foss.st.com>,
Alexei Starovoitov <ast@kernel.org>,
Daniel Borkmann <daniel@iogearbox.net>,
Jesper Dangaard Brouer <hawk@kernel.org>,
John Fastabend <john.fastabend@gmail.com>,
Stanislav Fomichev <sdf@fomichev.me>
Cc: netdev@vger.kernel.org, linux-stm32@st-md-mailman.stormreply.com,
linux-arm-kernel@lists.infradead.org, bpf@vger.kernel.org,
Lorenzo Bianconi <lorenzo.bianconi@oss.qualcomm.com>
Subject: [PATCH net-next v2] net: stmmac: rework stmmac_rx to support XDP rx multi-buff
Date: Mon, 21 Sep 2026 10:58:26 +0200 [thread overview]
Message-ID: <20260921-stmmac-rx-mb-v2-1-6e826e1ff306@oss.qualcomm.com> (raw)
Build the xdp_buff by accumulating all the descriptors that make up a
frame, so the XDP program runs on the full (possibly fragmented) packet
instead of just the first buffer. When the frame is not consumed by the
program, assemble the skb from the head buffer and the collected
fragments via napi_build_skb()/xdp_update_skb_frags_info().
To do so, store the in-progress xdp_buff in rx_q->state instead of the
partially built skb, so the accumulated head and fragments survive a
NAPI poll boundary (mid-frame dma_own or dirty_rx break). The state is
saved only while a frame is in progress and cleared once it completes,
leaving it untouched when the poll does not process anything (e.g.
netpoll invoked with a zero budget).
In addition:
- build the skb head with napi_build_skb() passing xdp->frame_sz, so
skb_shinfo() lands on the same shared_info the fragments were
accumulated into;
- attach fragments with xdp_buff_add_frag(), which initializes all the
shared_info fields and takes care of the pfmemalloc bit;
- release all buffers belonging to a frame when it is dropped on RX
errors, instead of leaking the ones already attached to the xdp_buff;
- drop the whole frame when the number of fragments exceeds
MAX_SKB_FRAGS, instead of delivering a truncated one;
- skip zero-length fragments, which can happen for non-first
descriptors when split-header (SPH) is enabled.
Signed-off-by: Lorenzo Bianconi <lorenzo.bianconi@oss.qualcomm.com>
---
Changes in v2:
- Rely on xdp_buff_add_frag() to create xdp fragments
- Fix bugs in rx_q state recording
- Fix the corner case where we receive more than MAX_SKB_FRAGS fragments
- Cosmetics
- Link to v1: https://lore.kernel.org/r/20260918-stmmac-rx-mb-v1-1-0b4517d404af@oss.qualcomm.com
---
drivers/net/ethernet/stmicro/stmmac/stmmac.h | 2 +-
drivers/net/ethernet/stmicro/stmmac/stmmac_main.c | 202 ++++++++++++++--------
2 files changed, 131 insertions(+), 73 deletions(-)
diff --git a/drivers/net/ethernet/stmicro/stmmac/stmmac.h b/drivers/net/ethernet/stmicro/stmmac/stmmac.h
index 4fc96b317d79..69bdbbf4b920 100644
--- a/drivers/net/ethernet/stmicro/stmmac/stmmac.h
+++ b/drivers/net/ethernet/stmicro/stmmac/stmmac.h
@@ -132,7 +132,7 @@ struct stmmac_rx_queue {
dma_addr_t dma_rx_phy;
unsigned int state_saved;
struct {
- struct sk_buff *skb;
+ struct xdp_buff xdp;
unsigned int len;
unsigned int error;
} state;
diff --git a/drivers/net/ethernet/stmicro/stmmac/stmmac_main.c b/drivers/net/ethernet/stmicro/stmmac/stmmac_main.c
index bf9e7e4cb1c3..7d1149511c55 100644
--- a/drivers/net/ethernet/stmicro/stmmac/stmmac_main.c
+++ b/drivers/net/ethernet/stmicro/stmmac/stmmac_main.c
@@ -1695,7 +1695,8 @@ static int stmmac_init_rx_buffers(struct stmmac_priv *priv,
if (!buf->sec_page)
return -ENOMEM;
- buf->sec_addr = page_pool_get_dma_addr(buf->sec_page);
+ buf->sec_addr = page_pool_get_dma_addr(buf->sec_page) +
+ buf->page_offset;
stmmac_set_desc_sec_addr(priv, p, buf->sec_addr, true);
} else {
buf->sec_page = NULL;
@@ -5140,7 +5141,8 @@ static inline void stmmac_rx_refill(struct stmmac_priv *priv, u32 queue)
if (!buf->sec_page)
break;
- buf->sec_addr = page_pool_get_dma_addr(buf->sec_page);
+ buf->sec_addr = page_pool_get_dma_addr(buf->sec_page) +
+ buf->page_offset;
}
buf->addr = page_pool_get_dma_addr(buf->page) + buf->page_offset;
@@ -5735,6 +5737,70 @@ static int stmmac_rx_zc(struct stmmac_priv *priv, int limit, u32 queue)
return failure ? limit : (int)count;
}
+static void
+stmmac_xdp_put_buff(struct stmmac_rx_queue *rx_q, struct xdp_buff *xdp,
+ int sync_len)
+{
+ struct skb_shared_info *sinfo = xdp_get_shared_info_from_buff(xdp);
+ int i;
+
+ if (likely(!xdp_buff_has_frags(xdp)))
+ goto out;
+
+ for (i = 0; i < sinfo->nr_frags; i++)
+ page_pool_put_full_page(rx_q->page_pool,
+ skb_frag_page(&sinfo->frags[i]), true);
+out:
+ page_pool_put_page(rx_q->page_pool, virt_to_head_page(xdp->data),
+ sync_len, true);
+}
+
+static struct sk_buff *stmmac_build_skb(struct xdp_buff *xdp)
+{
+ struct skb_shared_info *sinfo = xdp_get_shared_info_from_buff(xdp);
+ u32 metasize = xdp->data - xdp->data_meta;
+ struct sk_buff *skb;
+ u8 num_frags = 0;
+
+ if (unlikely(xdp_buff_has_frags(xdp)))
+ num_frags = sinfo->nr_frags;
+
+ skb = napi_build_skb(xdp->data_hard_start, xdp->frame_sz);
+ if (!skb)
+ return NULL;
+
+ skb_mark_for_recycle(skb);
+ skb_reserve(skb, xdp->data - xdp->data_hard_start);
+ skb_put(skb, xdp->data_end - xdp->data);
+ if (metasize)
+ skb_metadata_set(skb, metasize);
+
+ if (unlikely(xdp_buff_has_frags(xdp)))
+ xdp_update_skb_frags_info(skb, num_frags, sinfo->xdp_frags_size,
+ num_frags * xdp->frame_sz,
+ xdp_buff_get_skb_flags(xdp));
+ return skb;
+}
+
+static bool stmmac_build_xdp_frags(struct stmmac_priv *priv,
+ struct stmmac_rx_queue *rx_q,
+ unsigned int len, struct page *page,
+ unsigned int offset,
+ enum dma_data_direction dma_dir,
+ struct xdp_buff *xdp)
+{
+ dma_addr_t dma_addr = page_pool_get_dma_addr(page) + offset;
+
+ dma_sync_single_for_cpu(priv->device, dma_addr, len, dma_dir);
+ if (!xdp_buff_add_frag(xdp, page_to_netmem(page), offset, len,
+ xdp->frame_sz)) {
+ page_pool_put_full_page(rx_q->page_pool, page, true);
+ return false;
+ }
+
+ return true;
+}
+
/**
* stmmac_rx - manage the receive process
* @priv: driver private structure
@@ -5756,11 +5822,12 @@ static int stmmac_rx(struct stmmac_priv *priv, int limit, u32 queue)
unsigned int desc_size;
struct sk_buff *skb = NULL;
struct stmmac_xdp_buff ctx;
+ bool first_desc = true;
int xdp_status = 0;
int bufsz;
dma_dir = page_pool_get_dma_dir(rx_q->page_pool);
- bufsz = DIV_ROUND_UP(priv->dma_conf.dma_buf_sz, PAGE_SIZE) * PAGE_SIZE;
+ bufsz = rx_q->napi_skb_frag_size;
if (netif_msg_rx_status(priv)) {
void *rx_head = stmmac_get_rx_desc(priv, rx_q, 0);
@@ -5780,9 +5847,10 @@ static int stmmac_rx(struct stmmac_priv *priv, int limit, u32 queue)
u32 hash;
if (!count && rx_q->state_saved) {
- skb = rx_q->state.skb;
+ ctx.xdp = rx_q->state.xdp;
error = rx_q->state.error;
len = rx_q->state.len;
+ first_desc = false;
} else {
rx_q->state_saved = false;
skb = NULL;
@@ -5820,21 +5888,21 @@ static int stmmac_rx(struct stmmac_priv *priv, int limit, u32 queue)
if (priv->extend_desc)
stmmac_rx_extended_status(priv, &priv->xstats, rx_q->dma_erx + entry);
+
if (unlikely(status == discard_frame)) {
- page_pool_put_page(rx_q->page_pool, buf->page, 0, true);
- buf->page = NULL;
error = 1;
if (!priv->hwts_rx_en)
rx_errors++;
}
- if (unlikely(error && (status & rx_not_ls)))
- goto read_again;
if (unlikely(error)) {
- dev_kfree_skb(skb);
- skb = NULL;
- count++;
- continue;
+ page_pool_put_page(rx_q->page_pool, buf->page, 0, true);
+ buf->page = NULL;
+
+ if (status & rx_not_ls)
+ goto read_again;
+
+ goto error_free_frag;
}
/* Buffer is good. Go on. */
@@ -5855,9 +5923,7 @@ static int stmmac_rx(struct stmmac_priv *priv, int limit, u32 queue)
}
}
- if (!skb) {
- unsigned int pre_len, sync_len;
-
+ if (first_desc) {
dma_sync_single_for_cpu(priv->device, buf->addr,
buf1_len, dma_dir);
net_prefetch(page_address(buf->page) +
@@ -5866,6 +5932,33 @@ static int stmmac_rx(struct stmmac_priv *priv, int limit, u32 queue)
xdp_init_buff(&ctx.xdp, bufsz, &rx_q->xdp_rxq);
xdp_prepare_buff(&ctx.xdp, page_address(buf->page),
buf->page_offset, buf1_len, true);
+ first_desc = false;
+ buf->page = NULL;
+ } else if (buf1_len) {
+ error |= !stmmac_build_xdp_frags(priv, rx_q, buf1_len,
+ buf->page,
+ buf->page_offset,
+ dma_dir, &ctx.xdp);
+ buf->page = NULL;
+ }
+
+ if (buf2_len) {
+ error |= !stmmac_build_xdp_frags(priv, rx_q, buf2_len,
+ buf->sec_page,
+ buf->page_offset,
+ dma_dir, &ctx.xdp);
+ buf->sec_page = NULL;
+ }
+
+ if (likely(status & rx_not_ls))
+ goto read_again;
+
+ if (unlikely(error))
+ goto error_free_frag;
+
+ first_desc = true;
+ if (!skb) {
+ unsigned int pre_len, sync_len;
pre_len = ctx.xdp.data_end - ctx.xdp.data_hard_start -
buf->page_offset;
@@ -5887,26 +5980,18 @@ static int stmmac_rx(struct stmmac_priv *priv, int limit, u32 queue)
unsigned int xdp_res = -PTR_ERR(skb);
if (xdp_res & STMMAC_XDP_CONSUMED) {
- page_pool_put_page(rx_q->page_pool,
- virt_to_head_page(ctx.xdp.data),
- sync_len, true);
- buf->page = NULL;
+ stmmac_xdp_put_buff(rx_q, &ctx.xdp, sync_len);
rx_dropped++;
/* Clear skb as it was set as
* status by XDP program.
*/
skb = NULL;
-
- if (unlikely((status & rx_not_ls)))
- goto read_again;
-
count++;
continue;
} else if (xdp_res & (STMMAC_XDP_TX |
STMMAC_XDP_REDIRECT)) {
xdp_status |= xdp_res;
- buf->page = NULL;
skb = NULL;
count++;
continue;
@@ -5914,51 +5999,13 @@ static int stmmac_rx(struct stmmac_priv *priv, int limit, u32 queue)
}
}
+ skb = stmmac_build_skb(&ctx.xdp);
if (!skb) {
- unsigned int head_pad_len;
-
- /* XDP program may expand or reduce tail */
- buf1_len = ctx.xdp.data_end - ctx.xdp.data;
-
- skb = napi_build_skb(page_address(buf->page),
- rx_q->napi_skb_frag_size);
- if (!skb) {
- page_pool_recycle_direct(rx_q->page_pool,
- buf->page);
- rx_dropped++;
- count++;
- goto drain_data;
- }
-
- /* XDP program may adjust header */
- head_pad_len = ctx.xdp.data - ctx.xdp.data_hard_start;
- skb_reserve(skb, head_pad_len);
- skb_put(skb, buf1_len);
- skb_mark_for_recycle(skb);
- buf->page = NULL;
- } else if (buf1_len) {
- dma_sync_single_for_cpu(priv->device, buf->addr,
- buf1_len, dma_dir);
- skb_add_rx_frag(skb, skb_shinfo(skb)->nr_frags,
- buf->page, buf->page_offset, buf1_len,
- priv->dma_conf.dma_buf_sz);
- buf->page = NULL;
- }
-
- if (buf2_len) {
- dma_sync_single_for_cpu(priv->device, buf->sec_addr,
- buf2_len, dma_dir);
- skb_add_rx_frag(skb, skb_shinfo(skb)->nr_frags,
- buf->sec_page, 0, buf2_len,
- priv->dma_conf.dma_buf_sz);
- buf->sec_page = NULL;
- }
-
-drain_data:
- if (likely(status & rx_not_ls))
- goto read_again;
- if (!skb)
+ stmmac_xdp_put_buff(rx_q, &ctx.xdp, -1);
+ rx_dropped++;
+ count++;
continue;
+ }
/* Got entire packet into SKB. Finish it. */
@@ -5989,13 +6036,24 @@ static int stmmac_rx(struct stmmac_priv *priv, int limit, u32 queue)
rx_packets++;
rx_bytes += len;
count++;
+ continue;
+error_free_frag:
+ if (!first_desc) {
+ stmmac_xdp_put_buff(rx_q, &ctx.xdp, -1);
+ first_desc = true;
+ }
+ dev_kfree_skb(skb);
+ skb = NULL;
+ count++;
}
- if (status & rx_not_ls || skb) {
- rx_q->state_saved = true;
- rx_q->state.skb = skb;
- rx_q->state.error = error;
- rx_q->state.len = len;
+ if (count || !first_desc) {
+ rx_q->state_saved = !first_desc;
+ if (!first_desc) {
+ rx_q->state.xdp = ctx.xdp;
+ rx_q->state.error = error;
+ rx_q->state.len = len;
+ }
}
stmmac_finalize_xdp_rx(priv, xdp_status);
---
base-commit: 8830e65ed46de41f849eefb8ba227d4852c460f6
change-id: 20260918-stmmac-rx-mb-16469a2714ea
Best regards,
--
Lorenzo Bianconi <lorenzo.bianconi@oss.qualcomm.com>
next reply other threads:[~2026-09-21 8:58 UTC|newest]
Thread overview: 2+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-21 8:58 Lorenzo Bianconi [this message]
2026-09-23 23:59 ` [PATCH net-next v2] net: stmmac: rework stmmac_rx to support XDP rx multi-buff netdev-bot+sashiko
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260921-stmmac-rx-mb-v2-1-6e826e1ff306@oss.qualcomm.com \
--to=lorenzo.bianconi@oss.qualcomm.com \
--cc=alexandre.torgue@foss.st.com \
--cc=andrew+netdev@lunn.ch \
--cc=ast@kernel.org \
--cc=bpf@vger.kernel.org \
--cc=daniel@iogearbox.net \
--cc=davem@davemloft.net \
--cc=edumazet@google.com \
--cc=hawk@kernel.org \
--cc=john.fastabend@gmail.com \
--cc=kuba@kernel.org \
--cc=linux-arm-kernel@lists.infradead.org \
--cc=linux-stm32@st-md-mailman.stormreply.com \
--cc=maxime.chevallier@bootlin.com \
--cc=mcoquelin.stm32@gmail.com \
--cc=netdev@vger.kernel.org \
--cc=pabeni@redhat.com \
--cc=sdf@fomichev.me \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox