From: Jiayuan Chen <jiayuan.chen@linux.dev>
To: bpf@vger.kernel.org
Cc: "Jiayuan Chen" <jiayuan.chen@linux.dev>,
"Andrew Lunn" <andrew+netdev@lunn.ch>,
"David S. Miller" <davem@davemloft.net>,
"Eric Dumazet" <edumazet@google.com>,
"Jakub Kicinski" <kuba@kernel.org>,
"Paolo Abeni" <pabeni@redhat.com>,
"Alexei Starovoitov" <ast@kernel.org>,
"Daniel Borkmann" <daniel@iogearbox.net>,
"Jesper Dangaard Brouer" <hawk@kernel.org>,
"John Fastabend" <john.fastabend@gmail.com>,
"Stanislav Fomichev" <sdf@fomichev.me>,
"Simon Horman" <horms@kernel.org>,
"Andrii Nakryiko" <andrii@kernel.org>,
"Eduard Zingerman" <eddyz87@gmail.com>,
"Kumar Kartikeya Dwivedi" <memxor@gmail.com>,
"Martin KaFai Lau" <martin.lau@linux.dev>,
"Song Liu" <song@kernel.org>,
"Yonghong Song" <yonghong.song@linux.dev>,
"Jiri Olsa" <jolsa@kernel.org>,
"Emil Tsalapatis" <emil@etsalapatis.com>,
"Ihor Solodrai" <ihor.solodrai@linux.dev>,
"Shuah Khan" <shuah@kernel.org>,
"Kuniyuki Iwashima" <kuniyu@google.com>,
"Hangbin Liu" <liuhangbin@gmail.com>,
"Martin Karsten" <mkarsten@uwaterloo.ca>,
"Toke Høiland-Jørgensen" <toke@redhat.com>,
"Lorenzo Bianconi" <lorenzo.bianconi@oss.qualcomm.com>,
"Eelco Chaudron" <echaudro@redhat.com>,
linux-kernel@vger.kernel.org, netdev@vger.kernel.org,
linux-kselftest@vger.kernel.org
Subject: [PATCH bpf v3 2/2] selftests/bpf: add xdp_shrink_frags
Date: Fri, 11 Sep 2026 21:56:52 +0800 [thread overview]
Message-ID: <20260911135711.109338-3-jiayuan.chen@linux.dev> (raw)
In-Reply-To: <20260911135711.109338-1-jiayuan.chen@linux.dev>
Add a test that attaches an xdp.frags program which shrinks a whole frag
away, so bpf_xdp_shrink_data() frees a page_pool frag.
test_tun triggers the page-type mismatch on the generic XDP path.
test_veth triggers the same mismatch on the veth path.
test_veth_tx bounces the cow'd buff with XDP_TX so the peer runs a frags
program on the resulting frame, covering the buff -> frame -> buff path.
A kernel without the fix will report "Bad page state ... page_pool leak".
The packet sizes only leave a frag after the cow on 4K pages, so the
subtests skip elsewhere.
Signed-off-by: Jiayuan Chen <jiayuan.chen@linux.dev>
---
.../bpf/prog_tests/xdp_shrink_frags.c | 288 ++++++++++++++++++
.../selftests/bpf/progs/xdp_shrink_frags.c | 34 +++
2 files changed, 322 insertions(+)
create mode 100644 tools/testing/selftests/bpf/prog_tests/xdp_shrink_frags.c
create mode 100644 tools/testing/selftests/bpf/progs/xdp_shrink_frags.c
diff --git a/tools/testing/selftests/bpf/prog_tests/xdp_shrink_frags.c b/tools/testing/selftests/bpf/prog_tests/xdp_shrink_frags.c
new file mode 100644
index 0000000000000..ee8f9034a2684
--- /dev/null
+++ b/tools/testing/selftests/bpf/prog_tests/xdp_shrink_frags.c
@@ -0,0 +1,288 @@
+// SPDX-License-Identifier: GPL-2.0
+#include <test_progs.h>
+#include <network_helpers.h>
+#include <linux/if_tun.h>
+#include <linux/if_ether.h>
+#include <sys/uio.h>
+#include <net/if.h>
+#include <arpa/inet.h>
+#include "xdp_shrink_frags.skel.h"
+
+/*
+ * A generic-XDP program that shrinks into the frags frees a page_pool frag.
+ * skb-backed XDP first cow's the nonlinear skb into page_pool memory
+ * (skb_cow_data_for_xdp() for generic XDP, skb_pp_cow_data() for veth), but
+ * the shared rxq is registered as MEM_TYPE_PAGE_SHARED, so a buggy kernel
+ * frees the frag with page_frag_free() -> "Bad page state ... page_pool leak".
+ */
+
+#define TAP_NAME "xdp_shrink0"
+#define TAP_NETNS "xdp_shrink_tap"
+
+#define NS_NAME_MAX_LEN 32
+#define VETH_LOCAL "xdp_shrinkA"
+#define VETH_PEER "xdp_shrinkB"
+#define VETH_LOCAL_IP "10.9.9.1"
+#define VETH_PEER_IP "10.9.9.2"
+#define VETH_LOCAL_MAC "02:00:00:00:00:01"
+
+/*
+ * skb_pp_cow_data() keeps up to one page in the linear part, so the packet
+ * sizes below only leave a frag (smaller than the 3000-byte shrink, so it is
+ * released as a whole) on 4K pages. Skip elsewhere rather than run a test
+ * that cannot tell a fixed kernel from a buggy one.
+ */
+#define PAGE_SIZE_4K 4096
+
+/*
+ * Generous, so a loaded CI does not fail the assert prematurely; normally
+ * the first check already succeeds.
+ */
+#define WAIT_ITERS 10000
+#define WAIT_US 1000
+
+static void wait_for_prog(struct xdp_shrink_frags *skel)
+{
+ int i;
+
+ for (i = 0; i < WAIT_ITERS && !skel->bss->shrink_ran; i++)
+ usleep(WAIT_US);
+}
+
+static int create_tap_napi_frags(const char *ifname)
+{
+ struct ifreq ifr = {
+ .ifr_flags = IFF_TAP | IFF_NO_PI | IFF_NAPI | IFF_NAPI_FRAGS,
+ };
+ int fd, err;
+
+ strscpy(ifr.ifr_name, ifname);
+
+ fd = open("/dev/net/tun", O_RDWR);
+ if (fd < 0)
+ return -errno;
+
+ err = ioctl(fd, TUNSETIFF, &ifr);
+ if (err) {
+ err = -errno;
+ close(fd);
+ return err;
+ }
+
+ return fd;
+}
+
+/*
+ * Similar to flow_dissector.c: writev() an IFF_NAPI_FRAGS tap to build a
+ * nonlinear skb that tun runs through do_xdp_generic().
+ */
+static void test_tun(struct xdp_shrink_frags *skel)
+{
+ __u8 head[74], frag1[2048], frag2[2048];
+ struct ethhdr *eth = (void *)head;
+ int tap_fd = -1, ifindex, err;
+ struct netns_obj *ns = NULL;
+ struct iovec iov[3];
+ ssize_t n;
+
+ if (getpagesize() != PAGE_SIZE_4K) {
+ test__skip();
+ return;
+ }
+
+ ns = netns_new(TAP_NETNS, true);
+ if (!ASSERT_OK_PTR(ns, "netns_new"))
+ return;
+
+ tap_fd = create_tap_napi_frags(TAP_NAME);
+ if (!ASSERT_GE(tap_fd, 0, "create_tap"))
+ goto out;
+
+ SYS(out, "ip link set dev " TAP_NAME " up");
+
+ ifindex = if_nametoindex(TAP_NAME);
+ if (!ASSERT_GT(ifindex, 0, "if_nametoindex"))
+ goto out;
+
+ skel->bss->shrink_ran = 0;
+
+ err = bpf_xdp_attach(ifindex, bpf_program__fd(skel->progs.xdp_shrink),
+ 0, NULL);
+ if (!ASSERT_OK(err, "bpf_xdp_attach"))
+ goto out;
+
+ memset(head, 0, sizeof(head));
+ memset(frag1, 0x41, sizeof(frag1));
+ memset(frag2, 0x42, sizeof(frag2));
+ eth->h_proto = htons(ETH_P_IP);
+
+ iov[0].iov_base = head; iov[0].iov_len = sizeof(head);
+ iov[1].iov_base = frag1; iov[1].iov_len = sizeof(frag1);
+ iov[2].iov_base = frag2; iov[2].iov_len = sizeof(frag2);
+
+ n = writev(tap_fd, iov, ARRAY_SIZE(iov));
+ ASSERT_EQ(n, sizeof(head) + sizeof(frag1) + sizeof(frag2), "writev");
+
+ wait_for_prog(skel);
+ /* a buggy kernel only splats "page_pool leak", it does not fail here */
+ ASSERT_GT(skel->bss->shrink_ran, 0, "xdp_prog_ran");
+
+ bpf_xdp_detach(ifindex, 0, NULL);
+out:
+ if (tap_fd >= 0)
+ close(tap_fd);
+ netns_free(ns);
+}
+
+/*
+ * Both veth ends live in their own namespace, so the traffic really crosses
+ * the pair and nothing is created in the caller's namespace.
+ */
+static int veth_setup(char *ns0, char *ns1)
+{
+ if (!ASSERT_OK(append_tid(ns0, NS_NAME_MAX_LEN), "append_tid ns0"))
+ return -1;
+ if (!ASSERT_OK(append_tid(ns1, NS_NAME_MAX_LEN), "append_tid ns1"))
+ return -1;
+
+ SYS(fail, "ip netns add %s", ns0);
+ SYS(fail_ns0, "ip netns add %s", ns1);
+ SYS(fail_ns1, "ip -n %s link add %s mtu 8000 type veth peer name %s mtu 8000",
+ ns0, VETH_LOCAL, VETH_PEER);
+ SYS(fail_ns1, "ip -n %s link set %s netns %s", ns0, VETH_PEER, ns1);
+ SYS(fail_ns1, "ip -n %s link set %s address %s", ns0, VETH_LOCAL,
+ VETH_LOCAL_MAC);
+ SYS(fail_ns1, "ip -n %s addr add %s/24 dev %s", ns0, VETH_LOCAL_IP,
+ VETH_LOCAL);
+ SYS(fail_ns1, "ip -n %s link set %s up", ns0, VETH_LOCAL);
+ SYS(fail_ns1, "ip -n %s addr add %s/24 dev %s", ns1, VETH_PEER_IP,
+ VETH_PEER);
+ SYS(fail_ns1, "ip -n %s link set %s up", ns1, VETH_PEER);
+
+ return 0;
+
+fail_ns1:
+ SYS_NOFAIL("ip netns del %s", ns1);
+fail_ns0:
+ SYS_NOFAIL("ip netns del %s", ns0);
+fail:
+ return -1;
+}
+
+static void veth_cleanup(const char *ns0, const char *ns1)
+{
+ /* Dropping the namespaces takes the veth pair and its XDP programs. */
+ SYS_NOFAIL("ip netns del %s", ns1);
+ SYS_NOFAIL("ip netns del %s", ns0);
+}
+
+static int veth_attach(const char *ns, const char *dev, int prog_fd)
+{
+ struct nstoken *nstoken;
+ int ifindex, err;
+
+ nstoken = open_netns(ns);
+ if (!ASSERT_OK_PTR(nstoken, "open_netns"))
+ return -1;
+
+ ifindex = if_nametoindex(dev);
+ if (!ASSERT_GT(ifindex, 0, "if_nametoindex")) {
+ close_netns(nstoken);
+ return -1;
+ }
+
+ err = bpf_xdp_attach(ifindex, prog_fd, 0, NULL);
+ close_netns(nstoken);
+
+ return ASSERT_OK(err, "bpf_xdp_attach") ? 0 : -1;
+}
+
+/* A large ping builds a nonlinear skb that veth cow's into its page_pool. */
+static void test_veth(struct xdp_shrink_frags *skel)
+{
+ char ns0[NS_NAME_MAX_LEN] = "xdp_shrink_ns0-";
+ char ns1[NS_NAME_MAX_LEN] = "xdp_shrink_ns1-";
+
+ if (getpagesize() != PAGE_SIZE_4K) {
+ test__skip();
+ return;
+ }
+
+ if (veth_setup(ns0, ns1))
+ return;
+
+ skel->bss->shrink_ran = 0;
+
+ if (veth_attach(ns0, VETH_LOCAL, bpf_program__fd(skel->progs.xdp_shrink)))
+ goto out;
+
+ SYS_NOFAIL("ip netns exec %s ping -q -s 5000 -c 3 -W 1 %s",
+ ns1, VETH_LOCAL_IP);
+
+ wait_for_prog(skel);
+ /* a buggy kernel only splats "page_pool leak", it does not fail here */
+ ASSERT_GT(skel->bss->shrink_ran, 0, "xdp_prog_ran");
+out:
+ veth_cleanup(ns0, ns1);
+}
+
+/*
+ * LOCAL cow's the incoming skb and returns XDP_TX, so the buff is turned into
+ * an xdp_frame and bounced to PEER, which shrinks a frag. A page_pool tag
+ * recorded on the buff must not leak into the frame, or PEER frees a plain
+ * page (whose pp was already cleared on the XDP_TX side) as page_pool memory.
+ */
+static void test_veth_tx(struct xdp_shrink_frags *skel)
+{
+ char ns0[NS_NAME_MAX_LEN] = "xdp_shrink_tx0-";
+ char ns1[NS_NAME_MAX_LEN] = "xdp_shrink_tx1-";
+
+ if (getpagesize() != PAGE_SIZE_4K) {
+ test__skip();
+ return;
+ }
+
+ if (veth_setup(ns0, ns1))
+ return;
+
+ /*
+ * LOCAL bounces everything (incl. ARP) with XDP_TX, so pin a static
+ * neighbour to let the ping's payload actually reach it.
+ */
+ SYS(out, "ip -n %s neigh add %s lladdr %s dev %s nud permanent",
+ ns1, VETH_LOCAL_IP, VETH_LOCAL_MAC, VETH_PEER);
+
+ skel->bss->shrink_ran = 0;
+
+ if (veth_attach(ns0, VETH_LOCAL, bpf_program__fd(skel->progs.xdp_tx)))
+ goto out;
+ if (veth_attach(ns1, VETH_PEER, bpf_program__fd(skel->progs.xdp_shrink)))
+ goto out;
+
+ SYS_NOFAIL("ip netns exec %s ping -q -s 5000 -c 3 -W 1 %s",
+ ns1, VETH_LOCAL_IP);
+
+ wait_for_prog(skel);
+ /* a buggy kernel only splats "page_pool leak", it does not fail here */
+ ASSERT_GT(skel->bss->shrink_ran, 0, "xdp_prog_ran");
+out:
+ veth_cleanup(ns0, ns1);
+}
+
+void test_xdp_shrink_frags(void)
+{
+ struct xdp_shrink_frags *skel;
+
+ skel = xdp_shrink_frags__open_and_load();
+ if (!ASSERT_OK_PTR(skel, "skel_open_load"))
+ return;
+
+ if (test__start_subtest("tun"))
+ test_tun(skel);
+ if (test__start_subtest("veth"))
+ test_veth(skel);
+ if (test__start_subtest("veth_tx"))
+ test_veth_tx(skel);
+
+ xdp_shrink_frags__destroy(skel);
+}
diff --git a/tools/testing/selftests/bpf/progs/xdp_shrink_frags.c b/tools/testing/selftests/bpf/progs/xdp_shrink_frags.c
new file mode 100644
index 0000000000000..ad895ab699303
--- /dev/null
+++ b/tools/testing/selftests/bpf/progs/xdp_shrink_frags.c
@@ -0,0 +1,34 @@
+// SPDX-License-Identifier: GPL-2.0
+#include "vmlinux.h"
+#include <bpf/bpf_helpers.h>
+
+int shrink_ran;
+
+SEC("xdp.frags")
+int xdp_shrink(struct xdp_md *ctx)
+{
+ /*
+ * Runs on both skb-backed XDP paths (generic XDP via tun, and veth):
+ * the nonlinear skb is cow'd into page_pool memory before we run, so
+ * shrinking the tail releases a whole frag that has to go back to that
+ * pool. The counter only tells us the program ran and the helper
+ * succeeded -- a linear buff returns 0 as well -- it does not prove a
+ * frag was released.
+ */
+ if (bpf_xdp_adjust_tail(ctx, -3000) == 0)
+ __sync_fetch_and_add(&shrink_ran, 1);
+ return XDP_PASS;
+}
+
+SEC("xdp.frags")
+int xdp_tx(struct xdp_md *ctx)
+{
+ /*
+ * Bounce the frame back. On veth this turns the buff into an
+ * xdp_frame, which is where a buff-scoped page_pool tag must not leak
+ * into the frame handed to the peer.
+ */
+ return XDP_TX;
+}
+
+char _license[] SEC("license") = "GPL";
--
2.43.0
next prev parent reply other threads:[~2026-09-11 13:58 UTC|newest]
Thread overview: 8+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-11 13:56 [PATCH bpf v3 0/2] net: xdp: fix bpf_xdp_shrink_data() page handling on generic XDP and veth Jiayuan Chen
2026-09-11 13:56 ` [PATCH bpf v3 1/2] bpf, veth: xdp: fix page_pool page leak on skb-backed XDP Jiayuan Chen
2026-09-11 14:36 ` sashiko-bot
2026-09-12 2:28 ` Jiayuan Chen
2026-09-13 13:05 ` Lorenzo Bianconi
2026-09-11 13:56 ` Jiayuan Chen [this message]
2026-09-11 14:46 ` [PATCH bpf v3 2/2] selftests/bpf: add xdp_shrink_frags sashiko-bot
2026-09-12 2:34 ` Jiayuan Chen
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260911135711.109338-3-jiayuan.chen@linux.dev \
--to=jiayuan.chen@linux.dev \
--cc=andrew+netdev@lunn.ch \
--cc=andrii@kernel.org \
--cc=ast@kernel.org \
--cc=bpf@vger.kernel.org \
--cc=daniel@iogearbox.net \
--cc=davem@davemloft.net \
--cc=echaudro@redhat.com \
--cc=eddyz87@gmail.com \
--cc=edumazet@google.com \
--cc=emil@etsalapatis.com \
--cc=hawk@kernel.org \
--cc=horms@kernel.org \
--cc=ihor.solodrai@linux.dev \
--cc=john.fastabend@gmail.com \
--cc=jolsa@kernel.org \
--cc=kuba@kernel.org \
--cc=kuniyu@google.com \
--cc=linux-kernel@vger.kernel.org \
--cc=linux-kselftest@vger.kernel.org \
--cc=liuhangbin@gmail.com \
--cc=lorenzo.bianconi@oss.qualcomm.com \
--cc=martin.lau@linux.dev \
--cc=memxor@gmail.com \
--cc=mkarsten@uwaterloo.ca \
--cc=netdev@vger.kernel.org \
--cc=pabeni@redhat.com \
--cc=sdf@fomichev.me \
--cc=shuah@kernel.org \
--cc=song@kernel.org \
--cc=toke@redhat.com \
--cc=yonghong.song@linux.dev \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.