From: Chenguang Zhao <chenguang.zhao@linux.dev>
To: ast@kernel.org, daniel@iogearbox.net, andrii@kernel.org,
eddyz87@gmail.com, memxor@gmail.com, shuah@kernel.org
Cc: martin.lau@linux.dev, song@kernel.org, yonghong.song@linux.dev,
jolsa@kernel.org, emil@etsalapatis.com, ihor.solodrai@linux.dev,
bpf@vger.kernel.org, linux-kselftest@vger.kernel.org,
chenguang.zhao@liux.dev,
Chenguang Zhao <zhaochenguang@kylinos.cn>
Subject: [PATCH bpf-next] selftests/bpf: Add kptr-xchg benchmark
Date: Mon, 10 Aug 2026 13:57:09 +0800 [thread overview]
Message-ID: <20260810055709.729203-1-chenguang.zhao@linux.dev> (raw)
From: Chenguang Zhao <zhaochenguang@kylinos.cn>
Add a microbenchmark that repeatedly calls bpf_kptr_xchg() from an
fentry program. Architectures that advertise bpf_jit_supports_ptr_xchg()
inline the helper to BPF_XCHG; others keep the helper call. This bench
makes it easy to compare the two paths under the same workload.
Usage:
sudo ./bench -d 30 -w 5 -p 1 kptr-xchg --nr_loops 256
Signed-off-by: Chenguang Zhao <zhaochenguang@kylinos.cn>
---
Example results on LoongArch64 (helper vs inlined):
1. kptr xchg is helper:
./bench -d 30 -w 5 -p 1 kptr-xchg --nr_loops 256
Summary: throughput 68.612 ± 0.249 M ops/s ( 68.612M ops/prod), latency 14.575 ns/op
kptr xchg is inlined:
./bench -d 30 -w 5 -p 1 kptr-xchg --nr_loops 256
Summary: throughput 82.983 ± 0.268 M ops/s ( 82.983M ops/prod), latency 12.051 ns/op
This allows comparing the throughput difference.
2. To check whether bpf_kptr_xchg() was inlined:
bpftool prog show | grep -A4 -B1 'name benchmark'
xlated 64B jited 196B memlock 16384B
46: tracing name benchmark tag a9c8498a6197e8db gpl
loaded_at 2026-05-29T16:29:47+0800 uid 0
xlated 232B jited 380B memlock 16384B map_ids 8,10,9
bpftool prog dump xlated id 46 | grep -A6 -B6 -E 'xchg|atomic|call'
helper:
; __sync_add_and_fetch(&hits, i);
13: (18) r2 = map[id:10][0]+0
15: (db) lock *(u64 *)(r2 +0) += r1
; return 0;
16: (b4) w0 = 0
17: (95) exit
; old = bpf_kptr_xchg(&ptr, NULL);
18: (18) r1 = map[id:9][0]+0
20: (b7) r2 = 0
21: (85) call bpf_kptr_xchg#244684 ------ there call helper function
; if (old)
22: (15) if r0 == 0x0 goto pc-16
; bpf_obj_drop(old);
23: (bf) r1 = r0
24: (18) r2 = 0x0
26: (85) call 0x900000000046760c#92204
27: (05) goto pc-21
inlined:
; __sync_add_and_fetch(&hits, i);
13: (18) r2 = map[id:10][0]+0
15: (db) lock *(u64 *)(r2 +0) += r1
; return 0;
16: (b4) w0 = 0
17: (95) exit
; old = bpf_kptr_xchg(&ptr, NULL);
18: (18) r1 = map[id:9][0]+0
20: (b7) r2 = 0
21: (bf) r0 = r2
22: (db) r0 = atomic64_xchg((u64 *)(r1 +0), r0) ---- there inlining 'kptr xchg'
; if (old)
23: (15) if r0 == 0x0 goto pc-17
; bpf_obj_drop(old);
24: (bf) r1 = r0
25: (18) r2 = 0x0
27: (85) call 0x900000000046760e#92206
28: (05) goto pc-22
tools/testing/selftests/bpf/Makefile | 2 +
tools/testing/selftests/bpf/bench.c | 2 +
.../selftests/bpf/benchs/bench_kptr_xchg.c | 96 +++++++++++++++++++
.../selftests/bpf/progs/kptr_xchg_bench.c | 49 ++++++++++
4 files changed, 149 insertions(+)
create mode 100644 tools/testing/selftests/bpf/benchs/bench_kptr_xchg.c
create mode 100644 tools/testing/selftests/bpf/progs/kptr_xchg_bench.c
diff --git a/tools/testing/selftests/bpf/Makefile b/tools/testing/selftests/bpf/Makefile
index d3655a706482..76782b442aa9 100644
--- a/tools/testing/selftests/bpf/Makefile
+++ b/tools/testing/selftests/bpf/Makefile
@@ -988,6 +988,7 @@ $(OUTPUT)/bench_lpm_trie_map.o: $(OUTPUT)/lpm_trie_bench.skel.h $(OUTPUT)/lpm_tr
$(OUTPUT)/bench_bpf_nop.o: $(OUTPUT)/bpf_nop_bench.skel.h bench_bpf_timing.h
$(OUTPUT)/bench_xdp_lb.o: $(OUTPUT)/xdp_lb_bench.skel.h bench_bpf_timing.h
$(OUTPUT)/bench_bpf_timing.o: bench_bpf_timing.h
+$(OUTPUT)/bench_kptr_xchg.o: $(OUTPUT)/kptr_xchg_bench.skel.h
$(OUTPUT)/bench.o: bench.h testing_helpers.h $(BPFOBJ)
$(OUTPUT)/bench: LDLIBS += -lm
$(OUTPUT)/bench: $(OUTPUT)/bench.o \
@@ -1014,6 +1015,7 @@ $(OUTPUT)/bench: $(OUTPUT)/bench.o \
$(OUTPUT)/bench_bpf_timing.o \
$(OUTPUT)/bench_bpf_nop.o \
$(OUTPUT)/bench_xdp_lb.o \
+ $(OUTPUT)/bench_kptr_xchg.o \
$(OUTPUT)/usdt_1.o \
$(OUTPUT)/usdt_2.o \
#
diff --git a/tools/testing/selftests/bpf/bench.c b/tools/testing/selftests/bpf/bench.c
index b86b73456d3c..96cbd31088b8 100644
--- a/tools/testing/selftests/bpf/bench.c
+++ b/tools/testing/selftests/bpf/bench.c
@@ -585,6 +585,7 @@ extern const struct bench bench_lpm_trie_delete;
extern const struct bench bench_lpm_trie_free;
extern const struct bench bench_bpf_nop;
extern const struct bench bench_xdp_lb;
+extern const struct bench bench_kptr_xchg;
static const struct bench *benchs[] = {
&bench_count_global,
@@ -669,6 +670,7 @@ static const struct bench *benchs[] = {
&bench_lpm_trie_free,
&bench_bpf_nop,
&bench_xdp_lb,
+ &bench_kptr_xchg,
};
static void find_benchmark(void)
diff --git a/tools/testing/selftests/bpf/benchs/bench_kptr_xchg.c b/tools/testing/selftests/bpf/benchs/bench_kptr_xchg.c
new file mode 100644
index 000000000000..b8a0d346fda6
--- /dev/null
+++ b/tools/testing/selftests/bpf/benchs/bench_kptr_xchg.c
@@ -0,0 +1,96 @@
+// SPDX-License-Identifier: GPL-2.0
+/* Copyright (C) 2026. Loongson Technology Corporation Limited */
+#include <argp.h>
+#include "bench.h"
+#include "kptr_xchg_bench.skel.h"
+
+static struct ctx {
+ struct kptr_xchg_bench *skel;
+} ctx;
+
+static struct {
+ __u32 nr_loops;
+} args = {
+ .nr_loops = 256,
+};
+
+enum {
+ ARG_NR_LOOPS = 7000,
+};
+
+static const struct argp_option opts[] = {
+ { "nr_loops", ARG_NR_LOOPS, "nr_loops", 0,
+ "Set number of bpf_kptr_xchg() calls per trigger"},
+ {},
+};
+
+static error_t parse_arg(int key, char *arg, struct argp_state *state)
+{
+ switch (key) {
+ case ARG_NR_LOOPS:
+ args.nr_loops = strtol(arg, NULL, 10);
+ break;
+ default:
+ return ARGP_ERR_UNKNOWN;
+ }
+
+ return 0;
+}
+
+static const struct argp bench_kptr_xchg_argp = {
+ .options = opts,
+ .parser = parse_arg,
+};
+
+static void validate(void)
+{
+ if (env.consumer_cnt != 0) {
+ fprintf(stderr, "benchmark doesn't support consumer!\n");
+ exit(1);
+ }
+}
+
+static void *producer(void *input)
+{
+ while (true)
+ syscall(__NR_getpgid);
+
+ return NULL;
+}
+
+static void measure(struct bench_res *res)
+{
+ res->hits = atomic_swap(&ctx.skel->bss->hits, 0);
+}
+
+static void setup(void)
+{
+ struct bpf_link *link;
+
+ setup_libbpf();
+
+ ctx.skel = kptr_xchg_bench__open_and_load();
+ if (!ctx.skel) {
+ fprintf(stderr, "failed to open skeleton\n");
+ exit(1);
+ }
+
+ ctx.skel->data->nr_loops = args.nr_loops;
+
+ link = bpf_program__attach(ctx.skel->progs.benchmark);
+ if (!link) {
+ fprintf(stderr, "failed to attach program!\n");
+ exit(1);
+ }
+}
+
+const struct bench bench_kptr_xchg = {
+ .name = "kptr-xchg",
+ .argp = &bench_kptr_xchg_argp,
+ .validate = validate,
+ .setup = setup,
+ .producer_thread = producer,
+ .measure = measure,
+ .report_progress = ops_report_progress,
+ .report_final = ops_report_final,
+};
diff --git a/tools/testing/selftests/bpf/progs/kptr_xchg_bench.c b/tools/testing/selftests/bpf/progs/kptr_xchg_bench.c
new file mode 100644
index 000000000000..363883073e2c
--- /dev/null
+++ b/tools/testing/selftests/bpf/progs/kptr_xchg_bench.c
@@ -0,0 +1,49 @@
+// SPDX-License-Identifier: GPL-2.0
+/* Copyright (C) 2026. Loongson Technology Corporation Limited */
+#include "vmlinux.h"
+#include <bpf/bpf_helpers.h>
+
+#include "bpf_experimental.h"
+#include "bpf_misc.h"
+
+char _license[] SEC("license") = "GPL";
+
+#define MAX_XCHG_LOOPS 4096
+
+struct bin_data {
+ char blob[32];
+};
+
+#define private(name) SEC(".bss." #name) __hidden __attribute__((aligned(8)))
+
+private(kptr) struct bin_data __kptr *ptr;
+u32 nr_loops = 256;
+long hits;
+
+SEC("fentry/" SYS_PREFIX "sys_getpgid")
+int benchmark(void *ctx)
+{
+ struct bin_data *old;
+ u32 i;
+
+ for (i = 0; i < MAX_XCHG_LOOPS; i++) {
+ if (i >= nr_loops)
+ break;
+
+ old = bpf_kptr_xchg(&ptr, NULL);
+ if (old)
+ bpf_obj_drop(old);
+ }
+
+ __sync_add_and_fetch(&hits, i);
+ return 0;
+}
+
+/*
+ * BTF FUNC records are not generated for kfuncs referenced only through
+ * optimized paths. Keep bpf_obj_drop() visible to libbpf's kfunc linker.
+ */
+void __btf_root(void)
+{
+ bpf_obj_drop(NULL);
+}
--
2.25.1
next reply other threads:[~2026-08-10 5:57 UTC|newest]
Thread overview: 2+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-08-10 5:57 Chenguang Zhao [this message]
2026-08-10 7:44 ` [PATCH bpf-next] selftests/bpf: Add kptr-xchg benchmark bot+bpf-ci
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260810055709.729203-1-chenguang.zhao@linux.dev \
--to=chenguang.zhao@linux.dev \
--cc=andrii@kernel.org \
--cc=ast@kernel.org \
--cc=bpf@vger.kernel.org \
--cc=chenguang.zhao@liux.dev \
--cc=daniel@iogearbox.net \
--cc=eddyz87@gmail.com \
--cc=emil@etsalapatis.com \
--cc=ihor.solodrai@linux.dev \
--cc=jolsa@kernel.org \
--cc=linux-kselftest@vger.kernel.org \
--cc=martin.lau@linux.dev \
--cc=memxor@gmail.com \
--cc=shuah@kernel.org \
--cc=song@kernel.org \
--cc=yonghong.song@linux.dev \
--cc=zhaochenguang@kylinos.cn \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox