BPF List
 help / color / mirror / Atom feed
From: Yonghong Song <yonghong.song@linux.dev>
To: bpf@vger.kernel.org
Cc: Alexei Starovoitov <ast@kernel.org>,
	Andrii Nakryiko <andrii@kernel.org>,
	Daniel Borkmann <daniel@iogearbox.net>,
	Eduard Zingerman <eddyz87@gmail.com>,
	kernel-team@fb.com
Subject: [PATCH bpf-next v6 12/21] bpf, arm64: Dispatch exception cleanup pads at run time
Date: Fri, 25 Sep 2026 22:01:07 -0700	[thread overview]
Message-ID: <20260926050107.2218786-1-yonghong.song@linux.dev> (raw)
In-Reply-To: <20260926050006.2213110-1-yonghong.song@linux.dev>

The JIT half: build the native cleanup table from the JIT's byte offsets
once the image is final, emit a BTI at each pad head, address the frame
through the private stack pointer where one is in use, and record the one
epilogue so a frame the unwind passes over can return straight through it.

The dispatch is arch_bpf_stack_walk_ra(), which hands the unwind the slot a
frame's return address came out of rather than just the address. arm64's
unwinder reads that address from the frame record the callee pushed, so the
slot belongs to the record the previous entry stepped through.

Writing it has to respect pointer authentication: a BPF prologue signs the
link register with PACIASP and the epilogue authenticates it, so what is
written back has to carry the same signature. Its modifier is the stack
pointer the owner was entered with, which is not known here -- so recover
it by re-signing the address the unwinder stripped until that matches what
the slot holds. Only where the CPU implements address authentication, and
only where the slot was signed to begin with.

Two frames are not redirected. The walk's own first frame is not returning
anywhere yet, and a frame whose return the function graph tracer or a
kretprobe has hooked holds the tracer's trampoline in its slot rather than
the address the unwinder reports, so the walk stops there.

bpf_jit_supports_cleanup_pads() can now say yes.

Signed-off-by: Yonghong Song <yonghong.song@linux.dev>
---
 arch/arm64/kernel/stacktrace.c | 94 ++++++++++++++++++++++++++++++++++
 arch/arm64/net/bpf_jit_comp.c  | 20 +++++++-
 2 files changed, 113 insertions(+), 1 deletion(-)

diff --git a/arch/arm64/kernel/stacktrace.c b/arch/arm64/kernel/stacktrace.c
index 3ebcf8c53fb0..c750c520f24d 100644
--- a/arch/arm64/kernel/stacktrace.c
+++ b/arch/arm64/kernel/stacktrace.c
@@ -445,6 +445,100 @@ noinline noinstr void arch_bpf_stack_walk(bool (*consume_entry)(void *cookie, u6
 	kunwind_stack_walk(arch_bpf_unwind_consume_entry, &data, current, NULL);
 }
 
+struct bpf_unwind_ra_consume_entry_data {
+	bool (*consume_entry)(void *cookie, u64 ip, u64 sp, u64 fp, u64 *ra);
+	void *cookie;
+	unsigned long record;
+	bool seen_first;
+};
+
+static u64 bpf_unwind_sign_ra(u64 ra, u64 modifier)
+{
+	asm volatile(ARM64_ASM_PREAMBLE
+		     ".arch_extension pauth\n"
+		     "	pacia %0, %1"
+		     : "+r" (ra) : "r" (modifier));
+	return ra;
+}
+
+/*
+ * PACIASP's modifier is the stack pointer the owner was entered with: record
+ * + 16 for a BPF prologue, but further up for bpf_unwind()'s own C frame.
+ * Recognise it by re-signing @pc, which the unwinder stripped from @stored.
+ */
+static bool bpf_unwind_ra_modifier(unsigned long record, unsigned long caller_fp,
+				   u64 stored, u64 pc, u64 *modifier)
+{
+	u64 m;
+
+	for (m = record + sizeof(struct frame_record); m <= caller_fp; m += 16) {
+		if (bpf_unwind_sign_ra(pc, m) == stored) {
+			*modifier = m;
+			return true;
+		}
+	}
+	return false;
+}
+
+static bool bpf_unwind_store_ra(unsigned long record, unsigned long caller_fp,
+				u64 pc, u64 ra)
+{
+	struct frame_record *rec = (struct frame_record *)record;
+	u64 stored = READ_ONCE(rec->lr);
+
+	if (system_supports_address_auth() && stored != pc) {
+		u64 modifier;
+
+		if (WARN_ON_ONCE(!bpf_unwind_ra_modifier(record, caller_fp,
+							 stored, pc, &modifier)))
+			return false;
+		ra = bpf_unwind_sign_ra(ra, modifier);
+	}
+	WRITE_ONCE(rec->lr, ra);
+	return true;
+}
+
+static bool
+arch_bpf_unwind_ra_consume_entry(const struct kunwind_state *state, void *cookie)
+{
+	struct bpf_unwind_ra_consume_entry_data *data = cookie;
+	unsigned long record = data->record;
+	bool seen_first = data->seen_first;
+	u64 ra = state->common.pc;
+	bool cont;
+
+	/* The record this frame's return address will have come out of. */
+	data->record = state->common.fp;
+	data->seen_first = true;
+
+	/* The first pc is where the walk runs, not an address it returns to. */
+	if (!seen_first)
+		return true;
+	/* A traced return: the slot holds the tracer's trampoline, not @pc. */
+	if (state->flags.fgraph || state->flags.kretprobe)
+		return false;
+
+	/* A consumer that stops still gets to redirect the frame it stopped on. */
+	cont = data->consume_entry(data->cookie, state->common.pc, 0,
+				   state->common.fp, &ra);
+	if (ra != state->common.pc &&
+	    !bpf_unwind_store_ra(record, state->common.fp, state->common.pc, ra))
+		return false;
+	return cont;
+}
+
+noinline noinstr void arch_bpf_stack_walk_ra(bool (*consume_entry)(void *cookie, u64 ip, u64 sp,
+								   u64 fp, u64 *ra),
+					     void *cookie)
+{
+	struct bpf_unwind_ra_consume_entry_data data = {
+		.consume_entry = consume_entry,
+		.cookie = cookie,
+	};
+
+	kunwind_stack_walk(arch_bpf_unwind_ra_consume_entry, &data, current, NULL);
+}
+
 static const char *state_source_string(const struct kunwind_state *state)
 {
 	switch (state->source) {
diff --git a/arch/arm64/net/bpf_jit_comp.c b/arch/arm64/net/bpf_jit_comp.c
index 475e70653454..2422a1ae1256 100644
--- a/arch/arm64/net/bpf_jit_comp.c
+++ b/arch/arm64/net/bpf_jit_comp.c
@@ -10,6 +10,7 @@
 #include <linux/arm-smccc.h>
 #include <linux/bitfield.h>
 #include <linux/bpf.h>
+#include <linux/bpf_verifier.h>
 #include <linux/cfi.h>
 #include <linux/filter.h>
 #include <linux/memory.h>
@@ -1380,7 +1381,8 @@ static int build_insn(const struct bpf_verifier_env *env, const struct bpf_insn
 	int ret;
 	bool sign_extend;
 
-	if (bpf_insn_is_indirect_target(env, ctx->prog, i))
+	if (bpf_insn_is_indirect_target(env, ctx->prog, i) ||
+	    bpf_exc_insn_is_pad(env, ctx->prog, i))
 		emit_bti(A64_BTI_J, ctx);
 
 	switch (code) {
@@ -2423,6 +2425,17 @@ struct bpf_prog *bpf_int_jit_compile(struct bpf_verifier_env *env, struct bpf_pr
 		 * reasons, expects to point to the next instruction)
 		 */
 		bpf_prog_update_insn_ptrs(prog, ctx.offset, ctx.ro_image);
+
+		/*
+		 * Same byte offsets, consumed by the bpf_unwind() walk:
+		 * turn the cleanup records into native address ranges now that
+		 * the image is final.
+		 */
+		bpf_exc_fill_native_ranges(prog, ctx.offset, ctx.ro_image);
+
+		/* Where an unwind sends a frame with no pad. */
+		prog->aux->epilogue_ip = (u64)ctx.ro_image +
+					 ctx.epilogue_offset * AARCH64_INSN_SIZE;
 out_off:
 		if (!ro_header && priv_stack_ptr) {
 			free_percpu(priv_stack_ptr);
@@ -3408,6 +3421,11 @@ bool bpf_jit_supports_exceptions(void)
 	return true;
 }
 
+bool bpf_jit_supports_cleanup_pads(void)
+{
+	return true;
+}
+
 bool bpf_jit_supports_arena(void)
 {
 	return true;
-- 
2.53.0-Meta


  parent reply	other threads:[~2026-09-26  5:01 UTC|newest]

Thread overview: 56+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-26  5:00 [PATCH bpf-next v6 00/21] bpf: Run exception cleanup landing pads when bpf_unwind() unwinds Yonghong Song
2026-09-26  5:00 ` [PATCH bpf-next v6 01/21] bpf: Pack bpf_insn_aux_data flags into bit fields Yonghong Song
2026-09-26  5:00 ` [PATCH bpf-next v6 02/21] bpf: Accept the compiler's exception cleanup table at program load Yonghong Song
2026-09-26  5:00 ` [PATCH bpf-next v6 03/21] bpf: Add the bpf_unwind() and bpf_unwind_resume() kfuncs Yonghong Song
2026-09-26  5:00 ` [PATCH bpf-next v6 04/21] bpf: Add lookups for exception cleanup resumes and landing pads Yonghong Song
2026-09-26  5:00 ` [PATCH bpf-next v6 05/21] bpf: Prepare for an exception cleanup table before the CFG walk Yonghong Song
2026-09-26  5:16   ` sashiko-bot
2026-09-26 23:54     ` Yonghong Song
2026-09-27 20:39   ` bot+bpf-ci
2026-09-28  0:01     ` Yonghong Song
2026-09-26  5:00 ` [PATCH bpf-next v6 06/21] bpf: Make exception landing pads reachable in the CFG Yonghong Song
2026-09-26  5:21   ` sashiko-bot
2026-09-27  0:02     ` Yonghong Song
2026-09-27 20:40   ` bot+bpf-ci
2026-09-28  0:12     ` Yonghong Song
2026-09-26  5:00 ` [PATCH bpf-next v6 07/21] bpf: Resume a covered call at its landing pad Yonghong Song
2026-09-26  5:15   ` sashiko-bot
2026-09-26  8:21     ` Alexei Starovoitov
2026-09-27  0:04       ` Yonghong Song
2026-09-27  0:41     ` Yonghong Song
2026-09-27 20:40   ` bot+bpf-ci
2026-09-28  0:17     ` Yonghong Song
2026-09-26  5:00 ` [PATCH bpf-next v6 08/21] bpf: Refuse a landing pad that does not resume Yonghong Song
2026-09-26  5:17   ` sashiko-bot
2026-09-27  3:06     ` Yonghong Song
2026-09-27 20:40   ` bot+bpf-ci
2026-09-28  0:29     ` Yonghong Song
2026-09-26  5:00 ` [PATCH bpf-next v6 09/21] bpf: Refuse a private stack for a program with an exception cleanup table Yonghong Song
2026-09-26  5:00 ` [PATCH bpf-next v6 10/21] bpf: Dispatch cleanup pads by rewriting return addresses Yonghong Song
2026-09-27 20:40   ` bot+bpf-ci
2026-09-28  1:08     ` Yonghong Song
2026-09-26  5:01 ` [PATCH bpf-next v6 11/21] bpf, x86: Dispatch exception cleanup pads at run time Yonghong Song
2026-09-26  5:15   ` sashiko-bot
2026-09-27  4:35     ` Yonghong Song
2026-09-27 20:39   ` bot+bpf-ci
2026-09-28  3:10     ` Yonghong Song
2026-09-26  5:01 ` Yonghong Song [this message]
2026-09-26  5:14   ` [PATCH bpf-next v6 12/21] bpf, arm64: " sashiko-bot
2026-09-27 20:40   ` bot+bpf-ci
2026-09-26  5:01 ` [PATCH bpf-next v6 13/21] libbpf: Resolve the compiler's _Unwind_Resume to the kernel's kfunc Yonghong Song
2026-09-26  5:01 ` [PATCH bpf-next v6 14/21] libbpf: Add cleanup_info to bpf_prog_load_opts Yonghong Song
2026-09-26  5:01 ` [PATCH bpf-next v6 15/21] libbpf: Collect .bpf_cleanup records and pass them to the kernel Yonghong Song
2026-09-27 20:39   ` bot+bpf-ci
2026-09-28  3:28     ` Yonghong Song
2026-09-26  5:01 ` [PATCH bpf-next v6 16/21] libbpf: Carry the exception cleanup table through the light skeleton Yonghong Song
2026-09-26  5:01 ` [PATCH bpf-next v6 17/21] libbpf: Let the static linker carry .bpf_cleanup relocations Yonghong Song
2026-09-26  5:01 ` [PATCH bpf-next v6 18/21] selftests/bpf: Add an end-to-end .bpf_cleanup exception test Yonghong Song
2026-09-26  5:01 ` [PATCH bpf-next v6 19/21] selftests/bpf: Add __set_global() and __ret_global() test tags Yonghong Song
2026-09-26  5:18   ` sashiko-bot
2026-09-27  4:58     ` Yonghong Song
2026-09-27 20:24   ` bot+bpf-ci
2026-09-28  3:36     ` Yonghong Song
2026-09-26  5:01 ` [PATCH bpf-next v6 20/21] selftests/bpf: Cover the exception cleanup shapes the chain does not reach Yonghong Song
2026-09-27 20:40   ` bot+bpf-ci
2026-09-28  3:49     ` Yonghong Song
2026-09-26  5:01 ` [PATCH bpf-next v6 21/21] selftests/bpf: Load an exception cleanup program from a light skeleton Yonghong Song

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260926050107.2218786-1-yonghong.song@linux.dev \
    --to=yonghong.song@linux.dev \
    --cc=andrii@kernel.org \
    --cc=ast@kernel.org \
    --cc=bpf@vger.kernel.org \
    --cc=daniel@iogearbox.net \
    --cc=eddyz87@gmail.com \
    --cc=kernel-team@fb.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox