From: Yonghong Song <yonghong.song@linux.dev>
To: bpf@vger.kernel.org
Cc: Alexei Starovoitov <ast@kernel.org>,
Andrii Nakryiko <andrii@kernel.org>,
Daniel Borkmann <daniel@iogearbox.net>,
Eduard Zingerman <eddyz87@gmail.com>,
kernel-team@fb.com
Subject: [PATCH bpf-next 11/20] bpf, x86: Dispatch exception cleanup pads at run time
Date: Wed, 16 Sep 2026 22:57:42 -0700 [thread overview]
Message-ID: <20260917055742.3932170-1-yonghong.song@linux.dev> (raw)
In-Reply-To: <20260917055645.3926444-1-yonghong.song@linux.dev>
Provide the arch half for x86-64: force the full callee-saved spill for a
program carrying a cleanup table so the walker can find a frame's r6-r9 in
its callee's prologue, record where that spill area starts, build the
native cleanup table from the JIT's addrs[], emit a bare return for a pad's
bpf_unwind_resume(), and hand control to a pad from
arch_bpf_run_cleanup_pad().
Support is gated on CONFIG_UNWINDER_ORC, the same requirement
arch_bpf_stack_walk() and therefore bpf_throw() already have here.
Signed-off-by: Yonghong Song <yonghong.song@linux.dev>
---
arch/x86/net/Makefile | 2 +-
arch/x86/net/bpf_cleanup_pad.S | 74 ++++++++++++++++++++++++++++++++
arch/x86/net/bpf_jit_comp.c | 78 ++++++++++++++++++++++++++++------
3 files changed, 141 insertions(+), 13 deletions(-)
create mode 100644 arch/x86/net/bpf_cleanup_pad.S
diff --git a/arch/x86/net/Makefile b/arch/x86/net/Makefile
index dddbefc0f439..9d574d972df3 100644
--- a/arch/x86/net/Makefile
+++ b/arch/x86/net/Makefile
@@ -6,5 +6,5 @@
ifeq ($(CONFIG_X86_32),y)
obj-$(CONFIG_BPF_JIT) += bpf_jit_comp32.o
else
- obj-$(CONFIG_BPF_JIT) += bpf_jit_comp.o bpf_timed_may_goto.o
+ obj-$(CONFIG_BPF_JIT) += bpf_jit_comp.o bpf_timed_may_goto.o bpf_cleanup_pad.o
endif
diff --git a/arch/x86/net/bpf_cleanup_pad.S b/arch/x86/net/bpf_cleanup_pad.S
new file mode 100644
index 000000000000..da4b448ecf09
--- /dev/null
+++ b/arch/x86/net/bpf_cleanup_pad.S
@@ -0,0 +1,74 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+/* Copyright (c) 2026 Meta Platforms, Inc. and affiliates. */
+
+#include <linux/bpf_cleanup_abi.h>
+#include <linux/linkage.h>
+#include <asm/nospec-branch.h>
+
+/*
+ * The x86-64 BPF JIT prologue spills, once bpf_cleanup_force_spill() makes it
+ * unconditional, r12, rbx, r13, r14 and r15 in that order -- so within the
+ * spill area the lowest address holds r15 and the highest r12. The throw-site
+ * spill the JIT emits uses the same layout, so the routine below reads both
+ * the same way:
+ *
+ * spill_base + 0 BPF r9 (r15)
+ * spill_base + 8 BPF r8 (r14)
+ * spill_base + 16 BPF r7 (r13)
+ * spill_base + 24 BPF r6 (rbx)
+ * spill_base + 32 r12 (arena base, not a BPF register)
+ */
+
+ .code64
+ .section .text, "ax"
+
+/*
+ * void arch_bpf_run_cleanup_pad(u64 pad, u64 frame_fp, u64 spill_base)
+ *
+ * rdi = native address of the landing pad
+ * rsi = frame pointer of the frame the pad belongs to
+ * rdx = spill area holding that frame's BPF callee-saved registers
+ *
+ * Give the pad the register state of its own frame and call it. The pad ends
+ * in the bare return the JIT emits for its bpf_unwind_resume(), so it comes
+ * back here rather than returning to its frame's caller. It runs on this
+ * stack, far below the frame it is cleaning up after, so nothing it calls can
+ * reach into that frame.
+ */
+SYM_FUNC_START(arch_bpf_run_cleanup_pad)
+ ANNOTATE_NOENDBR
+
+ pushq %rbp
+ pushq %rbx
+ pushq %r12
+ pushq %r13
+ pushq %r14
+ pushq %r15
+ /* Keep the pad's entry rsp congruent to a normal call's. */
+ subq $8, %rsp
+
+ movq 0(%rdx), %r15
+ movq 8(%rdx), %r14
+ movq 16(%rdx), %r13
+ movq 24(%rdx), %rbx
+ movq 32(%rdx), %r12
+ /* rbp is BPF r10, so this is the whole of the pad's frame setup. */
+ movq %rsi, %rbp
+
+ /* CALL_NOSPEC needs the target in a register; rcx is BPF r4, dead. */
+ movq %rdi, %rcx
+
+ /* BPF r0 on the way into a pad, not whatever the kernel left in rax. */
+ movl $BPF_PAD_ENTRY_R0, %eax
+
+ CALL_NOSPEC rcx
+
+ addq $8, %rsp
+ popq %r15
+ popq %r14
+ popq %r13
+ popq %r12
+ popq %rbx
+ popq %rbp
+ RET
+SYM_FUNC_END(arch_bpf_run_cleanup_pad)
diff --git a/arch/x86/net/bpf_jit_comp.c b/arch/x86/net/bpf_jit_comp.c
index d4a980140b48..9d0dd54773e8 100644
--- a/arch/x86/net/bpf_jit_comp.c
+++ b/arch/x86/net/bpf_jit_comp.c
@@ -357,6 +357,11 @@ struct jit_context {
/* Number of bytes that will be skipped on tailcall */
#define X86_TAIL_CALL_OFFSET (12 + ENDBR_INSN_SIZE)
+/* Throw-site spill: r15, r14, r13, rbx, r12 low to high, the layout the
+ * prologue's pushes leave, so arch_bpf_run_cleanup_pad() reads both alike.
+ */
+#define X86_CLEANUP_SPILL_SZ (5 * 8)
+
static void push_r9(u8 **pprog)
{
u8 *prog = *pprog;
@@ -832,7 +837,7 @@ static void emit_bpf_tail_call_indirect(struct bpf_prog *bpf_prog,
/* Inc tail_call_cnt if the slot is populated. */
EMIT4(0x48, 0x83, 0x00, 0x01); /* add qword ptr [rax], 1 */
- if (bpf_prog->aux->exception_boundary) {
+ if (bpf_prog->aux->exception_boundary || bpf_cleanup_force_spill(bpf_prog)) {
pop_callee_regs(&prog, all_callee_regs_used);
pop_r12(&prog);
} else {
@@ -899,7 +904,7 @@ static void emit_bpf_tail_call_direct(struct bpf_prog *bpf_prog,
/* Inc tail_call_cnt if the slot is populated. */
EMIT4(0x48, 0x83, 0x00, 0x01); /* add qword ptr [rax], 1 */
- if (bpf_prog->aux->exception_boundary) {
+ if (bpf_prog->aux->exception_boundary || bpf_cleanup_force_spill(bpf_prog)) {
pop_callee_regs(&prog, all_callee_regs_used);
pop_r12(&prog);
} else {
@@ -1977,6 +1982,7 @@ static int do_jit(struct bpf_verifier_env *env, struct bpf_prog *bpf_prog, int *
u8 *ip, *prog = temp;
u32 stack_depth;
int callee_saved_size;
+ u32 throw_spill, prologue_depth;
s32 outgoing_arg_base;
int err;
@@ -2015,7 +2021,10 @@ static int do_jit(struct bpf_verifier_env *env, struct bpf_prog *bpf_prog, int *
detect_reg_usage(insn, insn_cnt, callee_regs_used);
- emit_prologue(&prog, image, stack_depth,
+ throw_spill = bpf_cleanup_force_spill(bpf_prog) ? X86_CLEANUP_SPILL_SZ : 0;
+ prologue_depth = stack_depth + throw_spill;
+
+ emit_prologue(&prog, image, prologue_depth,
bpf_prog_was_classic(bpf_prog), tail_call_reachable,
bpf_is_subprog(bpf_prog), bpf_prog->aux->exception_cb);
@@ -2024,7 +2033,7 @@ static int do_jit(struct bpf_verifier_env *env, struct bpf_prog *bpf_prog, int *
/* Exception callback will clobber callee regs for its own use, and
* restore the original callee regs from main prog's stack frame.
*/
- if (bpf_prog->aux->exception_boundary) {
+ if (bpf_prog->aux->exception_boundary || bpf_cleanup_force_spill(bpf_prog)) {
/* We also need to save r12, which is not mapped to any BPF
* register, as we throw after entry into the kernel, which may
* overwrite r12.
@@ -2039,9 +2048,10 @@ static int do_jit(struct bpf_verifier_env *env, struct bpf_prog *bpf_prog, int *
/* Compute callee-saved register area size. */
callee_saved_size = 0;
- if (bpf_prog->aux->exception_boundary || arena_vm_start)
+ if (bpf_prog->aux->exception_boundary || bpf_cleanup_force_spill(bpf_prog) ||
+ arena_vm_start)
callee_saved_size += 8; /* r12 */
- if (bpf_prog->aux->exception_boundary) {
+ if (bpf_prog->aux->exception_boundary || bpf_cleanup_force_spill(bpf_prog)) {
callee_saved_size += 4 * 8; /* rbx, r13, r14, r15 */
} else {
int j;
@@ -2063,7 +2073,19 @@ static int do_jit(struct bpf_verifier_env *env, struct bpf_prog *bpf_prog, int *
* Note that tail_call_reachable is guaranteed to be false when
* stack args exist, so tcc pushes need not be accounted for.
*/
- outgoing_arg_base = -(round_up(stack_depth, 8) + callee_saved_size);
+ outgoing_arg_base = -(round_up(stack_depth, 8) + throw_spill + callee_saved_size);
+
+ /*
+ * Lowest address of each spill area, as an offset from rbp; see
+ * bpf_cleanup_pad.S for the layout. The 16 is the tail call counter
+ * pair emit_prologue_tail_call() pushes above the callee-saved one.
+ */
+ if (bpf_cleanup_force_spill(bpf_prog)) {
+ bpf_prog->aux->exc->spill_off = -(round_up(stack_depth, 8) + throw_spill +
+ (tail_call_reachable ? 16 : 0) +
+ callee_saved_size);
+ bpf_prog->aux->exc->throw_spill_off = -(round_up(stack_depth, 8) + throw_spill);
+ }
/*
* Allocate outgoing stack arg area for args 7+ only.
@@ -2110,7 +2132,8 @@ static int do_jit(struct bpf_verifier_env *env, struct bpf_prog *bpf_prog, int *
dst_reg = X86_REG_R9;
}
- if (bpf_insn_is_indirect_target(env, bpf_prog, i - 1))
+ if (bpf_insn_is_indirect_target(env, bpf_prog, i - 1) ||
+ bpf_cleanup_insn_is_pad(bpf_prog, i - 1))
EMIT_ENDBR();
ip = image + addrs[i - 1] + (prog - temp);
@@ -2903,9 +2926,27 @@ static int do_jit(struct bpf_verifier_env *env, struct bpf_prog *bpf_prog, int *
case BPF_JMP | BPF_CALL: {
const struct btf_func_model *fm = NULL;
+ if (bpf_cleanup_insn_is_throw(bpf_prog, i - 1)) {
+ /* Spill r6-r9 and r12 where the bpf_throw() walker looks. */
+ s32 off = bpf_prog->aux->exc->throw_spill_off;
+ u8 *spill = prog;
+
+ emit_stx(&prog, BPF_DW, BPF_REG_FP, BPF_REG_9, off + 0);
+ emit_stx(&prog, BPF_DW, BPF_REG_FP, BPF_REG_8, off + 8);
+ emit_stx(&prog, BPF_DW, BPF_REG_FP, BPF_REG_7, off + 16);
+ emit_stx(&prog, BPF_DW, BPF_REG_FP, BPF_REG_6, off + 24);
+ emit_stx(&prog, BPF_DW, BPF_REG_FP, X86_REG_R12, off + 32);
+ ip += prog - spill;
+ }
+
+ if (bpf_is_unwind_resume_kfunc(insn)) {
+ emit_return(&prog, image + addrs[i - 1] + (prog - temp));
+ break;
+ }
+
func = (u8 *) __bpf_call_base + imm32;
if (src_reg == BPF_PSEUDO_CALL && tail_call_reachable) {
- LOAD_TAIL_CALL_CNT_PTR(stack_depth);
+ LOAD_TAIL_CALL_CNT_PTR(prologue_depth);
ip += 7;
}
if (!imm32)
@@ -2948,13 +2989,13 @@ static int do_jit(struct bpf_verifier_env *env, struct bpf_prog *bpf_prog, int *
&prog,
ip,
callee_regs_used,
- stack_depth,
+ prologue_depth,
ctx);
else
emit_bpf_tail_call_indirect(bpf_prog,
&prog,
callee_regs_used,
- stack_depth,
+ prologue_depth,
ip,
ctx);
break;
@@ -3215,7 +3256,8 @@ static int do_jit(struct bpf_verifier_env *env, struct bpf_prog *bpf_prog, int *
}
/* Deallocate outgoing args 7+ area. */
emit_add_rsp(&prog, outgoing_rsp);
- if (bpf_prog->aux->exception_boundary) {
+ if (bpf_prog->aux->exception_boundary ||
+ bpf_cleanup_force_spill(bpf_prog)) {
pop_callee_regs(&prog, all_callee_regs_used);
pop_r12(&prog);
} else {
@@ -4385,6 +4427,13 @@ struct bpf_prog *bpf_int_jit_compile(struct bpf_verifier_env *env, struct bpf_pr
*/
bpf_prog_update_insn_ptrs(prog, addrs, image);
+ /*
+ * Same mapping, consumed by the bpf_throw() frame walker:
+ * turn the cleanup records into native address ranges now
+ * that the image is final.
+ */
+ bpf_cleanup_fill_native_ranges(prog, addrs, image);
+
/*
* ctx.prog_offset is used when CFI preambles put code *before*
* the function. See emit_cfi(). For FineIBT specifically this code
@@ -4501,6 +4550,11 @@ bool bpf_jit_supports_exceptions(void)
return IS_ENABLED(CONFIG_UNWINDER_ORC);
}
+bool bpf_jit_supports_cleanup_pads(void)
+{
+ return IS_ENABLED(CONFIG_UNWINDER_ORC);
+}
+
bool bpf_jit_supports_private_stack(void)
{
return true;
--
2.53.0-Meta
next prev parent reply other threads:[~2026-09-17 5:57 UTC|newest]
Thread overview: 40+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-17 5:56 [PATCH bpf-next 00/20] bpf: Run exception cleanup landing pads when bpf_throw() unwinds Yonghong Song
2026-09-17 5:56 ` [PATCH bpf-next 01/20] bpf: Accept the compiler's exception cleanup table at program load Yonghong Song
2026-09-17 5:56 ` [PATCH bpf-next 02/20] bpf: Add the bpf_unwind_resume() kfunc Yonghong Song
2026-09-17 5:57 ` [PATCH bpf-next 03/20] bpf: Add lookups for exception cleanup resumes and landing pads Yonghong Song
2026-09-17 5:57 ` [PATCH bpf-next 04/20] bpf: Mark the call sites an exception cleanup table covers Yonghong Song
2026-09-17 5:57 ` [PATCH bpf-next 05/20] bpf: Make exception landing pads reachable in the CFG Yonghong Song
2026-09-17 6:22 ` sashiko-bot
2026-09-18 3:36 ` Yonghong Song
2026-09-17 5:57 ` [PATCH bpf-next 06/20] bpf: Explore the landing pads no call site reaches Yonghong Song
2026-09-17 5:57 ` [PATCH bpf-next 07/20] bpf: Refuse exception cleanup shapes bpf_throw() cannot dispatch Yonghong Song
2026-09-19 4:57 ` Alexei Starovoitov
2026-09-19 17:32 ` Yonghong Song
2026-09-17 5:57 ` [PATCH bpf-next 08/20] bpf: Walk the exception unwind in the verifier Yonghong Song
2026-09-19 4:57 ` Alexei Starovoitov
2026-09-19 17:36 ` Yonghong Song
2026-09-17 5:57 ` [PATCH bpf-next 09/20] bpf: Refuse a private stack for a program with an exception cleanup table Yonghong Song
2026-09-17 5:57 ` [PATCH bpf-next 10/20] bpf: Dispatch exception cleanup pads from bpf_throw() Yonghong Song
2026-09-17 5:57 ` Yonghong Song [this message]
2026-09-19 5:02 ` [PATCH bpf-next 11/20] bpf, x86: Dispatch exception cleanup pads at run time Alexei Starovoitov
2026-09-19 19:13 ` Yonghong Song
2026-09-17 5:57 ` [PATCH bpf-next 12/20] bpf, arm64: " Yonghong Song
2026-09-17 5:57 ` [PATCH bpf-next 13/20] libbpf: Resolve the compiler's _Unwind_Resume to the kernel's kfunc Yonghong Song
2026-09-17 6:08 ` sashiko-bot
2026-09-17 7:00 ` bot+bpf-ci
2026-09-18 3:40 ` Yonghong Song
2026-09-17 5:57 ` [PATCH bpf-next 14/20] libbpf: Add cleanup_info to bpf_prog_load_opts Yonghong Song
2026-09-17 5:58 ` [PATCH bpf-next 15/20] libbpf: Collect .bpf_cleanup records and pass them to the kernel Yonghong Song
2026-09-17 6:12 ` sashiko-bot
2026-09-18 3:44 ` Yonghong Song
2026-09-17 5:58 ` [PATCH bpf-next 16/20] libbpf: Carry the exception cleanup table through the light skeleton Yonghong Song
2026-09-17 6:18 ` sashiko-bot
2026-09-18 3:52 ` Yonghong Song
2026-09-17 5:58 ` [PATCH bpf-next 17/20] libbpf: Let the static linker carry .bpf_cleanup relocations Yonghong Song
2026-09-17 6:16 ` sashiko-bot
2026-09-18 3:58 ` Yonghong Song
2026-09-17 5:58 ` [PATCH bpf-next 18/20] selftests/bpf: Add an end-to-end .bpf_cleanup exception test Yonghong Song
2026-09-17 6:12 ` sashiko-bot
2026-09-17 5:58 ` [PATCH bpf-next 19/20] selftests/bpf: Cover the exception cleanup shapes the chain does not reach Yonghong Song
2026-09-17 6:16 ` sashiko-bot
2026-09-17 5:58 ` [PATCH bpf-next 20/20] selftests/bpf: Load an exception cleanup program from a light skeleton Yonghong Song
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260917055742.3932170-1-yonghong.song@linux.dev \
--to=yonghong.song@linux.dev \
--cc=andrii@kernel.org \
--cc=ast@kernel.org \
--cc=bpf@vger.kernel.org \
--cc=daniel@iogearbox.net \
--cc=eddyz87@gmail.com \
--cc=kernel-team@fb.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox;
as well as URLs for NNTP newsgroup(s).