* [PATCH v14 1/5] powerpc64/bpf: fix compare instruction emitted for tailcall
2026-09-28 5:13 [PATCH v14 0/5] powerpc/bpf: address missing verifier selftest coverage Saket Kumar Bhaskar
@ 2026-09-28 5:13 ` Saket Kumar Bhaskar
2026-09-28 5:13 ` [PATCH v14 2/5] powerpc64/bpf: fix percpu private stack leak on JIT failure Saket Kumar Bhaskar
` (3 subsequent siblings)
4 siblings, 0 replies; 9+ messages in thread
From: Saket Kumar Bhaskar @ 2026-09-28 5:13 UTC (permalink / raw)
To: bpf, linuxppc-dev
Cc: hbathini, maddy, ast, andrii, daniel, shuah, linux-kselftest,
stable, venkat88, yeswanth, skb99
From: Abhishek Dubey <adubey@linux.ibm.com>
The tail_call_info field can contain either a scalar counter
value or a 64-bit pointer to the counter, using a 32-bit
compare (cmplwi) only checks the lower 32 bits, which can lead
to incorrect comparisions when location of counter is near 4GB
boundary. Use instruction cmpldi/cmplwi for accurate comparision
in corresponding cases.
Reported-by: sashiko-bot@kernel.org
Closes: https://lore.kernel.org/bpf/20260517191450.85AE6C2BCB8@smtp.kernel.org/
Fixes: 2ed2d8f6fb38 ("powerpc64/bpf: Support tailcalls with subprogs")
Signed-off-by: Abhishek Dubey <adubey@linux.ibm.com>
Signed-off-by: Saket Kumar Bhaskar <skb99@linux.ibm.com>
Tested-by: Yeswanth Krishna Tellakula <yeswanth@linux.ibm.com>
Tested-by: R Nageswara Sastry <rnsastry@linux.ibm.com>
Acked-by: Hari Bathini <hbathini@linux.ibm.com>
---
arch/powerpc/net/bpf_jit.h | 6 ++++++
arch/powerpc/net/bpf_jit_comp.c | 2 +-
arch/powerpc/net/bpf_jit_comp64.c | 8 ++++----
3 files changed, 11 insertions(+), 5 deletions(-)
diff --git a/arch/powerpc/net/bpf_jit.h b/arch/powerpc/net/bpf_jit.h
index f32de8704d4d..35015d7ecb76 100644
--- a/arch/powerpc/net/bpf_jit.h
+++ b/arch/powerpc/net/bpf_jit.h
@@ -188,6 +188,12 @@ struct codegen_context {
#define bpf_to_ppc(r) (ctx->b2p[r])
+#ifdef CONFIG_PPC64
+#define PPC_RAW_CMPLLI(a, i) PPC_RAW_CMPLDI(a, i)
+#else
+#define PPC_RAW_CMPLLI(a, i) PPC_RAW_CMPLWI(a, i)
+#endif
+
#ifdef CONFIG_PPC32
#define BPF_FIXUP_LEN 3 /* Three instructions => 12 bytes */
#else
diff --git a/arch/powerpc/net/bpf_jit_comp.c b/arch/powerpc/net/bpf_jit_comp.c
index 7b07b43575f1..a875521ff904 100644
--- a/arch/powerpc/net/bpf_jit_comp.c
+++ b/arch/powerpc/net/bpf_jit_comp.c
@@ -739,7 +739,7 @@ static void bpf_trampoline_setup_tail_call_info(u32 *image, struct codegen_conte
* Setting the tail_call_info in trampoline's frame
* depending on if previous frame had value or reference.
*/
- EMIT(PPC_RAW_CMPLWI(_R3, MAX_TAIL_CALL_CNT));
+ EMIT(PPC_RAW_CMPLLI(_R3, MAX_TAIL_CALL_CNT));
PPC_BCC_CONST_SHORT(COND_GT, 8);
EMIT(PPC_RAW_ADDI(_R3, _R4, -BPF_PPC_TAILCALL));
diff --git a/arch/powerpc/net/bpf_jit_comp64.c b/arch/powerpc/net/bpf_jit_comp64.c
index dab106cae22b..59302cabc466 100644
--- a/arch/powerpc/net/bpf_jit_comp64.c
+++ b/arch/powerpc/net/bpf_jit_comp64.c
@@ -276,7 +276,7 @@ void bpf_jit_build_prologue(u32 *image, struct codegen_context *ctx)
*/
EMIT(PPC_RAW_LD(bpf_to_ppc(TMP_REG_2), _R1, 0));
EMIT(PPC_RAW_LD(bpf_to_ppc(TMP_REG_1), bpf_to_ppc(TMP_REG_2), -(BPF_PPC_TAILCALL)));
- EMIT(PPC_RAW_CMPLWI(bpf_to_ppc(TMP_REG_1), MAX_TAIL_CALL_CNT));
+ EMIT(PPC_RAW_CMPLDI(bpf_to_ppc(TMP_REG_1), MAX_TAIL_CALL_CNT));
PPC_BCC_CONST_SHORT(COND_GT, 8);
EMIT(PPC_RAW_ADDI(bpf_to_ppc(TMP_REG_1), bpf_to_ppc(TMP_REG_2),
-(BPF_PPC_TAILCALL)));
@@ -662,7 +662,7 @@ static int bpf_jit_emit_tail_call(u32 *image, struct codegen_context *ctx, u32 o
PPC_BCC_SHORT(COND_GE, out);
EMIT(PPC_RAW_LD(bpf_to_ppc(TMP_REG_1), _R1, bpf_jit_stack_tailcallinfo_offset(ctx)));
- EMIT(PPC_RAW_CMPLWI(bpf_to_ppc(TMP_REG_1), MAX_TAIL_CALL_CNT));
+ EMIT(PPC_RAW_CMPLDI(bpf_to_ppc(TMP_REG_1), MAX_TAIL_CALL_CNT));
PPC_BCC_CONST_SHORT(COND_LE, 8);
/* dereference TMP_REG_1 */
@@ -672,7 +672,7 @@ static int bpf_jit_emit_tail_call(u32 *image, struct codegen_context *ctx, u32 o
* if (tail_call_info == MAX_TAIL_CALL_CNT)
* goto out;
*/
- EMIT(PPC_RAW_CMPLWI(bpf_to_ppc(TMP_REG_1), MAX_TAIL_CALL_CNT));
+ EMIT(PPC_RAW_CMPLDI(bpf_to_ppc(TMP_REG_1), MAX_TAIL_CALL_CNT));
PPC_BCC_SHORT(COND_EQ, out);
/*
@@ -707,7 +707,7 @@ static int bpf_jit_emit_tail_call(u32 *image, struct codegen_context *ctx, u32 o
* tail_call_info.
*/
EMIT(PPC_RAW_LD(bpf_to_ppc(TMP_REG_2), _R1, bpf_jit_stack_tailcallinfo_offset(ctx)));
- EMIT(PPC_RAW_CMPLWI(bpf_to_ppc(TMP_REG_2), MAX_TAIL_CALL_CNT));
+ EMIT(PPC_RAW_CMPLDI(bpf_to_ppc(TMP_REG_2), MAX_TAIL_CALL_CNT));
PPC_BCC_CONST_SHORT(COND_GT, 8);
/* First get address of tail_call_info */
--
2.55.0
^ permalink raw reply related [flat|nested] 9+ messages in thread* [PATCH v14 2/5] powerpc64/bpf: fix percpu private stack leak on JIT failure
2026-09-28 5:13 [PATCH v14 0/5] powerpc/bpf: address missing verifier selftest coverage Saket Kumar Bhaskar
2026-09-28 5:13 ` [PATCH v14 1/5] powerpc64/bpf: fix compare instruction emitted for tailcall Saket Kumar Bhaskar
@ 2026-09-28 5:13 ` Saket Kumar Bhaskar
2026-09-28 5:13 ` [PATCH v14 3/5] powerpc/bpf: fix buffer overflow in JIT for large BPF programs Saket Kumar Bhaskar
` (2 subsequent siblings)
4 siblings, 0 replies; 9+ messages in thread
From: Saket Kumar Bhaskar @ 2026-09-28 5:13 UTC (permalink / raw)
To: bpf, linuxppc-dev
Cc: hbathini, maddy, ast, andrii, daniel, shuah, linux-kselftest,
stable, venkat88, yeswanth, skb99
From: Abhishek Dubey <adubey@linux.ibm.com>
The existing conditional statement in bpf_int_jit_compile() frees the
percpu private stack at out_addrs only when the image buffer was never
allocated.
If bpf_jit_build_body() fails during a code-generation pass, the
image buffer has already been allocated, so !image is false and the
percpu stack is not freed.
Because JIT compilation failed, fp->jited remains at 0. The subsequent
bpf_jit_free() path only frees priv_stack_ptr when fp->jited is set, so
freeing is skipped here too, leaking the percpu allocation.
Fix implements freeing the private stack whenever fp->jited was not set,
i.e. compilation did not succeed, instead of keying off !image. !fp->jited
already covers the !image case, since image is only NULL on early-failure
paths where fp->jited is likewise 0.
Reported-by: sashiko-bot@kernel.org
Closes: https://lore.kernel.org/bpf/20260616135426.A06B71F000E9@smtp.kernel.org
Fixes: 156d985123b6 ("powerpc64/bpf: Implement JIT support for private stack")
Cc: stable@vger.kernel.org
Signed-off-by: Abhishek Dubey <adubey@linux.ibm.com>
Signed-off-by: Saket Kumar Bhaskar <skb99@linux.ibm.com>
Tested-by: Yeswanth Krishna Tellakula <yeswanth@linux.ibm.com>
Tested-by: R Nageswara Sastry <rnsastry@linux.ibm.com>
Acked-by: Hari Bathini <hbathini@linux.ibm.com>
---
arch/powerpc/net/bpf_jit_comp.c | 2 +-
1 file changed, 1 insertion(+), 1 deletion(-)
diff --git a/arch/powerpc/net/bpf_jit_comp.c b/arch/powerpc/net/bpf_jit_comp.c
index a875521ff904..8f7501954d9f 100644
--- a/arch/powerpc/net/bpf_jit_comp.c
+++ b/arch/powerpc/net/bpf_jit_comp.c
@@ -357,7 +357,7 @@ struct bpf_prog *bpf_int_jit_compile(struct bpf_verifier_env *env, struct bpf_pr
(void *)fimage + FUNCTION_DESCR_SIZE);
out_addrs:
- if (!image && priv_stack_ptr) {
+ if (!fp->jited && priv_stack_ptr) {
fp->aux->priv_stack_ptr = NULL;
free_percpu(priv_stack_ptr);
}
--
2.55.0
^ permalink raw reply related [flat|nested] 9+ messages in thread* [PATCH v14 3/5] powerpc/bpf: fix buffer overflow in JIT for large BPF programs
2026-09-28 5:13 [PATCH v14 0/5] powerpc/bpf: address missing verifier selftest coverage Saket Kumar Bhaskar
2026-09-28 5:13 ` [PATCH v14 1/5] powerpc64/bpf: fix compare instruction emitted for tailcall Saket Kumar Bhaskar
2026-09-28 5:13 ` [PATCH v14 2/5] powerpc64/bpf: fix percpu private stack leak on JIT failure Saket Kumar Bhaskar
@ 2026-09-28 5:13 ` Saket Kumar Bhaskar
2026-09-28 5:13 ` [PATCH v14 4/5] powerpc/bpf: fix alignment of long branch trampoline address Saket Kumar Bhaskar
2026-09-28 5:13 ` [PATCH v14 5/5] powerpc/bpf: Move out dummy_tramp_addr after Long branch stub Saket Kumar Bhaskar
4 siblings, 0 replies; 9+ messages in thread
From: Saket Kumar Bhaskar @ 2026-09-28 5:13 UTC (permalink / raw)
To: bpf, linuxppc-dev
Cc: hbathini, maddy, ast, andrii, daniel, shuah, linux-kselftest,
stable, venkat88, yeswanth, skb99
From: Abhishek Dubey <adubey@linux.ibm.com>
During size calculation in pass-0, exit_addr is 0 since addrs[fp->len]
is not yet populated. bpf_jit_emit_exit_insn() treats a zero exit_addr
as in-range and skips bpf_jit_build_epilogue(), so the alternate inline
epilogue instructions are not counted in alloclen.
In later passes, if the real exit_addr falls outside the 32MB branch
range, the full inline epilogue is emitted into the already-allocated
buffer, writing past its end and corrupting adjacent memory.
Fix by ensuring exit_addr is non-zero before treating it as in-range,
so pass-0 always falls through to bpf_jit_build_epilogue() and
conservatively accounts for all epilogue instructions in alloclen.
Also range check alt_exit_addr directly in the else-if condition.
Since exit_addr handling now falls through to the epilogue, two
related issues in bpf_int_jit_compile() must also be addressed:
1. Reset cgctx.alt_exit_addr before the second size-calculation pass.
Without this, a stale alt_exit_addr from the first pass causes the
second pass to emit a single jump instead of the full epilogue,
undercounting alloclen and reintroducing the overflow.
2. Recompute addrs[fp->len] at the end of each code-generation pass.
The larger pass-0 body can shrink in later passes as out-of-range
exits settle into in-range jumps; a stale addrs[fp->len] would
leave exit branches targeting past the real (shrunken) epilogue.
Because shrinkage in a later pass can move the epilogue offset, the
fixed two-pass loop is no longer sufficient: an exit that was out of
range in an earlier pass may fall in range once the epilogue offset
shrinks, shrinking the body further and overwriting the start of the
epilogue. Convert the code-generation loop to iterate until the
program size converges, bounded by CODEGEN_MAX_PASSES, and fail the
JIT if it does not converge.
Reported-by: sashiko-bot@kernel.org
Closes: https://lore.kernel.org/bpf/20260529015855.364704-2-adubey@linux.ibm.com/T/#mfcb23909d977b949727cca4f59ee56a13fd69b92
Fixes: 0ffdbce6f4a8 ("powerpc/bpf: Handle large branch ranges with BPF_EXIT")
Cc: stable@vger.kernel.org
Signed-off-by: Hari Bathini <hbathini@linux.ibm.com>
Signed-off-by: Abhishek Dubey <adubey@linux.ibm.com>
Signed-off-by: Saket Kumar Bhaskar <skb99@linux.ibm.com>
Link: https://lore.kernel.org/bpf/20260529015855.364704-2-adubey@linux.ibm.com/T/#mfcb23909d977b949727cca4f59ee56a13fd69b92
Tested-by: Yeswanth Krishna Tellakula <yeswanth@linux.ibm.com>
Tested-by: R Nageswara Sastry <rnsastry@linux.ibm.com>
---
arch/powerpc/net/bpf_jit.h | 7 +++++++
arch/powerpc/net/bpf_jit_comp.c | 34 +++++++++++++++++++++++++--------
2 files changed, 33 insertions(+), 8 deletions(-)
diff --git a/arch/powerpc/net/bpf_jit.h b/arch/powerpc/net/bpf_jit.h
index 35015d7ecb76..6d58df361648 100644
--- a/arch/powerpc/net/bpf_jit.h
+++ b/arch/powerpc/net/bpf_jit.h
@@ -14,6 +14,13 @@
#include <asm/ppc-opcode.h>
#include <linux/build_bug.h>
+/*
+ * We need at least 2 passes for proper code generation, and may need
+ * additional passes if code size changes between passes.
+ */
+#define CODEGEN_MIN_PASSES 2
+#define CODEGEN_MAX_PASSES 3
+
#ifdef CONFIG_PPC64_ELF_ABI_V1
#define FUNCTION_DESCR_SIZE 24
#else
diff --git a/arch/powerpc/net/bpf_jit_comp.c b/arch/powerpc/net/bpf_jit_comp.c
index 8f7501954d9f..11981d2270a9 100644
--- a/arch/powerpc/net/bpf_jit_comp.c
+++ b/arch/powerpc/net/bpf_jit_comp.c
@@ -99,11 +99,10 @@ void bpf_jit_build_fentry_stubs(u32 *image, struct codegen_context *ctx)
int bpf_jit_emit_exit_insn(u32 *image, struct codegen_context *ctx, int tmp_reg, long exit_addr)
{
- if (!exit_addr || is_offset_in_branch_range(exit_addr - (ctx->idx * 4))) {
+ if (exit_addr && is_offset_in_branch_range(exit_addr - (long)(ctx->idx * 4))) {
PPC_JMP(exit_addr);
- } else if (ctx->alt_exit_addr) {
- if (WARN_ON(!is_offset_in_branch_range((long)ctx->alt_exit_addr - (ctx->idx * 4))))
- return -1;
+ } else if (ctx->alt_exit_addr && is_offset_in_branch_range(
+ (long)(ctx->alt_exit_addr) - (long)(ctx->idx * 4))) {
PPC_JMP(ctx->alt_exit_addr);
} else {
ctx->alt_exit_addr = ctx->idx * 4;
@@ -274,6 +273,7 @@ struct bpf_prog *bpf_int_jit_compile(struct bpf_verifier_env *env, struct bpf_pr
*/
if (cgctx.seen & SEEN_TAILCALL || !is_offset_in_branch_range((long)cgctx.idx * 4)) {
cgctx.idx = 0;
+ cgctx.alt_exit_addr = 0;
if (bpf_jit_build_body(fp, NULL, NULL, &cgctx, addrs, 0, false))
goto out_err;
}
@@ -306,10 +306,13 @@ struct bpf_prog *bpf_int_jit_compile(struct bpf_verifier_env *env, struct bpf_pr
code_base = (u32 *)(image + FUNCTION_DESCR_SIZE);
fcode_base = (u32 *)(fimage + FUNCTION_DESCR_SIZE);
- /* Code generation passes 1-2 */
- for (pass = 1; pass < 3; pass++) {
+ /* Code generation passes 1-2+, loop until program size converges. */
+ for (pass = 1; pass <= CODEGEN_MAX_PASSES; pass++) {
+ u32 prev_proglen = proglen;
+
/* Now build the prologue, body code & epilogue for real. */
cgctx.idx = 0;
+ cgctx.exentry_idx = 0;
cgctx.alt_exit_addr = 0;
bpf_jit_build_prologue(code_base, &cgctx);
if (bpf_jit_build_body(fp, code_base, fcode_base, &cgctx, addrs, pass,
@@ -318,11 +321,26 @@ struct bpf_prog *bpf_int_jit_compile(struct bpf_verifier_env *env, struct bpf_pr
bpf_jit_binary_pack_free(fhdr, hdr);
goto out_err;
}
+ addrs[fp->len] = cgctx.idx * 4;
bpf_jit_build_epilogue(code_base, &cgctx);
+ proglen = cgctx.idx * 4;
+
if (bpf_jit_enable > 1)
pr_info("Pass %d: shrink = %d, seen = 0x%x\n", pass,
- proglen - (cgctx.idx * 4), cgctx.seen);
+ prev_proglen - proglen, cgctx.seen);
+
+ /* Check if program size has converged, but ensure minimum passes */
+ if (pass >= CODEGEN_MIN_PASSES && proglen == prev_proglen)
+ break;
+
+ if (pass == CODEGEN_MAX_PASSES && proglen != prev_proglen) {
+ pr_err("BPF JIT: Program did not converge after %d passes\n",
+ CODEGEN_MAX_PASSES);
+ bpf_arch_text_copy(&fhdr->size, &hdr->size, sizeof(hdr->size));
+ bpf_jit_binary_pack_free(fhdr, hdr);
+ goto out_err;
+ }
}
if (bpf_jit_enable > 1)
@@ -399,7 +417,7 @@ int bpf_add_extable_entry(struct bpf_prog *fp, u32 *image, u32 *fimage, int pass
u32 *fixup;
/* Populate extable entries only in the last pass */
- if (pass != 2)
+ if (pass < CODEGEN_MIN_PASSES)
return 0;
if (!fp->aux->extable ||
--
2.55.0
^ permalink raw reply related [flat|nested] 9+ messages in thread* [PATCH v14 4/5] powerpc/bpf: fix alignment of long branch trampoline address
2026-09-28 5:13 [PATCH v14 0/5] powerpc/bpf: address missing verifier selftest coverage Saket Kumar Bhaskar
` (2 preceding siblings ...)
2026-09-28 5:13 ` [PATCH v14 3/5] powerpc/bpf: fix buffer overflow in JIT for large BPF programs Saket Kumar Bhaskar
@ 2026-09-28 5:13 ` Saket Kumar Bhaskar
2026-09-28 5:53 ` bot+bpf-ci
2026-09-28 5:13 ` [PATCH v14 5/5] powerpc/bpf: Move out dummy_tramp_addr after Long branch stub Saket Kumar Bhaskar
4 siblings, 1 reply; 9+ messages in thread
From: Saket Kumar Bhaskar @ 2026-09-28 5:13 UTC (permalink / raw)
To: bpf, linuxppc-dev
Cc: hbathini, maddy, ast, andrii, daniel, shuah, linux-kselftest,
stable, venkat88, yeswanth, skb99
From: Abhishek Dubey <adubey@linux.ibm.com>
Ensure the dummy trampoline address field present between the OOL stub
and the long branch stub is 4/8-byte aligned, for memory compatibility
when content loaded to a register.
Reported-by: Hari Bathini <hbathini@linux.ibm.com>
Fixes: d243b62b7bd3 ("powerpc64/bpf: Add support for bpf trampolines")
Cc: stable@vger.kernel.org
Signed-off-by: Abhishek Dubey <adubey@linux.ibm.com>
Signed-off-by: Saket Kumar Bhaskar <skb99@linux.ibm.com>
Tested-by: Yeswanth Krishna Tellakula <yeswanth@linux.ibm.com>
Tested-by: R Nageswara Sastry <rnsastry@linux.ibm.com>
Reviewed-by: Hari Bathini <hbathini@linux.ibm.com>
---
arch/powerpc/net/bpf_jit.h | 7 +++---
arch/powerpc/net/bpf_jit_comp.c | 38 ++++++++++++++++++++++++++-----
arch/powerpc/net/bpf_jit_comp32.c | 7 +++---
arch/powerpc/net/bpf_jit_comp64.c | 7 +++---
4 files changed, 44 insertions(+), 15 deletions(-)
diff --git a/arch/powerpc/net/bpf_jit.h b/arch/powerpc/net/bpf_jit.h
index 6d58df361648..4da8bde92e1e 100644
--- a/arch/powerpc/net/bpf_jit.h
+++ b/arch/powerpc/net/bpf_jit.h
@@ -227,10 +227,11 @@ int bpf_jit_emit_func_call_rel(u32 *image, u32 *fimage, struct codegen_context *
int bpf_jit_build_body(struct bpf_prog *fp, u32 *image, u32 *fimage, struct codegen_context *ctx,
u32 *addrs, int pass, bool extra_pass);
void bpf_jit_build_prologue(u32 *image, struct codegen_context *ctx);
-void bpf_jit_build_epilogue(u32 *image, struct codegen_context *ctx);
-void bpf_jit_build_fentry_stubs(u32 *image, struct codegen_context *ctx);
+void bpf_jit_build_epilogue(u32 *image, u32 *fimage, struct codegen_context *ctx);
+void bpf_jit_build_fentry_stubs(u32 *image, u32 *fimage, struct codegen_context *ctx);
void bpf_jit_realloc_regs(struct codegen_context *ctx);
-int bpf_jit_emit_exit_insn(u32 *image, struct codegen_context *ctx, int tmp_reg, long exit_addr);
+int bpf_jit_emit_exit_insn(u32 *image, u32 *fimage, struct codegen_context *ctx, int tmp_reg,
+ long exit_addr);
void prepare_for_fsession_fentry(u32 *image, struct codegen_context *ctx, int cookie_cnt,
int cookie_off, int retval_off);
void store_func_meta(u32 *image, struct codegen_context *ctx, u64 func_meta, int func_meta_off);
diff --git a/arch/powerpc/net/bpf_jit_comp.c b/arch/powerpc/net/bpf_jit_comp.c
index 11981d2270a9..8ca36a933c7a 100644
--- a/arch/powerpc/net/bpf_jit_comp.c
+++ b/arch/powerpc/net/bpf_jit_comp.c
@@ -49,11 +49,35 @@ asm (
" .popsection ;"
);
-void bpf_jit_build_fentry_stubs(u32 *image, struct codegen_context *ctx)
+void bpf_jit_build_fentry_stubs(u32 *image, u32 *fimage, struct codegen_context *ctx)
{
int ool_stub_idx, long_branch_stub_idx;
+ int ool_stub_sz;
/*
+ * Align the mis-aligned dummy_tramp_addr field in the fimage.
+ * The alignment NOP must appear before OOL stub, to make
+ * ool_stub_idx & long_branch_stub_idx constant from end.
+ *
+ * The fimage can be non 8-byte aligned, so final alignment depends
+ * on start of fimage and the stub's instruction count offset. The
+ * OOL stub size is 4 instructions (with CONFIG_PPC_FTRACE_OUT_OF_LINE)
+ * or 3 instructions (without) before dummy_tramp_addr.
+ *
+ * Emit a NOP here if address is not SZL aligned.
+ *
+ * In pass=0 when image==NULL, conservatively account for space
+ * required to accommodate alignment NOP. In case final pass skips
+ * emitting alignment NOP, the image buffer have 4 spare bytes and
+ * jited_len signifies correct program size.
+ */
+
+ ool_stub_sz = IS_ENABLED(CONFIG_PPC_FTRACE_OUT_OF_LINE) ? 16 : 12;
+ if (!image || !IS_ALIGNED((unsigned long)fimage + ctx->idx*4 + ool_stub_sz, SZL))
+ EMIT(PPC_RAW_NOP());
+
+ /*
+ * nop // optional, for alignment of dummy_tramp_addr
* Out-of-line stub:
* mflr r0
* [b|bl] tramp
@@ -70,7 +94,7 @@ void bpf_jit_build_fentry_stubs(u32 *image, struct codegen_context *ctx)
/*
* Long branch stub:
- * .long <dummy_tramp_addr>
+ * .long <dummy_tramp_addr> // 8-byte aligned
* mflr r11
* bcl 20,31,$+4
* mflr r12
@@ -81,6 +105,7 @@ void bpf_jit_build_fentry_stubs(u32 *image, struct codegen_context *ctx)
*/
if (image)
*((unsigned long *)&image[ctx->idx]) = (unsigned long)dummy_tramp;
+
ctx->idx += SZL / 4;
long_branch_stub_idx = ctx->idx;
EMIT(PPC_RAW_MFLR(_R11));
@@ -97,7 +122,8 @@ void bpf_jit_build_fentry_stubs(u32 *image, struct codegen_context *ctx)
}
}
-int bpf_jit_emit_exit_insn(u32 *image, struct codegen_context *ctx, int tmp_reg, long exit_addr)
+int bpf_jit_emit_exit_insn(u32 *image, u32 *fimage, struct codegen_context *ctx,
+ int tmp_reg, long exit_addr)
{
if (exit_addr && is_offset_in_branch_range(exit_addr - (long)(ctx->idx * 4))) {
PPC_JMP(exit_addr);
@@ -106,7 +132,7 @@ int bpf_jit_emit_exit_insn(u32 *image, struct codegen_context *ctx, int tmp_reg,
PPC_JMP(ctx->alt_exit_addr);
} else {
ctx->alt_exit_addr = ctx->idx * 4;
- bpf_jit_build_epilogue(image, ctx);
+ bpf_jit_build_epilogue(image, fimage, ctx);
}
return 0;
@@ -286,7 +312,7 @@ struct bpf_prog *bpf_int_jit_compile(struct bpf_verifier_env *env, struct bpf_pr
*/
bpf_jit_build_prologue(NULL, &cgctx);
addrs[fp->len] = cgctx.idx * 4;
- bpf_jit_build_epilogue(NULL, &cgctx);
+ bpf_jit_build_epilogue(NULL, NULL, &cgctx);
fixup_len = fp->aux->num_exentries * BPF_FIXUP_LEN * 4;
extable_len = fp->aux->num_exentries * sizeof(struct exception_table_entry);
@@ -322,7 +348,7 @@ struct bpf_prog *bpf_int_jit_compile(struct bpf_verifier_env *env, struct bpf_pr
goto out_err;
}
addrs[fp->len] = cgctx.idx * 4;
- bpf_jit_build_epilogue(code_base, &cgctx);
+ bpf_jit_build_epilogue(code_base, fcode_base, &cgctx);
proglen = cgctx.idx * 4;
diff --git a/arch/powerpc/net/bpf_jit_comp32.c b/arch/powerpc/net/bpf_jit_comp32.c
index bfdc50740da8..f5b9441cf46a 100644
--- a/arch/powerpc/net/bpf_jit_comp32.c
+++ b/arch/powerpc/net/bpf_jit_comp32.c
@@ -229,7 +229,7 @@ static void bpf_jit_emit_common_epilogue(u32 *image, struct codegen_context *ctx
}
-void bpf_jit_build_epilogue(u32 *image, struct codegen_context *ctx)
+void bpf_jit_build_epilogue(u32 *image, u32 *fimage, struct codegen_context *ctx)
{
EMIT(PPC_RAW_MR(_R3, bpf_to_ppc(BPF_REG_0)));
@@ -237,7 +237,7 @@ void bpf_jit_build_epilogue(u32 *image, struct codegen_context *ctx)
EMIT(PPC_RAW_BLR());
- bpf_jit_build_fentry_stubs(image, ctx);
+ bpf_jit_build_fentry_stubs(image, fimage, ctx);
}
/* Relative offset needs to be calculated based on final image location */
@@ -1149,7 +1149,8 @@ int bpf_jit_build_body(struct bpf_prog *fp, u32 *image, u32 *fimage, struct code
* we'll just fall through to the epilogue.
*/
if (i != flen - 1) {
- ret = bpf_jit_emit_exit_insn(image, ctx, _R0, exit_addr);
+ ret = bpf_jit_emit_exit_insn(image, fimage,
+ ctx, _R0, exit_addr);
if (ret)
return ret;
}
diff --git a/arch/powerpc/net/bpf_jit_comp64.c b/arch/powerpc/net/bpf_jit_comp64.c
index 59302cabc466..e80312171aa7 100644
--- a/arch/powerpc/net/bpf_jit_comp64.c
+++ b/arch/powerpc/net/bpf_jit_comp64.c
@@ -398,7 +398,7 @@ static void bpf_jit_emit_common_epilogue(u32 *image, struct codegen_context *ctx
}
}
-void bpf_jit_build_epilogue(u32 *image, struct codegen_context *ctx)
+void bpf_jit_build_epilogue(u32 *image, u32 *fimage, struct codegen_context *ctx)
{
bpf_jit_emit_common_epilogue(image, ctx);
@@ -407,7 +407,7 @@ void bpf_jit_build_epilogue(u32 *image, struct codegen_context *ctx)
EMIT(PPC_RAW_BLR());
- bpf_jit_build_fentry_stubs(image, ctx);
+ bpf_jit_build_fentry_stubs(image, fimage, ctx);
}
/*
@@ -1737,7 +1737,8 @@ int bpf_jit_build_body(struct bpf_prog *fp, u32 *image, u32 *fimage, struct code
* we'll just fall through to the epilogue.
*/
if (i != flen - 1) {
- ret = bpf_jit_emit_exit_insn(image, ctx, tmp1_reg, exit_addr);
+ ret = bpf_jit_emit_exit_insn(image, fimage, ctx,
+ tmp1_reg, exit_addr);
if (ret)
return ret;
}
--
2.55.0
^ permalink raw reply related [flat|nested] 9+ messages in thread* Re: [PATCH v14 4/5] powerpc/bpf: fix alignment of long branch trampoline address
2026-09-28 5:13 ` [PATCH v14 4/5] powerpc/bpf: fix alignment of long branch trampoline address Saket Kumar Bhaskar
@ 2026-09-28 5:53 ` bot+bpf-ci
0 siblings, 0 replies; 9+ messages in thread
From: bot+bpf-ci @ 2026-09-28 5:53 UTC (permalink / raw)
To: skb99, bpf, linuxppc-dev
Cc: hbathini, maddy, ast, andrii, daniel, shuah, linux-kselftest,
stable, venkat88, yeswanth, skb99, ast, andrii, daniel,
martin.lau, eddyz87, yonghong.song, mason, ihor.solodrai
[-- Attachment #1: Type: text/plain, Size: 2107 bytes --]
> commit 6ef500644d23a07941d6ec45def92b9391c03a10
> Author: Abhishek Dubey <adubey@linux.ibm.com>
> Date: Mon Sep 28 10:43:53 2026 +0530
>
> powerpc/bpf: fix alignment of long branch trampoline address
>
> Ensure the dummy trampoline address field present between the OOL stub
> and the long branch stub is 4/8-byte aligned, for memory compatibility
> when content loaded to a register.
>
> Reported-by: Hari Bathini <hbathini@linux.ibm.com>
> Fixes: d243b62b7bd3 ("powerpc64/bpf: Add support for bpf trampolines")
> Cc: stable@vger.kernel.org
> Signed-off-by: Abhishek Dubey <adubey@linux.ibm.com>
> Signed-off-by: Saket Kumar Bhaskar <skb99@linux.ibm.com>
> Tested-by: Yeswanth Krishna Tellakula <yeswanth@linux.ibm.com>
> Tested-by: R Nageswara Sastry <rnsastry@linux.ibm.com>
> Reviewed-by: Hari Bathini <hbathini@linux.ibm.com>
This isn't a bug, but could the changelog say what fails when the slot
is only 4-byte aligned, and briefly explain why sizing with an
unconditional NOP and then conditionally dropping it in the final pass
is safe?
The load-bearing rationale seems to be missing. From the code,
bpf_jit_binary_pack_alloc() only guarantees 4-byte alignment of the
final image, so the SZL-sized slot read by the long branch stub and
rewritten by patch_ulong() from bpf_arch_text_poke() may be only 4-byte
aligned on ppc64, where an 8-byte access is then not guaranteed to be
single-copy atomic against a concurrent patch.
The non-obvious part of the fix is that the NOP is emitted
unconditionally in the sizing pass when image is NULL and conditionally
on the final fimage address in the codegen passes, so the allocated
buffer may have 4 spare bytes while jited_len reflects the actual size.
The phrase "4/8-byte aligned" also blurs that only the 64-bit case needs
the NOP since SZL is 4 on ppc32.
---
AI reviewed your patch. Please fix the bug or email reply why it's not a bug.
See: https://github.com/kernel-patches/vmtest/blob/master/ci/claude/README.md
CI run summary: https://github.com/kernel-patches/bpf/actions/runs/36381917401
^ permalink raw reply [flat|nested] 9+ messages in thread
* [PATCH v14 5/5] powerpc/bpf: Move out dummy_tramp_addr after Long branch stub
2026-09-28 5:13 [PATCH v14 0/5] powerpc/bpf: address missing verifier selftest coverage Saket Kumar Bhaskar
` (3 preceding siblings ...)
2026-09-28 5:13 ` [PATCH v14 4/5] powerpc/bpf: fix alignment of long branch trampoline address Saket Kumar Bhaskar
@ 2026-09-28 5:13 ` Saket Kumar Bhaskar
2026-09-28 5:53 ` bot+bpf-ci
2026-09-28 6:51 ` Hari Bathini
4 siblings, 2 replies; 9+ messages in thread
From: Saket Kumar Bhaskar @ 2026-09-28 5:13 UTC (permalink / raw)
To: bpf, linuxppc-dev
Cc: hbathini, maddy, ast, andrii, daniel, shuah, linux-kselftest,
stable, venkat88, yeswanth, skb99
From: Abhishek Dubey <adubey@linux.ibm.com>
Move the long branch address field to the bottom of the long
branch stub. This allows uninterrupted disassembly until the
last 4/8 bytes. The last bytes exclusion is logically necessary to
prevent disassembly failure, otherwise the actual program layout
is never altered. Hence no effect on overall program size.
Also, align dummy_tramp_addr field with 8-byte boundary.
Following is disassembler output for test program with moved down
dummy_tramp_addr field:
.....
.....
pc:68 left:44 a6 03 08 7c : mtlr 0
pc:72 left:40 bc ff ff 4b : b .-68
pc:76 left:36 a6 02 68 7d : mflr 11
pc:80 left:32 05 00 9f 42 : bcl 20, 31, .+4
pc:84 left:28 a6 02 88 7d : mflr 12
pc:88 left:24 14 00 8c e9 : ld 12, 20(12)
pc:92 left:20 a6 03 89 7d : mtctr 12
pc:96 left:16 a6 03 68 7d : mtlr 11
pc:100 left:12 20 04 80 4e : bctr
pc:104 left:8 c0 34 1d 00 :
Failure log:
Can't disasm instruction at offset 104: c0 34 1d 00 00 00 00 c0
Disassembly logic can truncate at 104, ignoring last 8 bytes.
Update the dummy_tramp_addr field offset calculation from the end
of the program to reflect its new location, for bpf_arch_text_poke()
to update the actual trampoline's address in this field.
All BPF trampoline selftests continue to pass with this patch applied.
Fixes: d243b62b7bd3 ("powerpc64/bpf: Add support for bpf trampolines")
Signed-off-by: Abhishek Dubey <adubey@linux.ibm.com>
Signed-off-by: Saket Kumar Bhaskar <skb99@linux.ibm.com>
Tested-by: Yeswanth Krishna Tellakula <yeswanth@linux.ibm.com>
Tested-by: R Nageswara Sastry <rnsastry@linux.ibm.com>
---
arch/powerpc/net/bpf_jit_comp.c | 44 ++++++++++++++++++---------------
1 file changed, 24 insertions(+), 20 deletions(-)
diff --git a/arch/powerpc/net/bpf_jit_comp.c b/arch/powerpc/net/bpf_jit_comp.c
index 8ca36a933c7a..c2717f0d9cdd 100644
--- a/arch/powerpc/net/bpf_jit_comp.c
+++ b/arch/powerpc/net/bpf_jit_comp.c
@@ -52,19 +52,19 @@ asm (
void bpf_jit_build_fentry_stubs(u32 *image, u32 *fimage, struct codegen_context *ctx)
{
int ool_stub_idx, long_branch_stub_idx;
- int ool_stub_sz;
+ int stub_sz;
/*
+ * The dummy_tramp_addr field is placed at bottom of Long branch stub.
* Align the mis-aligned dummy_tramp_addr field in the fimage.
* The alignment NOP must appear before OOL stub, to make
* ool_stub_idx & long_branch_stub_idx constant from end.
*
* The fimage can be non 8-byte aligned, so final alignment depends
- * on start of fimage and the stub's instruction count offset. The
- * OOL stub size is 4 instructions (with CONFIG_PPC_FTRACE_OUT_OF_LINE)
- * or 3 instructions (without) before dummy_tramp_addr.
- *
- * Emit a NOP here if address is not SZL aligned.
+ * on start of fimage and the stub's instruction count. The
+ * stubs block has 11 instructions (with CONFIG_PPC_FTRACE_OUT_OF_LINE)
+ * or 10 instructions (without) before dummy_tramp_addr field. Emit a
+ * NOP if the address of dummy_tramp_addr is non aligned.
*
* In pass=0 when image==NULL, conservatively account for space
* required to accommodate alignment NOP. In case final pass skips
@@ -72,8 +72,8 @@ void bpf_jit_build_fentry_stubs(u32 *image, u32 *fimage, struct codegen_context
* jited_len signifies correct program size.
*/
- ool_stub_sz = IS_ENABLED(CONFIG_PPC_FTRACE_OUT_OF_LINE) ? 16 : 12;
- if (!image || !IS_ALIGNED((unsigned long)fimage + ctx->idx*4 + ool_stub_sz, SZL))
+ stub_sz = IS_ENABLED(CONFIG_PPC_FTRACE_OUT_OF_LINE) ? 44 : 40;
+ if (!image || !IS_ALIGNED((unsigned long)fimage + ctx->idx*4 + stub_sz, SZL))
EMIT(PPC_RAW_NOP());
/*
@@ -94,28 +94,29 @@ void bpf_jit_build_fentry_stubs(u32 *image, u32 *fimage, struct codegen_context
/*
* Long branch stub:
- * .long <dummy_tramp_addr> // 8-byte aligned
* mflr r11
* bcl 20,31,$+4
- * mflr r12
- * ld r12, -8-SZL(r12)
+ * mflr r12 // lr/r12 stores pc of current(this) inst.
+ * ld r12, 20(r12) // offset(dummy_tramp_addr) from prev inst. is 20
* mtctr r12
- * mtlr r11 // needed to retain ftrace ABI
+ * mtlr r11 // needed to retain ftrace ABI
* bctr
+ * .long <dummy_tramp_addr> // SZL bytes aligned
*/
- if (image)
- *((unsigned long *)&image[ctx->idx]) = (unsigned long)dummy_tramp;
-
- ctx->idx += SZL / 4;
long_branch_stub_idx = ctx->idx;
EMIT(PPC_RAW_MFLR(_R11));
EMIT(PPC_RAW_BCL4());
EMIT(PPC_RAW_MFLR(_R12));
- EMIT(PPC_RAW_LL(_R12, _R12, -8-SZL));
+ EMIT(PPC_RAW_LL(_R12, _R12, 20));
EMIT(PPC_RAW_MTCTR(_R12));
EMIT(PPC_RAW_MTLR(_R11));
EMIT(PPC_RAW_BCTR());
+ if (image)
+ *((unsigned long *)&image[ctx->idx]) = (unsigned long)dummy_tramp;
+
+ ctx->idx += SZL / 4;
+
if (!bpf_jit_ool_stub) {
bpf_jit_ool_stub = (ctx->idx - ool_stub_idx) * 4;
bpf_jit_long_branch_stub = (ctx->idx - long_branch_stub_idx) * 4;
@@ -1309,6 +1310,7 @@ static void do_isync(void *info __maybe_unused)
* bpf_func:
* [nop|b] ool_stub
* 2. Out-of-line stub:
+ * nop // optional nop for alignment
* ool_stub:
* mflr r0
* [b|bl] <bpf_prog>/<long_branch_stub>
@@ -1316,14 +1318,14 @@ static void do_isync(void *info __maybe_unused)
* b bpf_func + 4
* 3. Long branch stub:
* long_branch_stub:
- * .long <branch_addr>/<dummy_tramp>
* mflr r11
* bcl 20,31,$+4
* mflr r12
- * ld r12, -16(r12)
+ * ld r12, 20(r12)
* mtctr r12
* mtlr r11 // needed to retain ftrace ABI
* bctr
+ * .long <branch_addr>/<dummy_tramp>
*
* dummy_tramp is used to reduce synchronization requirements.
*
@@ -1425,10 +1427,12 @@ int bpf_arch_text_poke(void *ip, enum bpf_text_poke_type old_t,
* 1. Update the address in the long branch stub:
* If new_addr is out of range, we will have to use the long branch stub, so patch new_addr
* here. Otherwise, revert to dummy_tramp, but only if we had patched old_addr here.
+ *
+ * dummy_tramp_addr moved to bottom of long branch stub.
*/
if ((new_addr && !is_offset_in_branch_range(new_addr - ip)) ||
(old_addr && !is_offset_in_branch_range(old_addr - ip)))
- ret = patch_ulong((void *)(bpf_func_end - bpf_jit_long_branch_stub - SZL),
+ ret = patch_ulong((void *)(bpf_func_end - SZL), /* SZL: dummy_tramp_addr offset */
(new_addr && !is_offset_in_branch_range(new_addr - ip)) ?
(unsigned long)new_addr : (unsigned long)dummy_tramp);
if (ret)
--
2.55.0
^ permalink raw reply related [flat|nested] 9+ messages in thread* Re: [PATCH v14 5/5] powerpc/bpf: Move out dummy_tramp_addr after Long branch stub
2026-09-28 5:13 ` [PATCH v14 5/5] powerpc/bpf: Move out dummy_tramp_addr after Long branch stub Saket Kumar Bhaskar
@ 2026-09-28 5:53 ` bot+bpf-ci
2026-09-28 6:51 ` Hari Bathini
1 sibling, 0 replies; 9+ messages in thread
From: bot+bpf-ci @ 2026-09-28 5:53 UTC (permalink / raw)
To: skb99, bpf, linuxppc-dev
Cc: hbathini, maddy, ast, andrii, daniel, shuah, linux-kselftest,
stable, venkat88, yeswanth, skb99, ast, andrii, daniel,
martin.lau, eddyz87, yonghong.song, mason, ihor.solodrai
[-- Attachment #1: Type: text/plain, Size: 2103 bytes --]
> diff --git a/arch/powerpc/net/bpf_jit_comp.c b/arch/powerpc/net/bpf_jit_comp.c
> index 8ca36a933c7a..c2717f0d9cdd 100644
> --- a/arch/powerpc/net/bpf_jit_comp.c
> +++ b/arch/powerpc/net/bpf_jit_comp.c
[ ... ]
> @@ -1425,10 +1427,12 @@ int bpf_arch_text_poke(void *ip, enum bpf_text_poke_type old_t,
> * 1. Update the address in the long branch stub:
> * If new_addr is out of range, we will have to use the long branch stub, so patch new_addr
> * here. Otherwise, revert to dummy_tramp, but only if we had patched old_addr here.
> + *
> + * dummy_tramp_addr moved to bottom of long branch stub.
> */
This isn't a bug, but the comment "dummy_tramp_addr moved to bottom of
long branch stub" describes how the layout changed rather than what it
is now. Once merged, readers have nothing "before" to compare against.
The updated diagram above bpf_arch_text_poke() already shows the
.long <branch_addr>/<dummy_tramp> at the end of the long branch stub.
Could the "moved to bottom" note go into the changelog, with the comment
instead saying that dummy_tramp_addr is the last SZL bytes of the JITed
image (so it lives at bpf_func_end - SZL)?
> if ((new_addr && !is_offset_in_branch_range(new_addr - ip)) ||
> (old_addr && !is_offset_in_branch_range(old_addr - ip)))
> - ret = patch_ulong((void *)(bpf_func_end - bpf_jit_long_branch_stub - SZL),
> + ret = patch_ulong((void *)(bpf_func_end - SZL), /* SZL: dummy_tramp_addr offset */
> (new_addr && !is_offset_in_branch_range(new_addr - ip)) ?
> (unsigned long)new_addr : (unsigned long)dummy_tramp);
The trailing comment "/* SZL: dummy_tramp_addr offset */" could also be
clearer. SZL is the size of the field, and the field sits SZL bytes
before the end of the image. The combination of the expression
bpf_func_end - SZL and the updated diagram already communicate this.
---
AI reviewed your patch. Please fix the bug or email reply why it's not a bug.
See: https://github.com/kernel-patches/vmtest/blob/master/ci/claude/README.md
CI run summary: https://github.com/kernel-patches/bpf/actions/runs/36381917401
^ permalink raw reply [flat|nested] 9+ messages in thread
* Re: [PATCH v14 5/5] powerpc/bpf: Move out dummy_tramp_addr after Long branch stub
2026-09-28 5:13 ` [PATCH v14 5/5] powerpc/bpf: Move out dummy_tramp_addr after Long branch stub Saket Kumar Bhaskar
2026-09-28 5:53 ` bot+bpf-ci
@ 2026-09-28 6:51 ` Hari Bathini
1 sibling, 0 replies; 9+ messages in thread
From: Hari Bathini @ 2026-09-28 6:51 UTC (permalink / raw)
To: Saket Kumar Bhaskar, bpf, linuxppc-dev
Cc: maddy, ast, andrii, daniel, shuah, linux-kselftest, stable,
venkat88, yeswanth
On 28/09/26 10:43 am, Saket Kumar Bhaskar wrote:
> From: Abhishek Dubey <adubey@linux.ibm.com>
>
> Move the long branch address field to the bottom of the long
> branch stub. This allows uninterrupted disassembly until the
> last 4/8 bytes. The last bytes exclusion is logically necessary to
> prevent disassembly failure, otherwise the actual program layout
> is never altered. Hence no effect on overall program size.
> Also, align dummy_tramp_addr field with 8-byte boundary.
>
> Following is disassembler output for test program with moved down
> dummy_tramp_addr field:
> .....
> .....
> pc:68 left:44 a6 03 08 7c : mtlr 0
> pc:72 left:40 bc ff ff 4b : b .-68
> pc:76 left:36 a6 02 68 7d : mflr 11
> pc:80 left:32 05 00 9f 42 : bcl 20, 31, .+4
> pc:84 left:28 a6 02 88 7d : mflr 12
> pc:88 left:24 14 00 8c e9 : ld 12, 20(12)
> pc:92 left:20 a6 03 89 7d : mtctr 12
> pc:96 left:16 a6 03 68 7d : mtlr 11
> pc:100 left:12 20 04 80 4e : bctr
> pc:104 left:8 c0 34 1d 00 :
>
> Failure log:
> Can't disasm instruction at offset 104: c0 34 1d 00 00 00 00 c0
> Disassembly logic can truncate at 104, ignoring last 8 bytes.
>
> Update the dummy_tramp_addr field offset calculation from the end
> of the program to reflect its new location, for bpf_arch_text_poke()
> to update the actual trampoline's address in this field.
>
> All BPF trampoline selftests continue to pass with this patch applied.
>
Looks good to me.
Acked-by: Hari Bathini <hbathini@linux.ibm.com>
> Fixes: d243b62b7bd3 ("powerpc64/bpf: Add support for bpf trampolines")
> Signed-off-by: Abhishek Dubey <adubey@linux.ibm.com>
> Signed-off-by: Saket Kumar Bhaskar <skb99@linux.ibm.com>
> Tested-by: Yeswanth Krishna Tellakula <yeswanth@linux.ibm.com>
> Tested-by: R Nageswara Sastry <rnsastry@linux.ibm.com>
> ---
> arch/powerpc/net/bpf_jit_comp.c | 44 ++++++++++++++++++---------------
> 1 file changed, 24 insertions(+), 20 deletions(-)
>
> diff --git a/arch/powerpc/net/bpf_jit_comp.c b/arch/powerpc/net/bpf_jit_comp.c
> index 8ca36a933c7a..c2717f0d9cdd 100644
> --- a/arch/powerpc/net/bpf_jit_comp.c
> +++ b/arch/powerpc/net/bpf_jit_comp.c
> @@ -52,19 +52,19 @@ asm (
> void bpf_jit_build_fentry_stubs(u32 *image, u32 *fimage, struct codegen_context *ctx)
> {
> int ool_stub_idx, long_branch_stub_idx;
> - int ool_stub_sz;
> + int stub_sz;
>
> /*
> + * The dummy_tramp_addr field is placed at bottom of Long branch stub.
> * Align the mis-aligned dummy_tramp_addr field in the fimage.
> * The alignment NOP must appear before OOL stub, to make
> * ool_stub_idx & long_branch_stub_idx constant from end.
> *
> * The fimage can be non 8-byte aligned, so final alignment depends
> - * on start of fimage and the stub's instruction count offset. The
> - * OOL stub size is 4 instructions (with CONFIG_PPC_FTRACE_OUT_OF_LINE)
> - * or 3 instructions (without) before dummy_tramp_addr.
> - *
> - * Emit a NOP here if address is not SZL aligned.
> + * on start of fimage and the stub's instruction count. The
> + * stubs block has 11 instructions (with CONFIG_PPC_FTRACE_OUT_OF_LINE)
> + * or 10 instructions (without) before dummy_tramp_addr field. Emit a
> + * NOP if the address of dummy_tramp_addr is non aligned.
> *
> * In pass=0 when image==NULL, conservatively account for space
> * required to accommodate alignment NOP. In case final pass skips
> @@ -72,8 +72,8 @@ void bpf_jit_build_fentry_stubs(u32 *image, u32 *fimage, struct codegen_context
> * jited_len signifies correct program size.
> */
>
> - ool_stub_sz = IS_ENABLED(CONFIG_PPC_FTRACE_OUT_OF_LINE) ? 16 : 12;
> - if (!image || !IS_ALIGNED((unsigned long)fimage + ctx->idx*4 + ool_stub_sz, SZL))
> + stub_sz = IS_ENABLED(CONFIG_PPC_FTRACE_OUT_OF_LINE) ? 44 : 40;
> + if (!image || !IS_ALIGNED((unsigned long)fimage + ctx->idx*4 + stub_sz, SZL))
> EMIT(PPC_RAW_NOP());
>
> /*
> @@ -94,28 +94,29 @@ void bpf_jit_build_fentry_stubs(u32 *image, u32 *fimage, struct codegen_context
>
> /*
> * Long branch stub:
> - * .long <dummy_tramp_addr> // 8-byte aligned
> * mflr r11
> * bcl 20,31,$+4
> - * mflr r12
> - * ld r12, -8-SZL(r12)
> + * mflr r12 // lr/r12 stores pc of current(this) inst.
> + * ld r12, 20(r12) // offset(dummy_tramp_addr) from prev inst. is 20
> * mtctr r12
> - * mtlr r11 // needed to retain ftrace ABI
> + * mtlr r11 // needed to retain ftrace ABI
> * bctr
> + * .long <dummy_tramp_addr> // SZL bytes aligned
> */
> - if (image)
> - *((unsigned long *)&image[ctx->idx]) = (unsigned long)dummy_tramp;
> -
> - ctx->idx += SZL / 4;
> long_branch_stub_idx = ctx->idx;
> EMIT(PPC_RAW_MFLR(_R11));
> EMIT(PPC_RAW_BCL4());
> EMIT(PPC_RAW_MFLR(_R12));
> - EMIT(PPC_RAW_LL(_R12, _R12, -8-SZL));
> + EMIT(PPC_RAW_LL(_R12, _R12, 20));
> EMIT(PPC_RAW_MTCTR(_R12));
> EMIT(PPC_RAW_MTLR(_R11));
> EMIT(PPC_RAW_BCTR());
>
> + if (image)
> + *((unsigned long *)&image[ctx->idx]) = (unsigned long)dummy_tramp;
> +
> + ctx->idx += SZL / 4;
> +
> if (!bpf_jit_ool_stub) {
> bpf_jit_ool_stub = (ctx->idx - ool_stub_idx) * 4;
> bpf_jit_long_branch_stub = (ctx->idx - long_branch_stub_idx) * 4;
> @@ -1309,6 +1310,7 @@ static void do_isync(void *info __maybe_unused)
> * bpf_func:
> * [nop|b] ool_stub
> * 2. Out-of-line stub:
> + * nop // optional nop for alignment
> * ool_stub:
> * mflr r0
> * [b|bl] <bpf_prog>/<long_branch_stub>
> @@ -1316,14 +1318,14 @@ static void do_isync(void *info __maybe_unused)
> * b bpf_func + 4
> * 3. Long branch stub:
> * long_branch_stub:
> - * .long <branch_addr>/<dummy_tramp>
> * mflr r11
> * bcl 20,31,$+4
> * mflr r12
> - * ld r12, -16(r12)
> + * ld r12, 20(r12)
> * mtctr r12
> * mtlr r11 // needed to retain ftrace ABI
> * bctr
> + * .long <branch_addr>/<dummy_tramp>
> *
> * dummy_tramp is used to reduce synchronization requirements.
> *
> @@ -1425,10 +1427,12 @@ int bpf_arch_text_poke(void *ip, enum bpf_text_poke_type old_t,
> * 1. Update the address in the long branch stub:
> * If new_addr is out of range, we will have to use the long branch stub, so patch new_addr
> * here. Otherwise, revert to dummy_tramp, but only if we had patched old_addr here.
> + *
> + * dummy_tramp_addr moved to bottom of long branch stub.
> */
> if ((new_addr && !is_offset_in_branch_range(new_addr - ip)) ||
> (old_addr && !is_offset_in_branch_range(old_addr - ip)))
> - ret = patch_ulong((void *)(bpf_func_end - bpf_jit_long_branch_stub - SZL),
> + ret = patch_ulong((void *)(bpf_func_end - SZL), /* SZL: dummy_tramp_addr offset */
> (new_addr && !is_offset_in_branch_range(new_addr - ip)) ?
> (unsigned long)new_addr : (unsigned long)dummy_tramp);
> if (ret)
^ permalink raw reply [flat|nested] 9+ messages in thread