BPF List
 help / color / mirror / Atom feed
From: Kumar Kartikeya Dwivedi <memxor@gmail.com>
To: bpf@vger.kernel.org
Cc: Alexei Starovoitov <ast@kernel.org>,
	Andrii Nakryiko <andrii@kernel.org>,
	Daniel Borkmann <daniel@iogearbox.net>,
	Eduard Zingerman <eddyz87@gmail.com>,
	Emil Tsalapatis <emil@etsalapatis.com>, Tejun Heo <tj@kernel.org>,
	kkd@meta.com, kernel-team@meta.com
Subject: [PATCH bpf-next v3 15/18] selftests/bpf: Give the 512-byte stack boundary tests a 2 KiB twin
Date: Thu, 24 Sep 2026 18:31:29 +0200	[thread overview]
Message-ID: <20260924163144.1945455-16-memxor@gmail.com> (raw)
In-Reply-To: <20260924163144.1945455-1-memxor@gmail.com>

A number of tests pin the 512-byte stack limit: accesses just past it,
call chains that add up to more than it, private stack frames and
async callbacks that exceed it. Once a JIT raises the budget to 2 KiB
those programs load, so mark them __load_if_no_large_stack() and add a
counterpart at the 2 KiB boundary under __load_if_large_stack(), so
that each kernel runs the pair that matches its budget. The
combined-depth tests that were built from frames of a few hundred
bytes now chain five 480-byte frames, which exceeds both budgets and
keeps them valid on every architecture; the number of frames reported
in the error then differs, so those messages match any count. The C
tests are limited to 512 bytes per function by the compiler, hence the
chains. The callx stack depth tests, whose callee frames add up to 608
bytes, chain four 480-byte frames behind the callx target for the same
reason.

No kernel grants the larger budget yet, so the 512-byte tests still run
everywhere and the 2 KiB twins are skipped.

Signed-off-by: Kumar Kartikeya Dwivedi <memxor@gmail.com>
---
 .../selftests/bpf/progs/async_stack_depth.c   |  75 +++++++++++
 .../bpf/progs/struct_ops_private_stack_fail.c |  47 ++++++-
 .../selftests/bpf/progs/test_global_func1.c   |  65 ++++++++++
 .../bpf/progs/test_global_func_deep_stack.c   |  33 ++++-
 .../selftests/bpf/progs/verifier_callx.c      |  61 +++++++--
 .../bpf/progs/verifier_callx_rodata.c         |  47 ++++++-
 .../selftests/bpf/progs/verifier_live_stack.c |   4 +-
 .../selftests/bpf/progs/verifier_raw_stack.c  |  21 +++
 .../selftests/bpf/progs/verifier_stack_ptr.c  |  53 ++++++++
 .../selftests/bpf/progs/verifier_var_off.c    |  32 +++++
 tools/testing/selftests/bpf/verifier/calls.c  | 122 ++++++++++++++----
 11 files changed, 515 insertions(+), 45 deletions(-)

diff --git a/tools/testing/selftests/bpf/progs/async_stack_depth.c b/tools/testing/selftests/bpf/progs/async_stack_depth.c
index 36734683acbd..9cd874a90b39 100644
--- a/tools/testing/selftests/bpf/progs/async_stack_depth.c
+++ b/tools/testing/selftests/bpf/progs/async_stack_depth.c
@@ -29,7 +29,49 @@ static int bad_timer_cb(void *map, int *key, struct bpf_timer *timer)
 	return buf[255] + timer_cb(NULL, NULL, NULL);
 }
 
+/*
+ * The same shapes scaled to the 2 KiB budget of JITs with large stacks. The
+ * compiler caps a single function at 512 bytes, so the depth comes from a
+ * chain of 480-byte frames.
+ */
+__attribute__((noinline))
+static int timer_cb_large_0(void *map, int *key, struct bpf_timer *timer)
+{
+	volatile char buf[480] = {};
+	return buf[69];
+}
+
+__attribute__((noinline))
+static int timer_cb_large_1(void *map, int *key, struct bpf_timer *timer)
+{
+	volatile char buf[480] = {};
+	return buf[69] + timer_cb_large_0(map, key, timer);
+}
+
+__attribute__((noinline))
+static int timer_cb_large_2(void *map, int *key, struct bpf_timer *timer)
+{
+	volatile char buf[480] = {};
+	return buf[69] + timer_cb_large_1(map, key, timer);
+}
+
+__attribute__((noinline))
+static int timer_cb_large_3(void *map, int *key, struct bpf_timer *timer)
+{
+	volatile char buf[480] = {};
+	return buf[69] + timer_cb_large_2(map, key, timer);
+}
+
+/* 5 * 480 = 2400 bytes on its own */
+__attribute__((noinline))
+static int bad_timer_cb_large(void *map, int *key, struct bpf_timer *timer)
+{
+	volatile char buf[480] = {};
+	return buf[255] + timer_cb_large_3(map, key, timer);
+}
+
 SEC("tc")
+__load_if_no_large_stack()
 __failure __msg("combined stack size of 2 calls is")
 int pseudo_call_check(struct __sk_buff *ctx)
 {
@@ -44,7 +86,25 @@ int pseudo_call_check(struct __sk_buff *ctx)
 	return bpf_timer_set_callback(&elem->timer, timer_cb) + buf[0];
 }
 
+/* main plus the four frames under timer_cb_large_3: 2400 bytes */
 SEC("tc")
+__load_if_large_stack()
+__failure __msg("combined stack size of 5 calls is")
+int pseudo_call_check_large(struct __sk_buff *ctx)
+{
+	struct hmap_elem *elem;
+	volatile char buf[480] = {};
+
+	elem = bpf_map_lookup_elem(&hmap, &(int){0});
+	if (!elem)
+		return 0;
+
+	timer_cb_large_3(NULL, NULL, NULL);
+	return bpf_timer_set_callback(&elem->timer, timer_cb_large_3) + buf[0];
+}
+
+SEC("tc")
+__load_if_no_large_stack()
 __failure __msg("combined stack size of 2 calls is")
 int async_call_root_check(struct __sk_buff *ctx)
 {
@@ -58,4 +118,19 @@ int async_call_root_check(struct __sk_buff *ctx)
 	return bpf_timer_set_callback(&elem->timer, bad_timer_cb) + buf[0];
 }
 
+SEC("tc")
+__load_if_large_stack()
+__failure __msg("combined stack size of 5 calls is")
+int async_call_root_check_large(struct __sk_buff *ctx)
+{
+	struct hmap_elem *elem;
+	volatile char buf[480] = {};
+
+	elem = bpf_map_lookup_elem(&hmap, &(int){0});
+	if (!elem)
+		return 0;
+
+	return bpf_timer_set_callback(&elem->timer, bad_timer_cb_large) + buf[0];
+}
+
 char _license[] SEC("license") = "GPL";
diff --git a/tools/testing/selftests/bpf/progs/struct_ops_private_stack_fail.c b/tools/testing/selftests/bpf/progs/struct_ops_private_stack_fail.c
index 1442728f5604..c8cb35b37867 100644
--- a/tools/testing/selftests/bpf/progs/struct_ops_private_stack_fail.c
+++ b/tools/testing/selftests/bpf/progs/struct_ops_private_stack_fail.c
@@ -4,6 +4,7 @@
 #include <bpf/bpf_helpers.h>
 #include <bpf/bpf_tracing.h>
 #include "../test_kmods/bpf_testmod.h"
+#include "bpf_misc.h"
 
 char _license[] SEC("license") = "GPL";
 
@@ -25,6 +26,44 @@ __noinline static int subprog1(int *a)
 	return subprog2(a, b);
 }
 
+/*
+ * A chain of 480-byte frames under test_2, so that its call chain exceeds
+ * the 2 KiB budget of JITs with large stacks as well as the 512 bytes
+ * allowed elsewhere. The compiler caps a single function at 512 bytes, and
+ * the buffers are volatile so that it cannot shrink them.
+ */
+__noinline static int subprog_deep4(int *a)
+{
+	volatile char b[480] = {};
+
+	__sink(b[479]);
+	return a[10] + b[20];
+}
+
+__noinline static int subprog_deep3(int *a)
+{
+	volatile char b[480] = {};
+
+	__sink(b[479]);
+	return subprog_deep4(a) + b[20];
+}
+
+__noinline static int subprog_deep2(int *a)
+{
+	volatile char b[480] = {};
+
+	__sink(b[479]);
+	return subprog_deep3(a) + b[20];
+}
+
+__noinline static int subprog_deep1(int *a)
+{
+	volatile char b[480] = {};
+
+	__sink(b[479]);
+	return subprog_deep2(a) + b[20];
+}
+
 
 SEC("struct_ops")
 int BPF_PROG(test_1)
@@ -41,11 +80,13 @@ int BPF_PROG(test_1)
 SEC("struct_ops")
 int BPF_PROG(test_2)
 {
-	/* stack size 400 bytes */
-	int a[100] = {};
+	/* stack size 476 bytes, over 2 KiB with the four 480-byte deep subprogs */
+	volatile char buf[376] = {};
+	int a[25] = {};
 
+	__sink(buf[375]);
 	a[10] = 3;
-	val_j = subprog1(a);
+	val_j = subprog1(a) + subprog_deep1(a);
 	return 0;
 }
 
diff --git a/tools/testing/selftests/bpf/progs/test_global_func1.c b/tools/testing/selftests/bpf/progs/test_global_func1.c
index fc69ff18880d..f0eca282e0d4 100644
--- a/tools/testing/selftests/bpf/progs/test_global_func1.c
+++ b/tools/testing/selftests/bpf/progs/test_global_func1.c
@@ -48,8 +48,73 @@ int f3(int val, struct __sk_buff *skb, int var)
 }
 
 SEC("tc")
+__load_if_no_large_stack()
 __failure __msg("combined stack size of 3 calls is")
 int global_func1(struct __sk_buff *skb)
 {
 	return f0(1, skb) + f1(skb) + f2(2, skb) + f3(3, skb, 4);
 }
+
+/*
+ * A chain of five frames that stay under 512 bytes each but add up to more
+ * than the 2 KiB budget of JITs with large stacks; the chain also exceeds
+ * 512 bytes after two frames, so it is rejected everywhere.
+ */
+#define MAX_STACK_LARGE 480
+
+__attribute__ ((noinline))
+int g0(struct __sk_buff *skb)
+{
+	volatile char buf[MAX_STACK_LARGE] = {};
+
+	__sink(buf[MAX_STACK_LARGE - 1]);
+
+	return skb->len;
+}
+
+__attribute__ ((noinline))
+int g1(struct __sk_buff *skb)
+{
+	volatile char buf[MAX_STACK_LARGE] = {};
+
+	__sink(buf[MAX_STACK_LARGE - 1]);
+
+	return g0(skb) + skb->len;
+}
+
+__attribute__ ((noinline))
+int g2(struct __sk_buff *skb)
+{
+	volatile char buf[MAX_STACK_LARGE] = {};
+
+	__sink(buf[MAX_STACK_LARGE - 1]);
+
+	return g1(skb) + skb->len;
+}
+
+__attribute__ ((noinline))
+int g3(struct __sk_buff *skb)
+{
+	volatile char buf[MAX_STACK_LARGE] = {};
+
+	__sink(buf[MAX_STACK_LARGE - 1]);
+
+	return g2(skb) + skb->len;
+}
+
+__attribute__ ((noinline))
+int g4(struct __sk_buff *skb)
+{
+	volatile char buf[MAX_STACK_LARGE] = {};
+
+	__sink(buf[MAX_STACK_LARGE - 1]);
+
+	return g3(skb) + skb->len;
+}
+
+SEC("tc")
+__failure __msg("combined stack size of {{[0-9]+}} calls is")
+int global_func1_deep(struct __sk_buff *skb)
+{
+	return g4(skb);
+}
diff --git a/tools/testing/selftests/bpf/progs/test_global_func_deep_stack.c b/tools/testing/selftests/bpf/progs/test_global_func_deep_stack.c
index 1b634b543b62..edb8a223a3cb 100644
--- a/tools/testing/selftests/bpf/progs/test_global_func_deep_stack.c
+++ b/tools/testing/selftests/bpf/progs/test_global_func_deep_stack.c
@@ -67,12 +67,30 @@ int XCAT(f, n)(unsigned long a)                  \
 #define F_31 F_30       FN(31, 30)
 #define F_32 F_31       FN(32, 31)
 
+/* Same, with a 480-byte frame, to exceed the 2 KiB budget of large stacks. */
+#define FNB(n, prev) \
+__attribute__((noinline))                        \
+int XCAT(f, n)(unsigned long a)                  \
+{                                                \
+	volatile char buf[480] = {};             \
+	volatile long b = XCAT(f, prev)(a - 1);  \
+	if (!b)                                  \
+		return 0;                        \
+	return b + buf[479] + 1;                 \
+}
+
+#define F_33 F_32       FNB(33, 32)
+#define F_34 F_33       FNB(34, 33)
+#define F_35 F_34       FNB(35, 34)
+#define F_36 F_35       FNB(36, 35)
+#define F_37 F_36       FNB(37, 36)
+
 #define CAT2(a, b) a ## b
 #define XCAT2(a, b) CAT2(a, b)
 
 #define F(n) XCAT2(F_, n)
 
-F(32)
+F(37)
 
 /* Ensure that even 32 levels deep, the function verifies. */
 SEC("syscall")
@@ -88,8 +106,21 @@ int global_func_deep_stack_success(struct __sk_buff *skb)
  * the size.
  */
 SEC("syscall")
+__load_if_no_large_stack()
 __failure __msg("combined stack size of 34 calls")
 int global_func_deep_stack_fail(struct __sk_buff *skb)
 {
 	return f32(123);
 }
+
+/*
+ * Five 480-byte frames on top of the chain: 5 * 480 + 33 * 16 = 2928 bytes,
+ * more than the 2 KiB budget of JITs with large stacks, and more than 512
+ * bytes after the second frame everywhere else.
+ */
+SEC("syscall")
+__failure __msg("combined stack size of {{[0-9]+}} calls")
+int global_func_deep_stack_fail_large(struct __sk_buff *skb)
+{
+	return f37(123);
+}
diff --git a/tools/testing/selftests/bpf/progs/verifier_callx.c b/tools/testing/selftests/bpf/progs/verifier_callx.c
index 238fc75ad154..ea3d3f97da31 100644
--- a/tools/testing/selftests/bpf/progs/verifier_callx.c
+++ b/tools/testing/selftests/bpf/progs/verifier_callx.c
@@ -626,19 +626,64 @@ static unsigned long use_stack_304(void)
 	);
 }
 
+
+/* Four 480-byte frames, deeper than any budget together with their caller */
+__naked __noinline __used
+static unsigned long use_stack_480_0(void)
+{
+	asm volatile (
+		"r0 = 0;"
+		"*(u64 *)(r10 - 480) = r0;"
+		"exit;"
+	);
+}
+
+__naked __noinline __used
+static unsigned long use_stack_480_1(void)
+{
+	asm volatile (
+		"r0 = 0;"
+		"*(u64 *)(r10 - 480) = r0;"
+		"call use_stack_480_0;"
+		"exit;"
+	);
+}
+
+__naked __noinline __used
+static unsigned long use_stack_480_2(void)
+{
+	asm volatile (
+		"r0 = 0;"
+		"*(u64 *)(r10 - 480) = r0;"
+		"call use_stack_480_1;"
+		"exit;"
+	);
+}
+
+__naked __noinline __used
+static unsigned long use_stack_480_3(void)
+{
+	asm volatile (
+		"r0 = 0;"
+		"*(u64 *)(r10 - 480) = r0;"
+		"call use_stack_480_2;"
+		"exit;"
+	);
+}
+
 /* stack of the callee of callx is accounted */
 SEC("socket")
-__failure __msg("combined stack size of 2 calls is")
+__failure __msg("combined stack size of {{[0-9]+}} calls is")
 __naked void callx_stack_depth(void)
 {
 	asm volatile (
 		"r0 = 0;"
-		"*(u64 *)(r10 - 304) = r0;"
-		"r2 = %[use_stack_304] ll;"
+		"*(u64 *)(r10 - 480) = r0;"
+		"r2 = %[use_stack_480_3] ll;"
 		"callx r2;"
 		"exit;"
 		:
-		: __imm_addr(use_stack_304)
+		: __imm_addr(use_stack_480_3)
 		: __clobber_all);
 }
 
@@ -655,19 +700,19 @@ static unsigned long apply_stack_304(void)
 }
 
 /*
- * The address of use_stack_304() is taken by the main prog that doesn't
+ * The address of use_stack_480_3() is taken by the main prog that doesn't
  * use stack, but it is called from apply_stack_304().
  */
 SEC("socket")
-__failure __msg("combined stack size of 3 calls is")
+__failure __msg("combined stack size of {{[0-9]+}} calls is")
 __naked void callx_stack_depth_nested(void)
 {
 	asm volatile (
-		"r1 = %[use_stack_304] ll;"
+		"r1 = %[use_stack_480_3] ll;"
 		"call apply_stack_304;"
 		"exit;"
 		:
-		: __imm_addr(use_stack_304)
+		: __imm_addr(use_stack_480_3)
 		: __clobber_all);
 }
 
diff --git a/tools/testing/selftests/bpf/progs/verifier_callx_rodata.c b/tools/testing/selftests/bpf/progs/verifier_callx_rodata.c
index af1f9305da37..65d34769c8bf 100644
--- a/tools/testing/selftests/bpf/progs/verifier_callx_rodata.c
+++ b/tools/testing/selftests/bpf/progs/verifier_callx_rodata.c
@@ -579,23 +579,58 @@ __naked void callx_rodata_recursion(void)
 		::: __clobber_all);
 }
 
+
+/* Four 480-byte frames, deeper than any budget together with their caller */
 __naked __noinline __used
-static unsigned long use_stack_304(void)
+static unsigned long use_stack_480_0(void)
 {
 	asm volatile (
 		"r0 = 0;"
-		"*(u64 *)(r10 - 304) = r0;"
+		"*(u64 *)(r10 - 480) = r0;"
+		"exit;"
+	);
+}
+
+__naked __noinline __used
+static unsigned long use_stack_480_1(void)
+{
+	asm volatile (
+		"r0 = 0;"
+		"*(u64 *)(r10 - 480) = r0;"
+		"call use_stack_480_0;"
+		"exit;"
+	);
+}
+
+__naked __noinline __used
+static unsigned long use_stack_480_2(void)
+{
+	asm volatile (
+		"r0 = 0;"
+		"*(u64 *)(r10 - 480) = r0;"
+		"call use_stack_480_1;"
+		"exit;"
+	);
+}
+
+__naked __noinline __used
+static unsigned long use_stack_480_3(void)
+{
+	asm volatile (
+		"r0 = 0;"
+		"*(u64 *)(r10 - 480) = r0;"
+		"call use_stack_480_2;"
 		"exit;"
 	);
 }
 
 /* stack of all possible callees is accounted */
 SEC("socket")
-__failure __msg("combined stack size of 2 calls is")
+__failure __msg("combined stack size of {{[0-9]+}} calls is")
 __naked void callx_rodata_stack_depth(void)
 {
 	asm volatile (
-		FUNC_TABLE2(tbl, ret0, use_stack_304)
+		FUNC_TABLE2(tbl, ret0, use_stack_480_3)
 		"r0 = 0;"
 		"*(u64 *)(r10 - 304) = r0;"
 		"call %[bpf_get_prandom_u32];"
@@ -613,11 +648,11 @@ __naked void callx_rodata_stack_depth(void)
 
 /* stack of a callback that is read from the data is accounted too */
 SEC("socket")
-__failure __msg("combined stack size of 2 calls is")
+__failure __msg("combined stack size of {{[0-9]+}} calls is")
 __naked void callx_rodata_callback_stack_depth(void)
 {
 	asm volatile (
-		FUNC_TABLE2(tbl, use_stack_304, ret0)
+		FUNC_TABLE2(tbl, use_stack_480_3, ret0)
 		"r0 = 0;"
 		"*(u64 *)(r10 - 304) = r0;"
 		"r6 = tbl_%= ll;"
diff --git a/tools/testing/selftests/bpf/progs/verifier_live_stack.c b/tools/testing/selftests/bpf/progs/verifier_live_stack.c
index 4736bcca55da..a832df0b5bd2 100644
--- a/tools/testing/selftests/bpf/progs/verifier_live_stack.c
+++ b/tools/testing/selftests/bpf/progs/verifier_live_stack.c
@@ -318,7 +318,7 @@ struct {
 } map_array SEC(".maps");
 
 SEC("socket")
-__failure __msg("invalid read from stack R2 off=-1024 size=8")
+__failure __msg("invalid read from stack R2 off=-4096 size=8")
 __flag(BPF_F_TEST_STATE_FREQ)
 __naked unsigned long caller_stack_write_tail_call(void)
 {
@@ -329,7 +329,7 @@ __naked unsigned long caller_stack_write_tail_call(void)
         "if r0 != 42 goto 1f;"
         "goto 2f;"
   "1:"
-        "*(u64 *)(r10 - 8) = -1024;"
+        "*(u64 *)(r10 - 8) = -4096;"
   "2:"
         "r1 = r6;"
         "r2 = r10;"
diff --git a/tools/testing/selftests/bpf/progs/verifier_raw_stack.c b/tools/testing/selftests/bpf/progs/verifier_raw_stack.c
index 9f0f48ecb421..0fe631411b9c 100644
--- a/tools/testing/selftests/bpf/progs/verifier_raw_stack.c
+++ b/tools/testing/selftests/bpf/progs/verifier_raw_stack.c
@@ -240,6 +240,7 @@ __naked void load_bytes_spilled_regs_data(void)
 
 SEC("tc")
 __description("raw_stack: skb_load_bytes, invalid access 1")
+__load_if_no_large_stack()
 __failure __msg("invalid write to stack R3 off=-513 size=8")
 __naked void load_bytes_invalid_access_1(void)
 {
@@ -257,6 +258,26 @@ __naked void load_bytes_invalid_access_1(void)
 	: __clobber_all);
 }
 
+SEC("tc")
+__description("raw_stack: skb_load_bytes, invalid access 1, large stack")
+__load_if_large_stack()
+__failure __msg("invalid write to stack R3 off=-2049 size=8")
+__naked void load_bytes_invalid_access_1_large(void)
+{
+	asm volatile ("					\
+	r2 = 4;						\
+	r6 = r10;					\
+	r6 += -2049;					\
+	r3 = r6;					\
+	r4 = 8;						\
+	call %[bpf_skb_load_bytes];			\
+	r0 = *(u64*)(r6 + 0);				\
+	exit;						\
+"	:
+	: __imm(bpf_skb_load_bytes)
+	: __clobber_all);
+}
+
 SEC("tc")
 __description("raw_stack: skb_load_bytes, invalid access 2")
 __failure __msg("invalid write to stack R3 off=-1 size=8")
diff --git a/tools/testing/selftests/bpf/progs/verifier_stack_ptr.c b/tools/testing/selftests/bpf/progs/verifier_stack_ptr.c
index 8e8cf8232255..3e0bea9819ca 100644
--- a/tools/testing/selftests/bpf/progs/verifier_stack_ptr.c
+++ b/tools/testing/selftests/bpf/progs/verifier_stack_ptr.c
@@ -235,6 +235,7 @@ __naked void to_stack_check_low_1(void)
 
 SEC("socket")
 __description("PTR_TO_STACK check low 2")
+__load_if_no_large_stack()
 __success __failure_unpriv
 __msg_unpriv("R1 stack pointer arithmetic goes out of range")
 __retval(42)
@@ -250,8 +251,27 @@ __naked void to_stack_check_low_2(void)
 "	::: __clobber_all);
 }
 
+SEC("socket")
+__description("PTR_TO_STACK check low 2, large stack")
+__load_if_large_stack()
+__success __failure_unpriv
+__msg_unpriv("R1 stack pointer arithmetic goes out of range")
+__retval(42)
+__naked void to_stack_check_low_2_large(void)
+{
+	asm volatile ("					\
+	r1 = r10;					\
+	r1 += -2049;					\
+	r0 = 42;					\
+	*(u8*)(r1 + 1) = r0;				\
+	r0 = *(u8*)(r1 + 1);				\
+	exit;						\
+"	::: __clobber_all);
+}
+
 SEC("socket")
 __description("PTR_TO_STACK check low 3")
+__load_if_no_large_stack()
 __failure __msg("invalid write to stack R1 off=-513 size=1")
 __msg_unpriv("R1 stack pointer arithmetic goes out of range")
 __naked void to_stack_check_low_3(void)
@@ -266,6 +286,23 @@ __naked void to_stack_check_low_3(void)
 "	::: __clobber_all);
 }
 
+SEC("socket")
+__description("PTR_TO_STACK check low 3, large stack")
+__load_if_large_stack()
+__failure __msg("invalid write to stack R1 off=-2049 size=1")
+__msg_unpriv("R1 stack pointer arithmetic goes out of range")
+__naked void to_stack_check_low_3_large(void)
+{
+	asm volatile ("					\
+	r1 = r10;					\
+	r1 += -2049;					\
+	r0 = 42;					\
+	*(u8*)(r1 + 0) = r0;				\
+	r0 = *(u8*)(r1 + 0);				\
+	exit;						\
+"	::: __clobber_all);
+}
+
 SEC("socket")
 __description("PTR_TO_STACK check low 4")
 __failure __msg("math between fp pointer")
@@ -483,6 +520,7 @@ l1_%=:	r0 = 42;					\
 
 SEC("socket")
 __description("PTR_TO_STACK stack size > 512")
+__load_if_no_large_stack()
 __failure __msg("invalid write to stack R1 off=-520 size=8")
 __naked void stack_check_size_gt_512(void)
 {
@@ -495,6 +533,21 @@ __naked void stack_check_size_gt_512(void)
 "	::: __clobber_all);
 }
 
+SEC("socket")
+__description("PTR_TO_STACK stack size > 2048")
+__load_if_large_stack()
+__failure __msg("invalid write to stack R1 off=-2056 size=8")
+__naked void stack_check_size_gt_2048(void)
+{
+	asm volatile ("					\
+	r1 = r10;					\
+	r1 += -2056;					\
+	r0 = 42;					\
+	*(u64*)(r1 + 0) = r0;				\
+	exit;						\
+"	::: __clobber_all);
+}
+
 #ifdef __BPF_FEATURE_MAY_GOTO
 SEC("socket")
 __description("PTR_TO_STACK stack size 512 with may_goto with jit")
diff --git a/tools/testing/selftests/bpf/progs/verifier_var_off.c b/tools/testing/selftests/bpf/progs/verifier_var_off.c
index a63e33675091..399884911ea5 100644
--- a/tools/testing/selftests/bpf/progs/verifier_var_off.c
+++ b/tools/testing/selftests/bpf/progs/verifier_var_off.c
@@ -406,6 +406,7 @@ __naked void zero_sized_access_max_out_of_bound(void)
 
 SEC("lwt_in")
 __description("indirect variable-offset stack access, min out of bound")
+__load_if_no_large_stack()
 __failure __msg("invalid variable-offset read from stack R2")
 __naked void access_min_out_of_bound(void)
 {
@@ -433,6 +434,37 @@ __naked void access_min_out_of_bound(void)
 	: __clobber_all);
 }
 
+SEC("lwt_in")
+__description("indirect variable-offset stack access, min out of bound, large stack")
+__load_if_large_stack()
+__failure __msg("invalid variable-offset read from stack R2")
+__naked void access_min_out_of_bound_large(void)
+{
+	asm volatile ("					\
+	/* Fill the top 8 bytes of the stack */		\
+	r2 = 0;						\
+	*(u64*)(r10 - 8) = r2;				\
+	/* Get an unknown value */			\
+	r2 = *(u32*)(r1 + 0);				\
+	/* Make it small and 4-byte aligned */		\
+	r2 &= 4;					\
+	r2 -= 2052;					\
+	/*						\
+	 * add it to fp.  We now have either fp-2052 or fp-2048, but\
+	 * we don't know which				\
+	 */						\
+	r2 += r10;					\
+	/* dereference it indirectly */			\
+	r1 = %[map_hash_8b] ll;				\
+	call %[bpf_map_lookup_elem];			\
+	r0 = 0;						\
+	exit;						\
+"	:
+	: __imm(bpf_map_lookup_elem),
+	  __imm_addr(map_hash_8b)
+	: __clobber_all);
+}
+
 SEC("cgroup/skb")
 __description("indirect variable-offset stack access, min_off < min_initialized")
 __success
diff --git a/tools/testing/selftests/bpf/verifier/calls.c b/tools/testing/selftests/bpf/verifier/calls.c
index 8b94b87135bc..0af237c02ddf 100644
--- a/tools/testing/selftests/bpf/verifier/calls.c
+++ b/tools/testing/selftests/bpf/verifier/calls.c
@@ -1037,15 +1037,34 @@
 	.result = ACCEPT,
 },
 {
-	"calls: stack overflow using two frames (pre-call access)",
+	/*
+	 * Five 480-byte frames exceed the 2 KiB budget of JITs with large
+	 * stacks, and two of them the 512 bytes allowed elsewhere.
+	 */
+	"calls: stack overflow using five frames (pre-call access)",
 	.insns = {
 	/* prog 1 */
-	BPF_ST_MEM(BPF_B, BPF_REG_10, -300, 0),
+	BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
 	BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 1),
 	BPF_EXIT_INSN(),
 
 	/* prog 2 */
-	BPF_ST_MEM(BPF_B, BPF_REG_10, -300, 0),
+	BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
+	BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 1),
+	BPF_EXIT_INSN(),
+
+	/* prog 3 */
+	BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
+	BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 1),
+	BPF_EXIT_INSN(),
+
+	/* prog 4 */
+	BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
+	BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 1),
+	BPF_EXIT_INSN(),
+
+	/* prog 5 */
+	BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
 	BPF_MOV64_IMM(BPF_REG_0, 0),
 	BPF_EXIT_INSN(),
 	},
@@ -1054,15 +1073,30 @@
 	.result = REJECT,
 },
 {
-	"calls: stack overflow using two frames (post-call access)",
+	"calls: stack overflow using five frames (post-call access)",
 	.insns = {
 	/* prog 1 */
 	BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 2),
-	BPF_ST_MEM(BPF_B, BPF_REG_10, -300, 0),
+	BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
 	BPF_EXIT_INSN(),
 
 	/* prog 2 */
-	BPF_ST_MEM(BPF_B, BPF_REG_10, -300, 0),
+	BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 2),
+	BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
+	BPF_EXIT_INSN(),
+
+	/* prog 3 */
+	BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 2),
+	BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
+	BPF_EXIT_INSN(),
+
+	/* prog 4 */
+	BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 2),
+	BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
+	BPF_EXIT_INSN(),
+
+	/* prog 5 */
+	BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
 	BPF_MOV64_IMM(BPF_REG_0, 0),
 	BPF_EXIT_INSN(),
 	},
@@ -1127,7 +1161,7 @@
 	.result = ACCEPT,
 },
 {
-	"calls: stack depth check using three frames. test3",
+	"calls: stack depth check using five frames. test3",
 	.insns = {
 	/* main */
 	BPF_MOV64_REG(BPF_REG_6, BPF_REG_1),
@@ -1135,66 +1169,104 @@
 	BPF_MOV64_REG(BPF_REG_1, BPF_REG_6),
 	BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 8), /* call B */
 	BPF_JMP_IMM(BPF_JGE, BPF_REG_6, 0, 1),
-	BPF_ST_MEM(BPF_B, BPF_REG_10, -64, 0),
+	BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
 	BPF_MOV64_IMM(BPF_REG_0, 0),
 	BPF_EXIT_INSN(),
 	/* A */
 	BPF_JMP_IMM(BPF_JLT, BPF_REG_1, 10, 1),
 	BPF_EXIT_INSN(),
-	BPF_ST_MEM(BPF_B, BPF_REG_10, -224, 0),
+	BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
 	BPF_JMP_IMM(BPF_JA, 0, 0, -3),
 	/* B */
 	BPF_JMP_IMM(BPF_JGT, BPF_REG_1, 2, 1),
-	BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, -6), /* call A */
-	BPF_ST_MEM(BPF_B, BPF_REG_10, -256, 0),
+	BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 2), /* call C */
+	BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
+	BPF_EXIT_INSN(),
+	/* C */
+	BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 2), /* call D */
+	BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
+	BPF_EXIT_INSN(),
+	/* D */
+	BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, -12), /* call A */
+	BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
 	BPF_EXIT_INSN(),
 	},
 	.prog_type = BPF_PROG_TYPE_XDP,
-	/* stack_main=64, stack_A=224, stack_B=256
-	 * and max(main+A, main+A+B) > 512
+	/*
+	 * every frame is 480 bytes, main+A = 960 > 512 and
+	 * max(main+A, main+B+C+D+A) = 2400 > 2048
 	 */
 	.errstr = "combined stack",
 	.result = REJECT,
 },
 {
-	"calls: stack depth check using three frames. test4",
-	/* void main(void) {
+	"calls: stack depth check using five frames. test4",
+	/*
+	 * void main(void) {
 	 *   func1(0);
 	 *   func1(1);
 	 *   func2(1);
 	 * }
-	 * void func1(int alloc_or_recurse) {
+	 * void funcN(int alloc_or_recurse) {   N = 1..4
 	 *   if (alloc_or_recurse) {
-	 *     frame_pointer[-300] = 1;
+	 *     frame_pointer[-480] = 1;
 	 *   } else {
-	 *     func2(alloc_or_recurse);
+	 *     funcN+1(alloc_or_recurse);
 	 *   }
 	 * }
-	 * void func2(int alloc_or_recurse) {
+	 * void func5(int alloc_or_recurse) {
 	 *   if (alloc_or_recurse) {
-	 *     frame_pointer[-300] = 1;
+	 *     frame_pointer[-480] = 1;
 	 *   }
 	 * }
+	 * main also calls func2 to func5 with 1 so that every function has a
+	 * path allocating its 480 bytes, and the chain adds up to 2400 bytes,
+	 * more than the 2 KiB budget of JITs with large stacks, and to 960
+	 * bytes after two frames, more than the 512 bytes allowed elsewhere.
 	 */
 	.insns = {
 	/* main */
 	BPF_MOV64_IMM(BPF_REG_1, 0),
-	BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 6), /* call A */
+	BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 12), /* call A */
 	BPF_MOV64_IMM(BPF_REG_1, 1),
-	BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 4), /* call A */
+	BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 10), /* call A */
+	BPF_MOV64_IMM(BPF_REG_1, 1),
+	BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 13), /* call B */
+	BPF_MOV64_IMM(BPF_REG_1, 1),
+	BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 16), /* call C */
 	BPF_MOV64_IMM(BPF_REG_1, 1),
-	BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 7), /* call B */
+	BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 19), /* call D */
+	BPF_MOV64_IMM(BPF_REG_1, 1),
+	BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 22), /* call E */
 	BPF_MOV64_IMM(BPF_REG_0, 0),
 	BPF_EXIT_INSN(),
 	/* A */
 	BPF_JMP_IMM(BPF_JEQ, BPF_REG_1, 0, 2),
-	BPF_ST_MEM(BPF_B, BPF_REG_10, -300, 0),
+	BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
 	BPF_EXIT_INSN(),
 	BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 1), /* call B */
 	BPF_EXIT_INSN(),
 	/* B */
+	BPF_JMP_IMM(BPF_JEQ, BPF_REG_1, 0, 2),
+	BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
+	BPF_EXIT_INSN(),
+	BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 1), /* call C */
+	BPF_EXIT_INSN(),
+	/* C */
+	BPF_JMP_IMM(BPF_JEQ, BPF_REG_1, 0, 2),
+	BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
+	BPF_EXIT_INSN(),
+	BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 1), /* call D */
+	BPF_EXIT_INSN(),
+	/* D */
+	BPF_JMP_IMM(BPF_JEQ, BPF_REG_1, 0, 2),
+	BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
+	BPF_EXIT_INSN(),
+	BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 1), /* call E */
+	BPF_EXIT_INSN(),
+	/* E */
 	BPF_JMP_IMM(BPF_JEQ, BPF_REG_1, 0, 1),
-	BPF_ST_MEM(BPF_B, BPF_REG_10, -300, 0),
+	BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
 	BPF_EXIT_INSN(),
 	},
 	.prog_type = BPF_PROG_TYPE_XDP,
-- 
2.53.0


  parent reply	other threads:[~2026-09-24 16:32 UTC|newest]

Thread overview: 20+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-24 16:31 [PATCH bpf-next v3 00/18] Raise BPF program stack size to 2KiB Kumar Kartikeya Dwivedi
2026-09-24 16:31 ` [PATCH bpf-next v3 01/18] bpf: Add accessors for verifier stack slots Kumar Kartikeya Dwivedi
2026-09-24 16:31 ` [PATCH bpf-next v3 02/18] bpf: Widen the stack slot index in the jump history Kumar Kartikeya Dwivedi
2026-09-24 16:31 ` [PATCH bpf-next v3 03/18] bpf: Store linked registers in the jump history as an array Kumar Kartikeya Dwivedi
2026-09-24 16:31 ` [PATCH bpf-next v3 04/18] bpf: Track backtracking stack slots with bitmaps Kumar Kartikeya Dwivedi
2026-09-24 16:31 ` [PATCH bpf-next v3 05/18] bpf: Track scratched stack slots with a bitmap Kumar Kartikeya Dwivedi
2026-09-24 16:31 ` [PATCH bpf-next v3 06/18] bpf: Treat unknown-size stack reads as reaching the frame top Kumar Kartikeya Dwivedi
2026-09-24 16:31 ` [PATCH bpf-next v3 07/18] bpf: Size liveness stack masks by the stack each frame uses Kumar Kartikeya Dwivedi
2026-09-24 16:31 ` [PATCH bpf-next v3 08/18] bpf: Grow the verifier id scratch on demand Kumar Kartikeya Dwivedi
2026-09-24 16:31 ` [PATCH bpf-next v3 09/18] selftests/bpf: Cover the tail call caller stack depth limit Kumar Kartikeya Dwivedi
2026-09-24 16:31 ` [PATCH bpf-next v3 10/18] selftests/bpf: Check that narrow stack stores define no slot Kumar Kartikeya Dwivedi
2026-09-24 16:31 ` [PATCH bpf-next v3 11/18] selftests/bpf: Check liveness merge of masks with different widths Kumar Kartikeya Dwivedi
2026-09-24 16:31 ` [PATCH bpf-next v3 12/18] bpf: Size the per-frame verifier structures for a 2 KiB stack Kumar Kartikeya Dwivedi
2026-09-24 16:31 ` [PATCH bpf-next v3 13/18] bpf: Bound program stack use by a per-program limit Kumar Kartikeya Dwivedi
2026-09-24 17:09   ` sashiko-bot
2026-09-24 16:31 ` [PATCH bpf-next v3 14/18] selftests/bpf: Add load conditions on the program stack limit Kumar Kartikeya Dwivedi
2026-09-24 16:31 ` Kumar Kartikeya Dwivedi [this message]
2026-09-24 16:31 ` [PATCH bpf-next v3 16/18] bpf, x86: Allow programs 2 KiB of stack Kumar Kartikeya Dwivedi
2026-09-24 16:31 ` [PATCH bpf-next v3 17/18] bpf, arm64: " Kumar Kartikeya Dwivedi
2026-09-24 16:31 ` [PATCH bpf-next v3 18/18] selftests/bpf: Test the 2 KiB stack budget Kumar Kartikeya Dwivedi

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260924163144.1945455-16-memxor@gmail.com \
    --to=memxor@gmail.com \
    --cc=andrii@kernel.org \
    --cc=ast@kernel.org \
    --cc=bpf@vger.kernel.org \
    --cc=daniel@iogearbox.net \
    --cc=eddyz87@gmail.com \
    --cc=emil@etsalapatis.com \
    --cc=kernel-team@meta.com \
    --cc=kkd@meta.com \
    --cc=tj@kernel.org \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox