* [PATCH bpf-next v1 01/18] bpf: Add accessors for verifier stack slots
2026-09-23 19:11 [PATCH bpf-next v1 00/18] Raise BPF program stack size to 2KiB Kumar Kartikeya Dwivedi
@ 2026-09-23 19:11 ` Kumar Kartikeya Dwivedi
2026-09-23 19:57 ` bot+bpf-ci
2026-09-23 19:11 ` [PATCH bpf-next v1 02/18] bpf: Widen the stack slot index in the jump history Kumar Kartikeya Dwivedi
` (16 subsequent siblings)
17 siblings, 1 reply; 34+ messages in thread
From: Kumar Kartikeya Dwivedi @ 2026-09-23 19:11 UTC (permalink / raw)
To: bpf
Cc: Alexei Starovoitov, Andrii Nakryiko, Daniel Borkmann,
Eduard Zingerman, Emil Tsalapatis, Tejun Heo, kkd, kernel-team
The verifier indexes a frame's stack state directly through
state->stack[spi] and computes the number of tracked slots as
allocated_stack / BPF_REG_SIZE in every file that touches stack slots.
Route all of these through two helpers, bpf_stack_slot() and
bpf_stack_nr_slots(), so the layout of the per-frame stack state is
visible in one place and can change without touching every user.
Both take a const frame: the slot accessor returns the slot through the
frame's stack pointer, so read-only code such as the state printer can
use it without giving up its qualifiers. Functions that look up the
same slot repeatedly now fetch it once.
No functional change.
Signed-off-by: Kumar Kartikeya Dwivedi <memxor@gmail.com>
---
include/linux/bpf_verifier.h | 25 ++++--
kernel/bpf/backtrack.c | 14 +--
kernel/bpf/diagnostics.c | 6 +-
kernel/bpf/log.c | 13 +--
kernel/bpf/states.c | 84 +++++++++---------
kernel/bpf/verifier.c | 163 +++++++++++++++++++----------------
6 files changed, 166 insertions(+), 139 deletions(-)
diff --git a/include/linux/bpf_verifier.h b/include/linux/bpf_verifier.h
index 92f528c45605..36e0c6b97533 100644
--- a/include/linux/bpf_verifier.h
+++ b/include/linux/bpf_verifier.h
@@ -529,12 +529,27 @@ struct bpf_verifier_state {
u32 may_goto_depth;
};
+/* Number of BPF_REG_SIZE stack slots tracked for the frame so far. */
+static inline u32 bpf_stack_nr_slots(const struct bpf_func_state *frame)
+{
+ return frame->allocated_stack / BPF_REG_SIZE;
+}
+
+/*
+ * Stack slot @spi of @frame, covering bytes [fp - (spi + 1) * 8, fp - spi * 8).
+ * The caller must ensure spi < bpf_stack_nr_slots(frame), see grow_stack_state().
+ */
+static inline struct bpf_stack_state *bpf_stack_slot(const struct bpf_func_state *frame, u32 spi)
+{
+ return &frame->stack[spi];
+}
+
static inline struct bpf_reg_state *
bpf_get_spilled_reg(int slot, struct bpf_func_state *frame, u32 mask)
{
- if (slot < frame->allocated_stack / BPF_REG_SIZE &&
- (1 << frame->stack[slot].slot_type[BPF_REG_SIZE - 1]) & mask)
- return &frame->stack[slot].spilled_ptr;
+ if (slot < bpf_stack_nr_slots(frame) &&
+ (1 << bpf_stack_slot(frame, slot)->slot_type[BPF_REG_SIZE - 1]) & mask)
+ return &bpf_stack_slot(frame, slot)->spilled_ptr;
return NULL;
}
@@ -550,7 +565,7 @@ bpf_get_spilled_stack_arg(int slot, struct bpf_func_state *frame)
/* Iterate over 'frame', setting 'reg' to either NULL or a spilled register. */
#define bpf_for_each_spilled_reg(iter, frame, reg, mask) \
for (iter = 0, reg = bpf_get_spilled_reg(iter, frame, mask); \
- iter < frame->allocated_stack / BPF_REG_SIZE; \
+ iter < bpf_stack_nr_slots(frame); \
iter++, reg = bpf_get_spilled_reg(iter, frame, mask))
/* Iterate over 'frame', setting 'reg' to either NULL or a spilled stack arg. */
@@ -575,7 +590,7 @@ bpf_get_spilled_stack_arg(int slot, struct bpf_func_state *frame)
bpf_for_each_spilled_reg(___j, __state, __reg, __mask) { \
if (!__reg) \
continue; \
- __stack = &__state->stack[___j]; \
+ __stack = bpf_stack_slot(__state, ___j); \
(void)(__expr); \
} \
__stack = NULL; \
diff --git a/kernel/bpf/backtrack.c b/kernel/bpf/backtrack.c
index 507a366dffa4..4bec7b94796b 100644
--- a/kernel/bpf/backtrack.c
+++ b/kernel/bpf/backtrack.c
@@ -707,10 +707,10 @@ void bpf_mark_all_scalars_precise(struct bpf_verifier_env *env,
i, j);
}
}
- for (j = 0; j < func->allocated_stack / BPF_REG_SIZE; j++) {
- if (!bpf_is_spilled_reg(&func->stack[j]))
+ for (j = 0; j < bpf_stack_nr_slots(func); j++) {
+ if (!bpf_is_spilled_reg(bpf_stack_slot(func, j)))
continue;
- reg = &func->stack[j].spilled_ptr;
+ reg = &bpf_stack_slot(func, j)->spilled_ptr;
if (reg->type != SCALAR_VALUE || reg->precise)
continue;
reg->precise = true;
@@ -946,16 +946,16 @@ int bpf_mark_chain_precision(struct bpf_verifier_env *env,
bitmap_from_u64(mask, bt_frame_stack_mask(bt, fr));
for_each_set_bit(i, mask, 64) {
- if (verifier_bug_if(i >= func->allocated_stack / BPF_REG_SIZE,
+ if (verifier_bug_if(i >= bpf_stack_nr_slots(func),
env, "stack slot %d, total slots %d",
- i, func->allocated_stack / BPF_REG_SIZE))
+ i, bpf_stack_nr_slots(func)))
return -EFAULT;
- if (!bpf_is_spilled_scalar_reg(&func->stack[i])) {
+ if (!bpf_is_spilled_scalar_reg(bpf_stack_slot(func, i))) {
bt_clear_frame_slot(bt, fr, i);
continue;
}
- reg = &func->stack[i].spilled_ptr;
+ reg = &bpf_stack_slot(func, i)->spilled_ptr;
if (reg->precise) {
bt_clear_frame_slot(bt, fr, i);
} else {
diff --git a/kernel/bpf/diagnostics.c b/kernel/bpf/diagnostics.c
index 5ecfa86ed49f..a8ed6130c137 100644
--- a/kernel/bpf/diagnostics.c
+++ b/kernel/bpf/diagnostics.c
@@ -1600,9 +1600,9 @@ static struct bpf_reg_state *target_to_reg(struct bpf_verifier_env *env,
return NULL;
return &state->stack_arg_regs[target->stack_arg];
case BPF_DIAG_MOD_TARGET_STACK_SLOT:
- if (target->spi >= state->allocated_stack / BPF_REG_SIZE)
+ if (target->spi >= bpf_stack_nr_slots(state))
return NULL;
- return &state->stack[target->spi].spilled_ptr;
+ return &bpf_stack_slot(state, target->spi)->spilled_ptr;
default:
return NULL;
}
@@ -1618,7 +1618,7 @@ static bool reg_to_target(struct bpf_verifier_env *env, const struct bpf_reg_sta
for (frame = 0; frame <= vstate->curframe; frame++) {
struct bpf_func_state *state = vstate->frame[frame];
unsigned long start, end;
- u32 nslots = state->allocated_stack / BPF_REG_SIZE;
+ u32 nslots = bpf_stack_nr_slots(state);
int spi;
start = (unsigned long)state->regs;
diff --git a/kernel/bpf/log.c b/kernel/bpf/log.c
index fb032dfdc0de..d850a7863d2e 100644
--- a/kernel/bpf/log.c
+++ b/kernel/bpf/log.c
@@ -716,7 +716,8 @@ void print_verifier_state(struct bpf_verifier_env *env, const struct bpf_verifie
verbose(env, "=");
print_reg_state(env, state, reg);
}
- for (i = 0; i < state->allocated_stack / BPF_REG_SIZE; i++) {
+ for (i = 0; i < bpf_stack_nr_slots(state); i++) {
+ struct bpf_stack_state *slot = bpf_stack_slot(state, i);
char types_buf[BPF_REG_SIZE + 1];
const char *sep = "";
bool valid = false;
@@ -727,7 +728,7 @@ void print_verifier_state(struct bpf_verifier_env *env, const struct bpf_verifie
continue;
for (j = 0; j < BPF_REG_SIZE; j++) {
- slot_type = state->stack[i].slot_type[j];
+ slot_type = slot->slot_type[j];
if (slot_type != STACK_INVALID && slot_type != STACK_POISON)
valid = true;
types_buf[j] = slot_type_char[slot_type];
@@ -736,12 +737,12 @@ void print_verifier_state(struct bpf_verifier_env *env, const struct bpf_verifie
if (!valid)
continue;
- reg = &state->stack[i].spilled_ptr;
- switch (state->stack[i].slot_type[BPF_REG_SIZE - 1]) {
+ reg = &slot->spilled_ptr;
+ switch (slot->slot_type[BPF_REG_SIZE - 1]) {
case STACK_SPILL:
/* print MISC/ZERO/INVALID slots above subreg spill */
for (j = 0; j < BPF_REG_SIZE; j++)
- if (state->stack[i].slot_type[j] == STACK_SPILL)
+ if (slot->slot_type[j] == STACK_SPILL)
break;
types_buf[j] = '\0';
@@ -751,7 +752,7 @@ void print_verifier_state(struct bpf_verifier_env *env, const struct bpf_verifie
case STACK_DYNPTR:
/* skip to main dynptr slot */
i += BPF_DYNPTR_NR_SLOTS - 1;
- reg = &state->stack[i].spilled_ptr;
+ reg = &bpf_stack_slot(state, i)->spilled_ptr;
verbose(env, " fp%d", (-i - 1) * BPF_REG_SIZE);
verbose(env, "=dynptr_%s(", dynptr_type_str(reg->dynptr.type));
diff --git a/kernel/bpf/states.c b/kernel/bpf/states.c
index 66fb11b6c6a7..38795cf35247 100644
--- a/kernel/bpf/states.c
+++ b/kernel/bpf/states.c
@@ -415,14 +415,14 @@ static void __clean_func_state(struct bpf_verifier_env *env,
* half_spi 2*i → lower half: slot_type[0..3] (closer to FP)
* half_spi 2*i+1 → upper half: slot_type[4..7] (farther from FP)
*/
- for (i = 0; i < st->allocated_stack / BPF_REG_SIZE; i++) {
+ for (i = 0; i < bpf_stack_nr_slots(st); i++) {
bool lo_live = bpf_stack_slot_alive(env, frame, i * 2);
bool hi_live = bpf_stack_slot_alive(env, frame, i * 2 + 1);
if (!hi_live || !lo_live) {
int start = !lo_live ? 0 : BPF_REG_SIZE / 2;
int end = !hi_live ? BPF_REG_SIZE : BPF_REG_SIZE / 2;
- u8 stype = st->stack[i].slot_type[7];
+ u8 stype = bpf_stack_slot(st, i)->slot_type[7];
/*
* Don't clear special slots.
@@ -442,7 +442,7 @@ static void __clean_func_state(struct bpf_verifier_env *env,
* rejecting as non-scalar register fills.
*/
if (!hi_live) {
- struct bpf_reg_state *spill = &st->stack[i].spilled_ptr;
+ struct bpf_reg_state *spill = &bpf_stack_slot(st, i)->spilled_ptr;
if (lo_live && stype == STACK_SPILL) {
if (spill->type != SCALAR_VALUE)
@@ -454,7 +454,7 @@ static void __clean_func_state(struct bpf_verifier_env *env,
if (bpf_register_is_null(spill))
continue;
for (j = 0; j < 4; j++) {
- u8 *t = &st->stack[i].slot_type[j];
+ u8 *t = &bpf_stack_slot(st, i)->slot_type[j];
if (*t == STACK_SPILL)
*t = STACK_MISC;
@@ -463,7 +463,7 @@ static void __clean_func_state(struct bpf_verifier_env *env,
bpf_mark_reg_not_init(env, spill);
}
for (j = start; j < end; j++)
- st->stack[i].slot_type[j] = STACK_POISON;
+ bpf_stack_slot(st, i)->slot_type[j] = STACK_POISON;
}
}
}
@@ -707,37 +707,38 @@ static bool stacksafe(struct bpf_verifier_env *env, struct bpf_func_state *old,
* didn't use them
*/
for (i = 0; i < old->allocated_stack; i++) {
+ struct bpf_stack_state *old_slot, *cur_slot;
struct bpf_reg_state *old_reg, *cur_reg;
int im = i % BPF_REG_SIZE;
+ u8 old_type;
spi = i / BPF_REG_SIZE;
+ old_slot = bpf_stack_slot(old, spi);
+ old_type = old_slot->slot_type[im];
+ cur_slot = i < cur->allocated_stack ? bpf_stack_slot(cur, spi) : NULL;
if (exact == EXACT) {
- u8 old_type = old->stack[spi].slot_type[i % BPF_REG_SIZE];
- u8 cur_type = i < cur->allocated_stack ?
- cur->stack[spi].slot_type[i % BPF_REG_SIZE] : STACK_INVALID;
+ u8 cur_type = cur_slot ? cur_slot->slot_type[im] : STACK_INVALID;
/* STACK_INVALID and STACK_POISON are equivalent for pruning */
if (old_type == STACK_POISON)
old_type = STACK_INVALID;
if (cur_type == STACK_POISON)
cur_type = STACK_INVALID;
- if (i >= cur->allocated_stack || old_type != cur_type)
+ if (!cur_slot || old_type != cur_type)
return false;
}
- if (old->stack[spi].slot_type[i % BPF_REG_SIZE] == STACK_INVALID ||
- old->stack[spi].slot_type[i % BPF_REG_SIZE] == STACK_POISON)
+ if (old_type == STACK_INVALID || old_type == STACK_POISON)
continue;
- if (env->allow_uninit_stack &&
- old->stack[spi].slot_type[i % BPF_REG_SIZE] == STACK_MISC)
+ if (env->allow_uninit_stack && old_type == STACK_MISC)
continue;
/* explored stack has more populated slots than current stack
* and these slots were used
*/
- if (i >= cur->allocated_stack)
+ if (!cur_slot)
return false;
/*
@@ -747,8 +748,8 @@ static bool stacksafe(struct bpf_verifier_env *env, struct bpf_func_state *old,
* regsafe() to ensure scalar ids are compared.
*/
if (im == 0 || im == 4) {
- old_reg = scalar_reg_for_stack(env, &old->stack[spi], im);
- cur_reg = scalar_reg_for_stack(env, &cur->stack[spi], im);
+ old_reg = scalar_reg_for_stack(env, old_slot, im);
+ cur_reg = scalar_reg_for_stack(env, cur_slot, im);
if (old_reg && cur_reg) {
if (!regsafe(env, old_reg, cur_reg, idmap, exact))
return false;
@@ -761,21 +762,19 @@ static bool stacksafe(struct bpf_verifier_env *env, struct bpf_func_state *old,
* it will be safe with zero-initialized stack.
* The opposite is not true
*/
- if (old->stack[spi].slot_type[i % BPF_REG_SIZE] == STACK_MISC &&
- cur->stack[spi].slot_type[i % BPF_REG_SIZE] == STACK_ZERO)
+ if (old_type == STACK_MISC && cur_slot->slot_type[im] == STACK_ZERO)
continue;
- if (old->stack[spi].slot_type[i % BPF_REG_SIZE] !=
- cur->stack[spi].slot_type[i % BPF_REG_SIZE])
+ if (old_type != cur_slot->slot_type[im])
/* Ex: old explored (safe) state has STACK_SPILL in
* this stack slot, but current has STACK_MISC ->
* this verifier states are not equivalent,
* return false to continue verification of this path
*/
return false;
- if (i % BPF_REG_SIZE != BPF_REG_SIZE - 1)
+ if (im != BPF_REG_SIZE - 1)
continue;
/* Both old and cur are having same slot_type */
- switch (old->stack[spi].slot_type[BPF_REG_SIZE - 1]) {
+ switch (old_type) {
case STACK_SPILL:
/* when explored and current stack slot are both storing
* spilled registers, check that stored pointers types
@@ -787,13 +786,13 @@ static bool stacksafe(struct bpf_verifier_env *env, struct bpf_func_state *old,
* such verifier states are not equivalent.
* return false to continue verification of this path
*/
- if (!regsafe(env, &old->stack[spi].spilled_ptr,
- &cur->stack[spi].spilled_ptr, idmap, exact))
+ if (!regsafe(env, &old_slot->spilled_ptr, &cur_slot->spilled_ptr,
+ idmap, exact))
return false;
break;
case STACK_DYNPTR:
- old_reg = &old->stack[spi].spilled_ptr;
- cur_reg = &cur->stack[spi].spilled_ptr;
+ old_reg = &old_slot->spilled_ptr;
+ cur_reg = &cur_slot->spilled_ptr;
if (old_reg->dynptr.type != cur_reg->dynptr.type ||
old_reg->dynptr.first_slot != cur_reg->dynptr.first_slot ||
!check_ids(old_reg->id, cur_reg->id, idmap) ||
@@ -801,8 +800,8 @@ static bool stacksafe(struct bpf_verifier_env *env, struct bpf_func_state *old,
return false;
break;
case STACK_ITER:
- old_reg = &old->stack[spi].spilled_ptr;
- cur_reg = &cur->stack[spi].spilled_ptr;
+ old_reg = &old_slot->spilled_ptr;
+ cur_reg = &cur_slot->spilled_ptr;
/* iter.depth is not compared between states as it
* doesn't matter for correctness and would otherwise
* prevent convergence; we maintain it only to prevent
@@ -818,8 +817,8 @@ static bool stacksafe(struct bpf_verifier_env *env, struct bpf_func_state *old,
return false;
break;
case STACK_IRQ_FLAG:
- old_reg = &old->stack[spi].spilled_ptr;
- cur_reg = &cur->stack[spi].spilled_ptr;
+ old_reg = &old_slot->spilled_ptr;
+ cur_reg = &cur_slot->spilled_ptr;
if (!check_ids(old_reg->id, cur_reg->id, idmap) ||
old_reg->irq.kfunc_class != cur_reg->irq.kfunc_class)
return false;
@@ -1043,10 +1042,10 @@ static int propagate_precision(struct bpf_verifier_env *env,
first = false;
}
- for (i = 0; i < state->allocated_stack / BPF_REG_SIZE; i++) {
- if (!bpf_is_spilled_reg(&state->stack[i]))
+ for (i = 0; i < bpf_stack_nr_slots(state); i++) {
+ if (!bpf_is_spilled_reg(bpf_stack_slot(state, i)))
continue;
- state_reg = &state->stack[i].spilled_ptr;
+ state_reg = &bpf_stack_slot(state, i)->spilled_ptr;
if (state_reg->type != SCALAR_VALUE ||
!state_reg->precise)
continue;
@@ -1192,15 +1191,15 @@ static bool iter_active_depths_differ(struct bpf_verifier_state *old, struct bpf
for (fr = old->curframe; fr >= 0; fr--) {
state = old->frame[fr];
- for (i = 0; i < state->allocated_stack / BPF_REG_SIZE; i++) {
- if (state->stack[i].slot_type[0] != STACK_ITER)
+ for (i = 0; i < bpf_stack_nr_slots(state); i++) {
+ if (bpf_stack_slot(state, i)->slot_type[0] != STACK_ITER)
continue;
- slot = &state->stack[i].spilled_ptr;
+ slot = &bpf_stack_slot(state, i)->spilled_ptr;
if (slot->iter.state != BPF_ITER_STATE_ACTIVE)
continue;
- cur_slot = &cur->frame[fr]->stack[i].spilled_ptr;
+ cur_slot = &bpf_stack_slot(cur->frame[fr], i)->spilled_ptr;
if (cur_slot->iter.depth != slot->iter.depth)
return true;
}
@@ -1222,10 +1221,10 @@ static void mark_all_scalars_imprecise(struct bpf_verifier_env *env, struct bpf_
continue;
reg->precise = false;
}
- for (j = 0; j < func->allocated_stack / BPF_REG_SIZE; j++) {
- if (!bpf_is_spilled_reg(&func->stack[j]))
+ for (j = 0; j < bpf_stack_nr_slots(func); j++) {
+ if (!bpf_is_spilled_reg(bpf_stack_slot(func, j)))
continue;
- reg = &func->stack[j].spilled_ptr;
+ reg = &bpf_stack_slot(func, j)->spilled_ptr;
if (reg->type != SCALAR_VALUE)
continue;
reg->precise = false;
@@ -1328,7 +1327,7 @@ int bpf_is_state_visited(struct bpf_verifier_env *env, int insn_idx)
*/
if (is_iter_next_insn(env, insn_idx)) {
if (states_equal(env, &sl->state, cur, RANGE_WITHIN)) {
- struct bpf_func_state *cur_frame;
+ struct bpf_func_state *cur_frame, *iter_frame;
struct bpf_reg_state *iter_state, *iter_reg;
int spi;
@@ -1342,7 +1341,8 @@ int bpf_is_state_visited(struct bpf_verifier_env *env, int insn_idx)
* no need for extra (re-)validations
*/
spi = bpf_get_spi(iter_reg->var_off.value);
- iter_state = &bpf_func(env, iter_reg)->stack[spi].spilled_ptr;
+ iter_frame = bpf_func(env, iter_reg);
+ iter_state = &bpf_stack_slot(iter_frame, spi)->spilled_ptr;
if (iter_state->iter.state == BPF_ITER_STATE_ACTIVE) {
loop = true;
goto hit;
diff --git a/kernel/bpf/verifier.c b/kernel/bpf/verifier.c
index a7c9e2d8965d..058d128a369a 100644
--- a/kernel/bpf/verifier.c
+++ b/kernel/bpf/verifier.c
@@ -598,16 +598,17 @@ bool bpf_is_may_goto_insn(struct bpf_insn *insn)
static bool is_spi_bounds_valid(struct bpf_func_state *state, int spi, int nr_slots)
{
- int allocated_slots = state->allocated_stack / BPF_REG_SIZE;
+ int allocated_slots = bpf_stack_nr_slots(state);
- /* We need to check that slots between [spi - nr_slots + 1, spi] are
- * within [0, allocated_stack).
- *
- * Please note that the spi grows downwards. For example, a dynptr
- * takes the size of two stack slots; the first slot will be at
- * spi and the second slot will be at spi - 1.
- */
- return spi - nr_slots + 1 >= 0 && spi < allocated_slots;
+ /*
+ * We need to check that slots between [spi - nr_slots + 1, spi] are
+ * within [0, allocated_stack).
+ *
+ * Please note that the spi grows downwards. For example, a dynptr
+ * takes the size of two stack slots; the first slot will be at
+ * spi and the second slot will be at spi - 1.
+ */
+ return spi - nr_slots + 1 >= 0 && spi < allocated_slots;
}
static int stack_slot_obj_get_spi(struct bpf_verifier_env *env, struct bpf_reg_state *reg,
@@ -751,8 +752,8 @@ static int mark_stack_slots_dynptr(struct bpf_verifier_env *env, struct bpf_reg_
return err;
for (i = 0; i < BPF_REG_SIZE; i++) {
- state->stack[spi].slot_type[i] = STACK_DYNPTR;
- state->stack[spi - 1].slot_type[i] = STACK_DYNPTR;
+ bpf_stack_slot(state, spi)->slot_type[i] = STACK_DYNPTR;
+ bpf_stack_slot(state, spi - 1)->slot_type[i] = STACK_DYNPTR;
}
type = arg_to_dynptr_type(arg_type);
@@ -785,8 +786,8 @@ static int mark_stack_slots_dynptr(struct bpf_verifier_env *env, struct bpf_reg_
parent_id = dynptr->parent_id;
}
- mark_dynptr_stack_regs(env, &state->stack[spi].spilled_ptr,
- &state->stack[spi - 1].spilled_ptr, type, parent_id);
+ mark_dynptr_stack_regs(env, &bpf_stack_slot(state, spi)->spilled_ptr,
+ &bpf_stack_slot(state, spi - 1)->spilled_ptr, type, parent_id);
return 0;
}
@@ -818,7 +819,7 @@ static int unmark_stack_slots_dynptr(struct bpf_verifier_env *env, struct bpf_re
* all clones and derived slices. For non-referenced dynptr, only
* the dynptr and slices derived from it will be invalidated.
*/
- reg = &state->stack[spi].spilled_ptr;
+ reg = &bpf_stack_slot(state, spi)->spilled_ptr;
return release_reference(env, dynptr_type_referenced(reg->dynptr.type)
? reg->parent_id
: reg->id);
@@ -857,6 +858,7 @@ static int dynptr_ref_cnt(struct bpf_verifier_env *env, int v_parent_id)
static int destroy_if_dynptr_stack_slot(struct bpf_verifier_env *env,
struct bpf_func_state *state, int spi)
{
+ struct bpf_stack_state *slot = bpf_stack_slot(state, spi);
int err = 0;
/* We always ensure that STACK_DYNPTR is never set partially,
@@ -864,20 +866,22 @@ static int destroy_if_dynptr_stack_slot(struct bpf_verifier_env *env,
* different for STACK_SPILL, where it may be only set for
* 1 byte, so code has to use is_spilled_reg.
*/
- if (state->stack[spi].slot_type[0] != STACK_DYNPTR)
+ if (slot->slot_type[0] != STACK_DYNPTR)
return 0;
/* Reposition spi to first slot */
- if (!state->stack[spi].spilled_ptr.dynptr.first_slot)
+ if (!slot->spilled_ptr.dynptr.first_slot) {
spi = spi + 1;
+ slot = bpf_stack_slot(state, spi);
+ }
/*
* A referenced dynptr can be overwritten only if there is at
* least one other dynptr sharing the same virtual ref parent,
* ensuring the reference can still be properly released.
*/
- if (dynptr_type_referenced(state->stack[spi].spilled_ptr.dynptr.type) &&
- dynptr_ref_cnt(env, state->stack[spi].spilled_ptr.parent_id) <= 1) {
+ if (dynptr_type_referenced(slot->spilled_ptr.dynptr.type) &&
+ dynptr_ref_cnt(env, slot->spilled_ptr.parent_id) <= 1) {
verbose(env, "cannot overwrite referenced dynptr\n");
bpf_diag_res(
env, env->insn_idx, "referenced dynptr overwrite",
@@ -887,7 +891,7 @@ static int destroy_if_dynptr_stack_slot(struct bpf_verifier_env *env,
}
/* Invalidate the dynptr and any derived slices */
- err = release_reference(env, state->stack[spi].spilled_ptr.id);
+ err = release_reference(env, slot->spilled_ptr.id);
if (!err) {
mark_stack_slot_scratched(env, spi);
mark_stack_slot_scratched(env, spi - 1);
@@ -927,6 +931,7 @@ static bool is_dynptr_reg_valid_uninit(struct bpf_verifier_env *env, struct bpf_
static bool is_dynptr_reg_valid_init(struct bpf_verifier_env *env, struct bpf_reg_state *reg)
{
struct bpf_func_state *state = bpf_func(env, reg);
+ struct bpf_stack_state *slot;
int i, spi;
/* This already represents first slot of initialized bpf_dynptr.
@@ -941,12 +946,13 @@ static bool is_dynptr_reg_valid_init(struct bpf_verifier_env *env, struct bpf_re
spi = dynptr_get_spi(env, reg);
if (spi < 0)
return false;
- if (!state->stack[spi].spilled_ptr.dynptr.first_slot)
+ slot = bpf_stack_slot(state, spi);
+ if (!slot->spilled_ptr.dynptr.first_slot)
return false;
for (i = 0; i < BPF_REG_SIZE; i++) {
- if (state->stack[spi].slot_type[i] != STACK_DYNPTR ||
- state->stack[spi - 1].slot_type[i] != STACK_DYNPTR)
+ if (slot->slot_type[i] != STACK_DYNPTR ||
+ bpf_stack_slot(state, spi - 1)->slot_type[i] != STACK_DYNPTR)
return false;
}
@@ -965,7 +971,7 @@ static enum bpf_dynptr_type dynptr_reg_type(struct bpf_verifier_env *env, struct
if (spi < 0)
return BPF_DYNPTR_TYPE_INVALID;
state = bpf_func(env, reg);
- return state->stack[spi].spilled_ptr.dynptr.type;
+ return bpf_stack_slot(state, spi)->spilled_ptr.dynptr.type;
}
static bool is_dynptr_type_expected(struct bpf_verifier_env *env, struct bpf_reg_state *reg,
@@ -1005,7 +1011,7 @@ static int mark_stack_slots_iter(struct bpf_verifier_env *env,
return id;
for (i = 0; i < nr_slots; i++) {
- struct bpf_stack_state *slot = &state->stack[spi - i];
+ struct bpf_stack_state *slot = bpf_stack_slot(state, spi - i);
struct bpf_reg_state *st = &slot->spilled_ptr;
__mark_reg_known_zero(st);
@@ -1042,7 +1048,7 @@ static int unmark_stack_slots_iter(struct bpf_verifier_env *env,
return spi;
for (i = 0; i < nr_slots; i++) {
- struct bpf_stack_state *slot = &state->stack[spi - i];
+ struct bpf_stack_state *slot = bpf_stack_slot(state, spi - i);
struct bpf_reg_state *st = &slot->spilled_ptr;
if (i == 0)
@@ -1076,7 +1082,7 @@ static bool is_iter_reg_valid_uninit(struct bpf_verifier_env *env,
return false;
for (i = 0; i < nr_slots; i++) {
- struct bpf_stack_state *slot = &state->stack[spi - i];
+ struct bpf_stack_state *slot = bpf_stack_slot(state, spi - i);
for (j = 0; j < BPF_REG_SIZE; j++)
if (slot->slot_type[j] == STACK_ITER)
@@ -1097,7 +1103,7 @@ static int is_iter_reg_valid_init(struct bpf_verifier_env *env, struct bpf_reg_s
return -EINVAL;
for (i = 0; i < nr_slots; i++) {
- struct bpf_stack_state *slot = &state->stack[spi - i];
+ struct bpf_stack_state *slot = bpf_stack_slot(state, spi - i);
struct bpf_reg_state *st = &slot->spilled_ptr;
if (st->type & PTR_UNTRUSTED)
@@ -1139,7 +1145,7 @@ static int mark_stack_slot_irq_flag(struct bpf_verifier_env *env,
if (id < 0)
return id;
- slot = &state->stack[spi];
+ slot = bpf_stack_slot(state, spi);
st = &slot->spilled_ptr;
__mark_reg_known_zero(st);
@@ -1166,7 +1172,7 @@ static int unmark_stack_slot_irq_flag(struct bpf_verifier_env *env, struct bpf_r
if (spi < 0)
return spi;
- slot = &state->stack[spi];
+ slot = bpf_stack_slot(state, spi);
st = &slot->spilled_ptr;
if (st->irq.kfunc_class != kfunc_class) {
@@ -1235,7 +1241,7 @@ static bool is_irq_flag_reg_valid_uninit(struct bpf_verifier_env *env, struct bp
if (spi < 0)
return false;
- slot = &state->stack[spi];
+ slot = bpf_stack_slot(state, spi);
for (i = 0; i < BPF_REG_SIZE; i++)
if (slot->slot_type[i] == STACK_IRQ_FLAG)
@@ -1254,7 +1260,7 @@ static int is_irq_flag_reg_valid_init(struct bpf_verifier_env *env, struct bpf_r
if (spi < 0)
return -EINVAL;
- slot = &state->stack[spi];
+ slot = bpf_stack_slot(state, spi);
st = &slot->spilled_ptr;
if (!st->id)
@@ -1401,7 +1407,7 @@ static int copy_reference_state(struct bpf_verifier_state *dst, const struct bpf
static int copy_stack_state(struct bpf_func_state *dst, const struct bpf_func_state *src)
{
- size_t n = src->allocated_stack / BPF_REG_SIZE;
+ size_t n = bpf_stack_nr_slots(src);
dst->stack = copy_array(dst->stack, src->stack, n, sizeof(struct bpf_stack_state),
GFP_KERNEL_ACCOUNT);
@@ -1440,7 +1446,7 @@ static int resize_reference_state(struct bpf_verifier_state *state, size_t n)
*/
static int grow_stack_state(struct bpf_verifier_env *env, struct bpf_func_state *state, int size)
{
- size_t old_n = state->allocated_stack / BPF_REG_SIZE, n;
+ size_t old_n = bpf_stack_nr_slots(state), n;
/* The stack size is always a multiple of BPF_REG_SIZE. */
size = round_up(size, BPF_REG_SIZE);
@@ -3523,17 +3529,18 @@ static void save_register_state(struct bpf_verifier_env *env,
int spi, struct bpf_reg_state *reg,
int size)
{
+ struct bpf_stack_state *slot = bpf_stack_slot(state, spi);
int i;
- bpf_diag_mod_begin(env, &state->stack[spi].spilled_ptr, reg, BPF_DIAG_MOD_SPILL);
- state->stack[spi].spilled_ptr = *reg;
+ bpf_diag_mod_begin(env, &slot->spilled_ptr, reg, BPF_DIAG_MOD_SPILL);
+ slot->spilled_ptr = *reg;
for (i = BPF_REG_SIZE; i > BPF_REG_SIZE - size; i--)
- state->stack[spi].slot_type[i - 1] = STACK_SPILL;
+ slot->slot_type[i - 1] = STACK_SPILL;
/* size < 8 bytes spill */
for (; i; i--)
- mark_stack_slot_misc(env, &state->stack[spi].slot_type[i - 1]);
+ mark_stack_slot_misc(env, &slot->slot_type[i - 1]);
bpf_diag_mod_end(env);
}
@@ -3575,14 +3582,15 @@ static void check_fastcall_stack_contract(struct bpf_verifier_env *env,
static void scrub_special_slot(struct bpf_func_state *state, int spi)
{
+ struct bpf_stack_state *slot = bpf_stack_slot(state, spi);
int i;
/* regular write of data into stack destroys any spilled ptr */
- state->stack[spi].spilled_ptr.type = NOT_INIT;
+ slot->spilled_ptr.type = NOT_INIT;
/* Mark slots as STACK_MISC if they belonged to spilled ptr/dynptr/iter. */
- if (is_stack_slot_special(&state->stack[spi]))
+ if (is_stack_slot_special(slot))
for (i = 0; i < BPF_REG_SIZE; i++)
- scrub_spilled_slot(&state->stack[spi].slot_type[i]);
+ scrub_spilled_slot(&slot->slot_type[i]);
}
/* check_stack_{read,write}_fixed_off functions track spill/fill of registers,
@@ -3600,13 +3608,14 @@ static int check_stack_write_fixed_off(struct bpf_verifier_env *env,
struct bpf_reg_state *reg = NULL;
int insn_flags = INSN_F_STACK_ACCESS;
int hist_spi = spi, hist_frame = state->frameno;
+ struct bpf_stack_state *ss = bpf_stack_slot(state, spi);
/* caller checked that off % size == 0 and -MAX_BPF_STACK <= off < 0,
* so it's aligned access and [off, off + size) are within stack limits
*/
if (!env->allow_ptr_leaks &&
- bpf_is_spilled_reg(&state->stack[spi]) &&
- !bpf_is_spilled_scalar_reg(&state->stack[spi]) &&
+ bpf_is_spilled_reg(ss) &&
+ !bpf_is_spilled_scalar_reg(ss) &&
size != BPF_REG_SIZE) {
const char *reason;
@@ -3628,7 +3637,7 @@ static int check_stack_write_fixed_off(struct bpf_verifier_env *env,
bool sanitize = reg && is_pointer_regtype(reg->type);
for (i = 0; i < size; i++) {
- u8 type = state->stack[spi].slot_type[(slot - i) %
+ u8 type = ss->slot_type[(slot - i) %
BPF_REG_SIZE];
if (type != STACK_MISC && type != STACK_ZERO) {
@@ -3657,7 +3666,7 @@ static int check_stack_write_fixed_off(struct bpf_verifier_env *env,
save_register_state(env, state, spi, reg, size);
/* Break the relation on a narrowing spill. */
if (!reg_value_fits)
- state->stack[spi].spilled_ptr.id = 0;
+ ss->spilled_ptr.id = 0;
} else if (!reg && !(off % BPF_REG_SIZE) && is_bpf_st_mem(insn) &&
env->bpf_capable) {
struct bpf_reg_state *tmp_reg = &env->fake_reg[0];
@@ -3681,8 +3690,8 @@ static int check_stack_write_fixed_off(struct bpf_verifier_env *env,
} else {
u8 type = STACK_MISC;
- if (bpf_is_spilled_reg(&state->stack[spi]))
- bpf_diag_record_scrub(env, &state->stack[spi].spilled_ptr,
+ if (bpf_is_spilled_reg(ss))
+ bpf_diag_record_scrub(env, &ss->spilled_ptr,
BPF_DIAG_MOD_WRITE);
scrub_special_slot(state, spi);
@@ -3703,7 +3712,7 @@ static int check_stack_write_fixed_off(struct bpf_verifier_env *env,
/* Mark slots affected by this stack write. */
for (i = 0; i < size; i++)
- state->stack[spi].slot_type[(slot - i) % BPF_REG_SIZE] = type;
+ ss->slot_type[(slot - i) % BPF_REG_SIZE] = type;
insn_flags = 0; /* not a register spill */
}
@@ -3774,7 +3783,7 @@ static int check_stack_write_var_off(struct bpf_verifier_env *env,
slot = -i - 1;
spi = slot / BPF_REG_SIZE;
- stype = &state->stack[spi].slot_type[slot % BPF_REG_SIZE];
+ stype = &bpf_stack_slot(state, spi)->slot_type[slot % BPF_REG_SIZE];
mark_stack_slot_scratched(env, spi);
if (!env->allow_ptr_leaks && *stype != STACK_MISC && *stype != STACK_ZERO) {
@@ -3798,8 +3807,8 @@ static int check_stack_write_var_off(struct bpf_verifier_env *env,
* maintain the spill type.
*/
if (writing_zero && *stype == STACK_SPILL &&
- bpf_is_spilled_scalar_reg(&state->stack[spi])) {
- struct bpf_reg_state *spill_reg = &state->stack[spi].spilled_ptr;
+ bpf_is_spilled_scalar_reg(bpf_stack_slot(state, spi))) {
+ struct bpf_reg_state *spill_reg = &bpf_stack_slot(state, spi)->spilled_ptr;
if (tnum_is_const(spill_reg->var_off) && spill_reg->var_off.value == 0) {
zero_used = true;
@@ -3879,13 +3888,13 @@ static int mark_reg_stack_read(struct bpf_verifier_env *env,
slot = -i - 1;
spi = slot / BPF_REG_SIZE;
mark_stack_slot_scratched(env, spi);
- stype = ptr_state->stack[spi].slot_type;
+ stype = bpf_stack_slot(ptr_state, spi)->slot_type;
if (stype[slot % BPF_REG_SIZE] == STACK_ZERO) {
zeros++;
continue;
}
if (stype[slot % BPF_REG_SIZE] == STACK_SPILL &&
- bpf_register_is_null(&ptr_state->stack[spi].spilled_ptr)) {
+ bpf_register_is_null(&bpf_stack_slot(ptr_state, spi)->spilled_ptr)) {
zero_spill_mask |= 1ull << spi;
zeros++;
continue;
@@ -3946,9 +3955,10 @@ static int check_stack_read_fixed_off(struct bpf_verifier_env *env,
int err;
int insn_flags = INSN_F_STACK_ACCESS;
int hist_spi = spi, hist_frame = reg_state->frameno;
+ struct bpf_stack_state *ss = bpf_stack_slot(reg_state, spi);
- stype = reg_state->stack[spi].slot_type;
- reg = ®_state->stack[spi].spilled_ptr;
+ stype = ss->slot_type;
+ reg = &ss->spilled_ptr;
mark_stack_slot_scratched(env, spi);
check_fastcall_stack_contract(env, state, env->insn_idx, off);
@@ -3959,7 +3969,7 @@ static int check_stack_read_fixed_off(struct bpf_verifier_env *env,
if (dst_regno >= 0)
bpf_diag_mod_begin(env, &state->regs[dst_regno], reg, BPF_DIAG_MOD_WRITE);
- if (bpf_is_spilled_reg(®_state->stack[spi])) {
+ if (bpf_is_spilled_reg(ss)) {
u8 spill_size = 1;
for (i = BPF_REG_SIZE - 1; i > 0 && stype[i - 1] == STACK_SPILL; i--)
@@ -7048,6 +7058,7 @@ static int check_stack_range_initialized(
}
for (i = min_off; i < max_off + access_size; i++) {
+ struct bpf_stack_state *ss;
u8 *stype;
slot = -i - 1;
@@ -7057,7 +7068,8 @@ static int check_stack_range_initialized(
return -EFAULT;
}
- stype = &state->stack[spi].slot_type[slot % BPF_REG_SIZE];
+ ss = bpf_stack_slot(state, spi);
+ stype = &ss->slot_type[slot % BPF_REG_SIZE];
if (*stype == STACK_MISC)
goto mark;
if ((*stype == STACK_ZERO) ||
@@ -7069,13 +7081,13 @@ static int check_stack_range_initialized(
goto mark;
}
- if (bpf_is_spilled_reg(&state->stack[spi]) &&
- (state->stack[spi].spilled_ptr.type == SCALAR_VALUE ||
+ if (bpf_is_spilled_reg(ss) &&
+ (ss->spilled_ptr.type == SCALAR_VALUE ||
env->allow_ptr_leaks)) {
if (clobber) {
- __mark_reg_unknown(env, &state->stack[spi].spilled_ptr);
+ __mark_reg_unknown(env, &ss->spilled_ptr);
for (j = 0; j < BPF_REG_SIZE; j++)
- scrub_spilled_slot(&state->stack[spi].slot_type[j]);
+ scrub_spilled_slot(&ss->slot_type[j]);
}
goto mark;
}
@@ -7805,7 +7817,7 @@ static int process_dynptr_func(struct bpf_verifier_env *env, struct bpf_reg_stat
mark_stack_slots_scratched(env, spi, BPF_DYNPTR_NR_SLOTS);
- reg = &state->stack[spi].spilled_ptr;
+ reg = &bpf_stack_slot(state, spi)->spilled_ptr;
}
meta->dynptr.type = reg->dynptr.type;
@@ -7938,7 +7950,7 @@ static int process_iter_arg(struct bpf_verifier_env *env, struct bpf_reg_state *
/* remember meta->iter info for process_iter_next_call() */
meta->iter.spi = spi;
meta->iter.frameno = reg->frameno;
- update_ref_obj(&meta->ref_obj, &state->stack[spi].spilled_ptr);
+ update_ref_obj(&meta->ref_obj, &bpf_stack_slot(state, spi)->spilled_ptr);
if (is_iter_destroy_kfunc(meta)) {
err = unmark_stack_slots_iter(env, reg, nr_slots);
@@ -8015,16 +8027,15 @@ static int widen_imprecise_scalars(struct bpf_verifier_env *env,
&fold->regs[i],
&fcur->regs[i]);
- num_slots = min(fold->allocated_stack / BPF_REG_SIZE,
- fcur->allocated_stack / BPF_REG_SIZE);
+ num_slots = min(bpf_stack_nr_slots(fold), bpf_stack_nr_slots(fcur));
for (i = 0; i < num_slots; i++) {
- if (!bpf_is_spilled_reg(&fold->stack[i]) ||
- !bpf_is_spilled_reg(&fcur->stack[i]))
+ if (!bpf_is_spilled_reg(bpf_stack_slot(fold, i)) ||
+ !bpf_is_spilled_reg(bpf_stack_slot(fcur, i)))
continue;
maybe_widen_reg(env,
- &fold->stack[i].spilled_ptr,
- &fcur->stack[i].spilled_ptr);
+ &bpf_stack_slot(fold, i)->spilled_ptr,
+ &bpf_stack_slot(fcur, i)->spilled_ptr);
}
}
return 0;
@@ -8036,7 +8047,7 @@ static struct bpf_reg_state *get_iter_from_state(struct bpf_verifier_state *cur_
int iter_frameno = meta->iter.frameno;
int iter_spi = meta->iter.spi;
- return &cur_st->frame[iter_frameno]->stack[iter_spi].spilled_ptr;
+ return &bpf_stack_slot(cur_st->frame[iter_frameno], iter_spi)->spilled_ptr;
}
/* process_iter_next_call() is called when verifier gets to iterator's next
@@ -8841,7 +8852,7 @@ static int get_constant_map_key(struct bpf_verifier_env *env,
slot = -stack_off - 1;
spi = slot / BPF_REG_SIZE;
off = slot % BPF_REG_SIZE;
- stype = state->stack[spi].slot_type;
+ stype = bpf_stack_slot(state, spi)->slot_type;
/* First handle precisely tracked STACK_ZERO */
for (i = off; i >= 0 && stype[i] == STACK_ZERO; i--)
@@ -8852,14 +8863,14 @@ static int get_constant_map_key(struct bpf_verifier_env *env,
}
/* Check that stack contains a scalar spill of expected size */
- if (!bpf_is_spilled_scalar_reg(&state->stack[spi]))
+ if (!bpf_is_spilled_scalar_reg(bpf_stack_slot(state, spi)))
return -EOPNOTSUPP;
for (i = off; i >= 0 && stype[i] == STACK_SPILL; i--)
spill_size++;
if (spill_size != key_size)
return -EOPNOTSUPP;
- reg = &state->stack[spi].spilled_ptr;
+ reg = &bpf_stack_slot(state, spi)->spilled_ptr;
if (!tnum_is_const(reg->var_off))
/* Stack value not statically known */
return -EOPNOTSUPP;
@@ -17322,10 +17333,10 @@ static void collect_linked_regs(struct bpf_verifier_env *env,
reg = &func->regs[j];
__collect_linked_regs(linked_regs, reg, id, i, j, true);
}
- for (j = 0; j < func->allocated_stack / BPF_REG_SIZE; j++) {
- if (!bpf_is_spilled_reg(&func->stack[j]))
+ for (j = 0; j < bpf_stack_nr_slots(func); j++) {
+ if (!bpf_is_spilled_reg(bpf_stack_slot(func, j)))
continue;
- reg = &func->stack[j].spilled_ptr;
+ reg = &bpf_stack_slot(func, j)->spilled_ptr;
__collect_linked_regs(linked_regs, reg, id, i, j, false);
}
}
@@ -17345,7 +17356,7 @@ static void sync_linked_regs(struct bpf_verifier_env *env, struct bpf_verifier_s
for (i = 0; i < linked_regs->cnt; ++i) {
e = &linked_regs->entries[i];
reg = e->is_reg ? &vstate->frame[e->frameno]->regs[e->regno]
- : &vstate->frame[e->frameno]->stack[e->spi].spilled_ptr;
+ : &bpf_stack_slot(vstate->frame[e->frameno], e->spi)->spilled_ptr;
if (reg->type != SCALAR_VALUE || reg == known_reg)
continue;
if ((reg->id & ~BPF_ADD_CONST) != (known_reg->id & ~BPF_ADD_CONST))
--
2.53.0
^ permalink raw reply related [flat|nested] 34+ messages in thread* Re: [PATCH bpf-next v1 01/18] bpf: Add accessors for verifier stack slots
2026-09-23 19:11 ` [PATCH bpf-next v1 01/18] bpf: Add accessors for verifier stack slots Kumar Kartikeya Dwivedi
@ 2026-09-23 19:57 ` bot+bpf-ci
2026-09-23 20:04 ` Kumar Kartikeya Dwivedi
0 siblings, 1 reply; 34+ messages in thread
From: bot+bpf-ci @ 2026-09-23 19:57 UTC (permalink / raw)
To: memxor, bpf
Cc: ast, andrii, daniel, eddyz87, emil, tj, kkd, kernel-team, ast,
andrii, daniel, martin.lau, eddyz87, yonghong.song, mason,
ihor.solodrai
[-- Attachment #1: Type: text/plain, Size: 3563 bytes --]
> bpf: Add accessors for verifier stack slots
>
> The verifier indexes a frame's stack state directly through
> state->stack[spi] and computes the number of tracked slots as
> allocated_stack / BPF_REG_SIZE in every file that touches stack slots.
> Route all of these through two helpers, bpf_stack_slot() and
> bpf_stack_nr_slots(), so the layout of the per-frame stack state is
> visible in one place and can change without touching every user.
> Both take a const frame: the slot accessor returns the slot through the
> frame's stack pointer, so read-only code such as the state printer can
> use it without giving up its qualifiers. Functions that look up the
> same slot repeatedly now fetch it once.
>
> No functional change.
>
> Signed-off-by: Kumar Kartikeya Dwivedi <memxor@gmail.com>
The commit message makes two universal claims that the tree does not back
up after this patch.
1) "Route all of these through two helpers ... so the layout of the
per-frame stack state is visible in one place and can change without
touching every user."
At least two users of struct bpf_func_state::stack still depend on the raw
layout:
drivers/net/ethernet/netronome/nfp/bpf/verifier.c:nfp_bpf_map_update_value_ok():
state = env->cur_state->frame[reg3->frameno];
...
soff = -(off + i) - 1;
stack_entry = &state->stack[soff / BPF_REG_SIZE];
This is the exact state->stack[spi] indexing the message says was removed.
It was not converted.
kernel/bpf/diagnostics.c:reg_to_target(), which this patch edits (the nslots
line), still does raw pointer arithmetic that assumes stack is a flat array:
start = (unsigned long)state->stack;
end = (unsigned long)(state->stack + nslots);
...
spi = ((const char *)reg - (const char *)state->stack) /
sizeof(*state->stack);
If bpf_stack_slot() ever stops being &frame->stack[spi], both sites will
silently compute wrong slots. So the stated goal (the layout can change
without touching every user) does not hold yet.
Nothing later in 277f2757dee8..7757cddf610a changes the helpers, the layout,
or the nfp code, so this is not a runtime bug in the series. It is an
incomplete conversion, or a message that needs narrowing.
2) "Functions that look up the same slot repeatedly now fetch it once."
Several converted functions still call the accessor twice for the same
index:
bpf_get_spilled_reg():
(1 << bpf_stack_slot(frame, slot)->slot_type[BPF_REG_SIZE - 1]) & mask)
return &bpf_stack_slot(frame, slot)->spilled_ptr;
bpf_mark_all_scalars_precise():
if (!bpf_is_spilled_reg(bpf_stack_slot(func, j)))
reg = &bpf_stack_slot(func, j)->spilled_ptr;
bpf_mark_chain_precision():
if (!bpf_is_spilled_scalar_reg(bpf_stack_slot(func, i))) {
reg = &bpf_stack_slot(func, i)->spilled_ptr;
The same pattern appears in states.c (propagate_precision,
iter_active_depths_differ, mark_all_scalars_imprecise) and verifier.c
(check_stack_write_var_off, get_constant_map_key, collect_linked_regs,
widen_imprecise_scalars). Only some functions, such as
print_verifier_state(), were changed to fetch once.
Should nfp_bpf_map_update_value_ok() be converted to bpf_stack_slot(), and
reg_to_target() given a note or helper for its address-range check?
Otherwise the changelog could be narrowed to match what the patch actually
does.
---
AI reviewed your patch. Please fix the bug or email reply why it's not a bug.
See: https://github.com/kernel-patches/vmtest/blob/master/ci/claude/README.md
CI run summary: https://github.com/kernel-patches/bpf/actions/runs/35909853278
^ permalink raw reply [flat|nested] 34+ messages in thread* Re: [PATCH bpf-next v1 01/18] bpf: Add accessors for verifier stack slots
2026-09-23 19:57 ` bot+bpf-ci
@ 2026-09-23 20:04 ` Kumar Kartikeya Dwivedi
0 siblings, 0 replies; 34+ messages in thread
From: Kumar Kartikeya Dwivedi @ 2026-09-23 20:04 UTC (permalink / raw)
To: bot+bpf-ci, bpf
Cc: ast, andrii, daniel, eddyz87, emil, tj, kkd, kernel-team,
martin.lau, yonghong.song, mason, ihor.solodrai
On Wed Sep 23, 2026 at 9:57 PM CEST, bot+bpf-ci wrote:
>> bpf: Add accessors for verifier stack slots
>>
>> The verifier indexes a frame's stack state directly through
>> state->stack[spi] and computes the number of tracked slots as
>> allocated_stack / BPF_REG_SIZE in every file that touches stack slots.
>> Route all of these through two helpers, bpf_stack_slot() and
>> bpf_stack_nr_slots(), so the layout of the per-frame stack state is
>> visible in one place and can change without touching every user.
>> Both take a const frame: the slot accessor returns the slot through the
>> frame's stack pointer, so read-only code such as the state printer can
>> use it without giving up its qualifiers. Functions that look up the
>> same slot repeatedly now fetch it once.
>>
>> No functional change.
>>
>> Signed-off-by: Kumar Kartikeya Dwivedi <memxor@gmail.com>
>
> The commit message makes two universal claims that the tree does not back
> up after this patch.
>
> 1) "Route all of these through two helpers ... so the layout of the
> per-frame stack state is visible in one place and can change without
> touching every user."
>
> At least two users of struct bpf_func_state::stack still depend on the raw
> layout:
>
> drivers/net/ethernet/netronome/nfp/bpf/verifier.c:nfp_bpf_map_update_value_ok():
>
> state = env->cur_state->frame[reg3->frameno];
> ...
> soff = -(off + i) - 1;
> stack_entry = &state->stack[soff / BPF_REG_SIZE];
>
This is correct, will fix.
> This is the exact state->stack[spi] indexing the message says was removed.
> It was not converted.
>
> kernel/bpf/diagnostics.c:reg_to_target(), which this patch edits (the nslots
> line), still does raw pointer arithmetic that assumes stack is a flat array:
>
> start = (unsigned long)state->stack;
> end = (unsigned long)(state->stack + nslots);
> ...
> spi = ((const char *)reg - (const char *)state->stack) /
> sizeof(*state->stack);
>
> If bpf_stack_slot() ever stops being &frame->stack[spi], both sites will
> silently compute wrong slots. So the stated goal (the layout can change
> without touching every user) does not hold yet.
>
> Nothing later in 277f2757dee8..7757cddf610a changes the helpers, the layout,
> or the nfp code, so this is not a runtime bug in the series. It is an
> incomplete conversion, or a message that needs narrowing.
>
Yep, also needs fixing.
> 2) "Functions that look up the same slot repeatedly now fetch it once."
>
> Several converted functions still call the accessor twice for the same
> index:
>
> bpf_get_spilled_reg():
>
> (1 << bpf_stack_slot(frame, slot)->slot_type[BPF_REG_SIZE - 1]) & mask)
> return &bpf_stack_slot(frame, slot)->spilled_ptr;
>
> bpf_mark_all_scalars_precise():
>
> if (!bpf_is_spilled_reg(bpf_stack_slot(func, j)))
> reg = &bpf_stack_slot(func, j)->spilled_ptr;
>
> bpf_mark_chain_precision():
>
> if (!bpf_is_spilled_scalar_reg(bpf_stack_slot(func, i))) {
> reg = &bpf_stack_slot(func, i)->spilled_ptr;
>
> The same pattern appears in states.c (propagate_precision,
> iter_active_depths_differ, mark_all_scalars_imprecise) and verifier.c
> (check_stack_write_var_off, get_constant_map_key, collect_linked_regs,
> widen_imprecise_scalars). Only some functions, such as
> print_verifier_state(), were changed to fetch once.
>
This is less serious, but will fix.
> Should nfp_bpf_map_update_value_ok() be converted to bpf_stack_slot(), and
> reg_to_target() given a note or helper for its address-range check?
> Otherwise the changelog could be narrowed to match what the patch actually
> does.
>
>
> ---
> AI reviewed your patch. Please fix the bug or email reply why it's not a bug.
> See: https://github.com/kernel-patches/vmtest/blob/master/ci/claude/README.md
>
> CI run summary: https://github.com/kernel-patches/bpf/actions/runs/35909853278
^ permalink raw reply [flat|nested] 34+ messages in thread
* [PATCH bpf-next v1 02/18] bpf: Widen the stack slot index in the jump history
2026-09-23 19:11 [PATCH bpf-next v1 00/18] Raise BPF program stack size to 2KiB Kumar Kartikeya Dwivedi
2026-09-23 19:11 ` [PATCH bpf-next v1 01/18] bpf: Add accessors for verifier stack slots Kumar Kartikeya Dwivedi
@ 2026-09-23 19:11 ` Kumar Kartikeya Dwivedi
2026-09-23 19:11 ` [PATCH bpf-next v1 03/18] bpf: Store linked registers in the jump history as an array Kumar Kartikeya Dwivedi
` (15 subsequent siblings)
17 siblings, 0 replies; 34+ messages in thread
From: Kumar Kartikeya Dwivedi @ 2026-09-23 19:11 UTC (permalink / raw)
To: bpf
Cc: Alexei Starovoitov, Andrii Nakryiko, Daniel Borkmann,
Eduard Zingerman, Emil Tsalapatis, Tejun Heo, kkd, kernel-team
Jump history entries record the stack slot touched by a spill or fill in
a 6-bit field, which only fits the 64 slots of a 512-byte frame and is
pinned to that size by a static_assert on MAX_BPF_STACK. Move the flags
into the first word and give the slot index 12 bits of the second word
instead, so the entry stays 16 bytes while frames of up to 32 KiB can
be recorded.
Introduce MAX_BPF_STACK_SLOTS for the number of 8-byte slots a frame can
have and use it for the static_assert and for BPF_ID_MAP_SIZE, so the
verifier expresses per-frame capacity through one constant.
No functional change.
Signed-off-by: Kumar Kartikeya Dwivedi <memxor@gmail.com>
---
include/linux/bpf_verifier.h | 21 ++++++++++++++-------
1 file changed, 14 insertions(+), 7 deletions(-)
diff --git a/include/linux/bpf_verifier.h b/include/linux/bpf_verifier.h
index 36e0c6b97533..b1c0f981fe68 100644
--- a/include/linux/bpf_verifier.h
+++ b/include/linux/bpf_verifier.h
@@ -242,6 +242,14 @@ enum bpf_stack_slot_type {
#define BPF_REG_SIZE 8 /* size of eBPF register in bytes */
+/*
+ * Largest number of BPF_REG_SIZE stack slots a single frame can have. A frame
+ * may use any part of the MAX_BPF_STACK budget; check_max_stack_depth()
+ * enforces the bound on the combined depth of frames sharing the kernel stack
+ * and on each frame using a private stack.
+ */
+#define MAX_BPF_STACK_SLOTS (MAX_BPF_STACK / BPF_REG_SIZE)
+
/* 4-byte stack slot granularity for liveness analysis */
#define BPF_HALF_REG_SIZE 4
#define STACK_SLOT_SZ 4
@@ -424,12 +432,11 @@ struct bpf_jmp_history_entry {
/* insn idx can't be bigger than 1 million */
u32 idx : 20;
u32 frame : 4; /* stack access frame number */
- u32 spi : 6; /* stack slot index (0..63) */
- u32 : 2;
- u32 prev_idx : 20;
/* special INSN_F_xxx flags */
u32 flags : 4;
- u32 : 8;
+ u32 : 4;
+ u32 prev_idx : 20;
+ u32 spi : 12; /* stack slot index */
/*
* additional registers that need precision tracking when this
* jump is backtracked, vector of five 11-bit records
@@ -438,12 +445,12 @@ struct bpf_jmp_history_entry {
};
static_assert(MAX_CALL_FRAMES <= (1 << 4));
-static_assert(MAX_BPF_STACK / 8 <= (1 << 6));
+static_assert(MAX_BPF_STACK_SLOTS <= (1 << 12));
/* Maximum number of bpf_reg_state objects that can exist at once */
#define MAX_STACK_ARG_SLOTS (MAX_BPF_FUNC_ARGS - MAX_BPF_FUNC_REG_ARGS)
-#define BPF_ID_MAP_SIZE ((MAX_BPF_REG + MAX_BPF_STACK / BPF_REG_SIZE + \
- MAX_STACK_ARG_SLOTS) * MAX_CALL_FRAMES)
+#define BPF_ID_MAP_SIZE ((MAX_BPF_REG + MAX_BPF_STACK_SLOTS + MAX_STACK_ARG_SLOTS) * \
+ MAX_CALL_FRAMES)
struct bpf_verifier_state {
/* call stack tracking */
struct bpf_func_state *frame[MAX_CALL_FRAMES];
--
2.53.0
^ permalink raw reply related [flat|nested] 34+ messages in thread* [PATCH bpf-next v1 03/18] bpf: Store linked registers in the jump history as an array
2026-09-23 19:11 [PATCH bpf-next v1 00/18] Raise BPF program stack size to 2KiB Kumar Kartikeya Dwivedi
2026-09-23 19:11 ` [PATCH bpf-next v1 01/18] bpf: Add accessors for verifier stack slots Kumar Kartikeya Dwivedi
2026-09-23 19:11 ` [PATCH bpf-next v1 02/18] bpf: Widen the stack slot index in the jump history Kumar Kartikeya Dwivedi
@ 2026-09-23 19:11 ` Kumar Kartikeya Dwivedi
2026-09-23 19:11 ` [PATCH bpf-next v1 04/18] bpf: Track backtracking stack slots with bitmaps Kumar Kartikeya Dwivedi
` (14 subsequent siblings)
17 siblings, 0 replies; 34+ messages in thread
From: Kumar Kartikeya Dwivedi @ 2026-09-23 19:11 UTC (permalink / raw)
To: bpf
Cc: Alexei Starovoitov, Andrii Nakryiko, Daniel Borkmann,
Eduard Zingerman, Emil Tsalapatis, Tejun Heo, kkd, kernel-team
Linked scalar registers are recorded in the jump history packed into a
u64 as five 11-bit entries, each naming a frame and a register or stack
slot. The 6-bit slot field covers exactly the 64 slots of a 512-byte
frame, so a spilled scalar in a deeper slot could not be linked and
larger frames were ruled out by construction.
Store the linked registers as an array of five u16 entries plus a count
instead, each entry holding the frame number, a register-or-slot bit
and an 11-bit register or slot index, which covers frames of up to
16 KiB. Callers that record no linked registers pass NULL.
The history entry grows from 16 to 20 bytes, and the history is the
one verifier structure whose size follows the number of instructions a
loop iterates over rather than the state count. Measured over the 5075
selftest programs, peak verifier memory is unchanged for all but the
loop-heavy ones, which grow by 8 to 16%: loop1/nested_loops from 17.6 to
19.2 MiB, verifier_loops1/jumps_out_rather_than_in from 4.4 to 5.1 MiB,
strobemeta by 0.3% and pyperf600_nounroll by 0.8%. Verdicts, processed
instructions and state counts stay the same everywhere.
No functional change.
Signed-off-by: Kumar Kartikeya Dwivedi <memxor@gmail.com>
---
include/linux/bpf_verifier.h | 16 +++++--
kernel/bpf/backtrack.c | 18 +++++---
kernel/bpf/states.c | 2 +-
kernel/bpf/verifier.c | 86 ++++++++++++++++--------------------
4 files changed, 62 insertions(+), 60 deletions(-)
diff --git a/include/linux/bpf_verifier.h b/include/linux/bpf_verifier.h
index b1c0f981fe68..77cda3b3a3e9 100644
--- a/include/linux/bpf_verifier.h
+++ b/include/linux/bpf_verifier.h
@@ -428,6 +428,9 @@ enum {
INSN_F_STACK_ARG_ACCESS = BIT(3),
};
+/* Registers linked to one jump condition that a history entry can record */
+#define BPF_LINKED_REGS_MAX 5
+
struct bpf_jmp_history_entry {
/* insn idx can't be bigger than 1 million */
u32 idx : 20;
@@ -438,10 +441,14 @@ struct bpf_jmp_history_entry {
u32 prev_idx : 20;
u32 spi : 12; /* stack slot index */
/*
- * additional registers that need precision tracking when this
- * jump is backtracked, vector of five 11-bit records
+ * Scalar registers and spilled scalars linked to the condition of
+ * this jump, which need precision tracking together when the jump is
+ * backtracked. Each is packed as 4 bits of frame number, one bit
+ * telling a register from a stack slot and 11 bits of register or
+ * slot index, see linked_regs_pack().
*/
- u64 linked_regs;
+ u16 linked_regs[BPF_LINKED_REGS_MAX];
+ u8 linked_regs_cnt;
};
static_assert(MAX_CALL_FRAMES <= (1 << 4));
@@ -1249,7 +1256,8 @@ struct list_head *bpf_explored_state(struct bpf_verifier_env *env, int idx);
void bpf_free_verifier_state(struct bpf_verifier_state *state, bool free_self);
void bpf_free_backedges(struct bpf_scc_visit *visit);
int bpf_push_jmp_history(struct bpf_verifier_env *env, struct bpf_verifier_state *cur,
- int insn_flags, int spi, int frame, u64 linked_regs);
+ int insn_flags, int spi, int frame, const u16 *linked_regs,
+ u8 linked_regs_cnt);
void bpf_bt_sync_linked_regs(struct backtrack_state *bt, struct bpf_jmp_history_entry *hist);
void bpf_mark_reg_not_init(const struct bpf_verifier_env *env,
struct bpf_reg_state *reg);
diff --git a/kernel/bpf/backtrack.c b/kernel/bpf/backtrack.c
index 4bec7b94796b..6733d4078930 100644
--- a/kernel/bpf/backtrack.c
+++ b/kernel/bpf/backtrack.c
@@ -9,7 +9,8 @@
/* for any branch, call, exit record the history of jmps in the given state */
int bpf_push_jmp_history(struct bpf_verifier_env *env, struct bpf_verifier_state *cur,
- int insn_flags, int spi, int frame, u64 linked_regs)
+ int insn_flags, int spi, int frame, const u16 *linked_regs,
+ u8 linked_regs_cnt)
{
u32 cnt = cur->jmp_history_cnt;
struct bpf_jmp_history_entry *p;
@@ -27,10 +28,13 @@ int bpf_push_jmp_history(struct bpf_verifier_env *env, struct bpf_verifier_state
env->cur_hist_ent->flags |= insn_flags;
env->cur_hist_ent->spi = spi;
env->cur_hist_ent->frame = frame;
- verifier_bug_if(env->cur_hist_ent->linked_regs != 0, env,
- "insn history: insn_idx %d linked_regs: %#llx",
- env->insn_idx, env->cur_hist_ent->linked_regs);
- env->cur_hist_ent->linked_regs = linked_regs;
+ verifier_bug_if(env->cur_hist_ent->linked_regs_cnt != 0, env,
+ "insn history: insn_idx %d has %u linked regs",
+ env->insn_idx, env->cur_hist_ent->linked_regs_cnt);
+ if (linked_regs_cnt)
+ memcpy(env->cur_hist_ent->linked_regs, linked_regs,
+ linked_regs_cnt * sizeof(*linked_regs));
+ env->cur_hist_ent->linked_regs_cnt = linked_regs_cnt;
return 0;
}
@@ -47,7 +51,9 @@ int bpf_push_jmp_history(struct bpf_verifier_env *env, struct bpf_verifier_state
p->flags = insn_flags;
p->spi = spi;
p->frame = frame;
- p->linked_regs = linked_regs;
+ if (linked_regs_cnt)
+ memcpy(p->linked_regs, linked_regs, linked_regs_cnt * sizeof(*linked_regs));
+ p->linked_regs_cnt = linked_regs_cnt;
cur->jmp_history_cnt = cnt;
env->cur_hist_ent = p;
diff --git a/kernel/bpf/states.c b/kernel/bpf/states.c
index 38795cf35247..e84f37d724d5 100644
--- a/kernel/bpf/states.c
+++ b/kernel/bpf/states.c
@@ -1410,7 +1410,7 @@ int bpf_is_state_visited(struct bpf_verifier_env *env, int insn_idx)
*/
err = 0;
if (bpf_is_jmp_point(env, env->insn_idx))
- err = bpf_push_jmp_history(env, cur, 0, 0, 0, 0);
+ err = bpf_push_jmp_history(env, cur, 0, 0, 0, NULL, 0);
err = err ? : propagate_precision(env, &sl->state, cur, NULL);
if (err)
return err;
diff --git a/kernel/bpf/verifier.c b/kernel/bpf/verifier.c
index 058d128a369a..fa80f95105fb 100644
--- a/kernel/bpf/verifier.c
+++ b/kernel/bpf/verifier.c
@@ -3308,26 +3308,25 @@ static void mark_non_stack_access(struct bpf_verifier_env *env, int idx)
env->insn_aux_data[idx].non_stack_access = true;
}
+/* Layout of one packed linked register in the jump history, see linked_regs_pack() */
#define LR_FRAMENO_BITS 4
-#define LR_SPI_BITS 6
-#define LR_ENTRY_BITS (LR_SPI_BITS + LR_FRAMENO_BITS + 1)
-#define LR_SIZE_BITS 4
-#define LR_FRAMENO_MASK ((1ull << LR_FRAMENO_BITS) - 1)
-#define LR_SPI_MASK ((1ull << LR_SPI_BITS) - 1)
-#define LR_SIZE_MASK ((1ull << LR_SIZE_BITS) - 1)
-#define LR_SPI_OFF LR_FRAMENO_BITS
-#define LR_IS_REG_OFF (LR_SPI_BITS + LR_FRAMENO_BITS)
-#define LINKED_REGS_MAX 5
+#define LR_INDEX_BITS 11
+#define LR_FRAMENO_MASK ((1u << LR_FRAMENO_BITS) - 1)
+#define LR_IS_REG BIT(LR_FRAMENO_BITS)
+#define LR_INDEX_OFF (LR_FRAMENO_BITS + 1)
+#define LR_INDEX_MASK ((1u << LR_INDEX_BITS) - 1)
+#define LINKED_REGS_MAX BPF_LINKED_REGS_MAX
static_assert(MAX_CALL_FRAMES <= (1 << LR_FRAMENO_BITS));
-static_assert(LINKED_REGS_MAX < (1 << LR_SIZE_BITS));
-static_assert(LINKED_REGS_MAX * LR_ENTRY_BITS + LR_SIZE_BITS <= 64);
+static_assert(MAX_BPF_REG <= (1 << LR_INDEX_BITS));
+static_assert(MAX_BPF_STACK_SLOTS <= (1 << LR_INDEX_BITS));
+static_assert(LR_INDEX_OFF + LR_INDEX_BITS <= 16);
struct linked_reg {
u8 frameno;
union {
- u8 spi;
- u8 regno;
+ u16 spi;
+ u16 regno;
};
bool is_reg;
};
@@ -3346,48 +3345,34 @@ static struct linked_reg *linked_regs_push(struct linked_regs *s)
}
/*
- * Use u64 as a vector of 5 11-bit values, use first 4-bits to track
- * number of elements currently in stack.
- * Pack one history entry for linked registers as 11 bits in the following format:
- * - 4-bits frameno
- * - 6-bits spi_or_reg
- * - 1-bit is_reg
+ * Pack linked registers for a jump history entry, one u16 each:
+ * - 4 bits frameno
+ * - 1 bit is_reg
+ * - 11 bits register or stack slot index
*/
-static u64 linked_regs_pack(struct linked_regs *s)
+static void linked_regs_pack(const struct linked_regs *s, u16 *packed)
{
- u64 val = 0;
int i;
for (i = 0; i < s->cnt; ++i) {
- struct linked_reg *e = &s->entries[i];
- u64 tmp = 0;
-
- tmp |= e->frameno;
- tmp |= e->spi << LR_SPI_OFF;
- tmp |= (e->is_reg ? 1 : 0) << LR_IS_REG_OFF;
+ const struct linked_reg *e = &s->entries[i];
- val <<= LR_ENTRY_BITS;
- val |= tmp;
+ packed[i] = e->frameno | (e->is_reg ? LR_IS_REG : 0) | (e->spi << LR_INDEX_OFF);
}
- val <<= LR_SIZE_BITS;
- val |= s->cnt;
- return val;
}
-static void linked_regs_unpack(u64 val, struct linked_regs *s)
+static void linked_regs_unpack(const struct bpf_jmp_history_entry *hist, struct linked_regs *s)
{
int i;
- s->cnt = val & LR_SIZE_MASK;
- val >>= LR_SIZE_BITS;
-
+ s->cnt = hist->linked_regs_cnt;
for (i = 0; i < s->cnt; ++i) {
struct linked_reg *e = &s->entries[i];
+ u16 packed = hist->linked_regs[i];
- e->frameno = val & LR_FRAMENO_MASK;
- e->spi = (val >> LR_SPI_OFF) & LR_SPI_MASK;
- e->is_reg = (val >> LR_IS_REG_OFF) & 0x1;
- val >>= LR_ENTRY_BITS;
+ e->frameno = packed & LR_FRAMENO_MASK;
+ e->is_reg = packed & LR_IS_REG;
+ e->spi = (packed >> LR_INDEX_OFF) & LR_INDEX_MASK;
}
}
@@ -3429,10 +3414,10 @@ void bpf_bt_sync_linked_regs(struct backtrack_state *bt, struct bpf_jmp_history_
bool some_precise = false;
int i;
- if (!hist || hist->linked_regs == 0)
+ if (!hist || !hist->linked_regs_cnt)
return;
- linked_regs_unpack(hist->linked_regs, &linked_regs);
+ linked_regs_unpack(hist, &linked_regs);
for (i = 0; i < linked_regs.cnt; ++i) {
struct linked_reg *e = &linked_regs.entries[i];
@@ -3718,7 +3703,7 @@ static int check_stack_write_fixed_off(struct bpf_verifier_env *env,
if (insn_flags)
return bpf_push_jmp_history(env, env->cur_state, insn_flags,
- hist_spi, hist_frame, 0);
+ hist_spi, hist_frame, NULL, 0);
return 0;
}
@@ -4095,7 +4080,7 @@ static int check_stack_read_fixed_off(struct bpf_verifier_env *env,
}
if (insn_flags)
return bpf_push_jmp_history(env, env->cur_state, insn_flags,
- hist_spi, hist_frame, 0);
+ hist_spi, hist_frame, NULL, 0);
return 0;
}
@@ -4285,7 +4270,7 @@ static int check_stack_arg_write(struct bpf_verifier_env *env, struct bpf_func_s
bpf_diag_mod_end(env);
state->no_stack_arg_load = true;
return bpf_push_jmp_history(env, env->cur_state,
- INSN_F_STACK_ARG_ACCESS, spi, 0, 0);
+ INSN_F_STACK_ARG_ACCESS, spi, 0, NULL, 0);
}
/*
@@ -4319,7 +4304,7 @@ static int check_stack_arg_read(struct bpf_verifier_env *env, struct bpf_func_st
cur->regs[dst_regno] = *arg;
bpf_diag_mod_end(env);
return bpf_push_jmp_history(env, env->cur_state,
- INSN_F_STACK_ARG_ACCESS, spi, 0, 0);
+ INSN_F_STACK_ARG_ACCESS, spi, 0, NULL, 0);
}
static int mark_stack_arg_precision(struct bpf_verifier_env *env, int arg_idx)
@@ -17474,7 +17459,7 @@ static int check_cond_jmp_op(struct bpf_verifier_env *env,
}
if (insn_flags) {
- err = bpf_push_jmp_history(env, this_branch, insn_flags, 0, 0, 0);
+ err = bpf_push_jmp_history(env, this_branch, insn_flags, 0, 0, NULL, 0);
if (err)
return err;
}
@@ -17544,7 +17529,10 @@ static int check_cond_jmp_op(struct bpf_verifier_env *env,
* if parent state is created.
*/
if (linked_regs.cnt > 1) {
- err = bpf_push_jmp_history(env, this_branch, 0, 0, 0, linked_regs_pack(&linked_regs));
+ u16 packed[LINKED_REGS_MAX];
+
+ linked_regs_pack(&linked_regs, packed);
+ err = bpf_push_jmp_history(env, this_branch, 0, 0, 0, packed, linked_regs.cnt);
if (err)
return err;
}
@@ -18938,7 +18926,7 @@ static int do_check(struct bpf_verifier_env *env)
}
if (bpf_is_jmp_point(env, env->insn_idx)) {
- err = bpf_push_jmp_history(env, state, 0, 0, 0, 0);
+ err = bpf_push_jmp_history(env, state, 0, 0, 0, NULL, 0);
if (err)
return err;
}
--
2.53.0
^ permalink raw reply related [flat|nested] 34+ messages in thread* [PATCH bpf-next v1 04/18] bpf: Track backtracking stack slots with bitmaps
2026-09-23 19:11 [PATCH bpf-next v1 00/18] Raise BPF program stack size to 2KiB Kumar Kartikeya Dwivedi
` (2 preceding siblings ...)
2026-09-23 19:11 ` [PATCH bpf-next v1 03/18] bpf: Store linked registers in the jump history as an array Kumar Kartikeya Dwivedi
@ 2026-09-23 19:11 ` Kumar Kartikeya Dwivedi
2026-09-23 19:11 ` [PATCH bpf-next v1 05/18] bpf: Track scratched stack slots with a bitmap Kumar Kartikeya Dwivedi
` (13 subsequent siblings)
17 siblings, 0 replies; 34+ messages in thread
From: Kumar Kartikeya Dwivedi @ 2026-09-23 19:11 UTC (permalink / raw)
To: bpf
Cc: Alexei Starovoitov, Andrii Nakryiko, Daniel Borkmann,
Eduard Zingerman, Emil Tsalapatis, Tejun Heo, kkd, kernel-team
Precision backtracking keeps the stack slots that still need a precise
mark in one u64 per frame, which ties it to frames of at most 64 slots.
Turn the per-frame masks into bitmaps sized by MAX_BPF_STACK_SLOTS and
use the bitmap helpers for setting, clearing, testing and iterating
them. The formatting helper takes a bitmap and the leftover-slot bug
reports print the formatted slot list instead of a hex mask.
mark_reg_stack_read() collected zero spills in a u64 of its own before
handing it to the backtracker; it now counts them and revisits the
range to mark each slot, which drops the only remaining mask-typed
entry point.
No functional change.
Signed-off-by: Kumar Kartikeya Dwivedi <memxor@gmail.com>
---
include/linux/bpf_verifier.h | 13 +++-----
kernel/bpf/backtrack.c | 65 ++++++++++++++++++++++--------------
kernel/bpf/verifier.c | 15 ++++++---
3 files changed, 54 insertions(+), 39 deletions(-)
diff --git a/include/linux/bpf_verifier.h b/include/linux/bpf_verifier.h
index 77cda3b3a3e9..905a081d4b62 100644
--- a/include/linux/bpf_verifier.h
+++ b/include/linux/bpf_verifier.h
@@ -879,7 +879,7 @@ struct backtrack_state {
struct bpf_verifier_env *env;
u32 frame;
u32 reg_masks[MAX_CALL_FRAMES];
- u64 stack_masks[MAX_CALL_FRAMES];
+ unsigned long stack_masks[MAX_CALL_FRAMES][BITS_TO_LONGS(MAX_BPF_STACK_SLOTS)];
u8 stack_arg_masks[MAX_CALL_FRAMES];
};
@@ -1317,12 +1317,7 @@ static inline void bpf_bt_set_frame_reg(struct backtrack_state *bt, u32 frame, u
static inline void bpf_bt_set_frame_slot(struct backtrack_state *bt, u32 frame, u32 slot)
{
- bt->stack_masks[frame] |= 1ull << slot;
-}
-
-static inline void bpf_bt_set_frame_slot_mask(struct backtrack_state *bt, u32 frame, u64 mask)
-{
- bt->stack_masks[frame] |= mask;
+ __set_bit(slot, bt->stack_masks[frame]);
}
static inline void bt_set_frame_stack_arg_slot(struct backtrack_state *bt, u32 frame, u32 slot)
@@ -1337,7 +1332,7 @@ static inline bool bt_is_frame_reg_set(struct backtrack_state *bt, u32 frame, u3
static inline bool bt_is_frame_slot_set(struct backtrack_state *bt, u32 frame, u32 slot)
{
- return bt->stack_masks[frame] & (1ull << slot);
+ return test_bit(slot, bt->stack_masks[frame]);
}
bool bpf_map_is_rdonly(const struct bpf_map *map);
@@ -1531,7 +1526,7 @@ struct bpf_subprog_info *bpf_find_containing_subprog(struct bpf_verifier_env *en
const char *bpf_subprog_name(const struct bpf_verifier_env *env, int subprog);
int bpf_jmp_offset(struct bpf_insn *insn);
struct bpf_iarray *bpf_insn_successors(struct bpf_verifier_env *env, u32 idx);
-void bpf_fmt_stack_mask(char *buf, ssize_t buf_sz, u64 stack_mask);
+void bpf_fmt_stack_mask(char *buf, ssize_t buf_sz, const unsigned long *stack_mask);
bool bpf_subprog_is_global(const struct bpf_verifier_env *env, int subprog);
/* Kinds of member a by-value struct or union may be composed of. */
diff --git a/kernel/bpf/backtrack.c b/kernel/bpf/backtrack.c
index 6733d4078930..db1be14d0a68 100644
--- a/kernel/bpf/backtrack.c
+++ b/kernel/bpf/backtrack.c
@@ -129,13 +129,26 @@ static inline void bt_reset(struct backtrack_state *bt)
bt->env = env;
}
-static inline u32 bt_empty(struct backtrack_state *bt)
+static inline bool bt_frame_stack_empty(struct backtrack_state *bt, u32 frame)
{
- u64 mask = 0;
+ return bitmap_empty(bt->stack_masks[frame], MAX_BPF_STACK_SLOTS);
+}
+
+static inline bool bt_stack_empty(struct backtrack_state *bt)
+{
+ return bt_frame_stack_empty(bt, bt->frame);
+}
+
+static inline bool bt_empty(struct backtrack_state *bt)
+{
+ u32 mask = 0;
int i;
- for (i = 0; i <= bt->frame; i++)
- mask |= bt->reg_masks[i] | bt->stack_masks[i] | bt->stack_arg_masks[i];
+ for (i = 0; i <= bt->frame; i++) {
+ mask |= bt->reg_masks[i] | bt->stack_arg_masks[i];
+ if (!bt_frame_stack_empty(bt, i))
+ return false;
+ }
return mask == 0;
}
@@ -187,7 +200,7 @@ static inline void bt_clear_reg(struct backtrack_state *bt, u32 reg)
static inline void bt_clear_frame_slot(struct backtrack_state *bt, u32 frame, u32 slot)
{
- bt->stack_masks[frame] &= ~(1ull << slot);
+ __clear_bit(slot, bt->stack_masks[frame]);
}
static inline u32 bt_frame_reg_mask(struct backtrack_state *bt, u32 frame)
@@ -200,14 +213,14 @@ static inline u32 bt_reg_mask(struct backtrack_state *bt)
return bt->reg_masks[bt->frame];
}
-static inline u64 bt_frame_stack_mask(struct backtrack_state *bt, u32 frame)
+static inline unsigned long *bt_frame_stack_mask(struct backtrack_state *bt, u32 frame)
{
return bt->stack_masks[frame];
}
-static inline u64 bt_stack_mask(struct backtrack_state *bt)
+static inline unsigned long *bt_stack_mask(struct backtrack_state *bt)
{
- return bt->stack_masks[bt->frame];
+ return bt_frame_stack_mask(bt, bt->frame);
}
static inline u8 bt_stack_arg_mask(struct backtrack_state *bt)
@@ -239,17 +252,16 @@ static void fmt_reg_mask(char *buf, ssize_t buf_sz, u32 reg_mask)
break;
}
}
-/* format stack slots bitmask, e.g., "-8,-24,-40" for 0x15 mask */
-void bpf_fmt_stack_mask(char *buf, ssize_t buf_sz, u64 stack_mask)
+
+/* format stack slots bitmask, e.g., "-8,-24,-40" for slots 0, 2 and 4 */
+void bpf_fmt_stack_mask(char *buf, ssize_t buf_sz, const unsigned long *stack_mask)
{
- DECLARE_BITMAP(mask, 64);
bool first = true;
int i, n;
buf[0] = '\0';
- bitmap_from_u64(mask, stack_mask);
- for_each_set_bit(i, mask, 64) {
+ for_each_set_bit(i, stack_mask, MAX_BPF_STACK_SLOTS) {
n = snprintf(buf, buf_sz, "%s%d", first ? "" : ",", -(i + 1) * 8);
first = false;
buf += n;
@@ -458,10 +470,11 @@ static int backtrack_insn(struct bpf_verifier_env *env, int idx, int subseq_idx,
/* we are now tracking register spills correctly,
* so any instance of leftover slots is a bug
*/
- if (bt_stack_mask(bt) != 0) {
- verifier_bug(env,
- "static subprog leftover stack slots %llx",
- bt_stack_mask(bt));
+ if (!bt_stack_empty(bt)) {
+ bpf_fmt_stack_mask(env->tmp_str_buf, TMP_STR_BUF_LEN,
+ bt_stack_mask(bt));
+ verifier_bug(env, "static subprog leftover stack slots %s",
+ env->tmp_str_buf);
return -EFAULT;
}
/* propagate r1-r5 to the caller */
@@ -494,9 +507,11 @@ static int backtrack_insn(struct bpf_verifier_env *env, int idx, int subseq_idx,
bt_reg_mask(bt));
return -EFAULT;
}
- if (bt_stack_mask(bt) != 0) {
- verifier_bug(env, "callback leftover stack slots %llx",
- bt_stack_mask(bt));
+ if (!bt_stack_empty(bt)) {
+ bpf_fmt_stack_mask(env->tmp_str_buf, TMP_STR_BUF_LEN,
+ bt_stack_mask(bt));
+ verifier_bug(env, "callback leftover stack slots %s",
+ env->tmp_str_buf);
return -EFAULT;
}
/* clear r1-r5 in callback subprog's mask */
@@ -874,7 +889,7 @@ int bpf_mark_chain_precision(struct bpf_verifier_env *env,
if (st->curframe == 0 &&
st->frame[0]->subprogno > 0 &&
st->frame[0]->callsite == BPF_MAIN_FUNC &&
- bt_stack_mask(bt) == 0 &&
+ bt_stack_empty(bt) &&
(bt_reg_mask(bt) & ~BPF_REGMASK_ARGS) == 0) {
bitmap_from_u64(mask, bt_reg_mask(bt));
for_each_set_bit(i, mask, 32) {
@@ -888,8 +903,9 @@ int bpf_mark_chain_precision(struct bpf_verifier_env *env,
return 0;
}
- verifier_bug(env, "backtracking func entry subprog %d reg_mask %x stack_mask %llx",
- st->frame[0]->subprogno, bt_reg_mask(bt), bt_stack_mask(bt));
+ bpf_fmt_stack_mask(env->tmp_str_buf, TMP_STR_BUF_LEN, bt_stack_mask(bt));
+ verifier_bug(env, "backtracking func entry subprog %d reg_mask %x stack_mask %s",
+ st->frame[0]->subprogno, bt_reg_mask(bt), env->tmp_str_buf);
return -EFAULT;
}
@@ -950,8 +966,7 @@ int bpf_mark_chain_precision(struct bpf_verifier_env *env,
}
}
- bitmap_from_u64(mask, bt_frame_stack_mask(bt, fr));
- for_each_set_bit(i, mask, 64) {
+ for_each_set_bit(i, bt_frame_stack_mask(bt, fr), MAX_BPF_STACK_SLOTS) {
if (verifier_bug_if(i >= bpf_stack_nr_slots(func),
env, "stack slot %d, total slots %d",
i, bpf_stack_nr_slots(func)))
diff --git a/kernel/bpf/verifier.c b/kernel/bpf/verifier.c
index fa80f95105fb..a66f1688407c 100644
--- a/kernel/bpf/verifier.c
+++ b/kernel/bpf/verifier.c
@@ -3864,10 +3864,9 @@ static int mark_reg_stack_read(struct bpf_verifier_env *env,
{
struct bpf_verifier_state *vstate = env->cur_state;
struct bpf_func_state *state = vstate->frame[vstate->curframe];
- u64 zero_spill_mask = 0;
int i, slot, spi;
u8 *stype;
- int zeros = 0;
+ int zeros = 0, zero_spills = 0;
for (i = min_off; i < max_off; i++) {
slot = -i - 1;
@@ -3880,7 +3879,7 @@ static int mark_reg_stack_read(struct bpf_verifier_env *env,
}
if (stype[slot % BPF_REG_SIZE] == STACK_SPILL &&
bpf_register_is_null(&bpf_stack_slot(ptr_state, spi)->spilled_ptr)) {
- zero_spill_mask |= 1ull << spi;
+ zero_spills++;
zeros++;
continue;
}
@@ -3891,8 +3890,14 @@ static int mark_reg_stack_read(struct bpf_verifier_env *env,
* so the whole register == const_zero.
*/
__mark_reg_const_zero(env, &state->regs[dst_regno]);
- if (zero_spill_mask) {
- bpf_bt_set_frame_slot_mask(&env->bt, ptr_state->frameno, zero_spill_mask);
+ if (zero_spills) {
+ for (i = min_off; i < max_off; i++) {
+ slot = -i - 1;
+ spi = slot / BPF_REG_SIZE;
+ stype = bpf_stack_slot(ptr_state, spi)->slot_type;
+ if (stype[slot % BPF_REG_SIZE] == STACK_SPILL)
+ bpf_bt_set_frame_slot(&env->bt, ptr_state->frameno, spi);
+ }
return mark_chain_precision_batch(env, env->cur_state);
}
} else {
--
2.53.0
^ permalink raw reply related [flat|nested] 34+ messages in thread* [PATCH bpf-next v1 05/18] bpf: Track scratched stack slots with a bitmap
2026-09-23 19:11 [PATCH bpf-next v1 00/18] Raise BPF program stack size to 2KiB Kumar Kartikeya Dwivedi
` (3 preceding siblings ...)
2026-09-23 19:11 ` [PATCH bpf-next v1 04/18] bpf: Track backtracking stack slots with bitmaps Kumar Kartikeya Dwivedi
@ 2026-09-23 19:11 ` Kumar Kartikeya Dwivedi
2026-09-23 19:11 ` [PATCH bpf-next v1 06/18] bpf: Treat unknown-size stack reads as reaching the frame top Kumar Kartikeya Dwivedi
` (12 subsequent siblings)
17 siblings, 0 replies; 34+ messages in thread
From: Kumar Kartikeya Dwivedi @ 2026-09-23 19:11 UTC (permalink / raw)
To: bpf
Cc: Alexei Starovoitov, Andrii Nakryiko, Daniel Borkmann,
Eduard Zingerman, Emil Tsalapatis, Tejun Heo, kkd, kernel-team
The set of stack slots touched since the verifier state was last
printed lives in a u64, so the log could only ever mark the first 64
slots of a frame as scratched. Make it a bitmap sized by
MAX_BPF_STACK_SLOTS and go through the bitmap helpers for setting,
testing, clearing and filling it.
No functional change.
Signed-off-by: Kumar Kartikeya Dwivedi <memxor@gmail.com>
---
include/linux/bpf_verifier.h | 15 ++++++++-------
1 file changed, 8 insertions(+), 7 deletions(-)
diff --git a/include/linux/bpf_verifier.h b/include/linux/bpf_verifier.h
index 905a081d4b62..354463a74292 100644
--- a/include/linux/bpf_verifier.h
+++ b/include/linux/bpf_verifier.h
@@ -1057,7 +1057,7 @@ struct bpf_verifier_env {
*/
u32 scratched_regs;
/* Same as scratched_regs but for stack slots */
- u64 scratched_stack_slots;
+ DECLARE_BITMAP(scratched_stack_slots, MAX_BPF_STACK_SLOTS);
u64 prev_log_pos, prev_insn_print_pos;
/* buffer used to temporary hold constants as scalar registers */
struct bpf_reg_state fake_reg[1];
@@ -1463,7 +1463,7 @@ static inline void mark_reg_scratched(struct bpf_verifier_env *env, u32 regno)
static inline void mark_stack_slot_scratched(struct bpf_verifier_env *env, u32 spi)
{
- env->scratched_stack_slots |= 1ULL << spi;
+ __set_bit(spi, env->scratched_stack_slots);
}
static inline bool reg_scratched(const struct bpf_verifier_env *env, u32 regno)
@@ -1471,27 +1471,28 @@ static inline bool reg_scratched(const struct bpf_verifier_env *env, u32 regno)
return (env->scratched_regs >> regno) & 1;
}
-static inline bool stack_slot_scratched(const struct bpf_verifier_env *env, u64 regno)
+static inline bool stack_slot_scratched(const struct bpf_verifier_env *env, u32 spi)
{
- return (env->scratched_stack_slots >> regno) & 1;
+ return test_bit(spi, env->scratched_stack_slots);
}
static inline bool verifier_state_scratched(const struct bpf_verifier_env *env)
{
- return env->scratched_regs || env->scratched_stack_slots;
+ return env->scratched_regs ||
+ !bitmap_empty(env->scratched_stack_slots, MAX_BPF_STACK_SLOTS);
}
static inline void mark_verifier_state_clean(struct bpf_verifier_env *env)
{
env->scratched_regs = 0U;
- env->scratched_stack_slots = 0ULL;
+ bitmap_zero(env->scratched_stack_slots, MAX_BPF_STACK_SLOTS);
}
/* Used for printing the entire verifier state. */
static inline void mark_verifier_state_scratched(struct bpf_verifier_env *env)
{
env->scratched_regs = ~0U;
- env->scratched_stack_slots = ~0ULL;
+ bitmap_fill(env->scratched_stack_slots, MAX_BPF_STACK_SLOTS);
}
static inline bool bpf_stack_narrow_access_ok(int off, int fill_size, int spill_size)
--
2.53.0
^ permalink raw reply related [flat|nested] 34+ messages in thread* [PATCH bpf-next v1 06/18] bpf: Treat unknown-size stack reads as reaching the frame top
2026-09-23 19:11 [PATCH bpf-next v1 00/18] Raise BPF program stack size to 2KiB Kumar Kartikeya Dwivedi
` (4 preceding siblings ...)
2026-09-23 19:11 ` [PATCH bpf-next v1 05/18] bpf: Track scratched stack slots with a bitmap Kumar Kartikeya Dwivedi
@ 2026-09-23 19:11 ` Kumar Kartikeya Dwivedi
2026-09-23 19:11 ` [PATCH bpf-next v1 07/18] bpf: Size liveness stack masks by the stack each frame uses Kumar Kartikeya Dwivedi
` (11 subsequent siblings)
17 siblings, 0 replies; 34+ messages in thread
From: Kumar Kartikeya Dwivedi @ 2026-09-23 19:11 UTC (permalink / raw)
To: bpf
Cc: Alexei Starovoitov, Andrii Nakryiko, Daniel Borkmann,
Eduard Zingerman, Emil Tsalapatis, Tejun Heo, kkd, kernel-team
When the size of a helper or kfunc memory argument is not a constant on
the path, the stack liveness analysis is told the call reads
MAX_BPF_STACK bytes starting at the pointer's offset. Clipped at the
top of the frame this covers everything from the offset upwards, which
is the intent, but only as long as no frame is deeper than
MAX_BPF_STACK bytes.
Return the "unknown" marker instead, which the liveness analysis already
turns into a read of every slot between the offset and the frame top,
independent of how deep the frame is.
No functional change.
Signed-off-by: Kumar Kartikeya Dwivedi <memxor@gmail.com>
---
kernel/bpf/verifier.c | 7 ++++---
1 file changed, 4 insertions(+), 3 deletions(-)
diff --git a/kernel/bpf/verifier.c b/kernel/bpf/verifier.c
index a66f1688407c..d8f43f3a8991 100644
--- a/kernel/bpf/verifier.c
+++ b/kernel/bpf/verifier.c
@@ -13894,11 +13894,11 @@ s64 bpf_helper_stack_access_bytes(struct bpf_verifier_env *env, struct bpf_insn
}
/*
* Size arg is const on each path but differs across merged
- * paths. MAX_BPF_STACK is a safe upper bound for reads.
+ * paths. Reads may extend anywhere up to the frame top.
*/
if (full_write)
return 0;
- return MAX_BPF_STACK;
+ return S64_MIN;
}
return S64_MIN;
case ARG_PTR_TO_DYNPTR:
@@ -13984,7 +13984,8 @@ s64 bpf_kfunc_stack_access_bytes(struct bpf_verifier_env *env, struct bpf_insn *
size = (s64)aux->const_reg_vals[size_reg];
goto out;
}
- return MAX_BPF_STACK;
+ /* Unknown size: the read may extend anywhere up to the frame top. */
+ return S64_MIN;
}
/* fixed-size pointed-to type: resolve via BTF */
--
2.53.0
^ permalink raw reply related [flat|nested] 34+ messages in thread* [PATCH bpf-next v1 07/18] bpf: Size liveness stack masks by the stack each frame uses
2026-09-23 19:11 [PATCH bpf-next v1 00/18] Raise BPF program stack size to 2KiB Kumar Kartikeya Dwivedi
` (5 preceding siblings ...)
2026-09-23 19:11 ` [PATCH bpf-next v1 06/18] bpf: Treat unknown-size stack reads as reaching the frame top Kumar Kartikeya Dwivedi
@ 2026-09-23 19:11 ` Kumar Kartikeya Dwivedi
2026-09-23 19:11 ` [PATCH bpf-next v1 08/18] bpf: Grow the verifier id scratch on demand Kumar Kartikeya Dwivedi
` (10 subsequent siblings)
17 siblings, 0 replies; 34+ messages in thread
From: Kumar Kartikeya Dwivedi @ 2026-09-23 19:11 UTC (permalink / raw)
To: bpf
Cc: Alexei Starovoitov, Andrii Nakryiko, Daniel Borkmann,
Eduard Zingerman, Emil Tsalapatis, Tejun Heo, kkd, kernel-team
Stack liveness tracks three masks of 4-byte stack slots - may_read,
must_write and live_before - for every instruction of every frame of
every function instance. Each mask used to be a fixed 128-bit spis_t,
wide enough for the deepest frame MAX_BPF_STACK allows, so an
instruction paid 48 bytes per frame no matter how little stack the frame
actually touched. These per instruction arrays are the largest liveness
allocation: a 40k instruction program spends ~2 MiB on each frame array.
Sizing them for a larger stack budget would multiply that cost for every
program, while most frames use a fraction of the budget.
Replace the fixed masks with a variable width bitmap array per
(instance, frame). struct frame_masks holds the width shared by all of
its masks in @words plus a flat unsigned long array, where mask @kind of
the instruction at relative index @i lives at
&bits[(i * FM_MASK_CNT + kind) * words]. An array starts at the width
the first recorded half-slot needs (minimum one word, that is 256 bytes
of stack with 64-bit words) and is reallocated into a wider stride once
a deeper half-slot shows up. Only the marking functions widen an array,
and they all run during the static analysis before do_check(), so no
verification time path can move one.
The marking API now takes an inclusive half-slot range, which is what
record_stack_access_off() computed anyway, and clamps the upper end to
the deepest half-slot a frame can have. An access below the stack bound
is rejected by the main verifier pass later; liveness only has to stay
inside the masks. A range that covers no whole half-slot, as a byte
store at fp-1 produces, marks nothing.
A "read everything" mark, used for calls that are neither helpers nor
kfuncs and for pointers whose offset or frame identity was lost, widens
the array to the maximum width and sets every bit: recorded at a
narrower width it would lose the half-slots a later widening adds, and
no write can cancel it as no write covers the whole frame. A half-slot
past the width of the masks was never read by the frame and is never
live.
merge_instances() may see a different width on each side, so it widens
dst to cover src and counts a word only dst has as zero on the src
side, which unions into may_read as a no-op and intersects must_write
to empty, never claiming a write that did not happen.
The rest is mechanical: update_insn() propagates live_before word by
word over the array's width (all instructions of one array share it,
hence so do an instruction's successors), is_live_before() answers
false past the width, and the use:/def: log printing iterates up to the
width.
Liveness results are unchanged. A half-slot only ever gets a bit from a
recorded access and recording one widens the array to cover it, so every
bit the fixed masks could hold still fits. Frames whose deepest recorded
half-slot fits in a single word now cost 24 bytes per instruction per
frame instead of 48.
Signed-off-by: Kumar Kartikeya Dwivedi <memxor@gmail.com>
---
include/linux/bpf_verifier.h | 46 -----
kernel/bpf/liveness.c | 345 ++++++++++++++++++++++++-----------
2 files changed, 238 insertions(+), 153 deletions(-)
diff --git a/include/linux/bpf_verifier.h b/include/linux/bpf_verifier.h
index 354463a74292..0f348ce81fe2 100644
--- a/include/linux/bpf_verifier.h
+++ b/include/linux/bpf_verifier.h
@@ -253,52 +253,6 @@ enum bpf_stack_slot_type {
/* 4-byte stack slot granularity for liveness analysis */
#define BPF_HALF_REG_SIZE 4
#define STACK_SLOT_SZ 4
-#define STACK_SLOTS (MAX_BPF_STACK / BPF_HALF_REG_SIZE) /* 128 */
-
-typedef struct {
- u64 v[2];
-} spis_t;
-
-#define SPIS_ZERO ((spis_t){})
-#define SPIS_ALL ((spis_t){{ U64_MAX, U64_MAX }})
-
-static inline bool spis_is_zero(spis_t s)
-{
- return s.v[0] == 0 && s.v[1] == 0;
-}
-
-static inline bool spis_equal(spis_t a, spis_t b)
-{
- return a.v[0] == b.v[0] && a.v[1] == b.v[1];
-}
-
-static inline spis_t spis_or(spis_t a, spis_t b)
-{
- return (spis_t){{ a.v[0] | b.v[0], a.v[1] | b.v[1] }};
-}
-
-static inline spis_t spis_and(spis_t a, spis_t b)
-{
- return (spis_t){{ a.v[0] & b.v[0], a.v[1] & b.v[1] }};
-}
-
-static inline spis_t spis_not(spis_t s)
-{
- return (spis_t){{ ~s.v[0], ~s.v[1] }};
-}
-
-static inline bool spis_test_bit(spis_t s, u32 slot)
-{
- return s.v[slot / 64] & BIT_ULL(slot % 64);
-}
-
-static inline void spis_or_range(spis_t *mask, u32 lo, u32 hi)
-{
- u32 w;
-
- for (w = lo; w <= hi && w < STACK_SLOTS; w++)
- mask->v[w / 64] |= BIT_ULL(w % 64);
-}
#define BPF_REGMASK_ARGS ((1 << BPF_REG_1) | (1 << BPF_REG_2) | \
(1 << BPF_REG_3) | (1 << BPF_REG_4) | \
diff --git a/kernel/bpf/liveness.c b/kernel/bpf/liveness.c
index 44ecdc5b4ec2..07eb84ae0b7f 100644
--- a/kernel/bpf/liveness.c
+++ b/kernel/bpf/liveness.c
@@ -10,10 +10,39 @@
#define verbose(env, fmt, args...) bpf_verifier_log_write(env, fmt, ##args)
-struct per_frame_masks {
- spis_t may_read; /* stack slots that may be read by this instruction */
- spis_t must_write; /* stack slots written by this instruction */
- spis_t live_before; /* stack slots that may be read by this insn and its successors */
+/*
+ * Stack liveness is tracked with a 4-byte (half register) granularity.
+ * Half-slot 0 covers [fp-4, fp), half-slot 1 covers [fp-8, fp-4), and so on,
+ * hence FRAME_HALF_SPIS - 1 is the deepest half-slot a frame can have.
+ */
+#define FRAME_HALF_SPIS (MAX_BPF_STACK / BPF_HALF_REG_SIZE)
+#define FRAME_MAX_WORDS BITS_TO_LONGS(FRAME_HALF_SPIS)
+
+/* Masks tracked for each instruction of a frame */
+enum {
+ FM_MAY_READ, /* stack slots that may be read by this instruction */
+ FM_MUST_WRITE, /* stack slots written by this instruction */
+ FM_LIVE_BEFORE, /* stack slots that may be read by this insn and its successors */
+ FM_MASK_CNT,
+};
+
+/*
+ * Per instruction stack masks for one frame of a function instance.
+ *
+ * Most frames use only a fraction of the stack budget, so instead of masks
+ * wide enough for every half-slot of the largest possible frame, all masks
+ * of one array share the width @words, and the marking functions widen the
+ * array as deeper half-slots are recorded. Mask @kind of the instruction at
+ * relative index @i is at &bits[(i * FM_MASK_CNT + kind) * words]. A
+ * half-slot at or past @words * BITS_PER_LONG is never read by this frame,
+ * hence never live. An instruction that may read the whole frame, such as a
+ * call passing a frame pointer to another subprog, widens the array to
+ * FRAME_MAX_WORDS, so that the read cannot lose half-slots to a later
+ * widening.
+ */
+struct frame_masks {
+ u32 words;
+ unsigned long bits[];
};
/*
@@ -29,7 +58,7 @@ struct func_instance {
u32 subprog_start; /* cached env->subprog_info[subprog].start */
u32 insn_cnt; /* cached number of insns in the function */
/* Per frame, per instruction masks, frames allocated lazily. */
- struct per_frame_masks *frames[MAX_CALL_FRAMES];
+ struct frame_masks *frames[MAX_CALL_FRAMES];
bool must_write_initialized;
};
@@ -151,50 +180,118 @@ static int relative_idx(struct func_instance *instance, u32 insn_idx)
return insn_idx - instance->subprog_start;
}
-static struct per_frame_masks *get_frame_masks(struct func_instance *instance,
- u32 frame, u32 insn_idx)
+static u32 frame_mask_bits(struct frame_masks *fm)
{
- if (!instance->frames[frame])
- return NULL;
+ return fm->words * BITS_PER_LONG;
+}
- return &instance->frames[frame][relative_idx(instance, insn_idx)];
+static size_t frame_mask_words(struct func_instance *instance, u32 words)
+{
+ return (size_t)instance->insn_cnt * FM_MASK_CNT * words;
}
-static struct per_frame_masks *alloc_frame_masks(struct func_instance *instance,
- u32 frame, u32 insn_idx)
+/* Mask @kind of the instruction at relative index @rel */
+static unsigned long *rel_mask(struct frame_masks *fm, u32 rel, u32 kind)
{
- struct per_frame_masks *arr;
+ return fm->bits + ((size_t)rel * FM_MASK_CNT + kind) * fm->words;
+}
- if (!instance->frames[frame]) {
- arr = kvzalloc_objs(*arr, instance->insn_cnt,
- GFP_KERNEL_ACCOUNT);
- instance->frames[frame] = arr;
- if (!arr)
- return ERR_PTR(-ENOMEM);
+/*
+ * Make sure @frame has a mask array at least @words wide, allocating it or
+ * copying the existing masks into the wider stride as needed.
+ * @words must be in range [1, FRAME_MAX_WORDS].
+ */
+static struct frame_masks *widen_frame_masks(struct func_instance *instance,
+ u32 frame, u32 words)
+{
+ struct frame_masks *old = instance->frames[frame], *new;
+ u32 i, kind;
+
+ if (old && old->words >= words)
+ return old;
+ new = kvzalloc_flex(*new, bits, frame_mask_words(instance, words), GFP_KERNEL_ACCOUNT);
+ if (!new)
+ return NULL;
+ new->words = words;
+ if (old) {
+ for (i = 0; i < instance->insn_cnt; i++)
+ for (kind = 0; kind < FM_MASK_CNT; kind++)
+ memcpy(rel_mask(new, i, kind), rel_mask(old, i, kind),
+ old->words * sizeof(*old->bits));
+ kvfree(old);
}
- return get_frame_masks(instance, frame, insn_idx);
+ instance->frames[frame] = new;
+ return new;
}
-/* Accumulate may_read masks for @frame at @insn_idx */
-static int mark_stack_read(struct func_instance *instance, u32 frame, u32 insn_idx, spis_t mask)
+/*
+ * Set the inclusive half-slot range [lo, hi] in mask @kind of @frame at @insn_idx.
+ * An empty range, including one with a negative @hi as computed for a write
+ * that does not fully cover any half-slot, marks nothing.
+ */
+static int mark_stack_range(struct func_instance *instance, u32 frame, u32 insn_idx,
+ u32 kind, s32 lo, s32 hi)
{
- struct per_frame_masks *masks;
+ struct frame_masks *fm;
- masks = alloc_frame_masks(instance, frame, insn_idx);
- if (IS_ERR(masks))
- return PTR_ERR(masks);
- masks->may_read = spis_or(masks->may_read, mask);
+ /*
+ * An access past the frame bottom is rejected by the main verifier
+ * pass later, liveness only has to avoid running off the masks.
+ */
+ hi = min_t(s32, hi, FRAME_HALF_SPIS - 1);
+ if (lo > hi)
+ return 0;
+ fm = widen_frame_masks(instance, frame, BITS_TO_LONGS(hi + 1));
+ if (!fm)
+ return -ENOMEM;
+ bitmap_set(rel_mask(fm, relative_idx(instance, insn_idx), kind), lo, hi - lo + 1);
return 0;
}
-static int mark_stack_write(struct func_instance *instance, u32 frame, u32 insn_idx, spis_t mask)
+/* Accumulate may_read for half-slots [lo, hi] of @frame at @insn_idx */
+static int mark_stack_read(struct func_instance *instance, u32 frame, u32 insn_idx,
+ s32 lo, s32 hi)
+{
+ return mark_stack_range(instance, frame, insn_idx, FM_MAY_READ, lo, hi);
+}
+
+/* Accumulate must_write for half-slots [lo, hi] of @frame at @insn_idx */
+static int mark_stack_write(struct func_instance *instance, u32 frame, u32 insn_idx,
+ s32 lo, s32 hi)
{
- struct per_frame_masks *masks;
+ return mark_stack_range(instance, frame, insn_idx, FM_MUST_WRITE, lo, hi);
+}
- masks = alloc_frame_masks(instance, frame, insn_idx);
- if (IS_ERR(masks))
- return PTR_ERR(masks);
- masks->must_write = spis_or(masks->must_write, mask);
+/*
+ * Mark every half-slot of @frame as possibly read by @insn_idx. This widens
+ * the masks to the maximum width: a full read recorded at a narrower width
+ * would leave the bits added by a later widening clear and lose part of it.
+ */
+static int mark_stack_read_all(struct func_instance *instance, u32 frame, u32 insn_idx)
+{
+ return mark_stack_read(instance, frame, insn_idx, 0, FRAME_HALF_SPIS - 1);
+}
+
+/* Accumulate @src, a mask @src_words wide, into may_read of @frame at @insn_idx */
+static int mark_stack_read_mask(struct func_instance *instance, u32 frame, u32 insn_idx,
+ const unsigned long *src, u32 src_words)
+{
+ u32 nbits = src_words * BITS_PER_LONG;
+ struct frame_masks *fm;
+ unsigned long *dst;
+ u32 last, w;
+
+ last = find_last_bit(src, nbits);
+ if (last == nbits)
+ return 0;
+ fm = widen_frame_masks(instance, frame, BITS_TO_LONGS(last + 1));
+ if (!fm)
+ return -ENOMEM;
+ dst = rel_mask(fm, relative_idx(instance, insn_idx), FM_MAY_READ);
+ /* @src has no bits set past @last, hence none past @fm->words either */
+ src_words = min(src_words, fm->words);
+ for (w = 0; w < src_words; w++)
+ dst[w] |= src[w];
return 0;
}
@@ -272,33 +369,42 @@ __diag_pop();
static inline bool update_insn(struct bpf_verifier_env *env,
struct func_instance *instance, u32 frame, u32 insn_idx)
{
- spis_t new_before, new_after;
- struct per_frame_masks *insn, *succ_insn;
+ unsigned long new_after[FRAME_MAX_WORDS] = {};
+ unsigned long *may_read, *must_write, *live_before;
+ struct frame_masks *fm = instance->frames[frame];
+ u32 rel = relative_idx(instance, insn_idx);
struct bpf_iarray *succ;
- u32 s;
- bool changed;
+ bool changed = false;
+ u32 s, w;
succ = bpf_insn_successors(env, insn_idx);
if (succ->cnt == 0)
return false;
- changed = false;
- insn = get_frame_masks(instance, frame, insn_idx);
- new_before = SPIS_ZERO;
- new_after = SPIS_ZERO;
+ /* All instructions of one frame array share the same mask width */
for (s = 0; s < succ->cnt; ++s) {
- succ_insn = get_frame_masks(instance, frame, succ->items[s]);
- new_after = spis_or(new_after, succ_insn->live_before);
+ unsigned long *succ_live;
+
+ succ_live = rel_mask(fm, relative_idx(instance, succ->items[s]), FM_LIVE_BEFORE);
+ for (w = 0; w < fm->words; w++)
+ new_after[w] |= succ_live[w];
}
+ may_read = rel_mask(fm, rel, FM_MAY_READ);
+ must_write = rel_mask(fm, rel, FM_MUST_WRITE);
+ live_before = rel_mask(fm, rel, FM_LIVE_BEFORE);
/*
* New "live_before" is a union of all "live_before" of successors
* minus slots written by instruction plus slots read by instruction.
* new_before = (new_after & ~insn->must_write) | insn->may_read
*/
- new_before = spis_or(spis_and(new_after, spis_not(insn->must_write)),
- insn->may_read);
- changed |= !spis_equal(new_before, insn->live_before);
- insn->live_before = new_before;
+ for (w = 0; w < fm->words; w++) {
+ unsigned long new_before = (new_after[w] & ~must_write[w]) | may_read[w];
+
+ if (new_before != live_before[w]) {
+ live_before[w] = new_before;
+ changed = true;
+ }
+ }
return changed;
}
@@ -329,10 +435,12 @@ static void update_instance(struct bpf_verifier_env *env, struct func_instance *
static bool is_live_before(struct func_instance *instance, u32 insn_idx, u32 frameno, u32 half_spi)
{
- struct per_frame_masks *masks;
+ struct frame_masks *fm = instance->frames[frameno];
- masks = get_frame_masks(instance, frameno, insn_idx);
- return masks && spis_test_bit(masks->live_before, half_spi);
+ /* No recorded access reaches past the masks, so nothing there is live */
+ if (!fm || half_spi >= frame_mask_bits(fm))
+ return false;
+ return test_bit(half_spi, rel_mask(fm, relative_idx(instance, insn_idx), FM_LIVE_BEFORE));
}
int bpf_live_stack_query_init(struct bpf_verifier_env *env, struct bpf_verifier_state *st)
@@ -430,17 +538,19 @@ static int spi_off(int spi)
* When only one half is set, print as "-4h","-8h",...
* Runs of 3+ consecutive fully-set SPIs are collapsed: "fp0-8..-24"
*/
-static char *fmt_spis_mask(struct bpf_verifier_env *env, int frame, bool first, spis_t spis)
+static char *fmt_spis_mask(struct bpf_verifier_env *env, int frame, bool first,
+ const unsigned long *spis, u32 words)
{
int buf_sz = sizeof(env->tmp_str_buf);
+ int spi_cnt = words * BITS_PER_LONG / 2;
char *buf = env->tmp_str_buf;
int spi, n, run_start;
buf[0] = '\0';
- for (spi = 0; spi < STACK_SLOTS / 2 && buf_sz > 0; spi++) {
- bool lo = spis_test_bit(spis, spi * 2);
- bool hi = spis_test_bit(spis, spi * 2 + 1);
+ for (spi = 0; spi < spi_cnt && buf_sz > 0; spi++) {
+ bool lo = test_bit(spi * 2, spis);
+ bool hi = test_bit(spi * 2 + 1, spis);
const char *space = first ? "" : " ";
if (!lo && !hi)
@@ -450,16 +560,16 @@ static char *fmt_spis_mask(struct bpf_verifier_env *env, int frame, bool first,
/* half-spi */
n = scnprintf(buf, buf_sz, "%sfp%d%d%s",
space, frame, spi_off(spi) + (lo ? STACK_SLOT_SZ : 0), "h");
- } else if (spi + 2 < STACK_SLOTS / 2 &&
- spis_test_bit(spis, spi * 2 + 2) &&
- spis_test_bit(spis, spi * 2 + 3) &&
- spis_test_bit(spis, spi * 2 + 4) &&
- spis_test_bit(spis, spi * 2 + 5)) {
+ } else if (spi + 2 < spi_cnt &&
+ test_bit(spi * 2 + 2, spis) &&
+ test_bit(spi * 2 + 3, spis) &&
+ test_bit(spi * 2 + 4, spis) &&
+ test_bit(spi * 2 + 5, spis)) {
/* 3+ consecutive full spis */
run_start = spi;
- while (spi + 1 < STACK_SLOTS / 2 &&
- spis_test_bit(spis, (spi + 1) * 2) &&
- spis_test_bit(spis, (spi + 1) * 2 + 1))
+ while (spi + 1 < spi_cnt &&
+ test_bit((spi + 1) * 2, spis) &&
+ test_bit((spi + 1) * 2 + 1, spis))
spi++;
n = scnprintf(buf, buf_sz, "%sfp%d%d..%d",
space, frame, spi_off(run_start), spi_off(spi));
@@ -478,7 +588,8 @@ static void print_instance(struct bpf_verifier_env *env, struct func_instance *i
{
int start = env->subprog_info[instance->subprog].start;
struct bpf_insn *insns = env->prog->insnsi;
- struct per_frame_masks *masks;
+ struct frame_masks *fm;
+ unsigned long *mask;
int len = instance->insn_cnt;
int insn_idx, frame, i;
bool has_use, has_def;
@@ -501,10 +612,13 @@ static void print_instance(struct bpf_verifier_env *env, struct func_instance *i
pos = env->log.end_pos;
verbose(env, " use: ");
for (frame = instance->depth; frame >= 0; --frame) {
- masks = get_frame_masks(instance, frame, insn_idx);
- if (!masks || spis_is_zero(masks->may_read))
+ fm = instance->frames[frame];
+ if (!fm)
+ continue;
+ mask = rel_mask(fm, i, FM_MAY_READ);
+ if (bitmap_empty(mask, frame_mask_bits(fm)))
continue;
- verbose(env, "%s", fmt_spis_mask(env, frame, !has_use, masks->may_read));
+ verbose(env, "%s", fmt_spis_mask(env, frame, !has_use, mask, fm->words));
has_use = true;
}
if (!has_use)
@@ -512,10 +626,13 @@ static void print_instance(struct bpf_verifier_env *env, struct func_instance *i
pos = env->log.end_pos;
verbose(env, " def: ");
for (frame = instance->depth; frame >= 0; --frame) {
- masks = get_frame_masks(instance, frame, insn_idx);
- if (!masks || spis_is_zero(masks->must_write))
+ fm = instance->frames[frame];
+ if (!fm)
+ continue;
+ mask = rel_mask(fm, i, FM_MUST_WRITE);
+ if (bitmap_empty(mask, frame_mask_bits(fm)))
continue;
- verbose(env, "%s", fmt_spis_mask(env, frame, !has_def, masks->must_write));
+ verbose(env, "%s", fmt_spis_mask(env, frame, !has_def, mask, fm->words));
has_def = true;
}
if (!has_def)
@@ -584,9 +701,9 @@ static int print_instances(struct bpf_verifier_env *env)
* - same frame + different offset -> offset-imprecise
* - different frames -> fully-imprecise (bitmask OR)
*
- * At memory access sites (LDX/STX/ST), offset-imprecise marks only
- * the known frame's access mask as SPIS_ALL, while fully-imprecise
- * iterates bits in the bitmask and routes each frame to its target.
+ * At memory access sites (LDX/STX/ST), offset-imprecise marks the known
+ * frame as fully read, while fully-imprecise iterates bits in the bitmask
+ * and routes each frame to its target.
*/
#define MAX_ARG_OFFSETS 4
@@ -1235,7 +1352,6 @@ static int record_stack_access_off(struct func_instance *instance, s64 fp_off,
s64 access_bytes, u32 frame, u32 insn_idx)
{
s32 slot_hi, slot_lo;
- spis_t mask;
if (fp_off >= 0)
/*
@@ -1247,27 +1363,19 @@ static int record_stack_access_off(struct func_instance *instance, s64 fp_off,
if (access_bytes == S64_MIN) {
/* helper/kfunc read unknown amount of bytes from fp_off until fp+0 */
slot_hi = (-fp_off - 1) / STACK_SLOT_SZ;
- mask = SPIS_ZERO;
- spis_or_range(&mask, 0, slot_hi);
- return mark_stack_read(instance, frame, insn_idx, mask);
+ return mark_stack_read(instance, frame, insn_idx, 0, slot_hi);
}
if (access_bytes > 0) {
/* Mark any touched slot as use */
slot_hi = (-fp_off - 1) / STACK_SLOT_SZ;
slot_lo = max_t(s32, (-fp_off - access_bytes) / STACK_SLOT_SZ, 0);
- mask = SPIS_ZERO;
- spis_or_range(&mask, slot_lo, slot_hi);
- return mark_stack_read(instance, frame, insn_idx, mask);
+ return mark_stack_read(instance, frame, insn_idx, slot_lo, slot_hi);
} else if (access_bytes < 0) {
/* Mark only fully covered slots as def */
access_bytes = -access_bytes;
slot_hi = (-fp_off) / STACK_SLOT_SZ - 1;
slot_lo = max_t(s32, (-fp_off - access_bytes + STACK_SLOT_SZ - 1) / STACK_SLOT_SZ, 0);
- if (slot_lo <= slot_hi) {
- mask = SPIS_ZERO;
- spis_or_range(&mask, slot_lo, slot_hi);
- return mark_stack_write(instance, frame, insn_idx, mask);
- }
+ return mark_stack_write(instance, frame, insn_idx, slot_lo, slot_hi);
}
return 0;
}
@@ -1286,7 +1394,7 @@ static int record_stack_access(struct func_instance *instance,
return 0;
if (arg->off_cnt == 0) {
if (access_bytes > 0 || access_bytes == S64_MIN)
- return mark_stack_read(instance, frame, insn_idx, SPIS_ALL);
+ return mark_stack_read_all(instance, frame, insn_idx);
return 0;
}
if (access_bytes != S64_MIN && access_bytes < 0 && arg->off_cnt != 1)
@@ -1314,7 +1422,7 @@ static int record_imprecise(struct func_instance *instance, u32 mask, u32 insn_i
if (!(mask & 1))
continue;
if (f <= depth) {
- err = mark_stack_read(instance, f, insn_idx, SPIS_ALL);
+ err = mark_stack_read_all(instance, f, insn_idx);
if (err)
return err;
}
@@ -1410,7 +1518,7 @@ static int record_arg_access(struct bpf_verifier_env *env,
bytes = bpf_kfunc_stack_access_bytes(env, insn, arg_idx, insn_idx);
} else {
for (int f = 0; f <= depth; f++) {
- err = mark_stack_read(instance, f, insn_idx, SPIS_ALL);
+ err = mark_stack_read_all(instance, f, insn_idx);
if (err)
return err;
}
@@ -1772,36 +1880,52 @@ static bool has_fp_args(struct arg_track *args)
* may_read: union (any pass might read the slot).
* must_write: intersection (only slots written on ALL passes are guaranteed).
* live_before is recomputed by a subsequent update_instance() on @dst.
+ *
+ * The two instances may have settled on different mask widths for the same
+ * frame, so @dst is widened to cover @src first. A word only @dst has counts
+ * as zero on the @src side: it unions into may_read as a no-op and intersects
+ * must_write to empty.
*/
-static void merge_instances(struct func_instance *dst, struct func_instance *src)
+static int merge_instances(struct func_instance *dst, struct func_instance *src)
{
- int f, i;
+ struct frame_masks *d, *s;
+ u32 f, i, w;
for (f = 0; f <= dst->depth; f++) {
- if (!src->frames[f]) {
+ s = src->frames[f];
+ d = dst->frames[f];
+ if (!s) {
/* This pass didn't touch frame f — must_write intersects with empty. */
- if (dst->frames[f])
+ if (d)
for (i = 0; i < dst->insn_cnt; i++)
- dst->frames[f][i].must_write = SPIS_ZERO;
+ bitmap_zero(rel_mask(d, i, FM_MUST_WRITE),
+ frame_mask_bits(d));
continue;
}
- if (!dst->frames[f]) {
+ if (!d) {
/* Previous pass didn't touch frame f — take src, zero must_write. */
- dst->frames[f] = src->frames[f];
+ dst->frames[f] = s;
src->frames[f] = NULL;
for (i = 0; i < dst->insn_cnt; i++)
- dst->frames[f][i].must_write = SPIS_ZERO;
+ bitmap_zero(rel_mask(s, i, FM_MUST_WRITE), frame_mask_bits(s));
continue;
}
+ d = widen_frame_masks(dst, f, s->words);
+ if (!d)
+ return -ENOMEM;
for (i = 0; i < dst->insn_cnt; i++) {
- dst->frames[f][i].may_read =
- spis_or(dst->frames[f][i].may_read,
- src->frames[f][i].may_read);
- dst->frames[f][i].must_write =
- spis_and(dst->frames[f][i].must_write,
- src->frames[f][i].must_write);
+ unsigned long *dst_read = rel_mask(d, i, FM_MAY_READ);
+ unsigned long *dst_write = rel_mask(d, i, FM_MUST_WRITE);
+ unsigned long *src_read = rel_mask(s, i, FM_MAY_READ);
+ unsigned long *src_write = rel_mask(s, i, FM_MUST_WRITE);
+
+ for (w = 0; w < d->words; w++) {
+ dst_read[w] |= w < s->words ? src_read[w] : 0;
+ dst_write[w] &= w < s->words ? src_write[w] : 0;
+ }
}
}
+ return 0;
}
static struct func_instance *fresh_instance(struct func_instance *src)
@@ -1916,7 +2040,7 @@ static int analyze_subprog(struct bpf_verifier_env *env,
if (info[subprog].at_in[j][caller_reg].frame == ARG_NONE)
continue;
for (int f = 0; f <= depth; f++) {
- err = mark_stack_read(instance, f, idx, SPIS_ALL);
+ err = mark_stack_read_all(instance, f, idx);
if (err)
goto out_free;
}
@@ -1955,13 +2079,18 @@ static int analyze_subprog(struct bpf_verifier_env *env,
/* Pull callee's entry liveness back to caller's callsite */
{
u32 callee_start = callee_instance->subprog_start;
- struct per_frame_masks *entry;
+ struct frame_masks *callee_fm;
for (int f = 0; f < callee_instance->depth; f++) {
- entry = get_frame_masks(callee_instance, f, callee_start);
- if (!entry)
+ callee_fm = callee_instance->frames[f];
+ if (!callee_fm)
continue;
- err = mark_stack_read(instance, f, idx, entry->live_before);
+ err = mark_stack_read_mask(instance, f, idx,
+ rel_mask(callee_fm,
+ relative_idx(callee_instance,
+ callee_start),
+ FM_LIVE_BEFORE),
+ callee_fm->words);
if (err)
goto out_free;
}
@@ -1969,9 +2098,11 @@ static int analyze_subprog(struct bpf_verifier_env *env,
}
if (prev_instance) {
- merge_instances(prev_instance, instance);
+ err = merge_instances(prev_instance, instance);
free_instance(instance);
instance = prev_instance;
+ if (err)
+ return err;
}
update_instance(env, instance);
return 0;
--
2.53.0
^ permalink raw reply related [flat|nested] 34+ messages in thread* [PATCH bpf-next v1 08/18] bpf: Grow the verifier id scratch on demand
2026-09-23 19:11 [PATCH bpf-next v1 00/18] Raise BPF program stack size to 2KiB Kumar Kartikeya Dwivedi
` (6 preceding siblings ...)
2026-09-23 19:11 ` [PATCH bpf-next v1 07/18] bpf: Size liveness stack masks by the stack each frame uses Kumar Kartikeya Dwivedi
@ 2026-09-23 19:11 ` Kumar Kartikeya Dwivedi
2026-09-23 19:24 ` sashiko-bot
2026-09-23 19:11 ` [PATCH bpf-next v1 09/18] selftests/bpf: Cover the tail call caller stack depth limit Kumar Kartikeya Dwivedi
` (9 subsequent siblings)
17 siblings, 1 reply; 34+ messages in thread
From: Kumar Kartikeya Dwivedi @ 2026-09-23 19:11 UTC (permalink / raw)
To: bpf
Cc: Alexei Starovoitov, Andrii Nakryiko, Daniel Borkmann,
Eduard Zingerman, Emil Tsalapatis, Tejun Heo, kkd, kernel-team
The id map used to compare the ids of two states, which also serves as
the id stack of release_reference() and as the id set of
bpf_clear_singular_ids(), is a fixed array embedded in struct
bpf_verifier_env and sized for the most registers and stack slots a
state can possibly hold: 1312 entries, 10 KiB, for 512-byte frames, and
four times that once frames may reach 2 KiB, which pushes the env
allocation from 64 KiB to 128 KiB for every program verified. States
compare a few dozen ids in practice.
Turn the map and the set into arrays grown on demand, starting at 64
entries and doubling, and free them with the env. A map that cannot
grow treats the states as different, an id set that cannot grow keeps
the id, and the id stack reports -ENOMEM, so an allocation failure is
never unsafe. This removes the last structure whose size scaled with
the stack bound and shrinks the env by 10 KiB.
Signed-off-by: Kumar Kartikeya Dwivedi <memxor@gmail.com>
---
include/linux/bpf_verifier.h | 29 ++++++++++++++----------
kernel/bpf/states.c | 44 +++++++++++++++++++++++++-----------
kernel/bpf/verifier.c | 20 +++++++++-------
3 files changed, 60 insertions(+), 33 deletions(-)
diff --git a/include/linux/bpf_verifier.h b/include/linux/bpf_verifier.h
index 0f348ce81fe2..d8cafd3d74aa 100644
--- a/include/linux/bpf_verifier.h
+++ b/include/linux/bpf_verifier.h
@@ -408,10 +408,7 @@ struct bpf_jmp_history_entry {
static_assert(MAX_CALL_FRAMES <= (1 << 4));
static_assert(MAX_BPF_STACK_SLOTS <= (1 << 12));
-/* Maximum number of bpf_reg_state objects that can exist at once */
#define MAX_STACK_ARG_SLOTS (MAX_BPF_FUNC_ARGS - MAX_BPF_FUNC_REG_ARGS)
-#define BPF_ID_MAP_SIZE ((MAX_BPF_REG + MAX_BPF_STACK_SLOTS + MAX_STACK_ARG_SLOTS) * \
- MAX_CALL_FRAMES)
struct bpf_verifier_state {
/* call stack tracking */
struct bpf_func_state *frame[MAX_CALL_FRAMES];
@@ -842,18 +839,27 @@ struct bpf_id_pair {
u32 cur;
};
+/*
+ * Scratch map from the ids of one verifier state to those of another, also
+ * used as a stack of ids. Grown on demand by bpf_id_scratch_reserve().
+ */
struct bpf_idmap {
u32 tmp_id_gen;
u32 cnt;
- struct bpf_id_pair map[BPF_ID_MAP_SIZE];
+ u32 cap;
+ struct bpf_id_pair *map;
};
+struct bpf_idset_entry {
+ u32 id;
+ u32 cnt;
+};
+
+/* Scratch set of ids with a use count each, grown on demand */
struct bpf_idset {
u32 num_ids;
- struct {
- u32 id;
- u32 cnt;
- } entries[BPF_ID_MAP_SIZE];
+ u32 cap;
+ struct bpf_idset_entry *entries;
};
/* see verifier.c:compute_scc_callchain() */
@@ -948,10 +954,8 @@ struct bpf_verifier_env {
struct bpf_subprog_info subprog_info[BPF_MAX_SUBPROGS + 2]; /* max + 2 for the fake and exception subprogs */
/* subprog indices sorted in topological order: leaves first, callers last */
int subprog_topo_order[BPF_MAX_SUBPROGS + 2];
- union {
- struct bpf_idmap idmap_scratch;
- struct bpf_idset idset_scratch;
- };
+ struct bpf_idmap idmap_scratch;
+ struct bpf_idset idset_scratch;
struct {
int *insn_state;
int *insn_stack;
@@ -1209,6 +1213,7 @@ int bpf_copy_verifier_state(struct bpf_verifier_state *dst_state,
struct list_head *bpf_explored_state(struct bpf_verifier_env *env, int idx);
void bpf_free_verifier_state(struct bpf_verifier_state *state, bool free_self);
void bpf_free_backedges(struct bpf_scc_visit *visit);
+bool bpf_id_scratch_reserve(void **arr, u32 *cap, u32 cnt, size_t elem_size);
int bpf_push_jmp_history(struct bpf_verifier_env *env, struct bpf_verifier_state *cur,
int insn_flags, int spi, int frame, const u16 *linked_regs,
u8 linked_regs_cnt);
diff --git a/kernel/bpf/states.c b/kernel/bpf/states.c
index e84f37d724d5..5078e4832c6e 100644
--- a/kernel/bpf/states.c
+++ b/kernel/bpf/states.c
@@ -335,21 +335,18 @@ static bool check_ids(u32 old_id, u32 cur_id, struct bpf_idmap *idmap)
return false;
}
- /* Reached the end of known mappings; haven't seen this id before */
- if (idmap->cnt < BPF_ID_MAP_SIZE) {
- map[idmap->cnt].old = old_id;
- map[idmap->cnt].cur = cur_id;
- idmap->cnt++;
- return true;
- }
-
/*
- * idmap slots are bounded by the number of registers and stack slots.
- * Since referenced dynptrs acquire intermediate references that do
- * not live in either, so the map can be exhausted. Since it is unlikely,
- * fail the verification by treating the states as not equivalent.
+ * Reached the end of known mappings; haven't seen this id before. If
+ * the map cannot grow, treat the states as not equivalent, which only
+ * costs pruning.
*/
- return false;
+ if (!bpf_id_scratch_reserve((void **)&idmap->map, &idmap->cap, idmap->cnt, sizeof(*map)))
+ return false;
+ map = idmap->map;
+ map[idmap->cnt].old = old_id;
+ map[idmap->cnt].cur = cur_id;
+ idmap->cnt++;
+ return true;
}
/*
@@ -965,6 +962,27 @@ static bool func_states_equal(struct bpf_verifier_env *env, struct bpf_func_stat
return true;
}
+/*
+ * Make room for one more entry in an id scratch array, doubling it as needed.
+ * Returns false if it could not grow; callers then treat the id as unknown
+ * or the states as different, which is always safe.
+ */
+bool bpf_id_scratch_reserve(void **arr, u32 *cap, u32 cnt, size_t elem_size)
+{
+ u32 new_cap;
+ void *p;
+
+ if (cnt < *cap)
+ return true;
+ new_cap = *cap ? *cap * 2 : 64;
+ p = krealloc_array(*arr, new_cap, elem_size, GFP_KERNEL_ACCOUNT | __GFP_NOWARN);
+ if (!p)
+ return false;
+ *arr = p;
+ *cap = new_cap;
+ return true;
+}
+
static void reset_idmap_scratch(struct bpf_verifier_env *env)
{
struct bpf_idmap *idmap = &env->idmap_scratch;
diff --git a/kernel/bpf/verifier.c b/kernel/bpf/verifier.c
index d8f43f3a8991..8641f1a8d017 100644
--- a/kernel/bpf/verifier.c
+++ b/kernel/bpf/verifier.c
@@ -10136,8 +10136,9 @@ static int idstack_push(struct bpf_idmap *idmap, u32 id)
if (idmap->map[i].old == id)
return 0;
- if (WARN_ON_ONCE(idmap->cnt >= BPF_ID_MAP_SIZE))
- return -EFAULT;
+ if (!bpf_id_scratch_reserve((void **)&idmap->map, &idmap->cap, idmap->cnt,
+ sizeof(*idmap->map)))
+ return -ENOMEM;
idmap->map[idmap->cnt++].old = id;
return 0;
@@ -18438,12 +18439,13 @@ static void idset_cnt_inc(struct bpf_idset *idset, u32 id)
return;
}
}
- /* New id */
- if (idset->num_ids < BPF_ID_MAP_SIZE) {
- idset->entries[idset->num_ids].id = id;
- idset->entries[idset->num_ids].cnt = 1;
- idset->num_ids++;
- }
+ /* New id; one that cannot be recorded counts as shared and is kept */
+ if (!bpf_id_scratch_reserve((void **)&idset->entries, &idset->cap, idset->num_ids,
+ sizeof(*idset->entries)))
+ return;
+ idset->entries[idset->num_ids].id = id;
+ idset->entries[idset->num_ids].cnt = 1;
+ idset->num_ids++;
}
/* Find id in idset and return its count, or 0 if not found */
@@ -22003,6 +22005,8 @@ int bpf_check(struct bpf_prog **prog, union bpf_attr *attr, bpfptr_t uattr,
kvfree(env->scc_info);
kvfree(env->succ);
kvfree(env->gotox_tmp_buf);
+ kfree(env->idmap_scratch.map);
+ kfree(env->idset_scratch.entries);
bpf_diag_free(env);
kvfree(env);
return ret;
--
2.53.0
^ permalink raw reply related [flat|nested] 34+ messages in thread* Re: [PATCH bpf-next v1 08/18] bpf: Grow the verifier id scratch on demand
2026-09-23 19:11 ` [PATCH bpf-next v1 08/18] bpf: Grow the verifier id scratch on demand Kumar Kartikeya Dwivedi
@ 2026-09-23 19:24 ` sashiko-bot
2026-09-23 19:29 ` Kumar Kartikeya Dwivedi
0 siblings, 1 reply; 34+ messages in thread
From: sashiko-bot @ 2026-09-23 19:24 UTC (permalink / raw)
To: Kumar Kartikeya Dwivedi; +Cc: bpf
Thank you for your contribution! Sashiko AI review found 1 potential issue(s) to consider:
- [Medium] bpf: OOM in release_reference() triggers WARN_ON_ONCE and is improperly ignored in unmark_stack_slots_iter()
--
commit bdefbd593b54c82f98de0288f1f6f5e793ee9fbb
Author: Kumar Kartikeya Dwivedi <memxor@gmail.com>
bpf: Grow the verifier id scratch on demand
The id map used to compare the ids of two states, which also serves as
the id stack of release_reference() and as the id set of
bpf_clear_singular_ids(), is a fixed array embedded in struct
bpf_verifier_env and sized for the most registers and stack slots a
state can possibly hold. Turn the map and the set into arrays grown on
demand, starting at 64 entries and doubling, and free them with the env.
> diff --git a/kernel/bpf/verifier.c b/kernel/bpf/verifier.c
> index d8f43f3a89912..8641f1a8d017d 100644
> --- a/kernel/bpf/verifier.c
> +++ b/kernel/bpf/verifier.c
> @@ -10136,8 +10136,9 @@ static int idstack_push(struct bpf_idmap *idmap, u32 id)
> if (idmap->map[i].old == id)
> return 0;
>
> - if (WARN_ON_ONCE(idmap->cnt >= BPF_ID_MAP_SIZE))
> - return -EFAULT;
> + if (!bpf_id_scratch_reserve((void **)&idmap->map, &idmap->cap, idmap->cnt,
> + sizeof(*idmap->map)))
> + return -ENOMEM;
[Severity: Medium]
With this change, idstack_push() can return -ENOMEM on a normal allocation
failure, which propagates up through release_reference().
However, in kernel/bpf/verifier.c:unmark_stack_slots_iter(), we have:
for (i = 0; i < nr_slots; i++) {
struct bpf_stack_state *slot = bpf_stack_slot(state, spi - i);
struct bpf_reg_state *st = &slot->spilled_ptr;
if (i == 0)
WARN_ON_ONCE(release_reference(env, st->id));
Does this mean a simple -ENOMEM during verification will trigger this
WARN_ON_ONCE splat? Since release_reference() can now legitimately fail under
memory pressure, should this warning be removed, and should the -ENOMEM error
be properly propagated here instead of discarded?
--
Sashiko AI review · https://sashiko.dev/#/patchset/20260923191139.2816206-1-memxor@gmail.com?part=8
^ permalink raw reply [flat|nested] 34+ messages in thread* Re: [PATCH bpf-next v1 08/18] bpf: Grow the verifier id scratch on demand
2026-09-23 19:24 ` sashiko-bot
@ 2026-09-23 19:29 ` Kumar Kartikeya Dwivedi
0 siblings, 0 replies; 34+ messages in thread
From: Kumar Kartikeya Dwivedi @ 2026-09-23 19:29 UTC (permalink / raw)
To: sashiko-reviews; +Cc: bpf
On Wed Sep 23, 2026 at 9:24 PM CEST, sashiko-bot wrote:
> Thank you for your contribution! Sashiko AI review found 1 potential issue(s) to consider:
> - [Medium] bpf: OOM in release_reference() triggers WARN_ON_ONCE and is improperly ignored in unmark_stack_slots_iter()
> --
>
> commit bdefbd593b54c82f98de0288f1f6f5e793ee9fbb
> Author: Kumar Kartikeya Dwivedi <memxor@gmail.com>
>
> bpf: Grow the verifier id scratch on demand
>
> The id map used to compare the ids of two states, which also serves as
> the id stack of release_reference() and as the id set of
> bpf_clear_singular_ids(), is a fixed array embedded in struct
> bpf_verifier_env and sized for the most registers and stack slots a
> state can possibly hold. Turn the map and the set into arrays grown on
> demand, starting at 64 entries and doubling, and free them with the env.
>
>> diff --git a/kernel/bpf/verifier.c b/kernel/bpf/verifier.c
>> index d8f43f3a89912..8641f1a8d017d 100644
>> --- a/kernel/bpf/verifier.c
>> +++ b/kernel/bpf/verifier.c
>> @@ -10136,8 +10136,9 @@ static int idstack_push(struct bpf_idmap *idmap, u32 id)
>> if (idmap->map[i].old == id)
>> return 0;
>>
>> - if (WARN_ON_ONCE(idmap->cnt >= BPF_ID_MAP_SIZE))
>> - return -EFAULT;
>> + if (!bpf_id_scratch_reserve((void **)&idmap->map, &idmap->cap, idmap->cnt,
>> + sizeof(*idmap->map)))
>> + return -ENOMEM;
>
> [Severity: Medium]
> With this change, idstack_push() can return -ENOMEM on a normal allocation
> failure, which propagates up through release_reference().
>
> However, in kernel/bpf/verifier.c:unmark_stack_slots_iter(), we have:
>
> for (i = 0; i < nr_slots; i++) {
> struct bpf_stack_state *slot = bpf_stack_slot(state, spi - i);
> struct bpf_reg_state *st = &slot->spilled_ptr;
>
> if (i == 0)
> WARN_ON_ONCE(release_reference(env, st->id));
>
> Does this mean a simple -ENOMEM during verification will trigger this
> WARN_ON_ONCE splat? Since release_reference() can now legitimately fail under
> memory pressure, should this warning be removed, and should the -ENOMEM error
> be properly propagated here instead of discarded?
This one is legit, probably needs more graceful handling.
^ permalink raw reply [flat|nested] 34+ messages in thread
* [PATCH bpf-next v1 09/18] selftests/bpf: Cover the tail call caller stack depth limit
2026-09-23 19:11 [PATCH bpf-next v1 00/18] Raise BPF program stack size to 2KiB Kumar Kartikeya Dwivedi
` (7 preceding siblings ...)
2026-09-23 19:11 ` [PATCH bpf-next v1 08/18] bpf: Grow the verifier id scratch on demand Kumar Kartikeya Dwivedi
@ 2026-09-23 19:11 ` Kumar Kartikeya Dwivedi
2026-09-23 19:11 ` [PATCH bpf-next v1 10/18] selftests/bpf: Check that narrow stack stores define no slot Kumar Kartikeya Dwivedi
` (8 subsequent siblings)
17 siblings, 0 replies; 34+ messages in thread
From: Kumar Kartikeya Dwivedi @ 2026-09-23 19:11 UTC (permalink / raw)
To: bpf
Cc: Alexei Starovoitov, Andrii Nakryiko, Daniel Borkmann,
Eduard Zingerman, Emil Tsalapatis, Tejun Heo, kkd, kernel-team
A tail call from a subprog only unwinds that subprog's frame, so the
verifier refuses tail calls once the frames of the callers add up to
256 bytes or more. Nothing exercised that rule. Add a pair of tests
with a caller using 240 and 256 bytes of stack respectively, the
latter expecting the "tail_calls are not allowed when call stack of
previous frames is 256 bytes" rejection.
Signed-off-by: Kumar Kartikeya Dwivedi <memxor@gmail.com>
---
.../selftests/bpf/progs/verifier_tailcall.c | 57 +++++++++++++++++++
1 file changed, 57 insertions(+)
diff --git a/tools/testing/selftests/bpf/progs/verifier_tailcall.c b/tools/testing/selftests/bpf/progs/verifier_tailcall.c
index b4acce60fb9b..51687da97225 100644
--- a/tools/testing/selftests/bpf/progs/verifier_tailcall.c
+++ b/tools/testing/selftests/bpf/progs/verifier_tailcall.c
@@ -28,4 +28,61 @@ __naked void invalid_map_for_tail_call(void)
: __clobber_all);
}
+struct {
+ __uint(type, BPF_MAP_TYPE_PROG_ARRAY);
+ __uint(max_entries, 1);
+ __uint(key_size, sizeof(__u32));
+ __uint(value_size, sizeof(__u32));
+} jmp_table SEC(".maps");
+
+__used __naked
+static int subprog_tail_call(void)
+{
+ asm volatile (" \
+ r2 = %[jmp_table] ll; \
+ r3 = 0; \
+ call %[bpf_tail_call]; \
+ r0 = 0; \
+ exit; \
+" :
+ : __imm(bpf_tail_call),
+ __imm_addr(jmp_table)
+ : __clobber_all);
+}
+
+/*
+ * A tail call unwinds only the frame of the subprog doing it, so the
+ * frames of its callers stay on the stack. With up to 33 tail calls in
+ * a chain the verifier caps the stack those frames may add up to at
+ * 256 bytes.
+ */
+SEC("tc")
+__description("tail call from subprog with 240 bytes of caller stack")
+__success
+__naked void tail_call_caller_stack_ok(void)
+{
+ asm volatile (" \
+ r2 = 42; \
+ *(u64 *)(r10 - 240) = r2; \
+ call subprog_tail_call; \
+ r0 = 0; \
+ exit; \
+" ::: __clobber_all);
+}
+
+SEC("tc")
+__description("tail call from subprog with 256 bytes of caller stack")
+__failure
+__msg("tail_calls are not allowed when call stack of previous frames is 256 bytes. Too large")
+__naked void tail_call_caller_stack_too_large(void)
+{
+ asm volatile (" \
+ r2 = 42; \
+ *(u64 *)(r10 - 256) = r2; \
+ call subprog_tail_call; \
+ r0 = 0; \
+ exit; \
+" ::: __clobber_all);
+}
+
char _license[] SEC("license") = "GPL";
--
2.53.0
^ permalink raw reply related [flat|nested] 34+ messages in thread* [PATCH bpf-next v1 10/18] selftests/bpf: Check that narrow stack stores define no slot
2026-09-23 19:11 [PATCH bpf-next v1 00/18] Raise BPF program stack size to 2KiB Kumar Kartikeya Dwivedi
` (8 preceding siblings ...)
2026-09-23 19:11 ` [PATCH bpf-next v1 09/18] selftests/bpf: Cover the tail call caller stack depth limit Kumar Kartikeya Dwivedi
@ 2026-09-23 19:11 ` Kumar Kartikeya Dwivedi
2026-09-23 19:11 ` [PATCH bpf-next v1 11/18] selftests/bpf: Check liveness merge of masks with different widths Kumar Kartikeya Dwivedi
` (7 subsequent siblings)
17 siblings, 0 replies; 34+ messages in thread
From: Kumar Kartikeya Dwivedi @ 2026-09-23 19:11 UTC (permalink / raw)
To: bpf
Cc: Alexei Starovoitov, Andrii Nakryiko, Daniel Borkmann,
Eduard Zingerman, Emil Tsalapatis, Tejun Heo, kkd, kernel-team
The stack liveness analysis records a store as a "def" only for the
4-byte half-slots it covers completely, so a one or two byte store near
the top of the frame must not define anything. Add a test that reads
fp-8, stores one byte at fp-1 and two bytes at fp-4, and reads fp-8
again, expecting no def mark on either store and the second read to
still use fp-8.
Signed-off-by: Kumar Kartikeya Dwivedi <memxor@gmail.com>
---
.../selftests/bpf/progs/verifier_live_stack.c | 22 +++++++++++++++++++
1 file changed, 22 insertions(+)
diff --git a/tools/testing/selftests/bpf/progs/verifier_live_stack.c b/tools/testing/selftests/bpf/progs/verifier_live_stack.c
index 7a1a0670f851..f8758eb62dac 100644
--- a/tools/testing/selftests/bpf/progs/verifier_live_stack.c
+++ b/tools/testing/selftests/bpf/progs/verifier_live_stack.c
@@ -2839,3 +2839,25 @@ static __used __naked void imprecise_dst_spill_join_sub(void)
:: __imm(bpf_get_prandom_u32)
: __clobber_all);
}
+
+/*
+ * A store that does not fully cover a 4-byte half-slot defines nothing, so a
+ * narrow store at the top of the frame must not turn any slot into a "def",
+ * least of all every slot of the frame: the earlier data at fp-8 stays live.
+ */
+SEC("socket")
+__log_level(2)
+__msg("0: (79) r0 = *(u64 *)(r10 -8) ; use: fp0-8")
+__msg("1: (73) *(u8 *)(r10 -1) = r0{{$}}")
+__msg("2: (6b) *(u16 *)(r10 -4) = r0{{$}}")
+__msg("3: (79) r0 = *(u64 *)(r10 -8) ; use: fp0-8")
+__naked void narrow_store_defines_nothing(void)
+{
+ asm volatile (
+ "r0 = *(u64 *)(r10 - 8);"
+ "*(u8 *)(r10 - 1) = r0;"
+ "*(u16 *)(r10 - 4) = r0;"
+ "r0 = *(u64 *)(r10 - 8);"
+ "exit;"
+ ::: __clobber_all);
+}
--
2.53.0
^ permalink raw reply related [flat|nested] 34+ messages in thread* [PATCH bpf-next v1 11/18] selftests/bpf: Check liveness merge of masks with different widths
2026-09-23 19:11 [PATCH bpf-next v1 00/18] Raise BPF program stack size to 2KiB Kumar Kartikeya Dwivedi
` (9 preceding siblings ...)
2026-09-23 19:11 ` [PATCH bpf-next v1 10/18] selftests/bpf: Check that narrow stack stores define no slot Kumar Kartikeya Dwivedi
@ 2026-09-23 19:11 ` Kumar Kartikeya Dwivedi
2026-09-23 19:26 ` sashiko-bot
2026-09-23 19:11 ` [PATCH bpf-next v1 12/18] bpf: Size the per-frame verifier structures for a 2 KiB stack Kumar Kartikeya Dwivedi
` (6 subsequent siblings)
17 siblings, 1 reply; 34+ messages in thread
From: Kumar Kartikeya Dwivedi @ 2026-09-23 19:11 UTC (permalink / raw)
To: bpf
Cc: Alexei Starovoitov, Andrii Nakryiko, Daniel Borkmann,
Eduard Zingerman, Emil Tsalapatis, Tejun Heo, kkd, kernel-team
The liveness masks of a function instance are as wide as the deepest
half-slot the instance was seen to access. When the same instance is
analyzed again through another call site, the new pass may have settled
on a different width, and merge_instances() has to widen the original
before combining the two. Add a test where the first pass of a callee
reads through a pointer 264 bytes into the main frame and the second
one through a pointer of unknown offset, a whole-frame read at the
maximum width, and check that the merged result keeps the whole-frame
read.
Signed-off-by: Kumar Kartikeya Dwivedi <memxor@gmail.com>
---
.../selftests/bpf/progs/verifier_live_stack.c | 49 +++++++++++++++++++
1 file changed, 49 insertions(+)
diff --git a/tools/testing/selftests/bpf/progs/verifier_live_stack.c b/tools/testing/selftests/bpf/progs/verifier_live_stack.c
index f8758eb62dac..a94365decd5f 100644
--- a/tools/testing/selftests/bpf/progs/verifier_live_stack.c
+++ b/tools/testing/selftests/bpf/progs/verifier_live_stack.c
@@ -2861,3 +2861,52 @@ __naked void narrow_store_defines_nothing(void)
"exit;"
::: __clobber_all);
}
+
+/*
+ * The same callee instance is analyzed twice: the call sites are visited in
+ * postorder, so the second one goes first with a precise pointer 264 bytes
+ * into the main frame, and the first one then passes a pointer of unknown
+ * offset, which reads the whole frame. The masks of the two passes differ in
+ * width; merging the second into the first must keep the whole-frame read.
+ */
+SEC("socket")
+__log_level(2)
+__msg("stack use/def subprog#{{[0-9]+}} merge_read_all_callee (d2,cs{{[0-9]+}}):")
+__msg("(79) r0 = *(u64 *)(r1 +0){{.*}}; use: fp0-8..-512")
+__naked void merge_keeps_whole_frame_read(void)
+{
+ asm volatile (
+ "r1 = 0;"
+ "*(u64 *)(r10 - 8) = r1;"
+ "*(u64 *)(r10 - 16) = r1;"
+ "*(u64 *)(r10 - 264) = r1;"
+ "call %[bpf_get_prandom_u32];"
+ "r0 &= 8;"
+ "r1 = r10;"
+ "r1 += -16;"
+ "r1 += r0;"
+ "call merge_read_all_mid;"
+ "r1 = r10;"
+ "r1 += -264;"
+ "call merge_read_all_mid;"
+ "r0 = 0;"
+ "exit;"
+ :: __imm(bpf_get_prandom_u32)
+ : __clobber_all);
+}
+
+static __used __naked void merge_read_all_mid(void)
+{
+ asm volatile (
+ "call merge_read_all_callee;"
+ "exit;"
+ ::: __clobber_all);
+}
+
+static __used __naked void merge_read_all_callee(void)
+{
+ asm volatile (
+ "r0 = *(u64 *)(r1 + 0);"
+ "exit;"
+ ::: __clobber_all);
+}
--
2.53.0
^ permalink raw reply related [flat|nested] 34+ messages in thread* Re: [PATCH bpf-next v1 11/18] selftests/bpf: Check liveness merge of masks with different widths
2026-09-23 19:11 ` [PATCH bpf-next v1 11/18] selftests/bpf: Check liveness merge of masks with different widths Kumar Kartikeya Dwivedi
@ 2026-09-23 19:26 ` sashiko-bot
2026-09-23 19:29 ` Kumar Kartikeya Dwivedi
0 siblings, 1 reply; 34+ messages in thread
From: sashiko-bot @ 2026-09-23 19:26 UTC (permalink / raw)
To: Kumar Kartikeya Dwivedi; +Cc: bpf
Thank you for your contribution! Sashiko AI review found 1 potential issue(s) to consider:
- [Low] bpf: merge_keeps_whole_frame_read test is vacuous on some 64-bit archs
--
commit 1299c778ee1ba2ffa8d447fa3ce55a6d2aec24cd
Author: Kumar Kartikeya Dwivedi <memxor@gmail.com>
selftests/bpf: Check liveness merge of masks with different widths
This commit adds a selftest to verify that merge_instances() correctly
widens liveness masks of different widths when a function instance is
analyzed through different call sites.
> diff --git a/tools/testing/selftests/bpf/progs/verifier_live_stack.c b/tools/testing/selftests/bpf/progs/verifier_live_stack.c
> index f8758eb62dac5..a94365decd5f1 100644
> --- a/tools/testing/selftests/bpf/progs/verifier_live_stack.c
> +++ b/tools/testing/selftests/bpf/progs/verifier_live_stack.c
> @@ -2861,3 +2861,52 @@ __naked void narrow_store_defines_nothing(void)
[ ... ]
> +SEC("socket")
> +__log_level(2)
> +__msg("stack use/def subprog#{{[0-9]+}} merge_read_all_callee (d2,cs{{[0-9]+}}):")
> +__msg("(79) r0 = *(u64 *)(r1 +0){{.*}}; use: fp0-8..-512")
> +__naked void merge_keeps_whole_frame_read(void)
> +{
> + asm volatile (
> + "r1 = 0;"
> + "*(u64 *)(r10 - 8) = r1;"
> + "*(u64 *)(r10 - 16) = r1;"
> + "*(u64 *)(r10 - 264) = r1;"
[Severity: Low]
Is this test effectively vacuous on 64-bit architectures with a 512-byte stack
limit?
The precise 8-byte read at offset -264 touches half-slot 66. This means the
mask requires 66 bits, which on a 64-bit architecture occupies 2 words.
The second pass performs an unknown-offset read, forcing a whole-frame mask.
For a 512-byte stack limit, this covers 128 half-slots (128 bits), which
also occupies 2 words.
Because both masks occupy 2 words, the actual widening logic in
widen_frame_masks() is silently skipped, missing the intended test coverage.
While later patches in this series might raise the stack limit for some
architectures, the test remains mathematically vacuous on 64-bit architectures
that retain the 512-byte limit (e.g., RISC-V, s390x, or when JIT is off).
Would changing the precise read to an offset <= 256 (e.g., -248) correctly
produce a 1-word mask and test the intended logic universally?
> + "call %[bpf_get_prandom_u32];"
> + "r0 &= 8;"
> + "r1 = r10;"
> + "r1 += -16;"
> + "r1 += r0;"
> + "call merge_read_all_mid;"
> + "r1 = r10;"
> + "r1 += -264;"
> + "call merge_read_all_mid;"
> + "r0 = 0;"
> + "exit;"
> + :: __imm(bpf_get_prandom_u32)
> + : __clobber_all);
> +}
--
Sashiko AI review · https://sashiko.dev/#/patchset/20260923191139.2816206-1-memxor@gmail.com?part=11
^ permalink raw reply [flat|nested] 34+ messages in thread* Re: [PATCH bpf-next v1 11/18] selftests/bpf: Check liveness merge of masks with different widths
2026-09-23 19:26 ` sashiko-bot
@ 2026-09-23 19:29 ` Kumar Kartikeya Dwivedi
0 siblings, 0 replies; 34+ messages in thread
From: Kumar Kartikeya Dwivedi @ 2026-09-23 19:29 UTC (permalink / raw)
To: sashiko-reviews; +Cc: bpf
On Wed Sep 23, 2026 at 9:26 PM CEST, sashiko-bot wrote:
> Thank you for your contribution! Sashiko AI review found 1 potential issue(s) to consider:
> - [Low] bpf: merge_keeps_whole_frame_read test is vacuous on some 64-bit archs
> --
>
> commit 1299c778ee1ba2ffa8d447fa3ce55a6d2aec24cd
> Author: Kumar Kartikeya Dwivedi <memxor@gmail.com>
>
> selftests/bpf: Check liveness merge of masks with different widths
>
> This commit adds a selftest to verify that merge_instances() correctly
> widens liveness masks of different widths when a function instance is
> analyzed through different call sites.
>
>> diff --git a/tools/testing/selftests/bpf/progs/verifier_live_stack.c b/tools/testing/selftests/bpf/progs/verifier_live_stack.c
>> index f8758eb62dac5..a94365decd5f1 100644
>> --- a/tools/testing/selftests/bpf/progs/verifier_live_stack.c
>> +++ b/tools/testing/selftests/bpf/progs/verifier_live_stack.c
>> @@ -2861,3 +2861,52 @@ __naked void narrow_store_defines_nothing(void)
>
> [ ... ]
>
>> +SEC("socket")
>> +__log_level(2)
>> +__msg("stack use/def subprog#{{[0-9]+}} merge_read_all_callee (d2,cs{{[0-9]+}}):")
>> +__msg("(79) r0 = *(u64 *)(r1 +0){{.*}}; use: fp0-8..-512")
>> +__naked void merge_keeps_whole_frame_read(void)
>> +{
>> + asm volatile (
>> + "r1 = 0;"
>> + "*(u64 *)(r10 - 8) = r1;"
>> + "*(u64 *)(r10 - 16) = r1;"
>> + "*(u64 *)(r10 - 264) = r1;"
>
> [Severity: Low]
> Is this test effectively vacuous on 64-bit architectures with a 512-byte stack
> limit?
>
> The precise 8-byte read at offset -264 touches half-slot 66. This means the
> mask requires 66 bits, which on a 64-bit architecture occupies 2 words.
>
> The second pass performs an unknown-offset read, forcing a whole-frame mask.
> For a 512-byte stack limit, this covers 128 half-slots (128 bits), which
> also occupies 2 words.
>
> Because both masks occupy 2 words, the actual widening logic in
> widen_frame_masks() is silently skipped, missing the intended test coverage.
>
> While later patches in this series might raise the stack limit for some
> architectures, the test remains mathematically vacuous on 64-bit architectures
> that retain the 512-byte limit (e.g., RISC-V, s390x, or when JIT is off).
>
> Would changing the precise read to an offset <= 256 (e.g., -248) correctly
> produce a 1-word mask and test the intended logic universally?
>
Plausible, will investigate.
>> + "call %[bpf_get_prandom_u32];"
>> + "r0 &= 8;"
>> + "r1 = r10;"
>> + "r1 += -16;"
>> + "r1 += r0;"
>> + "call merge_read_all_mid;"
>> + "r1 = r10;"
>> + "r1 += -264;"
>> + "call merge_read_all_mid;"
>> + "r0 = 0;"
>> + "exit;"
>> + :: __imm(bpf_get_prandom_u32)
>> + : __clobber_all);
>> +}
^ permalink raw reply [flat|nested] 34+ messages in thread
* [PATCH bpf-next v1 12/18] bpf: Size the per-frame verifier structures for a 2 KiB stack
2026-09-23 19:11 [PATCH bpf-next v1 00/18] Raise BPF program stack size to 2KiB Kumar Kartikeya Dwivedi
` (10 preceding siblings ...)
2026-09-23 19:11 ` [PATCH bpf-next v1 11/18] selftests/bpf: Check liveness merge of masks with different widths Kumar Kartikeya Dwivedi
@ 2026-09-23 19:11 ` Kumar Kartikeya Dwivedi
2026-09-23 20:12 ` bot+bpf-ci
2026-09-23 22:55 ` Alexei Starovoitov
2026-09-23 19:11 ` [PATCH bpf-next v1 13/18] bpf: Bound program stack use by a per-program limit Kumar Kartikeya Dwivedi
` (5 subsequent siblings)
17 siblings, 2 replies; 34+ messages in thread
From: Kumar Kartikeya Dwivedi @ 2026-09-23 19:11 UTC (permalink / raw)
To: bpf
Cc: Alexei Starovoitov, Andrii Nakryiko, Daniel Borkmann,
Eduard Zingerman, Emil Tsalapatis, Tejun Heo, kkd, kernel-team
The verifier keeps a few structures whose size follows the deepest
frame a program may have: the backtracking and scratched-slot bitmaps,
the jump history slot index and the clamp of the liveness masks. They
are all expressed through MAX_BPF_STACK_SLOTS, which derives from
MAX_BPF_STACK, the frame size of the interpreter.
Introduce MAX_BPF_STACK_JIT, the stack budget a program may get on a
JIT that can lay out frames of any size, and derive those structures
from it so that a frame may be as deep as that budget. Nothing grants
the budget yet, so no program verifies differently; the only visible
change is that the liveness log prints a whole-frame read up to the new
depth, so the three selftests matching such reads are updated.
The backtracking and scratched-slot bitmaps grow from one to four
words per frame, a fixed few hundred bytes per verifier environment, and
tmp_str_buf, which formats a frame's slot list for the log, grows from
320 to 1408 bytes so that all 256 slots still fit; the environment stays
within its 64 KiB allocation. The liveness masks are only as wide as
the stack a frame uses, so most frames cost the same as before; a frame
that is read as a whole, through a pointer of unknown offset or by
bpf_loop() with two callbacks, now carries masks of eight words, 192
bytes per instruction per frame instead of 48. Measured over the 5075
selftest programs, that is 0.2% of the total peak verifier memory:
strobemeta_bpf_loop and pyperf600_bpf_loop grow by 11% (1.2 MiB and
0.6 MiB), a few dozen small programs by 40 to 100 KiB each, everything
else is unchanged. The next patch bounds such reads by the program's
budget, so this cost is only paid once a JIT grants it.
Signed-off-by: Kumar Kartikeya Dwivedi <memxor@gmail.com>
---
include/linux/bpf_verifier.h | 20 ++++++++++---------
include/linux/filter.h | 5 +++++
kernel/bpf/liveness.c | 2 +-
.../selftests/bpf/progs/verifier_live_stack.c | 6 +++---
4 files changed, 20 insertions(+), 13 deletions(-)
diff --git a/include/linux/bpf_verifier.h b/include/linux/bpf_verifier.h
index d8cafd3d74aa..11fa9f1f087c 100644
--- a/include/linux/bpf_verifier.h
+++ b/include/linux/bpf_verifier.h
@@ -19,11 +19,12 @@
* that converting umax_value to int cannot overflow.
*/
#define BPF_MAX_VAR_SIZ (1 << 29)
-/* size of tmp_str_buf in bpf_verifier.
- * we need at least 306 bytes to fit full stack mask representation
- * (in the "-8,-16,...,-512" form)
+/*
+ * size of tmp_str_buf in bpf_verifier.
+ * we need at least 1399 bytes to fit full stack mask representation
+ * (in the "-8,-16,...,-2048" form)
*/
-#define TMP_STR_BUF_LEN 320
+#define TMP_STR_BUF_LEN 1408
/* Patch buffer size */
#define INSN_BUF_SIZE 32
@@ -243,12 +244,13 @@ enum bpf_stack_slot_type {
#define BPF_REG_SIZE 8 /* size of eBPF register in bytes */
/*
- * Largest number of BPF_REG_SIZE stack slots a single frame can have. A frame
- * may use any part of the MAX_BPF_STACK budget; check_max_stack_depth()
- * enforces the bound on the combined depth of frames sharing the kernel stack
- * and on each frame using a private stack.
+ * Largest number of BPF_REG_SIZE stack slots a single frame can have, sized
+ * for the largest stack budget any JIT supports. A frame may use any part of
+ * its program's budget; check_max_stack_depth() enforces the budget on the
+ * combined depth of frames sharing the kernel stack and on each frame using
+ * a private stack.
*/
-#define MAX_BPF_STACK_SLOTS (MAX_BPF_STACK / BPF_REG_SIZE)
+#define MAX_BPF_STACK_SLOTS (MAX_BPF_STACK_JIT / BPF_REG_SIZE)
/* 4-byte stack slot granularity for liveness analysis */
#define BPF_HALF_REG_SIZE 4
diff --git a/include/linux/filter.h b/include/linux/filter.h
index 422284b4fa96..5688bc647df6 100644
--- a/include/linux/filter.h
+++ b/include/linux/filter.h
@@ -98,6 +98,11 @@ struct ctl_table_header;
/* BPF program can access up to 512 bytes of stack space. */
#define MAX_BPF_STACK 512
+/*
+ * Stack budget of a program on a JIT that lays out frames of that size.
+ * The interpreter and JITs without such support keep MAX_BPF_STACK.
+ */
+#define MAX_BPF_STACK_JIT 2048
/* Helper macros for filter block array initializers. */
diff --git a/kernel/bpf/liveness.c b/kernel/bpf/liveness.c
index 07eb84ae0b7f..f83254d41043 100644
--- a/kernel/bpf/liveness.c
+++ b/kernel/bpf/liveness.c
@@ -15,7 +15,7 @@
* Half-slot 0 covers [fp-4, fp), half-slot 1 covers [fp-8, fp-4), and so on,
* hence FRAME_HALF_SPIS - 1 is the deepest half-slot a frame can have.
*/
-#define FRAME_HALF_SPIS (MAX_BPF_STACK / BPF_HALF_REG_SIZE)
+#define FRAME_HALF_SPIS (MAX_BPF_STACK_JIT / BPF_HALF_REG_SIZE)
#define FRAME_MAX_WORDS BITS_TO_LONGS(FRAME_HALF_SPIS)
/* Masks tracked for each instruction of a frame */
diff --git a/tools/testing/selftests/bpf/progs/verifier_live_stack.c b/tools/testing/selftests/bpf/progs/verifier_live_stack.c
index a94365decd5f..ec27cb40dbef 100644
--- a/tools/testing/selftests/bpf/progs/verifier_live_stack.c
+++ b/tools/testing/selftests/bpf/progs/verifier_live_stack.c
@@ -1953,7 +1953,7 @@ static __used __naked void fwd_parent_key_to_helper(void)
SEC("socket")
__log_level(2)
__success
-__msg("call bpf_map_update_elem{{.*}}; use: fp1-8..-512 fp0-8")
+__msg("call bpf_map_update_elem{{.*}}; use: fp1-8..-2048 fp0-8")
__naked void helper_arg_fallback_keeps_scanning(void)
{
asm volatile (
@@ -2267,7 +2267,7 @@ static __used __naked void merge_leaf_read(void)
SEC("socket")
__log_level(2)
__success
-__msg("call bpf_loop#181 ; use: fp2-8..-512 fp1-8..-512 fp0-8..-512")
+__msg("call bpf_loop#181 ; use: fp2-8..-2048 fp1-8..-2048 fp0-8..-2048")
__naked void bpf_loop_two_callbacks(void)
{
asm volatile (
@@ -2872,7 +2872,7 @@ __naked void narrow_store_defines_nothing(void)
SEC("socket")
__log_level(2)
__msg("stack use/def subprog#{{[0-9]+}} merge_read_all_callee (d2,cs{{[0-9]+}}):")
-__msg("(79) r0 = *(u64 *)(r1 +0){{.*}}; use: fp0-8..-512")
+__msg("(79) r0 = *(u64 *)(r1 +0){{.*}}; use: fp0-8..-2048")
__naked void merge_keeps_whole_frame_read(void)
{
asm volatile (
--
2.53.0
^ permalink raw reply related [flat|nested] 34+ messages in thread* Re: [PATCH bpf-next v1 12/18] bpf: Size the per-frame verifier structures for a 2 KiB stack
2026-09-23 19:11 ` [PATCH bpf-next v1 12/18] bpf: Size the per-frame verifier structures for a 2 KiB stack Kumar Kartikeya Dwivedi
@ 2026-09-23 20:12 ` bot+bpf-ci
2026-09-23 20:27 ` Kumar Kartikeya Dwivedi
2026-09-23 22:55 ` Alexei Starovoitov
1 sibling, 1 reply; 34+ messages in thread
From: bot+bpf-ci @ 2026-09-23 20:12 UTC (permalink / raw)
To: memxor, bpf
Cc: ast, andrii, daniel, eddyz87, emil, tj, kkd, kernel-team, ast,
andrii, daniel, martin.lau, eddyz87, yonghong.song, mason,
ihor.solodrai
[-- Attachment #1: Type: text/plain, Size: 3845 bytes --]
> diff --git a/include/linux/bpf_verifier.h b/include/linux/bpf_verifier.h
> index d8cafd3d74aa..11fa9f1f087c 100644
> --- a/include/linux/bpf_verifier.h
> +++ b/include/linux/bpf_verifier.h
> @@ -19,11 +19,12 @@
> * that converting umax_value to int cannot overflow.
> */
> #define BPF_MAX_VAR_SIZ (1 << 29)
> -/* size of tmp_str_buf in bpf_verifier.
> - * we need at least 306 bytes to fit full stack mask representation
> - * (in the "-8,-16,...,-512" form)
> +/*
> + * size of tmp_str_buf in bpf_verifier.
> + * we need at least 1399 bytes to fit full stack mask representation
> + * (in the "-8,-16,...,-2048" form)
> */
> -#define TMP_STR_BUF_LEN 320
> +#define TMP_STR_BUF_LEN 1408
TMP_STR_BUF_LEN is raised to 1408 so that bpf_fmt_stack_mask() can render
all 256 slots. The commit message says the buffer grows "so that all 256
slots still fit". But every place that consumes env->tmp_str_buf prints it
through verbose() -> bpf_verifier_vlog(), which formats into a fixed kbuf:
kernel/bpf/log.c:bpf_verifier_vlog() {
n = vscnprintf(log->kbuf, BPF_VERIFIER_TMP_LOG_SIZE, fmt, args);
}
BPF_VERIFIER_TMP_LOG_SIZE is still 1024 (include/linux/bpf_verifier.h).
The precision backtracking log lines are:
kernel/bpf/backtrack.c: {
verbose(env, "stack=%s before ", env->tmp_str_buf);
verbose(env, "stack=%s: ", env->tmp_str_buf);
}
With a full 256-slot mask, the "stack=%s before " line is 1412 bytes and is
silently cut at 1023 characters. The remaining slots and the trailing
" before " / ": " separator are lost, so the next log fragment runs into
the truncated mask.
The mask alone goes over the kbuf limit once roughly 168 or more slots are
set. The larger tmp_str_buf therefore does not deliver the complete log
line the comment and commit message promise. The extra 1 KiB only helps up
to the 1023-character kbuf limit.
This commit only resizes the structures; the verifier still caps the stack
at 512 bytes here. The problem becomes reachable later in the same series,
in d1a4a32c6444 ("bpf, x86: Allow programs 2 KiB of stack") and
5782325bd60a ("bpf, arm64: Allow programs 2 KiB of stack"), which let
bpf_jit_supports_large_stack() programs use more than 64 slots. No later
commit in 00a8442f56e5..7757cddf610a changes BPF_VERIFIER_TMP_LOG_SIZE or
the print sites.
The impact is limited to log readability (no memory-safety or
verification-result impact), but selftests or tools that parse the stack=
mask from level-2 logs will see incomplete data.
Should BPF_VERIFIER_TMP_LOG_SIZE be raised to at least TMP_STR_BUF_LEN
plus a small prefix, or should the mask be emitted in pieces with several
verbose() calls?
> /* Patch buffer size */
> #define INSN_BUF_SIZE 32
>
> @@ -243,12 +244,13 @@ enum bpf_stack_slot_type {
> #define BPF_REG_SIZE 8 /* size of eBPF register in bytes */
>
> /*
> - * Largest number of BPF_REG_SIZE stack slots a single frame can have. A frame
> - * may use any part of the MAX_BPF_STACK budget; check_max_stack_depth()
> - * enforces the bound on the combined depth of frames sharing the kernel stack
> - * and on each frame using a private stack.
> + * Largest number of BPF_REG_SIZE stack slots a single frame can have, sized
> + * for the largest stack budget any JIT supports. A frame may use any part of
> + * its program's budget; check_max_stack_depth() enforces the budget on the
> + * combined depth of frames sharing the kernel stack and on each frame using
> + * a private stack.
> */
> -#define MAX_BPF_STACK_SLOTS (MAX_BPF_STACK / BPF_REG_SIZE)
> +#define MAX_BPF_STACK_SLOTS (MAX_BPF_STACK_JIT / BPF_REG_SIZE)
[ ... ]
---
AI reviewed your patch. Please fix the bug or email reply why it's not a bug.
See: https://github.com/kernel-patches/vmtest/blob/master/ci/claude/README.md
CI run summary: https://github.com/kernel-patches/bpf/actions/runs/35909853278
^ permalink raw reply [flat|nested] 34+ messages in thread* Re: [PATCH bpf-next v1 12/18] bpf: Size the per-frame verifier structures for a 2 KiB stack
2026-09-23 20:12 ` bot+bpf-ci
@ 2026-09-23 20:27 ` Kumar Kartikeya Dwivedi
0 siblings, 0 replies; 34+ messages in thread
From: Kumar Kartikeya Dwivedi @ 2026-09-23 20:27 UTC (permalink / raw)
To: bot+bpf-ci, bpf
Cc: ast, andrii, daniel, eddyz87, emil, tj, kkd, kernel-team,
martin.lau, yonghong.song, mason, ihor.solodrai
On Wed Sep 23, 2026 at 10:12 PM CEST, bot+bpf-ci wrote:
>> diff --git a/include/linux/bpf_verifier.h b/include/linux/bpf_verifier.h
>> index d8cafd3d74aa..11fa9f1f087c 100644
>> --- a/include/linux/bpf_verifier.h
>> +++ b/include/linux/bpf_verifier.h
>> @@ -19,11 +19,12 @@
>> * that converting umax_value to int cannot overflow.
>> */
>> #define BPF_MAX_VAR_SIZ (1 << 29)
>> -/* size of tmp_str_buf in bpf_verifier.
>> - * we need at least 306 bytes to fit full stack mask representation
>> - * (in the "-8,-16,...,-512" form)
>> +/*
>> + * size of tmp_str_buf in bpf_verifier.
>> + * we need at least 1399 bytes to fit full stack mask representation
>> + * (in the "-8,-16,...,-2048" form)
>> */
>> -#define TMP_STR_BUF_LEN 320
>> +#define TMP_STR_BUF_LEN 1408
>
> TMP_STR_BUF_LEN is raised to 1408 so that bpf_fmt_stack_mask() can render
> all 256 slots. The commit message says the buffer grows "so that all 256
> slots still fit". But every place that consumes env->tmp_str_buf prints it
> through verbose() -> bpf_verifier_vlog(), which formats into a fixed kbuf:
>
> kernel/bpf/log.c:bpf_verifier_vlog() {
> n = vscnprintf(log->kbuf, BPF_VERIFIER_TMP_LOG_SIZE, fmt, args);
> }
>
> BPF_VERIFIER_TMP_LOG_SIZE is still 1024 (include/linux/bpf_verifier.h).
> The precision backtracking log lines are:
>
I will bump the size.
> kernel/bpf/backtrack.c: {
> verbose(env, "stack=%s before ", env->tmp_str_buf);
> verbose(env, "stack=%s: ", env->tmp_str_buf);
> }
>
> With a full 256-slot mask, the "stack=%s before " line is 1412 bytes and is
> silently cut at 1023 characters. The remaining slots and the trailing
> " before " / ": " separator are lost, so the next log fragment runs into
> the truncated mask.
>
> The mask alone goes over the kbuf limit once roughly 168 or more slots are
> set. The larger tmp_str_buf therefore does not deliver the complete log
> line the comment and commit message promise. The extra 1 KiB only helps up
> to the 1023-character kbuf limit.
>
> This commit only resizes the structures; the verifier still caps the stack
> at 512 bytes here. The problem becomes reachable later in the same series,
> in d1a4a32c6444 ("bpf, x86: Allow programs 2 KiB of stack") and
> 5782325bd60a ("bpf, arm64: Allow programs 2 KiB of stack"), which let
> bpf_jit_supports_large_stack() programs use more than 64 slots. No later
> commit in 00a8442f56e5..7757cddf610a changes BPF_VERIFIER_TMP_LOG_SIZE or
> the print sites.
>
> The impact is limited to log readability (no memory-safety or
> verification-result impact), but selftests or tools that parse the stack=
> mask from level-2 logs will see incomplete data.
>
> Should BPF_VERIFIER_TMP_LOG_SIZE be raised to at least TMP_STR_BUF_LEN
> plus a small prefix, or should the mask be emitted in pieces with several
> verbose() calls?
>
>> /* Patch buffer size */
>> #define INSN_BUF_SIZE 32
>>
>> @@ -243,12 +244,13 @@ enum bpf_stack_slot_type {
>> #define BPF_REG_SIZE 8 /* size of eBPF register in bytes */
>>
>> /*
>> - * Largest number of BPF_REG_SIZE stack slots a single frame can have. A frame
>> - * may use any part of the MAX_BPF_STACK budget; check_max_stack_depth()
>> - * enforces the bound on the combined depth of frames sharing the kernel stack
>> - * and on each frame using a private stack.
>> + * Largest number of BPF_REG_SIZE stack slots a single frame can have, sized
>> + * for the largest stack budget any JIT supports. A frame may use any part of
>> + * its program's budget; check_max_stack_depth() enforces the budget on the
>> + * combined depth of frames sharing the kernel stack and on each frame using
>> + * a private stack.
>> */
>> -#define MAX_BPF_STACK_SLOTS (MAX_BPF_STACK / BPF_REG_SIZE)
>> +#define MAX_BPF_STACK_SLOTS (MAX_BPF_STACK_JIT / BPF_REG_SIZE)
>
> [ ... ]
>
>
> ---
> AI reviewed your patch. Please fix the bug or email reply why it's not a bug.
> See: https://github.com/kernel-patches/vmtest/blob/master/ci/claude/README.md
>
> CI run summary: https://github.com/kernel-patches/bpf/actions/runs/35909853278
^ permalink raw reply [flat|nested] 34+ messages in thread
* Re: [PATCH bpf-next v1 12/18] bpf: Size the per-frame verifier structures for a 2 KiB stack
2026-09-23 19:11 ` [PATCH bpf-next v1 12/18] bpf: Size the per-frame verifier structures for a 2 KiB stack Kumar Kartikeya Dwivedi
2026-09-23 20:12 ` bot+bpf-ci
@ 2026-09-23 22:55 ` Alexei Starovoitov
1 sibling, 0 replies; 34+ messages in thread
From: Alexei Starovoitov @ 2026-09-23 22:55 UTC (permalink / raw)
To: Kumar Kartikeya Dwivedi, bpf
Cc: Andrii Nakryiko, Daniel Borkmann, Eduard Zingerman,
Emil Tsalapatis, Tejun Heo, kkd, kernel-team
On Wed, Sep 23, 2026 at 09:11 PM Kumar Kartikeya Dwivedi <memxor@gmail.com> wrote:
> -#define FRAME_HALF_SPIS (MAX_BPF_STACK / BPF_HALF_REG_SIZE)
> +#define FRAME_HALF_SPIS (MAX_BPF_STACK_JIT / BPF_HALF_REG_SIZE)
MAX_ARG_SPILL_SLOTS in liveness.c is still 64.
fp_off_to_slot() returns -1 for anything below fp-512, so
r6 = *(u64 *)(r10 - 520)
makes r6 ARG_IMPRECISE in fill_from_stack() no matter what was
spilled there. The next load or store through r6 goes to
record_imprecise() and marks the whole stack of every frame as read.
A prog that keeps spills below fp-512 gets no help from liveness
and all of its frames carry 8 word masks.
The numbers in the cover letter are for selftests that fit in 512.
Pls share veristat numbers for progs that use more than that.
^ permalink raw reply [flat|nested] 34+ messages in thread
* [PATCH bpf-next v1 13/18] bpf: Bound program stack use by a per-program limit
2026-09-23 19:11 [PATCH bpf-next v1 00/18] Raise BPF program stack size to 2KiB Kumar Kartikeya Dwivedi
` (11 preceding siblings ...)
2026-09-23 19:11 ` [PATCH bpf-next v1 12/18] bpf: Size the per-frame verifier structures for a 2 KiB stack Kumar Kartikeya Dwivedi
@ 2026-09-23 19:11 ` Kumar Kartikeya Dwivedi
2026-09-23 19:11 ` [PATCH bpf-next v1 14/18] selftests/bpf: Add load conditions on the program stack limit Kumar Kartikeya Dwivedi
` (4 subsequent siblings)
17 siblings, 0 replies; 34+ messages in thread
From: Kumar Kartikeya Dwivedi @ 2026-09-23 19:11 UTC (permalink / raw)
To: bpf
Cc: Alexei Starovoitov, Andrii Nakryiko, Daniel Borkmann,
Eduard Zingerman, Emil Tsalapatis, Tejun Heo, kkd, kernel-team
The verifier checks every stack access and the combined depth of a call
chain against MAX_BPF_STACK, which is also the frame size of the
interpreter and the frame that JITs without subprogram tail call
support set up for tail-call targets. A JIT that lays out frames of any
size and lets a tail-called program set up its own frame does not need
that limit; it only needs the verifier to bound how much stack a
program uses in total.
Add bpf_jit_supports_large_stack() for a JIT to claim that, and give
each program its budget through bpf_prog_stack_limit(): MAX_BPF_STACK_JIT
when the JIT is requested, the program is not offloaded and the JIT
supports large stacks as well as tail calls from subprograms,
MAX_BPF_STACK otherwise. The latter is what lets a tail-called program
set up its own frame: without it, do_misc_fixups() gives every program
with tail calls a MAX_BPF_STACK frame, which a deeper frame verified
against the larger budget would overrun. The verifier keeps the budget
in env->stack_limit and uses it for the bounds of fixed and variable
offset stack accesses, for unprivileged stack pointer arithmetic and its
speculation limit, and for the combined and private stack depth checks.
A frame may use any part of its program's budget. The interpreter paths
keep MAX_BPF_STACK: a program whose main frame is deeper falls back to
the JIT-required path of bpf_prog_select_runtime() and one with deeper
subprogram frames is rejected when patching calls for the interpreter.
The extra stack that may_goto and the timed may_goto instrumentation
add below a frame is, as before, not counted against the budget of a
JITed program and rejected past MAX_BPF_STACK for an interpreted one.
Stack liveness treats a read through a pointer of unknown offset, or a
call passing a frame pointer to a subprogram, as reaching the whole
frame, and widens the masks of that frame to the deepest half-slot such
a read can cover. Bound that by the program's budget too: no access
past it is accepted, so a program kept at MAX_BPF_STACK carries masks
of two words for such frames, as before, instead of the eight that
MAX_BPF_STACK_JIT needs. The three selftests matching a whole-frame
read in the liveness log accept either depth.
The capability is a boolean and the budget a single constant, in the
style of the other bpf_jit_supports_*() queries, rather than a per JIT
size: the budget is meant to be the same everywhere it is raised, so
that programs verify identically across those architectures.
No JIT declares support yet, so every program keeps its 512-byte budget.
Signed-off-by: Kumar Kartikeya Dwivedi <memxor@gmail.com>
---
include/linux/bpf_verifier.h | 20 ++++++++++
include/linux/filter.h | 1 +
kernel/bpf/core.c | 13 +++++++
kernel/bpf/liveness.c | 39 +++++++++++--------
kernel/bpf/verifier.c | 24 +++++++-----
.../selftests/bpf/progs/verifier_live_stack.c | 6 +--
6 files changed, 73 insertions(+), 30 deletions(-)
diff --git a/include/linux/bpf_verifier.h b/include/linux/bpf_verifier.h
index 11fa9f1f087c..a5d493b3876f 100644
--- a/include/linux/bpf_verifier.h
+++ b/include/linux/bpf_verifier.h
@@ -982,6 +982,8 @@ struct bpf_verifier_env {
u32 prev_jmps_processed, jmps_processed;
/* maximum combined stack depth */
u32 max_stack_depth;
+ /* stack budget of the program, see bpf_prog_stack_limit() */
+ u32 stack_limit;
/* total verification time */
u64 verification_time;
/* maximum number of verifier states kept in 'branching' instructions */
@@ -1235,6 +1237,24 @@ static inline int bpf_get_spi(s32 off)
return (-off - 1) / BPF_REG_SIZE;
}
+/*
+ * Stack a program may use in total: combined over the frames of a call
+ * chain on the kernel stack, or per frame on a private stack. Any single
+ * frame may reach that deep. Only a JIT that lays out such frames may go
+ * beyond MAX_BPF_STACK, the interpreter's frame size, and only one whose
+ * tail calls let the target set up its own frame: without subprogram
+ * tail calls, do_misc_fixups() gives every program with tail calls a
+ * MAX_BPF_STACK frame, which a deeper frame would overrun.
+ */
+static inline u32 bpf_prog_stack_limit(const struct bpf_prog *prog)
+{
+ /* an offloaded program never runs on the host JIT, whatever it supports */
+ if (prog->jit_requested && !bpf_prog_is_offloaded(prog->aux) &&
+ bpf_jit_supports_large_stack() && bpf_jit_supports_subprog_tailcalls())
+ return MAX_BPF_STACK_JIT;
+ return MAX_BPF_STACK;
+}
+
static inline struct bpf_func_state *bpf_func(struct bpf_verifier_env *env,
const struct bpf_reg_state *reg)
{
diff --git a/include/linux/filter.h b/include/linux/filter.h
index 5688bc647df6..cdd16bdd4dfb 100644
--- a/include/linux/filter.h
+++ b/include/linux/filter.h
@@ -1251,6 +1251,7 @@ bool bpf_jit_supports_ptr_xchg(void);
bool bpf_jit_supports_arena(void);
bool bpf_jit_supports_insn(struct bpf_insn *insn, bool in_arena);
bool bpf_jit_supports_private_stack(void);
+bool bpf_jit_supports_large_stack(void);
bool bpf_jit_supports_timed_may_goto(void);
bool bpf_jit_supports_fsession(void);
diff --git a/kernel/bpf/core.c b/kernel/bpf/core.c
index 227211166dcc..fbb2d8a840ef 100644
--- a/kernel/bpf/core.c
+++ b/kernel/bpf/core.c
@@ -3471,6 +3471,19 @@ bool __weak bpf_jit_supports_private_stack(void)
return false;
}
+/*
+ * Return TRUE if the JIT lays out frames of up to MAX_BPF_STACK_JIT bytes.
+ * Its prologue, epilogue and tail call sequences must encode such frame
+ * sizes and a private stack must be sized from the program's depth. The
+ * budget is only granted alongside bpf_jit_supports_subprog_tailcalls(),
+ * whose tail calls land before the target sets up its own frame; see
+ * bpf_prog_stack_limit().
+ */
+bool __weak bpf_jit_supports_large_stack(void)
+{
+ return false;
+}
+
void __weak arch_bpf_stack_walk(bool (*consume_fn)(void *cookie, u64 ip, u64 sp, u64 bp), void *cookie)
{
}
diff --git a/kernel/bpf/liveness.c b/kernel/bpf/liveness.c
index f83254d41043..f9beef6695c4 100644
--- a/kernel/bpf/liveness.c
+++ b/kernel/bpf/liveness.c
@@ -36,9 +36,9 @@ enum {
* relative index @i is at &bits[(i * FM_MASK_CNT + kind) * words]. A
* half-slot at or past @words * BITS_PER_LONG is never read by this frame,
* hence never live. An instruction that may read the whole frame, such as a
- * call passing a frame pointer to another subprog, widens the array to
- * FRAME_MAX_WORDS, so that the read cannot lose half-slots to a later
- * widening.
+ * call passing a frame pointer to another subprog, widens the array to the
+ * program's stack budget, the deepest an accepted program can reach, so
+ * that the read cannot lose half-slots to a later widening.
*/
struct frame_masks {
u32 words;
@@ -264,12 +264,16 @@ static int mark_stack_write(struct func_instance *instance, u32 frame, u32 insn_
/*
* Mark every half-slot of @frame as possibly read by @insn_idx. This widens
- * the masks to the maximum width: a full read recorded at a narrower width
- * would leave the bits added by a later widening clear and lose part of it.
+ * the masks to the program's stack budget: a full read recorded at a narrower
+ * width would leave the bits added by a later widening clear and lose part of
+ * it, and an access past the budget is rejected by the main pass later, so no
+ * widening of an accepted program goes further.
*/
-static int mark_stack_read_all(struct func_instance *instance, u32 frame, u32 insn_idx)
+static int mark_stack_read_all(struct bpf_verifier_env *env, struct func_instance *instance,
+ u32 frame, u32 insn_idx)
{
- return mark_stack_read(instance, frame, insn_idx, 0, FRAME_HALF_SPIS - 1);
+ return mark_stack_read(instance, frame, insn_idx, 0,
+ env->stack_limit / BPF_HALF_REG_SIZE - 1);
}
/* Accumulate @src, a mask @src_words wide, into may_read of @frame at @insn_idx */
@@ -1384,7 +1388,7 @@ static int record_stack_access_off(struct func_instance *instance, s64 fp_off,
* 'arg' is FP-derived argument to helper/kfunc or load/store that
* reads (positive) or writes (negative) 'access_bytes' into 'use' or 'def'.
*/
-static int record_stack_access(struct func_instance *instance,
+static int record_stack_access(struct bpf_verifier_env *env, struct func_instance *instance,
const struct arg_track *arg,
s64 access_bytes, u32 frame, u32 insn_idx)
{
@@ -1394,7 +1398,7 @@ static int record_stack_access(struct func_instance *instance,
return 0;
if (arg->off_cnt == 0) {
if (access_bytes > 0 || access_bytes == S64_MIN)
- return mark_stack_read_all(instance, frame, insn_idx);
+ return mark_stack_read_all(env, instance, frame, insn_idx);
return 0;
}
if (access_bytes != S64_MIN && access_bytes < 0 && arg->off_cnt != 1)
@@ -1413,7 +1417,8 @@ static int record_stack_access(struct func_instance *instance,
* When a pointer is ARG_IMPRECISE, conservatively mark every frame in
* the bitmask as fully used.
*/
-static int record_imprecise(struct func_instance *instance, u32 mask, u32 insn_idx)
+static int record_imprecise(struct bpf_verifier_env *env, struct func_instance *instance,
+ u32 mask, u32 insn_idx)
{
int depth = instance->depth;
int f, err;
@@ -1422,7 +1427,7 @@ static int record_imprecise(struct func_instance *instance, u32 mask, u32 insn_i
if (!(mask & 1))
continue;
if (f <= depth) {
- err = mark_stack_read_all(instance, f, insn_idx);
+ err = mark_stack_read_all(env, instance, f, insn_idx);
if (err)
return err;
}
@@ -1491,9 +1496,9 @@ static int record_load_store_access(struct bpf_verifier_env *env,
}
if (ptr->frame >= 0 && ptr->frame <= depth)
- return record_stack_access(instance, ptr, sz, ptr->frame, insn_idx);
+ return record_stack_access(env, instance, ptr, sz, ptr->frame, insn_idx);
if (ptr->frame == ARG_IMPRECISE)
- return record_imprecise(instance, ptr->mask, insn_idx);
+ return record_imprecise(env, instance, ptr->mask, insn_idx);
/* ARG_NONE: not derived from any frame pointer, skip */
return 0;
}
@@ -1518,7 +1523,7 @@ static int record_arg_access(struct bpf_verifier_env *env,
bytes = bpf_kfunc_stack_access_bytes(env, insn, arg_idx, insn_idx);
} else {
for (int f = 0; f <= depth; f++) {
- err = mark_stack_read_all(instance, f, insn_idx);
+ err = mark_stack_read_all(env, instance, f, insn_idx);
if (err)
return err;
}
@@ -1528,9 +1533,9 @@ static int record_arg_access(struct bpf_verifier_env *env,
return 0;
if (frame >= 0 && frame <= depth)
- err = record_stack_access(instance, at, bytes, frame, insn_idx);
+ err = record_stack_access(env, instance, at, bytes, frame, insn_idx);
else if (frame == ARG_IMPRECISE)
- err = record_imprecise(instance, at->mask, insn_idx);
+ err = record_imprecise(env, instance, at->mask, insn_idx);
return err;
}
@@ -2040,7 +2045,7 @@ static int analyze_subprog(struct bpf_verifier_env *env,
if (info[subprog].at_in[j][caller_reg].frame == ARG_NONE)
continue;
for (int f = 0; f <= depth; f++) {
- err = mark_stack_read_all(instance, f, idx);
+ err = mark_stack_read_all(env, instance, f, idx);
if (err)
goto out_free;
}
diff --git a/kernel/bpf/verifier.c b/kernel/bpf/verifier.c
index 8641f1a8d017..9bd7f9b3a67c 100644
--- a/kernel/bpf/verifier.c
+++ b/kernel/bpf/verifier.c
@@ -3595,7 +3595,8 @@ static int check_stack_write_fixed_off(struct bpf_verifier_env *env,
int hist_spi = spi, hist_frame = state->frameno;
struct bpf_stack_state *ss = bpf_stack_slot(state, spi);
- /* caller checked that off % size == 0 and -MAX_BPF_STACK <= off < 0,
+ /*
+ * caller checked that off % size == 0 and -env->stack_limit <= off < 0,
* so it's aligned access and [off, off + size) are within stack limits
*/
if (!env->allow_ptr_leaks &&
@@ -5448,7 +5449,7 @@ static int check_max_stack_depth_subprog(struct bpf_verifier_env *env, int idx,
if (subprog[idx].priv_stack_mode == PRIV_STACK_ADAPTIVE) {
if (subprog_depth > env->max_stack_depth)
env->max_stack_depth = subprog_depth;
- if (subprog_depth > MAX_BPF_STACK) {
+ if (subprog_depth > env->stack_limit) {
verbose(env, "stack size of subprog %d is %d. Too large\n",
idx, subprog_depth);
return -EACCES;
@@ -5457,7 +5458,7 @@ static int check_max_stack_depth_subprog(struct bpf_verifier_env *env, int idx,
depth += subprog_depth;
if (depth > env->max_stack_depth)
env->max_stack_depth = depth;
- if (depth > MAX_BPF_STACK) {
+ if (depth > env->stack_limit) {
total = 0;
for (tmp = idx; tmp >= 0; tmp = dinfo[tmp].caller)
total++;
@@ -6308,10 +6309,11 @@ static int check_ptr_to_map_access(struct bpf_verifier_env *env,
return 0;
}
-/* Check that the stack access at the given offset is within bounds. The
+/*
+ * Check that the stack access at the given offset is within bounds. The
* maximum valid offset is -1.
*
- * The minimum valid offset is -MAX_BPF_STACK for writes, and
+ * The minimum valid offset is -env->stack_limit for writes, and
* -state->allocated_stack for reads.
*/
static int check_stack_slot_within_bounds(struct bpf_verifier_env *env,
@@ -6322,7 +6324,7 @@ static int check_stack_slot_within_bounds(struct bpf_verifier_env *env,
int min_valid_off;
if (t == BPF_WRITE || env->allow_uninit_stack)
- min_valid_off = -MAX_BPF_STACK;
+ min_valid_off = -(int)env->stack_limit;
else
min_valid_off = -state->allocated_stack;
@@ -14728,7 +14730,8 @@ enum {
REASON_STACK = -5,
};
-static int retrieve_ptr_limit(const struct bpf_reg_state *ptr_reg,
+static int retrieve_ptr_limit(const struct bpf_verifier_env *env,
+ const struct bpf_reg_state *ptr_reg,
u32 *alu_limit, bool mask_to_left)
{
u32 max = 0, ptr_limit = 0;
@@ -14740,7 +14743,7 @@ static int retrieve_ptr_limit(const struct bpf_reg_state *ptr_reg,
* offset where we would need to deal with min/max bounds is
* currently prohibited for unprivileged.
*/
- max = MAX_BPF_STACK + mask_to_left;
+ max = env->stack_limit + mask_to_left;
ptr_limit = -ptr_reg->var_off.value;
break;
case PTR_TO_MAP_VALUE:
@@ -14860,7 +14863,7 @@ static int sanitize_ptr_alu(struct bpf_verifier_env *env,
(opcode == BPF_SUB && !off_is_neg);
}
- err = retrieve_ptr_limit(ptr_reg, &alu_limit, info->mask_to_left);
+ err = retrieve_ptr_limit(env, ptr_reg, &alu_limit, info->mask_to_left);
if (err < 0)
return err;
@@ -14990,7 +14993,7 @@ static int check_stack_access_for_ptr_arithmetic(
return -EACCES;
}
- if (off >= 0 || off < -MAX_BPF_STACK) {
+ if (off >= 0 || off < -(int)env->stack_limit) {
verbose(env, "R%d stack pointer arithmetic goes out of range, "
"prohibited for !root; off=%d\n", regno, off);
return -EACCES;
@@ -21693,6 +21696,7 @@ int bpf_check(struct bpf_prog **prog, union bpf_attr *attr, bpfptr_t uattr,
env->bt.env = env;
env->prog = *prog;
env->ops = bpf_verifier_ops[env->prog->type];
+ env->stack_limit = bpf_prog_stack_limit(env->prog);
env->allow_ptr_leaks = bpf_allow_ptr_leaks(env->prog->aux->token);
env->allow_uninit_stack = bpf_allow_uninit_stack(env->prog->aux->token);
diff --git a/tools/testing/selftests/bpf/progs/verifier_live_stack.c b/tools/testing/selftests/bpf/progs/verifier_live_stack.c
index ec27cb40dbef..dec2230f32aa 100644
--- a/tools/testing/selftests/bpf/progs/verifier_live_stack.c
+++ b/tools/testing/selftests/bpf/progs/verifier_live_stack.c
@@ -1953,7 +1953,7 @@ static __used __naked void fwd_parent_key_to_helper(void)
SEC("socket")
__log_level(2)
__success
-__msg("call bpf_map_update_elem{{.*}}; use: fp1-8..-2048 fp0-8")
+__msg("call bpf_map_update_elem{{.*}}; use: fp1-8..-{{(512|2048)}} fp0-8")
__naked void helper_arg_fallback_keeps_scanning(void)
{
asm volatile (
@@ -2267,7 +2267,7 @@ static __used __naked void merge_leaf_read(void)
SEC("socket")
__log_level(2)
__success
-__msg("call bpf_loop#181 ; use: fp2-8..-2048 fp1-8..-2048 fp0-8..-2048")
+__msg("call bpf_loop#181 ; use: fp2-8..-{{(512|2048)}} fp1-8..-{{(512|2048)}} fp0-8..-{{(512|2048)}}")
__naked void bpf_loop_two_callbacks(void)
{
asm volatile (
@@ -2872,7 +2872,7 @@ __naked void narrow_store_defines_nothing(void)
SEC("socket")
__log_level(2)
__msg("stack use/def subprog#{{[0-9]+}} merge_read_all_callee (d2,cs{{[0-9]+}}):")
-__msg("(79) r0 = *(u64 *)(r1 +0){{.*}}; use: fp0-8..-2048")
+__msg("(79) r0 = *(u64 *)(r1 +0){{.*}}; use: fp0-8..-{{(512|2048)}}")
__naked void merge_keeps_whole_frame_read(void)
{
asm volatile (
--
2.53.0
^ permalink raw reply related [flat|nested] 34+ messages in thread* [PATCH bpf-next v1 14/18] selftests/bpf: Add load conditions on the program stack limit
2026-09-23 19:11 [PATCH bpf-next v1 00/18] Raise BPF program stack size to 2KiB Kumar Kartikeya Dwivedi
` (12 preceding siblings ...)
2026-09-23 19:11 ` [PATCH bpf-next v1 13/18] bpf: Bound program stack use by a per-program limit Kumar Kartikeya Dwivedi
@ 2026-09-23 19:11 ` Kumar Kartikeya Dwivedi
2026-09-23 20:12 ` bot+bpf-ci
2026-09-23 19:11 ` [PATCH bpf-next v1 15/18] selftests/bpf: Give the 512-byte stack boundary tests a 2 KiB twin Kumar Kartikeya Dwivedi
` (3 subsequent siblings)
17 siblings, 1 reply; 34+ messages in thread
From: Kumar Kartikeya Dwivedi @ 2026-09-23 19:11 UTC (permalink / raw)
To: bpf
Cc: Alexei Starovoitov, Andrii Nakryiko, Daniel Borkmann,
Eduard Zingerman, Emil Tsalapatis, Tejun Heo, kkd, kernel-team
The stack a program may use will depend on the JIT: 2 KiB where the JIT
declares support for large stacks, 512 bytes elsewhere and for
interpreted programs. Tests that probe the limit therefore need to know
which one is in force. Add __load_if_large_stack() and
__load_if_no_large_stack() to test_loader, analogous to the JIT load
conditions, backed by a one-time probe that loads a program storing at
fp-2048. The probe caches only the verifier's verdict on that store: a
load that fails for another reason, such as a missing capability, is
reported and probed again on the next call.
Signed-off-by: Kumar Kartikeya Dwivedi <memxor@gmail.com>
---
tools/testing/selftests/bpf/progs/bpf_misc.h | 3 ++
tools/testing/selftests/bpf/test_loader.c | 24 +++++++++++
tools/testing/selftests/bpf/testing_helpers.c | 41 +++++++++++++++++++
tools/testing/selftests/bpf/testing_helpers.h | 1 +
4 files changed, 69 insertions(+)
diff --git a/tools/testing/selftests/bpf/progs/bpf_misc.h b/tools/testing/selftests/bpf/progs/bpf_misc.h
index 2ced1d751ace..f3dbc3b59bff 100644
--- a/tools/testing/selftests/bpf/progs/bpf_misc.h
+++ b/tools/testing/selftests/bpf/progs/bpf_misc.h
@@ -175,6 +175,9 @@
#define __prepare_priv __test_tag("test_prepare_priv")
#define __load_if_JITed() __test_tag("load_mode=jited")
#define __load_if_no_JITed() __test_tag("load_mode=no_jited")
+/* Whether programs may use more than 512 bytes of stack on this kernel and JIT */
+#define __load_if_large_stack() __test_tag("stack_mode=large")
+#define __load_if_no_large_stack() __test_tag("stack_mode=small")
#define __stderr(msg) __test_tag("test_expect_stderr=" msg)
#define __stderr_unpriv(msg) __test_tag("test_expect_stderr_unpriv=" msg)
#define __stdout(msg) __test_tag("test_expect_stdout=" msg)
diff --git a/tools/testing/selftests/bpf/test_loader.c b/tools/testing/selftests/bpf/test_loader.c
index a6e3fcc1079c..25eeb1c1248b 100644
--- a/tools/testing/selftests/bpf/test_loader.c
+++ b/tools/testing/selftests/bpf/test_loader.c
@@ -45,6 +45,11 @@ enum load_mode {
NO_JITED = 1 << 1,
};
+enum stack_mode {
+ LARGE_STACK = 1 << 0,
+ SMALL_STACK = 1 << 1,
+};
+
struct test_subspec {
char *name;
char *description;
@@ -70,6 +75,7 @@ struct test_spec {
int mode_mask;
int arch_mask;
int load_mask;
+ int stack_mask;
int linear_sz;
const char *skip_reason;
bool prepare_priv;
@@ -425,6 +431,7 @@ static int parse_test_spec(struct test_loader *tester,
int err = 0;
u32 arch_mask = 0;
u32 load_mask = 0;
+ u32 stack_mask = 0;
struct btf *btf;
enum arch arch;
@@ -620,6 +627,16 @@ static int parse_test_spec(struct test_loader *tester,
err = -EINVAL;
goto cleanup;
}
+ } else if ((val = str_has_pfx(s, "stack_mode="))) {
+ if (strcmp(val, "large") == 0) {
+ stack_mask = LARGE_STACK;
+ } else if (strcmp(val, "small") == 0) {
+ stack_mask = SMALL_STACK;
+ } else {
+ PRINT_FAIL("bad stack spec: '%s'", val);
+ err = -EINVAL;
+ goto cleanup;
+ }
} else if ((msg = str_has_pfx(s, "test_expect_stderr="))) {
err = push_disasm_msg(msg, &stderr_on_next_line,
&spec->priv.stderr);
@@ -659,6 +676,7 @@ static int parse_test_spec(struct test_loader *tester,
spec->arch_mask = arch_mask ?: -1;
spec->load_mask = load_mask ?: (JITED | NO_JITED);
+ spec->stack_mask = stack_mask ?: (LARGE_STACK | SMALL_STACK);
if (spec->mode_mask == 0)
spec->mode_mask = PRIV;
@@ -1331,6 +1349,7 @@ void run_subtest(struct test_loader *tester,
{
struct test_subspec *subspec = unpriv ? &spec->unpriv : &spec->priv;
int current_runtime = is_jit_enabled() ? JITED : NO_JITED;
+ int current_stack = is_large_stack_supported() ? LARGE_STACK : SMALL_STACK;
struct bpf_program *tprog = NULL, *tprog_iter;
struct bpf_link *link, *links[32] = {};
struct test_spec *spec_iter;
@@ -1360,6 +1379,11 @@ void run_subtest(struct test_loader *tester,
return;
}
+ if ((current_stack & spec->stack_mask) == 0) {
+ test__skip();
+ return;
+ }
+
if (unpriv) {
if (!can_execute_unpriv(tester, spec)) {
test__skip();
diff --git a/tools/testing/selftests/bpf/testing_helpers.c b/tools/testing/selftests/bpf/testing_helpers.c
index d1d60451c5bc..47fe61a1ebff 100644
--- a/tools/testing/selftests/bpf/testing_helpers.c
+++ b/tools/testing/selftests/bpf/testing_helpers.c
@@ -517,6 +517,47 @@ bool is_jit_enabled(void)
return enabled;
}
+/*
+ * Whether the kernel accepts a program using more than 512 bytes of stack,
+ * which depends on the JIT in use. Probed once with a program that stores
+ * at the 2 KiB depth. Only the verifier's verdict on that store is cached:
+ * a load that fails for another reason, such as a missing capability, is
+ * reported and probed again on the next call.
+ */
+bool is_large_stack_supported(void)
+{
+ static int supported = -1;
+ struct bpf_insn insns[] = {
+ BPF_ST_MEM(BPF_DW, BPF_REG_10, -2048, 0),
+ BPF_MOV64_IMM(BPF_REG_0, 0),
+ BPF_EXIT_INSN(),
+ };
+ char log[1024] = {};
+ LIBBPF_OPTS(bpf_prog_load_opts, opts,
+ .log_buf = log,
+ .log_size = sizeof(log),
+ .log_level = 1,
+ );
+ int fd;
+
+ if (supported >= 0)
+ return supported;
+
+ fd = bpf_prog_load(BPF_PROG_TYPE_SOCKET_FILTER, NULL, "GPL", insns, ARRAY_SIZE(insns),
+ &opts);
+ if (fd >= 0) {
+ close(fd);
+ supported = 1;
+ } else if (strstr(log, "invalid write to stack")) {
+ supported = 0;
+ } else {
+ fprintf(stderr, "%s: probe failed with errno %d, assuming 512 bytes:\n%s",
+ __func__, errno, log);
+ return false;
+ }
+ return supported;
+}
+
int stack_mprotect(void)
{
void *buf;
diff --git a/tools/testing/selftests/bpf/testing_helpers.h b/tools/testing/selftests/bpf/testing_helpers.h
index 1c58a2f08b64..f1505108e26a 100644
--- a/tools/testing/selftests/bpf/testing_helpers.h
+++ b/tools/testing/selftests/bpf/testing_helpers.h
@@ -59,6 +59,7 @@ struct bpf_insn;
int get_xlated_program(int fd_prog, struct bpf_insn **buf, __u32 *cnt);
int testing_prog_flags(void);
bool is_jit_enabled(void);
+bool is_large_stack_supported(void);
int stack_mprotect(void);
/* Runs diff(1) on mismatch */
--
2.53.0
^ permalink raw reply related [flat|nested] 34+ messages in thread* Re: [PATCH bpf-next v1 14/18] selftests/bpf: Add load conditions on the program stack limit
2026-09-23 19:11 ` [PATCH bpf-next v1 14/18] selftests/bpf: Add load conditions on the program stack limit Kumar Kartikeya Dwivedi
@ 2026-09-23 20:12 ` bot+bpf-ci
2026-09-23 20:27 ` Kumar Kartikeya Dwivedi
0 siblings, 1 reply; 34+ messages in thread
From: bot+bpf-ci @ 2026-09-23 20:12 UTC (permalink / raw)
To: memxor, bpf
Cc: ast, andrii, daniel, eddyz87, emil, tj, kkd, kernel-team, ast,
andrii, daniel, martin.lau, eddyz87, yonghong.song, mason,
ihor.solodrai
[-- Attachment #1: Type: text/plain, Size: 4442 bytes --]
> diff --git a/tools/testing/selftests/bpf/progs/bpf_misc.h b/tools/testing/selftests/bpf/progs/bpf_misc.h
> index 2ced1d751ace..f3dbc3b59bff 100644
> --- a/tools/testing/selftests/bpf/progs/bpf_misc.h
> +++ b/tools/testing/selftests/bpf/progs/bpf_misc.h
> @@ -175,6 +175,9 @@
> #define __prepare_priv __test_tag("test_prepare_priv")
> #define __load_if_JITed() __test_tag("load_mode=jited")
> #define __load_if_no_JITed() __test_tag("load_mode=no_jited")
> +/* Whether programs may use more than 512 bytes of stack on this kernel and JIT */
> +#define __load_if_large_stack() __test_tag("stack_mode=large")
> +#define __load_if_no_large_stack() __test_tag("stack_mode=small")
> #define __stderr(msg) __test_tag("test_expect_stderr=" msg)
> #define __stderr_unpriv(msg) __test_tag("test_expect_stderr_unpriv=" msg)
> #define __stdout(msg) __test_tag("test_expect_stdout=" msg)
> diff --git a/tools/testing/selftests/bpf/test_loader.c b/tools/testing/selftests/bpf/test_loader.c
> index a6e3fcc1079c..25eeb1c1248b 100644
> --- a/tools/testing/selftests/bpf/test_loader.c
> +++ b/tools/testing/selftests/bpf/test_loader.c
[ ... ]
> @@ -1331,6 +1349,7 @@ void run_subtest(struct test_loader *tester,
> {
> struct test_subspec *subspec = unpriv ? &spec->unpriv : &spec->priv;
> int current_runtime = is_jit_enabled() ? JITED : NO_JITED;
> + int current_stack = is_large_stack_supported() ? LARGE_STACK : SMALL_STACK;
> struct bpf_program *tprog = NULL, *tprog_iter;
> struct bpf_link *link, *links[32] = {};
> struct test_spec *spec_iter;
[ ... ]
> @@ -1360,6 +1379,11 @@ void run_subtest(struct test_loader *tester,
> return;
> }
>
> + if ((current_stack & spec->stack_mask) == 0) {
> + test__skip();
> + return;
> + }
> +
A subsystem pattern flags this as potentially concerning: the new
stack-limit load condition skips the subtest with a bare test__skip() that
prints no reason. When a user sees a verifier_large_stack or
verifier_stack_ptr subtest reported as SKIP, they get no hint whether it
was skipped because is_large_stack_supported() found the 2 KiB budget
missing or present.
This matters more because the probe can also fall back to "assuming 512
bytes" when it fails for an unrelated reason. That fallback message goes to
stderr before test__start_subtest_with_desc() runs, so it is not attached
to the subtest that gets skipped.
Does the skip need a message like the existing skip_reason and
jited-disassembly skips in run_subtest do? Or is it acceptable for these
stack-mode skips to match the arch_mask and load_mask pattern right above
this check, which also skip without a message?
> diff --git a/tools/testing/selftests/bpf/testing_helpers.c b/tools/testing/selftests/bpf/testing_helpers.c
> index d1d60451c5bc..47fe61a1ebff 100644
> --- a/tools/testing/selftests/bpf/testing_helpers.c
> +++ b/tools/testing/selftests/bpf/testing_helpers.c
> @@ -517,6 +517,47 @@ bool is_jit_enabled(void)
> return enabled;
> }
>
> +/*
> + * Whether the kernel accepts a program using more than 512 bytes of stack,
> + * which depends on the JIT in use. Probed once with a program that stores
> + * at the 2 KiB depth. Only the verifier's verdict on that store is cached:
> + * a load that fails for another reason, such as a missing capability, is
> + * reported and probed again on the next call.
> + */
> +bool is_large_stack_supported(void)
> +{
> + static int supported = -1;
> + struct bpf_insn insns[] = {
> + BPF_ST_MEM(BPF_DW, BPF_REG_10, -2048, 0),
> + BPF_MOV64_IMM(BPF_REG_0, 0),
> + BPF_EXIT_INSN(),
> + };
> + char log[1024] = {};
> + LIBBPF_OPTS(bpf_prog_load_opts, opts,
> + .log_buf = log,
> + .log_size = sizeof(log),
> + .log_level = 1,
> + );
> + int fd;
> +
> + if (supported >= 0)
> + return supported;
> +
> + fd = bpf_prog_load(BPF_PROG_TYPE_SOCKET_FILTER, NULL, "GPL", insns, ARRAY_SIZE(insns),
> + &opts);
> + if (fd >= 0) {
> + close(fd);
> + supported = 1;
> + } else if (strstr(log, "invalid write to stack")) {
> + supported = 0;
> + } else {
> + fprintf(stderr, "%s: probe failed with errno %d, assuming 512 bytes:\n%s",
> + __func__, errno, log);
> + return false;
> + }
> + return supported;
> +}
> +
[ ... ]
---
AI reviewed your patch. Please fix the bug or email reply why it's not a bug.
See: https://github.com/kernel-patches/vmtest/blob/master/ci/claude/README.md
CI run summary: https://github.com/kernel-patches/bpf/actions/runs/35909853278
^ permalink raw reply [flat|nested] 34+ messages in thread* Re: [PATCH bpf-next v1 14/18] selftests/bpf: Add load conditions on the program stack limit
2026-09-23 20:12 ` bot+bpf-ci
@ 2026-09-23 20:27 ` Kumar Kartikeya Dwivedi
0 siblings, 0 replies; 34+ messages in thread
From: Kumar Kartikeya Dwivedi @ 2026-09-23 20:27 UTC (permalink / raw)
To: bot+bpf-ci, bpf
Cc: ast, andrii, daniel, eddyz87, emil, tj, kkd, kernel-team,
martin.lau, yonghong.song, mason, ihor.solodrai
On Wed Sep 23, 2026 at 10:12 PM CEST, bot+bpf-ci wrote:
>> diff --git a/tools/testing/selftests/bpf/progs/bpf_misc.h b/tools/testing/selftests/bpf/progs/bpf_misc.h
>> index 2ced1d751ace..f3dbc3b59bff 100644
>> --- a/tools/testing/selftests/bpf/progs/bpf_misc.h
>> +++ b/tools/testing/selftests/bpf/progs/bpf_misc.h
>> @@ -175,6 +175,9 @@
>> #define __prepare_priv __test_tag("test_prepare_priv")
>> #define __load_if_JITed() __test_tag("load_mode=jited")
>> #define __load_if_no_JITed() __test_tag("load_mode=no_jited")
>> +/* Whether programs may use more than 512 bytes of stack on this kernel and JIT */
>> +#define __load_if_large_stack() __test_tag("stack_mode=large")
>> +#define __load_if_no_large_stack() __test_tag("stack_mode=small")
>> #define __stderr(msg) __test_tag("test_expect_stderr=" msg)
>> #define __stderr_unpriv(msg) __test_tag("test_expect_stderr_unpriv=" msg)
>> #define __stdout(msg) __test_tag("test_expect_stdout=" msg)
>> diff --git a/tools/testing/selftests/bpf/test_loader.c b/tools/testing/selftests/bpf/test_loader.c
>> index a6e3fcc1079c..25eeb1c1248b 100644
>> --- a/tools/testing/selftests/bpf/test_loader.c
>> +++ b/tools/testing/selftests/bpf/test_loader.c
>
> [ ... ]
>
>> @@ -1331,6 +1349,7 @@ void run_subtest(struct test_loader *tester,
>> {
>> struct test_subspec *subspec = unpriv ? &spec->unpriv : &spec->priv;
>> int current_runtime = is_jit_enabled() ? JITED : NO_JITED;
>> + int current_stack = is_large_stack_supported() ? LARGE_STACK : SMALL_STACK;
>> struct bpf_program *tprog = NULL, *tprog_iter;
>> struct bpf_link *link, *links[32] = {};
>> struct test_spec *spec_iter;
>
> [ ... ]
>
>> @@ -1360,6 +1379,11 @@ void run_subtest(struct test_loader *tester,
>> return;
>> }
>>
>> + if ((current_stack & spec->stack_mask) == 0) {
>> + test__skip();
>> + return;
>> + }
>> +
>
> A subsystem pattern flags this as potentially concerning: the new
> stack-limit load condition skips the subtest with a bare test__skip() that
> prints no reason. When a user sees a verifier_large_stack or
> verifier_stack_ptr subtest reported as SKIP, they get no hint whether it
> was skipped because is_large_stack_supported() found the 2 KiB budget
> missing or present.
>
> This matters more because the probe can also fall back to "assuming 512
> bytes" when it fails for an unrelated reason. That fallback message goes to
> stderr before test__start_subtest_with_desc() runs, so it is not attached
> to the subtest that gets skipped.
>
> Does the skip need a message like the existing skip_reason and
> jited-disassembly skips in run_subtest do? Or is it acceptable for these
> stack-mode skips to match the arch_mask and load_mask pattern right above
> this check, which also skip without a message?
>
Won't fix.
>> diff --git a/tools/testing/selftests/bpf/testing_helpers.c b/tools/testing/selftests/bpf/testing_helpers.c
>> index d1d60451c5bc..47fe61a1ebff 100644
>> --- a/tools/testing/selftests/bpf/testing_helpers.c
>> +++ b/tools/testing/selftests/bpf/testing_helpers.c
>> @@ -517,6 +517,47 @@ bool is_jit_enabled(void)
>> return enabled;
>> }
>>
>> +/*
>> + * Whether the kernel accepts a program using more than 512 bytes of stack,
>> + * which depends on the JIT in use. Probed once with a program that stores
>> + * at the 2 KiB depth. Only the verifier's verdict on that store is cached:
>> + * a load that fails for another reason, such as a missing capability, is
>> + * reported and probed again on the next call.
>> + */
>> +bool is_large_stack_supported(void)
>> +{
>> + static int supported = -1;
>> + struct bpf_insn insns[] = {
>> + BPF_ST_MEM(BPF_DW, BPF_REG_10, -2048, 0),
>> + BPF_MOV64_IMM(BPF_REG_0, 0),
>> + BPF_EXIT_INSN(),
>> + };
>> + char log[1024] = {};
>> + LIBBPF_OPTS(bpf_prog_load_opts, opts,
>> + .log_buf = log,
>> + .log_size = sizeof(log),
>> + .log_level = 1,
>> + );
>> + int fd;
>> +
>> + if (supported >= 0)
>> + return supported;
>> +
>> + fd = bpf_prog_load(BPF_PROG_TYPE_SOCKET_FILTER, NULL, "GPL", insns, ARRAY_SIZE(insns),
>> + &opts);
>> + if (fd >= 0) {
>> + close(fd);
>> + supported = 1;
>> + } else if (strstr(log, "invalid write to stack")) {
>> + supported = 0;
>> + } else {
>> + fprintf(stderr, "%s: probe failed with errno %d, assuming 512 bytes:\n%s",
>> + __func__, errno, log);
>> + return false;
>> + }
>> + return supported;
>> +}
>> +
>
> [ ... ]
>
>
> ---
> AI reviewed your patch. Please fix the bug or email reply why it's not a bug.
> See: https://github.com/kernel-patches/vmtest/blob/master/ci/claude/README.md
>
> CI run summary: https://github.com/kernel-patches/bpf/actions/runs/35909853278
^ permalink raw reply [flat|nested] 34+ messages in thread
* [PATCH bpf-next v1 15/18] selftests/bpf: Give the 512-byte stack boundary tests a 2 KiB twin
2026-09-23 19:11 [PATCH bpf-next v1 00/18] Raise BPF program stack size to 2KiB Kumar Kartikeya Dwivedi
` (13 preceding siblings ...)
2026-09-23 19:11 ` [PATCH bpf-next v1 14/18] selftests/bpf: Add load conditions on the program stack limit Kumar Kartikeya Dwivedi
@ 2026-09-23 19:11 ` Kumar Kartikeya Dwivedi
2026-09-23 19:11 ` [PATCH bpf-next v1 16/18] bpf, x86: Allow programs 2 KiB of stack Kumar Kartikeya Dwivedi
` (2 subsequent siblings)
17 siblings, 0 replies; 34+ messages in thread
From: Kumar Kartikeya Dwivedi @ 2026-09-23 19:11 UTC (permalink / raw)
To: bpf
Cc: Alexei Starovoitov, Andrii Nakryiko, Daniel Borkmann,
Eduard Zingerman, Emil Tsalapatis, Tejun Heo, kkd, kernel-team
A number of tests pin the 512-byte stack limit: accesses just past it,
call chains that add up to more than it, private stack frames and
async callbacks that exceed it. Once a JIT raises the budget to 2 KiB
those programs load, so mark them __load_if_no_large_stack() and add a
counterpart at the 2 KiB boundary under __load_if_large_stack(), so
that each kernel runs the pair that matches its budget. The
combined-depth tests that were built from frames of a few hundred
bytes now chain five 480-byte frames, which exceeds both budgets and
keeps them valid on every architecture; the number of frames reported
in the error then differs, so those messages match any count. The C
tests are limited to 512 bytes per function by the compiler, hence the
chains.
No kernel grants the larger budget yet, so the 512-byte tests still run
everywhere and the 2 KiB twins are skipped.
Signed-off-by: Kumar Kartikeya Dwivedi <memxor@gmail.com>
---
.../selftests/bpf/progs/async_stack_depth.c | 75 +++++++++++
.../bpf/progs/struct_ops_private_stack_fail.c | 47 ++++++-
.../selftests/bpf/progs/test_global_func1.c | 65 ++++++++++
.../bpf/progs/test_global_func_deep_stack.c | 33 ++++-
.../selftests/bpf/progs/verifier_live_stack.c | 4 +-
.../selftests/bpf/progs/verifier_raw_stack.c | 21 +++
.../selftests/bpf/progs/verifier_stack_ptr.c | 53 ++++++++
.../selftests/bpf/progs/verifier_var_off.c | 32 +++++
tools/testing/selftests/bpf/verifier/calls.c | 122 ++++++++++++++----
9 files changed, 421 insertions(+), 31 deletions(-)
diff --git a/tools/testing/selftests/bpf/progs/async_stack_depth.c b/tools/testing/selftests/bpf/progs/async_stack_depth.c
index 36734683acbd..9cd874a90b39 100644
--- a/tools/testing/selftests/bpf/progs/async_stack_depth.c
+++ b/tools/testing/selftests/bpf/progs/async_stack_depth.c
@@ -29,7 +29,49 @@ static int bad_timer_cb(void *map, int *key, struct bpf_timer *timer)
return buf[255] + timer_cb(NULL, NULL, NULL);
}
+/*
+ * The same shapes scaled to the 2 KiB budget of JITs with large stacks. The
+ * compiler caps a single function at 512 bytes, so the depth comes from a
+ * chain of 480-byte frames.
+ */
+__attribute__((noinline))
+static int timer_cb_large_0(void *map, int *key, struct bpf_timer *timer)
+{
+ volatile char buf[480] = {};
+ return buf[69];
+}
+
+__attribute__((noinline))
+static int timer_cb_large_1(void *map, int *key, struct bpf_timer *timer)
+{
+ volatile char buf[480] = {};
+ return buf[69] + timer_cb_large_0(map, key, timer);
+}
+
+__attribute__((noinline))
+static int timer_cb_large_2(void *map, int *key, struct bpf_timer *timer)
+{
+ volatile char buf[480] = {};
+ return buf[69] + timer_cb_large_1(map, key, timer);
+}
+
+__attribute__((noinline))
+static int timer_cb_large_3(void *map, int *key, struct bpf_timer *timer)
+{
+ volatile char buf[480] = {};
+ return buf[69] + timer_cb_large_2(map, key, timer);
+}
+
+/* 5 * 480 = 2400 bytes on its own */
+__attribute__((noinline))
+static int bad_timer_cb_large(void *map, int *key, struct bpf_timer *timer)
+{
+ volatile char buf[480] = {};
+ return buf[255] + timer_cb_large_3(map, key, timer);
+}
+
SEC("tc")
+__load_if_no_large_stack()
__failure __msg("combined stack size of 2 calls is")
int pseudo_call_check(struct __sk_buff *ctx)
{
@@ -44,7 +86,25 @@ int pseudo_call_check(struct __sk_buff *ctx)
return bpf_timer_set_callback(&elem->timer, timer_cb) + buf[0];
}
+/* main plus the four frames under timer_cb_large_3: 2400 bytes */
SEC("tc")
+__load_if_large_stack()
+__failure __msg("combined stack size of 5 calls is")
+int pseudo_call_check_large(struct __sk_buff *ctx)
+{
+ struct hmap_elem *elem;
+ volatile char buf[480] = {};
+
+ elem = bpf_map_lookup_elem(&hmap, &(int){0});
+ if (!elem)
+ return 0;
+
+ timer_cb_large_3(NULL, NULL, NULL);
+ return bpf_timer_set_callback(&elem->timer, timer_cb_large_3) + buf[0];
+}
+
+SEC("tc")
+__load_if_no_large_stack()
__failure __msg("combined stack size of 2 calls is")
int async_call_root_check(struct __sk_buff *ctx)
{
@@ -58,4 +118,19 @@ int async_call_root_check(struct __sk_buff *ctx)
return bpf_timer_set_callback(&elem->timer, bad_timer_cb) + buf[0];
}
+SEC("tc")
+__load_if_large_stack()
+__failure __msg("combined stack size of 5 calls is")
+int async_call_root_check_large(struct __sk_buff *ctx)
+{
+ struct hmap_elem *elem;
+ volatile char buf[480] = {};
+
+ elem = bpf_map_lookup_elem(&hmap, &(int){0});
+ if (!elem)
+ return 0;
+
+ return bpf_timer_set_callback(&elem->timer, bad_timer_cb_large) + buf[0];
+}
+
char _license[] SEC("license") = "GPL";
diff --git a/tools/testing/selftests/bpf/progs/struct_ops_private_stack_fail.c b/tools/testing/selftests/bpf/progs/struct_ops_private_stack_fail.c
index 1442728f5604..c8cb35b37867 100644
--- a/tools/testing/selftests/bpf/progs/struct_ops_private_stack_fail.c
+++ b/tools/testing/selftests/bpf/progs/struct_ops_private_stack_fail.c
@@ -4,6 +4,7 @@
#include <bpf/bpf_helpers.h>
#include <bpf/bpf_tracing.h>
#include "../test_kmods/bpf_testmod.h"
+#include "bpf_misc.h"
char _license[] SEC("license") = "GPL";
@@ -25,6 +26,44 @@ __noinline static int subprog1(int *a)
return subprog2(a, b);
}
+/*
+ * A chain of 480-byte frames under test_2, so that its call chain exceeds
+ * the 2 KiB budget of JITs with large stacks as well as the 512 bytes
+ * allowed elsewhere. The compiler caps a single function at 512 bytes, and
+ * the buffers are volatile so that it cannot shrink them.
+ */
+__noinline static int subprog_deep4(int *a)
+{
+ volatile char b[480] = {};
+
+ __sink(b[479]);
+ return a[10] + b[20];
+}
+
+__noinline static int subprog_deep3(int *a)
+{
+ volatile char b[480] = {};
+
+ __sink(b[479]);
+ return subprog_deep4(a) + b[20];
+}
+
+__noinline static int subprog_deep2(int *a)
+{
+ volatile char b[480] = {};
+
+ __sink(b[479]);
+ return subprog_deep3(a) + b[20];
+}
+
+__noinline static int subprog_deep1(int *a)
+{
+ volatile char b[480] = {};
+
+ __sink(b[479]);
+ return subprog_deep2(a) + b[20];
+}
+
SEC("struct_ops")
int BPF_PROG(test_1)
@@ -41,11 +80,13 @@ int BPF_PROG(test_1)
SEC("struct_ops")
int BPF_PROG(test_2)
{
- /* stack size 400 bytes */
- int a[100] = {};
+ /* stack size 476 bytes, over 2 KiB with the four 480-byte deep subprogs */
+ volatile char buf[376] = {};
+ int a[25] = {};
+ __sink(buf[375]);
a[10] = 3;
- val_j = subprog1(a);
+ val_j = subprog1(a) + subprog_deep1(a);
return 0;
}
diff --git a/tools/testing/selftests/bpf/progs/test_global_func1.c b/tools/testing/selftests/bpf/progs/test_global_func1.c
index fc69ff18880d..f0eca282e0d4 100644
--- a/tools/testing/selftests/bpf/progs/test_global_func1.c
+++ b/tools/testing/selftests/bpf/progs/test_global_func1.c
@@ -48,8 +48,73 @@ int f3(int val, struct __sk_buff *skb, int var)
}
SEC("tc")
+__load_if_no_large_stack()
__failure __msg("combined stack size of 3 calls is")
int global_func1(struct __sk_buff *skb)
{
return f0(1, skb) + f1(skb) + f2(2, skb) + f3(3, skb, 4);
}
+
+/*
+ * A chain of five frames that stay under 512 bytes each but add up to more
+ * than the 2 KiB budget of JITs with large stacks; the chain also exceeds
+ * 512 bytes after two frames, so it is rejected everywhere.
+ */
+#define MAX_STACK_LARGE 480
+
+__attribute__ ((noinline))
+int g0(struct __sk_buff *skb)
+{
+ volatile char buf[MAX_STACK_LARGE] = {};
+
+ __sink(buf[MAX_STACK_LARGE - 1]);
+
+ return skb->len;
+}
+
+__attribute__ ((noinline))
+int g1(struct __sk_buff *skb)
+{
+ volatile char buf[MAX_STACK_LARGE] = {};
+
+ __sink(buf[MAX_STACK_LARGE - 1]);
+
+ return g0(skb) + skb->len;
+}
+
+__attribute__ ((noinline))
+int g2(struct __sk_buff *skb)
+{
+ volatile char buf[MAX_STACK_LARGE] = {};
+
+ __sink(buf[MAX_STACK_LARGE - 1]);
+
+ return g1(skb) + skb->len;
+}
+
+__attribute__ ((noinline))
+int g3(struct __sk_buff *skb)
+{
+ volatile char buf[MAX_STACK_LARGE] = {};
+
+ __sink(buf[MAX_STACK_LARGE - 1]);
+
+ return g2(skb) + skb->len;
+}
+
+__attribute__ ((noinline))
+int g4(struct __sk_buff *skb)
+{
+ volatile char buf[MAX_STACK_LARGE] = {};
+
+ __sink(buf[MAX_STACK_LARGE - 1]);
+
+ return g3(skb) + skb->len;
+}
+
+SEC("tc")
+__failure __msg("combined stack size of {{[0-9]+}} calls is")
+int global_func1_deep(struct __sk_buff *skb)
+{
+ return g4(skb);
+}
diff --git a/tools/testing/selftests/bpf/progs/test_global_func_deep_stack.c b/tools/testing/selftests/bpf/progs/test_global_func_deep_stack.c
index 1b634b543b62..edb8a223a3cb 100644
--- a/tools/testing/selftests/bpf/progs/test_global_func_deep_stack.c
+++ b/tools/testing/selftests/bpf/progs/test_global_func_deep_stack.c
@@ -67,12 +67,30 @@ int XCAT(f, n)(unsigned long a) \
#define F_31 F_30 FN(31, 30)
#define F_32 F_31 FN(32, 31)
+/* Same, with a 480-byte frame, to exceed the 2 KiB budget of large stacks. */
+#define FNB(n, prev) \
+__attribute__((noinline)) \
+int XCAT(f, n)(unsigned long a) \
+{ \
+ volatile char buf[480] = {}; \
+ volatile long b = XCAT(f, prev)(a - 1); \
+ if (!b) \
+ return 0; \
+ return b + buf[479] + 1; \
+}
+
+#define F_33 F_32 FNB(33, 32)
+#define F_34 F_33 FNB(34, 33)
+#define F_35 F_34 FNB(35, 34)
+#define F_36 F_35 FNB(36, 35)
+#define F_37 F_36 FNB(37, 36)
+
#define CAT2(a, b) a ## b
#define XCAT2(a, b) CAT2(a, b)
#define F(n) XCAT2(F_, n)
-F(32)
+F(37)
/* Ensure that even 32 levels deep, the function verifies. */
SEC("syscall")
@@ -88,8 +106,21 @@ int global_func_deep_stack_success(struct __sk_buff *skb)
* the size.
*/
SEC("syscall")
+__load_if_no_large_stack()
__failure __msg("combined stack size of 34 calls")
int global_func_deep_stack_fail(struct __sk_buff *skb)
{
return f32(123);
}
+
+/*
+ * Five 480-byte frames on top of the chain: 5 * 480 + 33 * 16 = 2928 bytes,
+ * more than the 2 KiB budget of JITs with large stacks, and more than 512
+ * bytes after the second frame everywhere else.
+ */
+SEC("syscall")
+__failure __msg("combined stack size of {{[0-9]+}} calls")
+int global_func_deep_stack_fail_large(struct __sk_buff *skb)
+{
+ return f37(123);
+}
diff --git a/tools/testing/selftests/bpf/progs/verifier_live_stack.c b/tools/testing/selftests/bpf/progs/verifier_live_stack.c
index dec2230f32aa..7be813f06a5a 100644
--- a/tools/testing/selftests/bpf/progs/verifier_live_stack.c
+++ b/tools/testing/selftests/bpf/progs/verifier_live_stack.c
@@ -318,7 +318,7 @@ struct {
} map_array SEC(".maps");
SEC("socket")
-__failure __msg("invalid read from stack R2 off=-1024 size=8")
+__failure __msg("invalid read from stack R2 off=-4096 size=8")
__flag(BPF_F_TEST_STATE_FREQ)
__naked unsigned long caller_stack_write_tail_call(void)
{
@@ -329,7 +329,7 @@ __naked unsigned long caller_stack_write_tail_call(void)
"if r0 != 42 goto 1f;"
"goto 2f;"
"1:"
- "*(u64 *)(r10 - 8) = -1024;"
+ "*(u64 *)(r10 - 8) = -4096;"
"2:"
"r1 = r6;"
"r2 = r10;"
diff --git a/tools/testing/selftests/bpf/progs/verifier_raw_stack.c b/tools/testing/selftests/bpf/progs/verifier_raw_stack.c
index 9f0f48ecb421..0fe631411b9c 100644
--- a/tools/testing/selftests/bpf/progs/verifier_raw_stack.c
+++ b/tools/testing/selftests/bpf/progs/verifier_raw_stack.c
@@ -240,6 +240,7 @@ __naked void load_bytes_spilled_regs_data(void)
SEC("tc")
__description("raw_stack: skb_load_bytes, invalid access 1")
+__load_if_no_large_stack()
__failure __msg("invalid write to stack R3 off=-513 size=8")
__naked void load_bytes_invalid_access_1(void)
{
@@ -257,6 +258,26 @@ __naked void load_bytes_invalid_access_1(void)
: __clobber_all);
}
+SEC("tc")
+__description("raw_stack: skb_load_bytes, invalid access 1, large stack")
+__load_if_large_stack()
+__failure __msg("invalid write to stack R3 off=-2049 size=8")
+__naked void load_bytes_invalid_access_1_large(void)
+{
+ asm volatile (" \
+ r2 = 4; \
+ r6 = r10; \
+ r6 += -2049; \
+ r3 = r6; \
+ r4 = 8; \
+ call %[bpf_skb_load_bytes]; \
+ r0 = *(u64*)(r6 + 0); \
+ exit; \
+" :
+ : __imm(bpf_skb_load_bytes)
+ : __clobber_all);
+}
+
SEC("tc")
__description("raw_stack: skb_load_bytes, invalid access 2")
__failure __msg("invalid write to stack R3 off=-1 size=8")
diff --git a/tools/testing/selftests/bpf/progs/verifier_stack_ptr.c b/tools/testing/selftests/bpf/progs/verifier_stack_ptr.c
index 8e8cf8232255..3e0bea9819ca 100644
--- a/tools/testing/selftests/bpf/progs/verifier_stack_ptr.c
+++ b/tools/testing/selftests/bpf/progs/verifier_stack_ptr.c
@@ -235,6 +235,7 @@ __naked void to_stack_check_low_1(void)
SEC("socket")
__description("PTR_TO_STACK check low 2")
+__load_if_no_large_stack()
__success __failure_unpriv
__msg_unpriv("R1 stack pointer arithmetic goes out of range")
__retval(42)
@@ -250,8 +251,27 @@ __naked void to_stack_check_low_2(void)
" ::: __clobber_all);
}
+SEC("socket")
+__description("PTR_TO_STACK check low 2, large stack")
+__load_if_large_stack()
+__success __failure_unpriv
+__msg_unpriv("R1 stack pointer arithmetic goes out of range")
+__retval(42)
+__naked void to_stack_check_low_2_large(void)
+{
+ asm volatile (" \
+ r1 = r10; \
+ r1 += -2049; \
+ r0 = 42; \
+ *(u8*)(r1 + 1) = r0; \
+ r0 = *(u8*)(r1 + 1); \
+ exit; \
+" ::: __clobber_all);
+}
+
SEC("socket")
__description("PTR_TO_STACK check low 3")
+__load_if_no_large_stack()
__failure __msg("invalid write to stack R1 off=-513 size=1")
__msg_unpriv("R1 stack pointer arithmetic goes out of range")
__naked void to_stack_check_low_3(void)
@@ -266,6 +286,23 @@ __naked void to_stack_check_low_3(void)
" ::: __clobber_all);
}
+SEC("socket")
+__description("PTR_TO_STACK check low 3, large stack")
+__load_if_large_stack()
+__failure __msg("invalid write to stack R1 off=-2049 size=1")
+__msg_unpriv("R1 stack pointer arithmetic goes out of range")
+__naked void to_stack_check_low_3_large(void)
+{
+ asm volatile (" \
+ r1 = r10; \
+ r1 += -2049; \
+ r0 = 42; \
+ *(u8*)(r1 + 0) = r0; \
+ r0 = *(u8*)(r1 + 0); \
+ exit; \
+" ::: __clobber_all);
+}
+
SEC("socket")
__description("PTR_TO_STACK check low 4")
__failure __msg("math between fp pointer")
@@ -483,6 +520,7 @@ l1_%=: r0 = 42; \
SEC("socket")
__description("PTR_TO_STACK stack size > 512")
+__load_if_no_large_stack()
__failure __msg("invalid write to stack R1 off=-520 size=8")
__naked void stack_check_size_gt_512(void)
{
@@ -495,6 +533,21 @@ __naked void stack_check_size_gt_512(void)
" ::: __clobber_all);
}
+SEC("socket")
+__description("PTR_TO_STACK stack size > 2048")
+__load_if_large_stack()
+__failure __msg("invalid write to stack R1 off=-2056 size=8")
+__naked void stack_check_size_gt_2048(void)
+{
+ asm volatile (" \
+ r1 = r10; \
+ r1 += -2056; \
+ r0 = 42; \
+ *(u64*)(r1 + 0) = r0; \
+ exit; \
+" ::: __clobber_all);
+}
+
#ifdef __BPF_FEATURE_MAY_GOTO
SEC("socket")
__description("PTR_TO_STACK stack size 512 with may_goto with jit")
diff --git a/tools/testing/selftests/bpf/progs/verifier_var_off.c b/tools/testing/selftests/bpf/progs/verifier_var_off.c
index a63e33675091..399884911ea5 100644
--- a/tools/testing/selftests/bpf/progs/verifier_var_off.c
+++ b/tools/testing/selftests/bpf/progs/verifier_var_off.c
@@ -406,6 +406,7 @@ __naked void zero_sized_access_max_out_of_bound(void)
SEC("lwt_in")
__description("indirect variable-offset stack access, min out of bound")
+__load_if_no_large_stack()
__failure __msg("invalid variable-offset read from stack R2")
__naked void access_min_out_of_bound(void)
{
@@ -433,6 +434,37 @@ __naked void access_min_out_of_bound(void)
: __clobber_all);
}
+SEC("lwt_in")
+__description("indirect variable-offset stack access, min out of bound, large stack")
+__load_if_large_stack()
+__failure __msg("invalid variable-offset read from stack R2")
+__naked void access_min_out_of_bound_large(void)
+{
+ asm volatile (" \
+ /* Fill the top 8 bytes of the stack */ \
+ r2 = 0; \
+ *(u64*)(r10 - 8) = r2; \
+ /* Get an unknown value */ \
+ r2 = *(u32*)(r1 + 0); \
+ /* Make it small and 4-byte aligned */ \
+ r2 &= 4; \
+ r2 -= 2052; \
+ /* \
+ * add it to fp. We now have either fp-2052 or fp-2048, but\
+ * we don't know which \
+ */ \
+ r2 += r10; \
+ /* dereference it indirectly */ \
+ r1 = %[map_hash_8b] ll; \
+ call %[bpf_map_lookup_elem]; \
+ r0 = 0; \
+ exit; \
+" :
+ : __imm(bpf_map_lookup_elem),
+ __imm_addr(map_hash_8b)
+ : __clobber_all);
+}
+
SEC("cgroup/skb")
__description("indirect variable-offset stack access, min_off < min_initialized")
__success
diff --git a/tools/testing/selftests/bpf/verifier/calls.c b/tools/testing/selftests/bpf/verifier/calls.c
index 8b94b87135bc..0af237c02ddf 100644
--- a/tools/testing/selftests/bpf/verifier/calls.c
+++ b/tools/testing/selftests/bpf/verifier/calls.c
@@ -1037,15 +1037,34 @@
.result = ACCEPT,
},
{
- "calls: stack overflow using two frames (pre-call access)",
+ /*
+ * Five 480-byte frames exceed the 2 KiB budget of JITs with large
+ * stacks, and two of them the 512 bytes allowed elsewhere.
+ */
+ "calls: stack overflow using five frames (pre-call access)",
.insns = {
/* prog 1 */
- BPF_ST_MEM(BPF_B, BPF_REG_10, -300, 0),
+ BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 1),
BPF_EXIT_INSN(),
/* prog 2 */
- BPF_ST_MEM(BPF_B, BPF_REG_10, -300, 0),
+ BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
+ BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 1),
+ BPF_EXIT_INSN(),
+
+ /* prog 3 */
+ BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
+ BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 1),
+ BPF_EXIT_INSN(),
+
+ /* prog 4 */
+ BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
+ BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 1),
+ BPF_EXIT_INSN(),
+
+ /* prog 5 */
+ BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
BPF_MOV64_IMM(BPF_REG_0, 0),
BPF_EXIT_INSN(),
},
@@ -1054,15 +1073,30 @@
.result = REJECT,
},
{
- "calls: stack overflow using two frames (post-call access)",
+ "calls: stack overflow using five frames (post-call access)",
.insns = {
/* prog 1 */
BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 2),
- BPF_ST_MEM(BPF_B, BPF_REG_10, -300, 0),
+ BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
BPF_EXIT_INSN(),
/* prog 2 */
- BPF_ST_MEM(BPF_B, BPF_REG_10, -300, 0),
+ BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 2),
+ BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
+ BPF_EXIT_INSN(),
+
+ /* prog 3 */
+ BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 2),
+ BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
+ BPF_EXIT_INSN(),
+
+ /* prog 4 */
+ BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 2),
+ BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
+ BPF_EXIT_INSN(),
+
+ /* prog 5 */
+ BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
BPF_MOV64_IMM(BPF_REG_0, 0),
BPF_EXIT_INSN(),
},
@@ -1127,7 +1161,7 @@
.result = ACCEPT,
},
{
- "calls: stack depth check using three frames. test3",
+ "calls: stack depth check using five frames. test3",
.insns = {
/* main */
BPF_MOV64_REG(BPF_REG_6, BPF_REG_1),
@@ -1135,66 +1169,104 @@
BPF_MOV64_REG(BPF_REG_1, BPF_REG_6),
BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 8), /* call B */
BPF_JMP_IMM(BPF_JGE, BPF_REG_6, 0, 1),
- BPF_ST_MEM(BPF_B, BPF_REG_10, -64, 0),
+ BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
BPF_MOV64_IMM(BPF_REG_0, 0),
BPF_EXIT_INSN(),
/* A */
BPF_JMP_IMM(BPF_JLT, BPF_REG_1, 10, 1),
BPF_EXIT_INSN(),
- BPF_ST_MEM(BPF_B, BPF_REG_10, -224, 0),
+ BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
BPF_JMP_IMM(BPF_JA, 0, 0, -3),
/* B */
BPF_JMP_IMM(BPF_JGT, BPF_REG_1, 2, 1),
- BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, -6), /* call A */
- BPF_ST_MEM(BPF_B, BPF_REG_10, -256, 0),
+ BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 2), /* call C */
+ BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
+ BPF_EXIT_INSN(),
+ /* C */
+ BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 2), /* call D */
+ BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
+ BPF_EXIT_INSN(),
+ /* D */
+ BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, -12), /* call A */
+ BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
BPF_EXIT_INSN(),
},
.prog_type = BPF_PROG_TYPE_XDP,
- /* stack_main=64, stack_A=224, stack_B=256
- * and max(main+A, main+A+B) > 512
+ /*
+ * every frame is 480 bytes, main+A = 960 > 512 and
+ * max(main+A, main+B+C+D+A) = 2400 > 2048
*/
.errstr = "combined stack",
.result = REJECT,
},
{
- "calls: stack depth check using three frames. test4",
- /* void main(void) {
+ "calls: stack depth check using five frames. test4",
+ /*
+ * void main(void) {
* func1(0);
* func1(1);
* func2(1);
* }
- * void func1(int alloc_or_recurse) {
+ * void funcN(int alloc_or_recurse) { N = 1..4
* if (alloc_or_recurse) {
- * frame_pointer[-300] = 1;
+ * frame_pointer[-480] = 1;
* } else {
- * func2(alloc_or_recurse);
+ * funcN+1(alloc_or_recurse);
* }
* }
- * void func2(int alloc_or_recurse) {
+ * void func5(int alloc_or_recurse) {
* if (alloc_or_recurse) {
- * frame_pointer[-300] = 1;
+ * frame_pointer[-480] = 1;
* }
* }
+ * main also calls func2 to func5 with 1 so that every function has a
+ * path allocating its 480 bytes, and the chain adds up to 2400 bytes,
+ * more than the 2 KiB budget of JITs with large stacks, and to 960
+ * bytes after two frames, more than the 512 bytes allowed elsewhere.
*/
.insns = {
/* main */
BPF_MOV64_IMM(BPF_REG_1, 0),
- BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 6), /* call A */
+ BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 12), /* call A */
BPF_MOV64_IMM(BPF_REG_1, 1),
- BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 4), /* call A */
+ BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 10), /* call A */
+ BPF_MOV64_IMM(BPF_REG_1, 1),
+ BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 13), /* call B */
+ BPF_MOV64_IMM(BPF_REG_1, 1),
+ BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 16), /* call C */
BPF_MOV64_IMM(BPF_REG_1, 1),
- BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 7), /* call B */
+ BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 19), /* call D */
+ BPF_MOV64_IMM(BPF_REG_1, 1),
+ BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 22), /* call E */
BPF_MOV64_IMM(BPF_REG_0, 0),
BPF_EXIT_INSN(),
/* A */
BPF_JMP_IMM(BPF_JEQ, BPF_REG_1, 0, 2),
- BPF_ST_MEM(BPF_B, BPF_REG_10, -300, 0),
+ BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
BPF_EXIT_INSN(),
BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 1), /* call B */
BPF_EXIT_INSN(),
/* B */
+ BPF_JMP_IMM(BPF_JEQ, BPF_REG_1, 0, 2),
+ BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
+ BPF_EXIT_INSN(),
+ BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 1), /* call C */
+ BPF_EXIT_INSN(),
+ /* C */
+ BPF_JMP_IMM(BPF_JEQ, BPF_REG_1, 0, 2),
+ BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
+ BPF_EXIT_INSN(),
+ BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 1), /* call D */
+ BPF_EXIT_INSN(),
+ /* D */
+ BPF_JMP_IMM(BPF_JEQ, BPF_REG_1, 0, 2),
+ BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
+ BPF_EXIT_INSN(),
+ BPF_RAW_INSN(BPF_JMP|BPF_CALL, 0, 1, 0, 1), /* call E */
+ BPF_EXIT_INSN(),
+ /* E */
BPF_JMP_IMM(BPF_JEQ, BPF_REG_1, 0, 1),
- BPF_ST_MEM(BPF_B, BPF_REG_10, -300, 0),
+ BPF_ST_MEM(BPF_B, BPF_REG_10, -480, 0),
BPF_EXIT_INSN(),
},
.prog_type = BPF_PROG_TYPE_XDP,
--
2.53.0
^ permalink raw reply related [flat|nested] 34+ messages in thread* [PATCH bpf-next v1 16/18] bpf, x86: Allow programs 2 KiB of stack
2026-09-23 19:11 [PATCH bpf-next v1 00/18] Raise BPF program stack size to 2KiB Kumar Kartikeya Dwivedi
` (14 preceding siblings ...)
2026-09-23 19:11 ` [PATCH bpf-next v1 15/18] selftests/bpf: Give the 512-byte stack boundary tests a 2 KiB twin Kumar Kartikeya Dwivedi
@ 2026-09-23 19:11 ` Kumar Kartikeya Dwivedi
2026-09-23 20:12 ` bot+bpf-ci
2026-09-23 19:11 ` [PATCH bpf-next v1 17/18] bpf, arm64: " Kumar Kartikeya Dwivedi
2026-09-23 19:11 ` [PATCH bpf-next v1 18/18] selftests/bpf: Test the 2 KiB stack budget Kumar Kartikeya Dwivedi
17 siblings, 1 reply; 34+ messages in thread
From: Kumar Kartikeya Dwivedi @ 2026-09-23 19:11 UTC (permalink / raw)
To: bpf
Cc: Alexei Starovoitov, Andrii Nakryiko, Daniel Borkmann,
Eduard Zingerman, Emil Tsalapatis, Tejun Heo, kkd, kernel-team
The x86-64 JIT encodes frame sizes as 32-bit immediates in its
prologue, epilogue and tail call sequences, a tail call pops the frame
of the program making it and lands in the target's prologue before the
target allocates its own frame, and private stacks are allocated from
each program's depth, so nothing in it depends on frames staying within
512 bytes. Report bpf_jit_supports_large_stack(), which raises the
budget of JITed programs to MAX_BPF_STACK_JIT: 2 KiB combined over a
call chain, or per frame on a private stack, with no separate limit on
a single frame. Interpreted programs keep 512 bytes.
The worst case kernel stack use of a chain of tail calls grows
accordingly: the callers of a tail call may still leave at most 256
bytes on the stack, so 33 programs can accumulate 8 KiB of dead frames
below the last one, which may now use 2 KiB instead of 512 bytes, for
a little over 10 KiB in total on a 16 KiB kernel stack. The per-cpu
memory behind a private stack grows in proportion to the frames a
program asks for, up to 2 KiB plus guards per frame.
The budget does not depend on the privileges of the loader: an
unprivileged program cannot call other BPF functions, so a tail call
from one leaves no frame behind and its worst case is a single 2 KiB
frame. Programs nested through helpers or attach points, such as a
tracing program entered from a helper of a networking program, are not
accounted against each other before or after this change; each level
of nesting may now add up to 1.5 KiB more.
The verifier state of a frame grows with the stack it uses, up to four
times as many stack slots as before; the allocation is on demand, so
only programs using deep frames pay for them.
Signed-off-by: Kumar Kartikeya Dwivedi <memxor@gmail.com>
---
Documentation/bpf/bpf_design_QA.rst | 10 +++++++---
arch/x86/net/bpf_jit_comp.c | 12 ++++++++++++
2 files changed, 19 insertions(+), 3 deletions(-)
diff --git a/Documentation/bpf/bpf_design_QA.rst b/Documentation/bpf/bpf_design_QA.rst
index eb19c945f4d5..be5fc4ac00d6 100644
--- a/Documentation/bpf/bpf_design_QA.rst
+++ b/Documentation/bpf/bpf_design_QA.rst
@@ -221,9 +221,13 @@ newer kernels. BPF programs need to change accordingly when this happens.
Q: How much stack space a BPF program uses?
-------------------------------------------
-A: Currently all program types are limited to 512 bytes of stack
-space, but the verifier computes the actual amount of stack used
-and both interpreter and most JITed code consume necessary amount.
+A: A program may use up to 2 KiB of stack, combined over its call
+chain, when the JIT of the architecture reports support for large
+stacks (currently x86-64); a single function may use all of it, and
+every frame of a program running on a private stack gets the whole
+amount. Elsewhere, and whenever the interpreter is used, the limit is
+512 bytes. The verifier computes the actual amount of stack used and
+both interpreter and most JITed code consume necessary amount.
Q: Can BPF be offloaded to HW?
------------------------------
diff --git a/arch/x86/net/bpf_jit_comp.c b/arch/x86/net/bpf_jit_comp.c
index d4a980140b48..d6998c909754 100644
--- a/arch/x86/net/bpf_jit_comp.c
+++ b/arch/x86/net/bpf_jit_comp.c
@@ -4452,6 +4452,18 @@ bool bpf_jit_supports_subprog_tailcalls(void)
return true;
}
+/*
+ * Frame sizes are 32-bit immediates in the prologue, epilogue and tail call
+ * sequences, a tail call pops the caller's frame and lands in the target's
+ * prologue before the target allocates its own, and private stacks are
+ * allocated from the program's own depth, so MAX_BPF_STACK_JIT frames need
+ * nothing special.
+ */
+bool bpf_jit_supports_large_stack(void)
+{
+ return true;
+}
+
bool bpf_jit_supports_percpu_insn(void)
{
return true;
--
2.53.0
^ permalink raw reply related [flat|nested] 34+ messages in thread* Re: [PATCH bpf-next v1 16/18] bpf, x86: Allow programs 2 KiB of stack
2026-09-23 19:11 ` [PATCH bpf-next v1 16/18] bpf, x86: Allow programs 2 KiB of stack Kumar Kartikeya Dwivedi
@ 2026-09-23 20:12 ` bot+bpf-ci
2026-09-23 20:28 ` Kumar Kartikeya Dwivedi
0 siblings, 1 reply; 34+ messages in thread
From: bot+bpf-ci @ 2026-09-23 20:12 UTC (permalink / raw)
To: memxor, bpf
Cc: ast, andrii, daniel, eddyz87, emil, tj, kkd, kernel-team, ast,
andrii, daniel, martin.lau, eddyz87, yonghong.song, mason,
ihor.solodrai
[-- Attachment #1: Type: text/plain, Size: 3053 bytes --]
> diff --git a/Documentation/bpf/bpf_design_QA.rst b/Documentation/bpf/bpf_design_QA.rst
> index eb19c945f4d5..be5fc4ac00d6 100644
> --- a/Documentation/bpf/bpf_design_QA.rst
> +++ b/Documentation/bpf/bpf_design_QA.rst
> @@ -221,9 +221,13 @@ newer kernels. BPF programs need to change accordingly when this happens.
>
> Q: How much stack space a BPF program uses?
> -------------------------------------------
> -A: Currently all program types are limited to 512 bytes of stack
> -space, but the verifier computes the actual amount of stack used
> -and both interpreter and most JITed code consume necessary amount.
> +A: A program may use up to 2 KiB of stack, combined over its call
> +chain, when the JIT of the architecture reports support for large
> +stacks (currently x86-64); a single function may use all of it, and
> +every frame of a program running on a private stack gets the whole
> +amount. Elsewhere, and whenever the interpreter is used, the limit is
> +512 bytes. The verifier computes the actual amount of stack used and
> +both interpreter and most JITed code consume necessary amount.
Is the interpreter limit stated correctly here? The text says "whenever
the interpreter is used, the limit is 512 bytes" in a paragraph where
2 KiB is defined as combined over the call chain.
When bpf_jit_supports_large_stack() returns true, a program with
jit_requested set is verified against MAX_BPF_STACK_JIT (2048) in
bpf_check() via env->stack_limit = bpf_prog_stack_limit(env->prog), and
check_max_stack_depth_subprog() bounds the combined depth of the call chain
by that limit.
If JIT compilation then fails with anything other than -EFAULT (for example,
a subprogram not being jited returns -ENOTSUPP in bpf_jit_subprogs(), or an
allocation failure during blinding occurs) and CONFIG_BPF_JIT_ALWAYS_ON is
not set, bpf_jit_subprogs() restores the instructions, clears
prog->jit_requested, and bpf_fixup_call_args() falls back to the
interpreter. The interpreter checks on that path are per frame in
bpf_patch_call_args():
int bpf_patch_call_args(struct bpf_insn *insn, u32 stack_depth)
{
stack_depth = max_t(u32, stack_depth, 1);
/* Prevent out-of-bounds read to interpreters_args */
if (stack_depth > MAX_BPF_STACK)
return -EINVAL;
and bpf_prog_select_interpreter() for the main frame.
A chain of four subprograms each using 512 bytes (2048 combined) would be
accepted by the verifier and run in the interpreter, whereas before this
patch the combined chain was capped at 512. Each interpreter frame has its
own stack array, so this is not an out-of-bounds access. Should the text
clarify that the interpreter's 512 byte limit applies per function, or
should the fallback path reject a chain whose combined depth exceeds
MAX_BPF_STACK?
[ ... ]
---
AI reviewed your patch. Please fix the bug or email reply why it's not a bug.
See: https://github.com/kernel-patches/vmtest/blob/master/ci/claude/README.md
CI run summary: https://github.com/kernel-patches/bpf/actions/runs/35909853278
^ permalink raw reply [flat|nested] 34+ messages in thread* Re: [PATCH bpf-next v1 16/18] bpf, x86: Allow programs 2 KiB of stack
2026-09-23 20:12 ` bot+bpf-ci
@ 2026-09-23 20:28 ` Kumar Kartikeya Dwivedi
0 siblings, 0 replies; 34+ messages in thread
From: Kumar Kartikeya Dwivedi @ 2026-09-23 20:28 UTC (permalink / raw)
To: bot+bpf-ci, bpf
Cc: ast, andrii, daniel, eddyz87, emil, tj, kkd, kernel-team,
martin.lau, yonghong.song, mason, ihor.solodrai
On Wed Sep 23, 2026 at 10:12 PM CEST, bot+bpf-ci wrote:
>> diff --git a/Documentation/bpf/bpf_design_QA.rst b/Documentation/bpf/bpf_design_QA.rst
>> index eb19c945f4d5..be5fc4ac00d6 100644
>> --- a/Documentation/bpf/bpf_design_QA.rst
>> +++ b/Documentation/bpf/bpf_design_QA.rst
>> @@ -221,9 +221,13 @@ newer kernels. BPF programs need to change accordingly when this happens.
>>
>> Q: How much stack space a BPF program uses?
>> -------------------------------------------
>> -A: Currently all program types are limited to 512 bytes of stack
>> -space, but the verifier computes the actual amount of stack used
>> -and both interpreter and most JITed code consume necessary amount.
>> +A: A program may use up to 2 KiB of stack, combined over its call
>> +chain, when the JIT of the architecture reports support for large
>> +stacks (currently x86-64); a single function may use all of it, and
>> +every frame of a program running on a private stack gets the whole
>> +amount. Elsewhere, and whenever the interpreter is used, the limit is
>> +512 bytes. The verifier computes the actual amount of stack used and
>> +both interpreter and most JITed code consume necessary amount.
>
> Is the interpreter limit stated correctly here? The text says "whenever
> the interpreter is used, the limit is 512 bytes" in a paragraph where
> 2 KiB is defined as combined over the call chain.
>
> When bpf_jit_supports_large_stack() returns true, a program with
> jit_requested set is verified against MAX_BPF_STACK_JIT (2048) in
> bpf_check() via env->stack_limit = bpf_prog_stack_limit(env->prog), and
> check_max_stack_depth_subprog() bounds the combined depth of the call chain
> by that limit.
>
> If JIT compilation then fails with anything other than -EFAULT (for example,
> a subprogram not being jited returns -ENOTSUPP in bpf_jit_subprogs(), or an
> allocation failure during blinding occurs) and CONFIG_BPF_JIT_ALWAYS_ON is
> not set, bpf_jit_subprogs() restores the instructions, clears
> prog->jit_requested, and bpf_fixup_call_args() falls back to the
> interpreter. The interpreter checks on that path are per frame in
> bpf_patch_call_args():
>
> int bpf_patch_call_args(struct bpf_insn *insn, u32 stack_depth)
> {
> stack_depth = max_t(u32, stack_depth, 1);
> /* Prevent out-of-bounds read to interpreters_args */
> if (stack_depth > MAX_BPF_STACK)
> return -EINVAL;
>
> and bpf_prog_select_interpreter() for the main frame.
>
> A chain of four subprograms each using 512 bytes (2048 combined) would be
> accepted by the verifier and run in the interpreter, whereas before this
> patch the combined chain was capped at 512. Each interpreter frame has its
> own stack array, so this is not an out-of-bounds access. Should the text
> clarify that the interpreter's 512 byte limit applies per function, or
> should the fallback path reject a chain whose combined depth exceeds
> MAX_BPF_STACK?
>
It is indeed slightly confusnig, will reword.
> [ ... ]
>
>
> ---
> AI reviewed your patch. Please fix the bug or email reply why it's not a bug.
> See: https://github.com/kernel-patches/vmtest/blob/master/ci/claude/README.md
>
> CI run summary: https://github.com/kernel-patches/bpf/actions/runs/35909853278
^ permalink raw reply [flat|nested] 34+ messages in thread
* [PATCH bpf-next v1 17/18] bpf, arm64: Allow programs 2 KiB of stack
2026-09-23 19:11 [PATCH bpf-next v1 00/18] Raise BPF program stack size to 2KiB Kumar Kartikeya Dwivedi
` (15 preceding siblings ...)
2026-09-23 19:11 ` [PATCH bpf-next v1 16/18] bpf, x86: Allow programs 2 KiB of stack Kumar Kartikeya Dwivedi
@ 2026-09-23 19:11 ` Kumar Kartikeya Dwivedi
2026-09-23 19:11 ` [PATCH bpf-next v1 18/18] selftests/bpf: Test the 2 KiB stack budget Kumar Kartikeya Dwivedi
17 siblings, 0 replies; 34+ messages in thread
From: Kumar Kartikeya Dwivedi @ 2026-09-23 19:11 UTC (permalink / raw)
To: bpf
Cc: Alexei Starovoitov, Andrii Nakryiko, Daniel Borkmann,
Eduard Zingerman, Emil Tsalapatis, Tejun Heo, kkd, kernel-team
The arm64 JIT sets up a frame with an add/sub immediate, which encodes
up to 4 KiB directly, addresses the BPF stack relative to SP or the
private stack pointer with offsets that fall back to a scratch register
when they do not fit the load/store immediate, and a tail call pops the
current frame and lands in the target's prologue before the target
sets up its own. The exception callback reuses the frame record of the
main program and does not depend on its size. Report
bpf_jit_supports_large_stack() to give JITed programs the same 2 KiB
budget as on x86-64.
Tested on an emulated arm64 guest with programs loaded and run through
the raw bpf() syscall: a 2 KiB frame, a 2 KiB frame with may_goto, a
chain of four 512-byte frames, and a tail call from a program with a
2 KiB frame into another one; a 2056-byte frame and five 512-byte frames
are rejected. The selftest suite was not run on arm64.
Signed-off-by: Kumar Kartikeya Dwivedi <memxor@gmail.com>
---
Documentation/bpf/bpf_design_QA.rst | 10 +++++-----
arch/arm64/net/bpf_jit_comp.c | 5 +++++
2 files changed, 10 insertions(+), 5 deletions(-)
diff --git a/Documentation/bpf/bpf_design_QA.rst b/Documentation/bpf/bpf_design_QA.rst
index be5fc4ac00d6..f4e4a7f4f9fd 100644
--- a/Documentation/bpf/bpf_design_QA.rst
+++ b/Documentation/bpf/bpf_design_QA.rst
@@ -223,11 +223,11 @@ Q: How much stack space a BPF program uses?
-------------------------------------------
A: A program may use up to 2 KiB of stack, combined over its call
chain, when the JIT of the architecture reports support for large
-stacks (currently x86-64); a single function may use all of it, and
-every frame of a program running on a private stack gets the whole
-amount. Elsewhere, and whenever the interpreter is used, the limit is
-512 bytes. The verifier computes the actual amount of stack used and
-both interpreter and most JITed code consume necessary amount.
+stacks (currently x86-64 and arm64); a single function may use all of
+it, and every frame of a program running on a private stack gets the
+whole amount. Elsewhere, and whenever the interpreter is used, the
+limit is 512 bytes. The verifier computes the actual amount of stack
+used and both interpreter and most JITed code consume necessary amount.
Q: Can BPF be offloaded to HW?
------------------------------
diff --git a/arch/arm64/net/bpf_jit_comp.c b/arch/arm64/net/bpf_jit_comp.c
index 6c04fee46876..d92d7754578a 100644
--- a/arch/arm64/net/bpf_jit_comp.c
+++ b/arch/arm64/net/bpf_jit_comp.c
@@ -2485,6 +2485,11 @@ bool bpf_jit_supports_subprog_tailcalls(void)
return true;
}
+bool bpf_jit_supports_large_stack(void)
+{
+ return true;
+}
+
static void invoke_bpf_prog(struct jit_ctx *ctx, struct bpf_tramp_node *node,
int bargs_off, int retval_off, int run_ctx_off,
bool save_ret)
--
2.53.0
^ permalink raw reply related [flat|nested] 34+ messages in thread* [PATCH bpf-next v1 18/18] selftests/bpf: Test the 2 KiB stack budget
2026-09-23 19:11 [PATCH bpf-next v1 00/18] Raise BPF program stack size to 2KiB Kumar Kartikeya Dwivedi
` (16 preceding siblings ...)
2026-09-23 19:11 ` [PATCH bpf-next v1 17/18] bpf, arm64: " Kumar Kartikeya Dwivedi
@ 2026-09-23 19:11 ` Kumar Kartikeya Dwivedi
2026-09-23 20:12 ` bot+bpf-ci
17 siblings, 1 reply; 34+ messages in thread
From: Kumar Kartikeya Dwivedi @ 2026-09-23 19:11 UTC (permalink / raw)
To: bpf
Cc: Alexei Starovoitov, Andrii Nakryiko, Daniel Borkmann,
Eduard Zingerman, Emil Tsalapatis, Tejun Heo, kkd, kernel-team
Add tests for the stack budget of JITs with large stack support: a
single 2 KiB frame, with and without may_goto, four 512-byte frames
that fit the budget and five that do not, a 512-byte frame calling a
1536-byte static subprog, global subprog or bpf_loop() callback and the
same with eight bytes too many, variable offset writes reaching exactly
the budget and past it, a tail call made from a 1 KiB frame, and two
2 KiB frames on a private stack, where each frame gets the whole budget.
A program with a 2 KiB frame is checked to be rejected wherever the
budget is 512 bytes.
Two tests run such programs: a tail call made from a subprog with a
1536-byte frame under a 240-byte caller into a program with a 2 KiB
frame, and a struct_ops program on a private stack calling a subprog
when both frames are 2 KiB.
Signed-off-by: Kumar Kartikeya Dwivedi <memxor@gmail.com>
---
.../bpf/prog_tests/struct_ops_private_stack.c | 31 ++
.../selftests/bpf/prog_tests/tailcalls.c | 42 ++
.../selftests/bpf/prog_tests/verifier.c | 2 +
.../progs/struct_ops_private_stack_large.c | 51 +++
.../bpf/progs/tailcall_large_stack.c | 62 +++
.../bpf/progs/verifier_large_stack.c | 377 ++++++++++++++++++
6 files changed, 565 insertions(+)
create mode 100644 tools/testing/selftests/bpf/progs/struct_ops_private_stack_large.c
create mode 100644 tools/testing/selftests/bpf/progs/tailcall_large_stack.c
create mode 100644 tools/testing/selftests/bpf/progs/verifier_large_stack.c
diff --git a/tools/testing/selftests/bpf/prog_tests/struct_ops_private_stack.c b/tools/testing/selftests/bpf/prog_tests/struct_ops_private_stack.c
index 98db9bafa44b..2b3ec2b79091 100644
--- a/tools/testing/selftests/bpf/prog_tests/struct_ops_private_stack.c
+++ b/tools/testing/selftests/bpf/prog_tests/struct_ops_private_stack.c
@@ -4,6 +4,7 @@
#include "struct_ops_private_stack.skel.h"
#include "struct_ops_private_stack_fail.skel.h"
#include "struct_ops_private_stack_recur.skel.h"
+#include "struct_ops_private_stack_large.skel.h"
#if defined(__x86_64__) || defined(__aarch64__) || defined(__powerpc64__)
static void test_private_stack(void)
@@ -78,6 +79,34 @@ static void test_private_stack_recur(void)
struct_ops_private_stack_recur__destroy(skel);
}
+/* Two frames of 2 KiB each on the private stack */
+static void test_private_stack_large(void)
+{
+ struct struct_ops_private_stack_large *skel;
+ struct bpf_link *link;
+
+ if (!is_large_stack_supported()) {
+ test__skip();
+ return;
+ }
+
+ skel = struct_ops_private_stack_large__open_and_load();
+ if (!ASSERT_OK_PTR(skel, "struct_ops_private_stack_large__open_and_load"))
+ return;
+
+ link = bpf_map__attach_struct_ops(skel->maps.testmod_1);
+ if (!ASSERT_OK_PTR(link, "attach_struct_ops"))
+ goto cleanup;
+
+ ASSERT_OK(trigger_module_test_read(256), "trigger_read");
+
+ ASSERT_EQ(skel->bss->val, 100 + 30 + 12, "val");
+
+ bpf_link__destroy(link);
+cleanup:
+ struct_ops_private_stack_large__destroy(skel);
+}
+
static void __test_struct_ops_private_stack(void)
{
if (test__start_subtest("private_stack"))
@@ -86,6 +115,8 @@ static void __test_struct_ops_private_stack(void)
test_private_stack_fail();
if (test__start_subtest("private_stack_recur"))
test_private_stack_recur();
+ if (test__start_subtest("private_stack_large"))
+ test_private_stack_large();
}
#else
static void __test_struct_ops_private_stack(void)
diff --git a/tools/testing/selftests/bpf/prog_tests/tailcalls.c b/tools/testing/selftests/bpf/prog_tests/tailcalls.c
index c5c9d6c359bb..c5e3f198ca7f 100644
--- a/tools/testing/selftests/bpf/prog_tests/tailcalls.c
+++ b/tools/testing/selftests/bpf/prog_tests/tailcalls.c
@@ -9,6 +9,7 @@
#include "tc_bpf2bpf.skel.h"
#include "tailcall_fail.skel.h"
#include "tailcall_cgrp_storage_owner.skel.h"
+#include "tailcall_large_stack.skel.h"
#include "tailcall_cgrp_storage_no_storage.skel.h"
#include "tailcall_cgrp_storage.skel.h"
#include "tailcall_sleepable.skel.h"
@@ -1953,6 +1954,45 @@ static void test_tailcall_bpf2bpf_fexit_links(void)
tailcall_bpf2bpf2__destroy(skel_tc);
}
+/*
+ * test_tailcall_large_stack runs a tail call made from a subprog with a 1536
+ * byte frame, under a 240-byte caller, into a program with a 2 KiB frame:
+ *
+ * entry (240) --call-> subprog_tail (1536) --tailcall-> classifier_0 (2048)
+ */
+static void test_tailcall_large_stack(void)
+{
+ struct tailcall_large_stack *skel;
+ int err, prog_fd, map_fd, key = 0;
+ char buff[128] = {};
+ LIBBPF_OPTS(bpf_test_run_opts, topts,
+ .data_in = buff,
+ .data_size_in = sizeof(buff),
+ .repeat = 1,
+ );
+
+ if (!is_large_stack_supported()) {
+ test__skip();
+ return;
+ }
+
+ skel = tailcall_large_stack__open_and_load();
+ if (!ASSERT_OK_PTR(skel, "tailcall_large_stack__open_and_load"))
+ return;
+
+ prog_fd = bpf_program__fd(skel->progs.classifier_0);
+ map_fd = bpf_map__fd(skel->maps.jmp_table);
+ err = bpf_map_update_elem(map_fd, &key, &prog_fd, BPF_ANY);
+ if (!ASSERT_OK(err, "update jmp_table"))
+ goto out;
+
+ err = bpf_prog_test_run_opts(bpf_program__fd(skel->progs.entry), &topts);
+ ASSERT_OK(err, "test_run");
+ ASSERT_EQ(topts.retval, 42 + 7, "retval");
+out:
+ tailcall_large_stack__destroy(skel);
+}
+
void test_tailcalls(void)
{
if (test__start_subtest("tailcall_1"))
@@ -2022,4 +2062,6 @@ void test_tailcalls(void)
test_tailcall_callback();
if (test__start_subtest("tailcall_bpf2bpf_fexit_links"))
test_tailcall_bpf2bpf_fexit_links();
+ if (test__start_subtest("tailcall_large_stack"))
+ test_tailcall_large_stack();
}
diff --git a/tools/testing/selftests/bpf/prog_tests/verifier.c b/tools/testing/selftests/bpf/prog_tests/verifier.c
index 4f1e1c1cd5ab..ced8a2f1c89c 100644
--- a/tools/testing/selftests/bpf/prog_tests/verifier.c
+++ b/tools/testing/selftests/bpf/prog_tests/verifier.c
@@ -59,6 +59,7 @@
#include "verifier_kfunc_uninit.skel.h"
#include "verifier_kfunc_uninit_multi.skel.h"
#include "verifier_ld_ind.skel.h"
+#include "verifier_large_stack.skel.h"
#include "verifier_ldsx.skel.h"
#include "verifier_leak_ptr.skel.h"
#include "verifier_linked_scalars.skel.h"
@@ -229,6 +230,7 @@ void test_verifier_kfunc_uninit(void) { RUN_TESTS(verifier_kfunc_uninit)
void test_verifier_kfunc_uninit_multi(void) { RUN_TESTS(verifier_kfunc_uninit_multi); }
void test_verifier_load_acquire(void) { RUN(verifier_load_acquire); }
void test_verifier_ld_ind(void) { RUN(verifier_ld_ind); }
+void test_verifier_large_stack(void) { RUN(verifier_large_stack); }
void test_verifier_ldsx(void) { RUN(verifier_ldsx); }
void test_verifier_leak_ptr(void) { RUN(verifier_leak_ptr); }
void test_verifier_linked_scalars(void) { RUN(verifier_linked_scalars); }
diff --git a/tools/testing/selftests/bpf/progs/struct_ops_private_stack_large.c b/tools/testing/selftests/bpf/progs/struct_ops_private_stack_large.c
new file mode 100644
index 000000000000..94a25a2cff6e
--- /dev/null
+++ b/tools/testing/selftests/bpf/progs/struct_ops_private_stack_large.c
@@ -0,0 +1,51 @@
+// SPDX-License-Identifier: GPL-2.0
+
+#include <vmlinux.h>
+#include <bpf/bpf_helpers.h>
+#include <bpf/bpf_tracing.h>
+#include "../test_kmods/bpf_testmod.h"
+#include "bpf_misc.h"
+
+char _license[] SEC("license") = "GPL";
+
+long val;
+
+/* On a private stack every frame gets the whole 2 KiB budget. */
+__used __naked
+static long frame_2048_leaf(void)
+{
+ asm volatile (" \
+ r1 = 30; \
+ *(u64 *)(r10 - 2048) = r1; \
+ r1 = 12; \
+ *(u64 *)(r10 - 8) = r1; \
+ r0 = *(u64 *)(r10 - 2048); \
+ r1 = *(u64 *)(r10 - 8); \
+ r0 += r1; \
+ exit; \
+" ::: __clobber_all);
+}
+
+/* test_1 is the member bpf_testmod requests a private stack for */
+SEC("struct_ops")
+__naked int test_1(void)
+{
+ asm volatile (" \
+ r1 = 100; \
+ *(u64 *)(r10 - 2048) = r1; \
+ call frame_2048_leaf; \
+ r1 = *(u64 *)(r10 - 2048); \
+ r0 += r1; \
+ r1 = %[val] ll; \
+ *(u64 *)(r1 + 0) = r0; \
+ r0 = 0; \
+ exit; \
+" :
+ : __imm_addr(val)
+ : __clobber_all);
+}
+
+SEC(".struct_ops")
+struct bpf_testmod_ops3 testmod_1 = {
+ .test_1 = (void *)test_1,
+};
diff --git a/tools/testing/selftests/bpf/progs/tailcall_large_stack.c b/tools/testing/selftests/bpf/progs/tailcall_large_stack.c
new file mode 100644
index 000000000000..977197dac5d3
--- /dev/null
+++ b/tools/testing/selftests/bpf/progs/tailcall_large_stack.c
@@ -0,0 +1,62 @@
+// SPDX-License-Identifier: GPL-2.0
+#include <linux/bpf.h>
+#include <bpf/bpf_helpers.h>
+#include "bpf_misc.h"
+
+struct {
+ __uint(type, BPF_MAP_TYPE_PROG_ARRAY);
+ __uint(max_entries, 1);
+ __uint(key_size, sizeof(__u32));
+ __uint(value_size, sizeof(__u32));
+} jmp_table SEC(".maps");
+
+/* The tail call target sets up a 2 KiB frame of its own and uses both of its ends. */
+SEC("tc")
+__naked int classifier_0(void)
+{
+ asm volatile (" \
+ r1 = 42; \
+ *(u64 *)(r10 - 2048) = r1; \
+ r1 = 7; \
+ *(u64 *)(r10 - 8) = r1; \
+ r0 = *(u64 *)(r10 - 2048); \
+ r1 = *(u64 *)(r10 - 8); \
+ r0 += r1; \
+ exit; \
+" ::: __clobber_all);
+}
+
+/*
+ * The frame of the subprog doing the tail call is unwound by it, so it may be
+ * large; only the frames of its callers stay behind and are limited to 256
+ * bytes in total. Returns 1 when the tail call falls through.
+ */
+__used __naked
+static int subprog_tail(void)
+{
+ asm volatile (" \
+ r2 = 1; \
+ *(u64 *)(r10 - 1536) = r2; \
+ r2 = %[jmp_table] ll; \
+ r3 = 0; \
+ call %[bpf_tail_call]; \
+ r0 = 1; \
+ exit; \
+" :
+ : __imm(bpf_tail_call),
+ __imm_addr(jmp_table)
+ : __clobber_all);
+}
+
+SEC("tc")
+__naked int entry(void)
+{
+ asm volatile (" \
+ r2 = 2; \
+ *(u64 *)(r10 - 240) = r2; \
+ call subprog_tail; \
+ exit; \
+" ::: __clobber_all);
+}
+
+char _license[] SEC("license") = "GPL";
diff --git a/tools/testing/selftests/bpf/progs/verifier_large_stack.c b/tools/testing/selftests/bpf/progs/verifier_large_stack.c
new file mode 100644
index 000000000000..2d4c81a3f0cb
--- /dev/null
+++ b/tools/testing/selftests/bpf/progs/verifier_large_stack.c
@@ -0,0 +1,377 @@
+// SPDX-License-Identifier: GPL-2.0
+
+#include <linux/bpf.h>
+#include <bpf/bpf_helpers.h>
+#include "bpf_misc.h"
+
+/*
+ * Programs may use MAX_BPF_STACK_JIT (2 KiB) of stack on JITs that support
+ * large stacks, combined over a call chain, with no separate limit on a
+ * single frame. Interpreted programs and other JITs keep 512 bytes.
+ */
+
+SEC("socket")
+__description("single frame of 2048 bytes")
+__load_if_large_stack()
+__success __success_unpriv __retval(42)
+__naked void single_frame_2048(void)
+{
+ asm volatile (" \
+ r1 = r10; \
+ r1 += -2048; \
+ r0 = 42; \
+ *(u64*)(r1 + 0) = r0; \
+ r0 = *(u64*)(r1 + 0); \
+ exit; \
+" ::: __clobber_all);
+}
+
+SEC("socket")
+__description("single frame of 2048 bytes without large stack support")
+__load_if_no_large_stack()
+__failure __msg("invalid write to stack R1 off=-2048 size=8")
+__naked void single_frame_2048_no_large_stack(void)
+{
+ asm volatile (" \
+ r1 = r10; \
+ r1 += -2048; \
+ r0 = 42; \
+ *(u64*)(r1 + 0) = r0; \
+ exit; \
+" ::: __clobber_all);
+}
+
+__used __naked
+static void frame_512_leaf(void)
+{
+ asm volatile (" \
+ r1 = 1; \
+ *(u64 *)(r10 - 512) = r1; \
+ exit; \
+" ::: __clobber_all);
+}
+
+__used __naked
+static void frame_512_depth_2(void)
+{
+ asm volatile (" \
+ r1 = 2; \
+ *(u64 *)(r10 - 512) = r1; \
+ call frame_512_leaf; \
+ exit; \
+" ::: __clobber_all);
+}
+
+__used __naked
+static void frame_512_depth_3(void)
+{
+ asm volatile (" \
+ r1 = 3; \
+ *(u64 *)(r10 - 512) = r1; \
+ call frame_512_depth_2; \
+ exit; \
+" ::: __clobber_all);
+}
+
+__used __naked
+static void frame_512_depth_4(void)
+{
+ asm volatile (" \
+ r1 = 4; \
+ *(u64 *)(r10 - 512) = r1; \
+ call frame_512_depth_3; \
+ exit; \
+" ::: __clobber_all);
+}
+
+SEC("socket")
+__description("four frames of 512 bytes fit the 2 KiB budget")
+__load_if_large_stack()
+__success __log_level(4) __msg("stack depth max 2048")
+__naked void four_frames_of_512(void)
+{
+ asm volatile (" \
+ call frame_512_depth_4; \
+ r0 = 0; \
+ exit; \
+" ::: __clobber_all);
+}
+
+SEC("socket")
+__description("five frames of 512 bytes exceed the 2 KiB budget")
+__load_if_large_stack()
+__failure __msg("combined stack size of 5 calls is 2560. Too large")
+__naked void five_frames_of_512(void)
+{
+ asm volatile (" \
+ r1 = 5; \
+ *(u64 *)(r10 - 512) = r1; \
+ call frame_512_depth_4; \
+ r0 = 0; \
+ exit; \
+" ::: __clobber_all);
+}
+
+__used __naked
+static void frame_1536_leaf(void)
+{
+ asm volatile (" \
+ r1 = 1; \
+ *(u64 *)(r10 - 1536) = r1; \
+ exit; \
+" ::: __clobber_all);
+}
+
+SEC("socket")
+__description("512-byte frame calling a 1536-byte frame")
+__load_if_large_stack()
+__success __log_level(4) __msg("stack depth max 2048")
+__naked void uneven_frames_fit(void)
+{
+ asm volatile (" \
+ r1 = 2; \
+ *(u64 *)(r10 - 512) = r1; \
+ call frame_1536_leaf; \
+ r0 = 0; \
+ exit; \
+" ::: __clobber_all);
+}
+
+SEC("socket")
+__description("520-byte frame calling a 1536-byte frame")
+__load_if_large_stack()
+__failure __msg("combined stack size of 2 calls is 2064. Too large")
+__naked void uneven_frames_exceed(void)
+{
+ asm volatile (" \
+ r1 = 2; \
+ *(u64 *)(r10 - 520) = r1; \
+ call frame_1536_leaf; \
+ r0 = 0; \
+ exit; \
+" ::: __clobber_all);
+}
+
+#ifdef __BPF_FEATURE_MAY_GOTO
+/* may_goto adds its counter below the frame; a JIT does not hold that against the budget */
+SEC("socket")
+__description("frame of 2048 bytes with may_goto")
+__load_if_large_stack()
+__success __retval(42)
+__naked void frame_2048_with_may_goto(void)
+{
+ asm volatile (" \
+ r1 = r10; \
+ r1 += -2048; \
+ r0 = 42; \
+ *(u32*)(r1 + 0) = r0; \
+ may_goto l0_%=; \
+ r2 = 100; \
+ l0_%=: \
+ exit; \
+" ::: __clobber_all);
+}
+#endif
+
+SEC("socket")
+__description("variable offset write reaching 2048 bytes deep")
+__load_if_large_stack()
+__success
+__naked void var_off_write_to_2048(void)
+{
+ asm volatile (" \
+ call %[bpf_get_prandom_u32]; \
+ r0 &= 8; \
+ r2 = r10; \
+ r2 += -2048; \
+ r2 += r0; \
+ r1 = 0; \
+ *(u64*)(r2 + 0) = r1; \
+ r0 = 0; \
+ exit; \
+" :
+ : __imm(bpf_get_prandom_u32)
+ : __clobber_all);
+}
+
+SEC("socket")
+__description("variable offset write reaching 2056 bytes deep")
+__load_if_large_stack()
+__failure __msg("invalid variable-offset write to stack R2")
+__naked void var_off_write_to_2056(void)
+{
+ asm volatile (" \
+ call %[bpf_get_prandom_u32]; \
+ r0 &= 8; \
+ r2 = r10; \
+ r2 += -2056; \
+ r2 += r0; \
+ r1 = 0; \
+ *(u64*)(r2 + 0) = r1; \
+ r0 = 0; \
+ exit; \
+" :
+ : __imm(bpf_get_prandom_u32)
+ : __clobber_all);
+}
+
+/* Each frame of a private stack gets the whole budget. */
+__used __naked
+static void priv_stack_frame_2048(void)
+{
+ asm volatile (" \
+ r1 = 1; \
+ *(u64 *)(r10 - 2048) = r1; \
+ exit; \
+" ::: __clobber_all);
+}
+
+SEC("kprobe")
+__description("private stack: two frames of 2048 bytes")
+__load_if_large_stack()
+__arch_x86_64
+__arch_arm64
+__success __log_level(4)
+__msg("stack depth max 2048")
+__msg("subprog 0 (private_stack_two_frames) main {{.*}} stack 2048")
+__msg("subprog 1 (priv_stack_frame_2048) static {{.*}} stack 2048")
+__naked void private_stack_two_frames(void)
+{
+ asm volatile (" \
+ r1 = 2; \
+ *(u64 *)(r10 - 2048) = r1; \
+ call priv_stack_frame_2048; \
+ r0 = 0; \
+ exit; \
+" ::: __clobber_all);
+}
+
+struct {
+ __uint(type, BPF_MAP_TYPE_PROG_ARRAY);
+ __uint(max_entries, 1);
+ __uint(key_size, sizeof(__u32));
+ __uint(value_size, sizeof(__u32));
+} jmp_table SEC(".maps");
+
+/*
+ * A tail call unwinds the frame of the program doing it, so a large main
+ * frame is fine; the 256-byte rule only concerns the frames of callers of a
+ * subprog that tail calls.
+ */
+SEC("tc")
+__description("tail call from a 1 KiB frame")
+__load_if_large_stack()
+__success
+__naked void tail_call_from_large_frame(void)
+{
+ asm volatile (" \
+ r2 = 42; \
+ *(u64 *)(r10 - 1024) = r2; \
+ r2 = %[jmp_table] ll; \
+ r3 = 0; \
+ call %[bpf_tail_call]; \
+ r0 = 0; \
+ exit; \
+" :
+ : __imm(bpf_tail_call),
+ __imm_addr(jmp_table)
+ : __clobber_all);
+}
+
+/* Global subprogs are verified on their own but share the call chain budget. */
+__used __naked int global_frame_1536(void)
+{
+ asm volatile (" \
+ r1 = 1; \
+ *(u64 *)(r10 - 1536) = r1; \
+ r0 = 0; \
+ exit; \
+" ::: __clobber_all);
+}
+
+SEC("socket")
+__description("512-byte frame calling a 1536-byte global subprog")
+__load_if_large_stack()
+__success __log_level(4) __msg("stack depth max 2048")
+__naked void global_subprog_fits(void)
+{
+ asm volatile (" \
+ r1 = 2; \
+ *(u64 *)(r10 - 512) = r1; \
+ call global_frame_1536; \
+ r0 = 0; \
+ exit; \
+" ::: __clobber_all);
+}
+
+SEC("socket")
+__description("520-byte frame calling a 1536-byte global subprog")
+__load_if_large_stack()
+__failure __msg("combined stack size of 2 calls is 2064. Too large")
+__naked void global_subprog_exceeds(void)
+{
+ asm volatile (" \
+ r1 = 2; \
+ *(u64 *)(r10 - 520) = r1; \
+ call global_frame_1536; \
+ r0 = 0; \
+ exit; \
+" ::: __clobber_all);
+}
+
+/* Callback frames are part of the chain of the helper that calls them. */
+static __naked int loop_cb_1536(void)
+{
+ asm volatile (" \
+ r1 = 1; \
+ *(u64 *)(r10 - 1536) = r1; \
+ r0 = 0; \
+ exit; \
+" ::: __clobber_all);
+}
+
+SEC("socket")
+__description("512-byte frame with a 1536-byte bpf_loop callback")
+__load_if_large_stack()
+__success __log_level(4) __msg("stack depth max 2048")
+__naked void loop_callback_fits(void)
+{
+ asm volatile (" \
+ r1 = 2; \
+ *(u64 *)(r10 - 512) = r1; \
+ r1 = 1; \
+ r2 = %[loop_cb_1536]; \
+ r3 = 0; \
+ r4 = 0; \
+ call %[bpf_loop]; \
+ r0 = 0; \
+ exit; \
+" :
+ : __imm_ptr(loop_cb_1536),
+ __imm(bpf_loop)
+ : __clobber_common);
+}
+
+SEC("socket")
+__description("520-byte frame with a 1536-byte bpf_loop callback")
+__load_if_large_stack()
+__failure __msg("combined stack size of 2 calls is 2064. Too large")
+__naked void loop_callback_exceeds(void)
+{
+ asm volatile (" \
+ r1 = 2; \
+ *(u64 *)(r10 - 520) = r1; \
+ r1 = 1; \
+ r2 = %[loop_cb_1536]; \
+ r3 = 0; \
+ r4 = 0; \
+ call %[bpf_loop]; \
+ r0 = 0; \
+ exit; \
+" :
+ : __imm_ptr(loop_cb_1536),
+ __imm(bpf_loop)
+ : __clobber_common);
+}
+
+char _license[] SEC("license") = "GPL";
--
2.53.0
^ permalink raw reply related [flat|nested] 34+ messages in thread* Re: [PATCH bpf-next v1 18/18] selftests/bpf: Test the 2 KiB stack budget
2026-09-23 19:11 ` [PATCH bpf-next v1 18/18] selftests/bpf: Test the 2 KiB stack budget Kumar Kartikeya Dwivedi
@ 2026-09-23 20:12 ` bot+bpf-ci
2026-09-23 20:28 ` Kumar Kartikeya Dwivedi
0 siblings, 1 reply; 34+ messages in thread
From: bot+bpf-ci @ 2026-09-23 20:12 UTC (permalink / raw)
To: memxor, bpf
Cc: ast, andrii, daniel, eddyz87, emil, tj, kkd, kernel-team, ast,
andrii, daniel, martin.lau, eddyz87, yonghong.song, mason,
ihor.solodrai
[-- Attachment #1: Type: text/plain, Size: 5595 bytes --]
> diff --git a/tools/testing/selftests/bpf/prog_tests/struct_ops_private_stack.c b/tools/testing/selftests/bpf/prog_tests/struct_ops_private_stack.c
> index 98db9bafa44b..2b3ec2b79091 100644
> --- a/tools/testing/selftests/bpf/prog_tests/struct_ops_private_stack.c
> +++ b/tools/testing/selftests/bpf/prog_tests/struct_ops_private_stack.c
[ ... ]
> @@ -78,6 +79,34 @@ static void test_private_stack_recur(void)
> struct_ops_private_stack_recur__destroy(skel);
> }
>
> +/* Two frames of 2 KiB each on the private stack */
> +static void test_private_stack_large(void)
> +{
> + struct struct_ops_private_stack_large *skel;
> + struct bpf_link *link;
> +
> + if (!is_large_stack_supported()) {
> + test__skip();
> + return;
> + }
A subsystem pattern flags this as potentially concerning: The BPF
selftests guide recommends always including a reason string in the skip
message so future debuggers know why the test was skipped. When the probe
gets a clean "invalid write to stack" rejection, is_large_stack_supported()
prints nothing, and the subtest shows only SKIP with no indication that
the reason is missing large stack support.
The test_loader's run_subtest() prints a reason
(printf("%s:SKIP: %s\n", __func__, spec->skip_reason)) before calling
test__skip(), but here the skip is silent. On the other hand, a bare
test__skip() with no printf is the most common idiom in prog_tests
(about 90 of 149 call sites follow this pattern, and the #else branch
of this same file does the same). Could this include a skip reason,
or is silent skipping acceptable BPF selftest convention?
[ ... ]
> diff --git a/tools/testing/selftests/bpf/prog_tests/tailcalls.c b/tools/testing/selftests/bpf/prog_tests/tailcalls.c
> index c5c9d6c359bb..c5e3f198ca7f 100644
> --- a/tools/testing/selftests/bpf/prog_tests/tailcalls.c
> +++ b/tools/testing/selftests/bpf/prog_tests/tailcalls.c
[ ... ]
> @@ -1953,6 +1954,45 @@ static void test_tailcall_bpf2bpf_fexit_links(void)
> tailcall_bpf2bpf2__destroy(skel_tc);
> }
>
> +/*
> + * test_tailcall_large_stack runs a tail call made from a subprog with a 1536
> + * byte frame, under a 240-byte caller, into a program with a 2 KiB frame:
> + *
> + * entry (240) --call-> subprog_tail (1536) --tailcall-> classifier_0 (2048)
> + */
> +static void test_tailcall_large_stack(void)
> +{
> + struct tailcall_large_stack *skel;
> + int err, prog_fd, map_fd, key = 0;
> + char buff[128] = {};
> + LIBBPF_OPTS(bpf_test_run_opts, topts,
> + .data_in = buff,
> + .data_size_in = sizeof(buff),
> + .repeat = 1,
> + );
> +
> + if (!is_large_stack_supported()) {
> + test__skip();
> + return;
> + }
The same pattern appears here: when is_large_stack_supported() returns
false, test__skip() is called with no reason string. Could this also
benefit from a skip message?
[ ... ]
> diff --git a/tools/testing/selftests/bpf/progs/verifier_large_stack.c b/tools/testing/selftests/bpf/progs/verifier_large_stack.c
> new file mode 100644
> index 000000000000..2d4c81a3f0cb
> --- /dev/null
> +++ b/tools/testing/selftests/bpf/progs/verifier_large_stack.c
> @@ -0,0 +1,377 @@
> +// SPDX-License-Identifier: GPL-2.0
> +
> +#include <linux/bpf.h>
> +#include <bpf/bpf_helpers.h>
> +#include "bpf_misc.h"
> +
> +/*
> + * Programs may use MAX_BPF_STACK_JIT (2 KiB) of stack on JITs that support
> + * large stacks, combined over a call chain, with no separate limit on a
> + * single frame. Interpreted programs and other JITs keep 512 bytes.
> + */
A subsystem pattern flags this as potentially concerning: Several tests
in this new file appear to be 2 KiB versions of tests that already
exist elsewhere in the same directory.
For instance, frame_2048_with_may_goto is the same program as
stack_check_size_512_with_may_goto_jit in verifier_stack_ptr.c
(r1 = r10; r1 += -512; r0 = 42; *(u32*)(r1 + 0) = r0; may_goto ...; exit),
with only -512 changed to -2048 and __load_if_JITed() changed to
__load_if_large_stack().
Similarly, single_frame_2048 / single_frame_2048_no_large_stack check
the single-frame boundary that verifier_stack_ptr.c already covers with
stack_check_size_gt_512 and the new stack_check_size_gt_2048 (fp-2056).
The var_off_write_to_2048/2056 tests check variable-offset stack bounds,
which verifier_var_off.c already covers (it gained
access_min_out_of_bound_large in the earlier patch in this series).
The private_stack_two_frames test follows verifier_private_stack.c's
"Private stack, subtree > MAX_BPF_STACK" test, including the same
__log_level(4) "stack depth max" / "subprog N (...) stack" messages.
The BPF selftests guide asks whether variants like these belong as new
cases in the existing files instead of a new file, especially since
the earlier patch in this series (2d0faaadf6a8) already places its
2 KiB versions next to the originals.
On the other hand, this new file does group tests for the 2 KiB stack
feature, and several behaviors (four 512-byte frames that fit, a tail
call from a 1 KiB frame, 1536-byte global subprogs and callbacks) cannot
occur under the 512-byte budget, so they have no existing counterpart.
Adding one verifier_<feature>.c file per feature is also common in BPF
selftests (verifier_private_stack.c is an example). Could you clarify
whether the overlap cases should be integrated into existing files, or
is a new file the intended organization here?
---
AI reviewed your patch. Please fix the bug or email reply why it's not a bug.
See: https://github.com/kernel-patches/vmtest/blob/master/ci/claude/README.md
CI run summary: https://github.com/kernel-patches/bpf/actions/runs/35909853278
^ permalink raw reply [flat|nested] 34+ messages in thread* Re: [PATCH bpf-next v1 18/18] selftests/bpf: Test the 2 KiB stack budget
2026-09-23 20:12 ` bot+bpf-ci
@ 2026-09-23 20:28 ` Kumar Kartikeya Dwivedi
0 siblings, 0 replies; 34+ messages in thread
From: Kumar Kartikeya Dwivedi @ 2026-09-23 20:28 UTC (permalink / raw)
To: bot+bpf-ci, bpf
Cc: ast, andrii, daniel, eddyz87, emil, tj, kkd, kernel-team,
martin.lau, yonghong.song, mason, ihor.solodrai
On Wed Sep 23, 2026 at 10:12 PM CEST, bot+bpf-ci wrote:
>> diff --git a/tools/testing/selftests/bpf/prog_tests/struct_ops_private_stack.c b/tools/testing/selftests/bpf/prog_tests/struct_ops_private_stack.c
>> index 98db9bafa44b..2b3ec2b79091 100644
>> --- a/tools/testing/selftests/bpf/prog_tests/struct_ops_private_stack.c
>> +++ b/tools/testing/selftests/bpf/prog_tests/struct_ops_private_stack.c
>
> [ ... ]
>
>> @@ -78,6 +79,34 @@ static void test_private_stack_recur(void)
>> struct_ops_private_stack_recur__destroy(skel);
>> }
>>
>> +/* Two frames of 2 KiB each on the private stack */
>> +static void test_private_stack_large(void)
>> +{
>> + struct struct_ops_private_stack_large *skel;
>> + struct bpf_link *link;
>> +
>> + if (!is_large_stack_supported()) {
>> + test__skip();
>> + return;
>> + }
>
> A subsystem pattern flags this as potentially concerning: The BPF
> selftests guide recommends always including a reason string in the skip
> message so future debuggers know why the test was skipped. When the probe
> gets a clean "invalid write to stack" rejection, is_large_stack_supported()
> prints nothing, and the subtest shows only SKIP with no indication that
> the reason is missing large stack support.
>
> The test_loader's run_subtest() prints a reason
> (printf("%s:SKIP: %s\n", __func__, spec->skip_reason)) before calling
> test__skip(), but here the skip is silent. On the other hand, a bare
> test__skip() with no printf is the most common idiom in prog_tests
> (about 90 of 149 call sites follow this pattern, and the #else branch
> of this same file does the same). Could this include a skip reason,
> or is silent skipping acceptable BPF selftest convention?
>
Won't fix.
> [ ... ]
>
>> diff --git a/tools/testing/selftests/bpf/prog_tests/tailcalls.c b/tools/testing/selftests/bpf/prog_tests/tailcalls.c
>> index c5c9d6c359bb..c5e3f198ca7f 100644
>> --- a/tools/testing/selftests/bpf/prog_tests/tailcalls.c
>> +++ b/tools/testing/selftests/bpf/prog_tests/tailcalls.c
>
> [ ... ]
>
>> @@ -1953,6 +1954,45 @@ static void test_tailcall_bpf2bpf_fexit_links(void)
>> tailcall_bpf2bpf2__destroy(skel_tc);
>> }
>>
>> +/*
>> + * test_tailcall_large_stack runs a tail call made from a subprog with a 1536
>> + * byte frame, under a 240-byte caller, into a program with a 2 KiB frame:
>> + *
>> + * entry (240) --call-> subprog_tail (1536) --tailcall-> classifier_0 (2048)
>> + */
>> +static void test_tailcall_large_stack(void)
>> +{
>> + struct tailcall_large_stack *skel;
>> + int err, prog_fd, map_fd, key = 0;
>> + char buff[128] = {};
>> + LIBBPF_OPTS(bpf_test_run_opts, topts,
>> + .data_in = buff,
>> + .data_size_in = sizeof(buff),
>> + .repeat = 1,
>> + );
>> +
>> + if (!is_large_stack_supported()) {
>> + test__skip();
>> + return;
>> + }
>
> The same pattern appears here: when is_large_stack_supported() returns
> false, test__skip() is called with no reason string. Could this also
> benefit from a skip message?
>
> [ ... ]
>
>> diff --git a/tools/testing/selftests/bpf/progs/verifier_large_stack.c b/tools/testing/selftests/bpf/progs/verifier_large_stack.c
>> new file mode 100644
>> index 000000000000..2d4c81a3f0cb
>> --- /dev/null
>> +++ b/tools/testing/selftests/bpf/progs/verifier_large_stack.c
>> @@ -0,0 +1,377 @@
>> +// SPDX-License-Identifier: GPL-2.0
>> +
>> +#include <linux/bpf.h>
>> +#include <bpf/bpf_helpers.h>
>> +#include "bpf_misc.h"
>> +
>> +/*
>> + * Programs may use MAX_BPF_STACK_JIT (2 KiB) of stack on JITs that support
>> + * large stacks, combined over a call chain, with no separate limit on a
>> + * single frame. Interpreted programs and other JITs keep 512 bytes.
>> + */
>
> A subsystem pattern flags this as potentially concerning: Several tests
> in this new file appear to be 2 KiB versions of tests that already
> exist elsewhere in the same directory.
>
> For instance, frame_2048_with_may_goto is the same program as
> stack_check_size_512_with_may_goto_jit in verifier_stack_ptr.c
> (r1 = r10; r1 += -512; r0 = 42; *(u32*)(r1 + 0) = r0; may_goto ...; exit),
> with only -512 changed to -2048 and __load_if_JITed() changed to
> __load_if_large_stack().
>
> Similarly, single_frame_2048 / single_frame_2048_no_large_stack check
> the single-frame boundary that verifier_stack_ptr.c already covers with
> stack_check_size_gt_512 and the new stack_check_size_gt_2048 (fp-2056).
>
> The var_off_write_to_2048/2056 tests check variable-offset stack bounds,
> which verifier_var_off.c already covers (it gained
> access_min_out_of_bound_large in the earlier patch in this series).
>
> The private_stack_two_frames test follows verifier_private_stack.c's
> "Private stack, subtree > MAX_BPF_STACK" test, including the same
> __log_level(4) "stack depth max" / "subprog N (...) stack" messages.
>
> The BPF selftests guide asks whether variants like these belong as new
> cases in the existing files instead of a new file, especially since
> the earlier patch in this series (2d0faaadf6a8) already places its
> 2 KiB versions next to the originals.
>
> On the other hand, this new file does group tests for the 2 KiB stack
> feature, and several behaviors (four 512-byte frames that fit, a tail
> call from a 1 KiB frame, 1536-byte global subprogs and callbacks) cannot
> occur under the 512-byte budget, so they have no existing counterpart.
> Adding one verifier_<feature>.c file per feature is also common in BPF
> selftests (verifier_private_stack.c is an example). Could you clarify
> whether the overlap cases should be integrated into existing files, or
> is a new file the intended organization here?
>
>
I think it is fine as is.
> ---
> AI reviewed your patch. Please fix the bug or email reply why it's not a bug.
> See: https://github.com/kernel-patches/vmtest/blob/master/ci/claude/README.md
>
> CI run summary: https://github.com/kernel-patches/bpf/actions/runs/35909853278
^ permalink raw reply [flat|nested] 34+ messages in thread