From: Jiayuan Chen <jiayuan.chen@linux.dev>
To: Emil Tsalapatis <emil@etsalapatis.com>, bpf@vger.kernel.org
Cc: ast@kernel.org, andrii@kernel.org, eddyz87@gmail.com,
memxor@gmail.com, daniel@iogearbox.net
Subject: Re: [PATCH bpf v2 1/2] bpf: Add sleepable arena page allocation path
Date: Thu, 24 Sep 2026 18:11:40 +0800 [thread overview]
Message-ID: <7fba6431-14ab-45de-99dd-17e38cae2841@linux.dev> (raw)
In-Reply-To: <20260924053621.7076-2-emil@etsalapatis.com>
On 9/24/26 1:36 PM, Emil Tsalapatis wrote:
> The bpf_arena_alloc_pages() function currently only allocates pages
> inside a spinlock critical section with IRQs off. This forces the use
> of alloc_pages_nolock() in the BPF allocator, even when the caller is
> a sleepable BPF function. This in turn causes allocation failures even
> in cases where falling into the allocator slow path and possibly
> sleeping would eventually succeed. This can be triggered consistently
> by heavy BPF arena users like scx.
>
> Add a separate arena page allocation path just for sleepable callers.
> The path preallocates the arena memory to be added to the tree before
> taking the critical section.
>
> Signed-off-by: Emil Tsalapatis <emil@etsalapatis.com>
> ---
> include/linux/bpf.h | 6 +++
> kernel/bpf/arena.c | 123 ++++++++++++++++++++++++++++++-------------
> kernel/bpf/syscall.c | 8 +--
> 3 files changed, 92 insertions(+), 45 deletions(-)
>
> diff --git a/include/linux/bpf.h b/include/linux/bpf.h
> index e7c5e203eddd..b37e34cf1086 100644
> --- a/include/linux/bpf.h
> +++ b/include/linux/bpf.h
> @@ -720,6 +720,12 @@ void bpf_map_free_internal_structs(struct bpf_map *map, void *obj);
> int bpf_dynptr_from_file_sleepable(struct file *file, u32 flags,
> struct bpf_dynptr *ptr__uninit);
>
> +static inline bool is_bpf_alloc_nonsleepable(void)
> +{
> + return preempt_count() > 0 || irqs_disabled() ||
> + IS_ENABLED(CONFIG_PREEMPT_RT);
> +}
> +
A runtime check cannot tell whether sleeping is allowed: without
CONFIG_PREEMPT_COUNT preempt_count() does not see spinlocks or
rcu_read_lock(), which is why preemptible() is 0 there.
How about passing 'sleepable' down to bpf_map_alloc_pages() instead?
The verifier already proves it, and the code gets simpler.
> #if defined(CONFIG_MMU) && defined(CONFIG_64BIT)
> void *bpf_arena_alloc_pages_non_sleepable(void *p__map, void *addr__ign, u32 page_cnt, int node_id,
> u64 flags);
> diff --git a/kernel/bpf/arena.c b/kernel/bpf/arena.c
> index c6369ea5e208..eaeae16bbe7a 100644
> --- a/kernel/bpf/arena.c
> +++ b/kernel/bpf/arena.c
> @@ -704,6 +704,27 @@ static u64 clear_lo32(u64 val)
> return val & ~(u64)~0U;
> }
>
> +static int arena_adjust_tree(struct bpf_arena *arena, long uaddr, long page_cnt, long *pgoff)
> +{
> + int ret;
> +
> + /* Special case where user is requesting specific range. */
> + if (uaddr) {
> + ret = is_range_tree_set(&arena->rt, *pgoff, page_cnt);
> + if (ret)
> + return ret;
> + return range_tree_clear(&arena->rt, *pgoff, page_cnt);
> + }
> +
> + ret = range_tree_find(&arena->rt, page_cnt);
> + if (ret < 0)
> + return ret;
> +
> + *pgoff = ret;
> +
> + return range_tree_clear(&arena->rt, *pgoff, page_cnt);
> +}
> +
> /*
> * Allocate pages and vmap them into kernel vmalloc area.
> * Later the pages will be mmaped into user space vma.
> @@ -721,7 +742,9 @@ static long arena_alloc_pages(struct bpf_arena *arena, long uaddr, long page_cnt
> long alloc_pages;
> unsigned long flags;
> long pgoff = 0;
> + bool can_sleep;
> u32 uaddr32;
> + long addr = 0;
> int ret, i;
>
> if (node_id != NUMA_NO_NODE &&
> @@ -741,31 +764,38 @@ static long arena_alloc_pages(struct bpf_arena *arena, long uaddr, long page_cnt
> }
>
> bpf_map_memcg_enter(&arena->map, &old_memcg, &new_memcg);
> - /* Cap allocation size to KMALLOC_MAX_CACHE_SIZE so kmalloc_nolock() can succeed. */
> - alloc_pages = min(page_cnt, KMALLOC_MAX_CACHE_SIZE / sizeof(struct page *));
> - pages = kmalloc_nolock(alloc_pages * sizeof(struct page *), __GFP_ACCOUNT, NUMA_NO_NODE);
> - if (!pages) {
> - bpf_map_memcg_exit(old_memcg, new_memcg);
> - return 0;
> +
> + can_sleep = sleepable && !is_bpf_alloc_nonsleepable();
> + if (can_sleep) {
> + alloc_pages = page_cnt;
> + pages = kvcalloc(page_cnt, sizeof(struct page *), GFP_KERNEL_ACCOUNT);
> + if (!pages)
> + goto out_memcg;
> +
> + ret = bpf_map_alloc_pages(&arena->map, node_id, page_cnt, pages);
> + if (ret)
> + goto out_free_array;
> + data.i = 0;
> + } else {
> + /* Cap allocation size so kmalloc_nolock() can succeed. */
> + alloc_pages = min(page_cnt, KMALLOC_MAX_CACHE_SIZE / sizeof(struct page *));
> + pages = kmalloc_nolock(alloc_pages * sizeof(struct page *), __GFP_ACCOUNT,
> + NUMA_NO_NODE);
> + if (!pages)
> + goto out_memcg;
> }
> +
> data.arena = arena;
> data.pages = pages;
>
> if (raw_res_spin_lock_irqsave(&arena->spinlock, flags))
> goto out_free_pages;
>
> - if (uaddr) {
> - ret = is_range_tree_set(&arena->rt, pgoff, page_cnt);
> - if (ret)
> - goto out_unlock_free_pages;
> - ret = range_tree_clear(&arena->rt, pgoff, page_cnt);
> - } else {
> - ret = pgoff = range_tree_find(&arena->rt, page_cnt);
> - if (pgoff >= 0)
> - ret = range_tree_clear(&arena->rt, pgoff, page_cnt);
> + ret = arena_adjust_tree(arena, uaddr, page_cnt, &pgoff);
> + if (ret) {
> + raw_res_spin_unlock_irqrestore(&arena->spinlock, flags);
> + goto out_free_pages;
> }
> - if (ret)
> - goto out_unlock_free_pages;
>
> remaining = page_cnt;
> uaddr32 = (u32)(arena->user_vm_start + pgoff * PAGE_SIZE);
> @@ -773,12 +803,14 @@ static long arena_alloc_pages(struct bpf_arena *arena, long uaddr, long page_cnt
> while (remaining) {
> long this_batch = min(remaining, alloc_pages);
>
> - /* zeroing is needed, since alloc_pages_bulk() only fills in non-zero entries */
> - memset(pages, 0, this_batch * sizeof(struct page *));
> + if (!can_sleep) {
> + /* alloc_pages_bulk() only fills in non-zero entries. */
> + memset(pages, 0, this_batch * sizeof(struct page *));
>
> - ret = bpf_map_alloc_pages(&arena->map, node_id, this_batch, pages);
> - if (ret)
> - goto out;
> + ret = bpf_map_alloc_pages(&arena->map, node_id, this_batch, pages);
> + if (ret)
> + goto out_unmap;
> + }
>
> /*
> * Earlier checks made sure that uaddr32 + page_cnt * PAGE_SIZE - 1
> @@ -793,35 +825,50 @@ static long arena_alloc_pages(struct bpf_arena *arena, long uaddr, long page_cnt
> kern_vm_start + uaddr32 + (mapped << PAGE_SHIFT),
> this_batch << PAGE_SHIFT, apply_range_set_cb, &data);
> if (ret) {
> - /* data.i pages were mapped, account them and free the remaining */
> + /* data.i pages were mapped, account them and free the remaining. */
> mapped += data.i;
> - for (i = data.i; i < this_batch; i++)
> - free_pages_nolock(pages[i], 0);
> - goto out;
> + if (!can_sleep)
> + for (i = data.i; i < this_batch; i++)
> + free_pages_nolock(pages[i], 0);
> + goto out_unmap;
> }
>
> mapped += this_batch;
> remaining -= this_batch;
> }
> +
> flush_vmap_cache(kern_vm_start + uaddr32, mapped << PAGE_SHIFT);
> raw_res_spin_unlock_irqrestore(&arena->spinlock, flags);
> - kfree_nolock(pages);
> - bpf_map_memcg_exit(old_memcg, new_memcg);
> - return clear_lo32(arena->user_vm_start) + uaddr32;
> -out:
> +
> + addr = clear_lo32(arena->user_vm_start) + uaddr32;
> + goto out_free_array;
> +
> +out_unmap:
> + if (can_sleep)
> + flush_vmap_cache(kern_vm_start + uaddr32, mapped << PAGE_SHIFT);
> range_tree_set(&arena->rt, pgoff + mapped, page_cnt - mapped);
> raw_res_spin_unlock_irqrestore(&arena->spinlock, flags);
> - if (mapped) {
> - flush_vmap_cache(kern_vm_start + uaddr32, mapped << PAGE_SHIFT);
> - arena_free_pages(arena, uaddr32, mapped, sleepable);
> + if (mapped || can_sleep) {
> + if (!can_sleep)
> + flush_vmap_cache(kern_vm_start + uaddr32, mapped << PAGE_SHIFT);
> + arena_free_pages(arena, uaddr32, mapped, can_sleep);
> }
> - goto out_free_pages;
> -out_unlock_free_pages:
> - raw_res_spin_unlock_irqrestore(&arena->spinlock, flags);
> +
> out_free_pages:
> - kfree_nolock(pages);
> + if (can_sleep)
> + for (i = data.i; i < page_cnt; i++)
> + __free_page(pages[i]);
> +
> +out_free_array:
> + if (can_sleep)
> + kvfree(pages);
> + else
> + kfree_nolock(pages);
> +
> +out_memcg:
> bpf_map_memcg_exit(old_memcg, new_memcg);
> - return 0;
> +
> + return addr;
> }
>
> /*
> diff --git a/kernel/bpf/syscall.c b/kernel/bpf/syscall.c
> index 74496fd716d3..d193d84cc33e 100644
> --- a/kernel/bpf/syscall.c
> +++ b/kernel/bpf/syscall.c
> @@ -596,15 +596,9 @@ static void bpf_map_release_memcg(struct bpf_map *map)
> }
> #endif
>
> -static bool can_alloc_pages(void)
> -{
> - return preempt_count() == 0 && !irqs_disabled() &&
> - !IS_ENABLED(CONFIG_PREEMPT_RT);
> -}
> -
> static struct page *__bpf_alloc_page(int nid)
> {
> - if (!can_alloc_pages())
> + if (is_bpf_alloc_nonsleepable())
> return alloc_pages_nolock(__GFP_ACCOUNT, nid, 0);
>
> return alloc_pages_node(nid,
next prev parent reply other threads:[~2026-09-24 10:11 UTC|newest]
Thread overview: 9+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-24 5:36 [PATCH bpf v2 0/2] Make sleepable arena paths use sleepable alloc_pages Emil Tsalapatis
2026-09-24 5:36 ` [PATCH bpf v2 1/2] bpf: Add sleepable arena page allocation path Emil Tsalapatis
2026-09-24 6:29 ` bot+bpf-ci
2026-09-24 18:29 ` Emil Tsalapatis
2026-09-24 10:11 ` Jiayuan Chen [this message]
2026-09-24 17:56 ` Emil Tsalapatis
2026-09-24 21:33 ` Alexei Starovoitov
2026-09-24 22:03 ` Emil Tsalapatis
2026-09-24 5:36 ` [PATCH bpf v2 2/2] selftests/bpf: Test large allocations for both sleepable/nonsleepable arena users Emil Tsalapatis
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=7fba6431-14ab-45de-99dd-17e38cae2841@linux.dev \
--to=jiayuan.chen@linux.dev \
--cc=andrii@kernel.org \
--cc=ast@kernel.org \
--cc=bpf@vger.kernel.org \
--cc=daniel@iogearbox.net \
--cc=eddyz87@gmail.com \
--cc=emil@etsalapatis.com \
--cc=memxor@gmail.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox