BPF List
 help / color / mirror / Atom feed
From: "Emil Tsalapatis" <emil@etsalapatis.com>
To: "Jiayuan Chen" <jiayuan.chen@linux.dev>,
	"Emil Tsalapatis" <emil@etsalapatis.com>, <bpf@vger.kernel.org>
Cc: <ast@kernel.org>, <andrii@kernel.org>, <eddyz87@gmail.com>,
	<memxor@gmail.com>, <daniel@iogearbox.net>
Subject: Re: [PATCH bpf v2 1/2] bpf: Add sleepable arena page allocation path
Date: Thu, 24 Sep 2026 17:56:31 +0000	[thread overview]
Message-ID: <DLNQO2KI4P26.3IJRH3BW8EQ13@etsalapatis.com> (raw)
In-Reply-To: <7fba6431-14ab-45de-99dd-17e38cae2841@linux.dev>

On Thu Sep 24, 2026 at 10:11 AM UTC, Jiayuan Chen wrote:
>
> On 9/24/26 1:36 PM, Emil Tsalapatis wrote:
>> The bpf_arena_alloc_pages() function currently only allocates pages
>> inside a spinlock critical section with IRQs off. This forces the use
>> of alloc_pages_nolock() in the BPF allocator, even when the caller is
>> a sleepable BPF function. This in turn causes allocation failures even
>> in cases where falling into the allocator slow path and possibly
>> sleeping would eventually succeed. This can be triggered consistently
>> by heavy BPF arena users like scx.
>>
>> Add a separate arena page allocation path just for sleepable callers.
>> The path preallocates the arena memory to be added to the tree before
>> taking the critical section.
>>
>> Signed-off-by: Emil Tsalapatis <emil@etsalapatis.com>
>> ---
>>   include/linux/bpf.h  |   6 +++
>>   kernel/bpf/arena.c   | 123 ++++++++++++++++++++++++++++++-------------
>>   kernel/bpf/syscall.c |   8 +--
>>   3 files changed, 92 insertions(+), 45 deletions(-)
>>
>> diff --git a/include/linux/bpf.h b/include/linux/bpf.h
>> index e7c5e203eddd..b37e34cf1086 100644
>> --- a/include/linux/bpf.h
>> +++ b/include/linux/bpf.h
>> @@ -720,6 +720,12 @@ void bpf_map_free_internal_structs(struct bpf_map *map, void *obj);
>>   int bpf_dynptr_from_file_sleepable(struct file *file, u32 flags,
>>   				   struct bpf_dynptr *ptr__uninit);
>>   
>> +static inline bool is_bpf_alloc_nonsleepable(void)
>> +{
>> +	return preempt_count() > 0 || irqs_disabled() ||
>> +		IS_ENABLED(CONFIG_PREEMPT_RT);
>> +}
>> +
>
>
> A runtime check cannot tell whether sleeping is allowed: without
> CONFIG_PREEMPT_COUNT preempt_count() does not see spinlocks or
> rcu_read_lock(), which is why preemptible() is 0 there.
>
> How about passing 'sleepable' down to bpf_map_alloc_pages() instead?
> The verifier already proves it, and the code gets simpler.

Sure, we can keep it there. Afaict what you're proposing is to keep
the can_alloc_pages() call where it is and pass the sleepable flag
down to bpf_map_alloc_pages() instead, where now we're hoisting
can_alloc_pages() up.

>
>
>>   #if defined(CONFIG_MMU) && defined(CONFIG_64BIT)
>>   void *bpf_arena_alloc_pages_non_sleepable(void *p__map, void *addr__ign, u32 page_cnt, int node_id,
>>   					  u64 flags);
>> diff --git a/kernel/bpf/arena.c b/kernel/bpf/arena.c
>> index c6369ea5e208..eaeae16bbe7a 100644
>> --- a/kernel/bpf/arena.c
>> +++ b/kernel/bpf/arena.c
>> @@ -704,6 +704,27 @@ static u64 clear_lo32(u64 val)
>>   	return val & ~(u64)~0U;
>>   }
>>   
>> +static int arena_adjust_tree(struct bpf_arena *arena, long uaddr, long page_cnt, long *pgoff)
>> +{
>> +	int ret;
>> +
>> +	/* Special case where user is requesting specific range. */
>> +	if (uaddr) {
>> +		ret = is_range_tree_set(&arena->rt, *pgoff, page_cnt);
>> +		if (ret)
>> +			return ret;
>> +		return range_tree_clear(&arena->rt, *pgoff, page_cnt);
>> +	}
>> +
>> +	ret = range_tree_find(&arena->rt, page_cnt);
>> +	if (ret < 0)
>> +		return ret;
>> +
>> +	*pgoff = ret;
>> +
>> +	return range_tree_clear(&arena->rt, *pgoff, page_cnt);
>> +}
>> +
>>   /*
>>    * Allocate pages and vmap them into kernel vmalloc area.
>>    * Later the pages will be mmaped into user space vma.
>> @@ -721,7 +742,9 @@ static long arena_alloc_pages(struct bpf_arena *arena, long uaddr, long page_cnt
>>   	long alloc_pages;
>>   	unsigned long flags;
>>   	long pgoff = 0;
>> +	bool can_sleep;
>>   	u32 uaddr32;
>> +	long addr = 0;
>>   	int ret, i;
>>   
>>   	if (node_id != NUMA_NO_NODE &&
>> @@ -741,31 +764,38 @@ static long arena_alloc_pages(struct bpf_arena *arena, long uaddr, long page_cnt
>>   	}
>>   
>>   	bpf_map_memcg_enter(&arena->map, &old_memcg, &new_memcg);
>> -	/* Cap allocation size to KMALLOC_MAX_CACHE_SIZE so kmalloc_nolock() can succeed. */
>> -	alloc_pages = min(page_cnt, KMALLOC_MAX_CACHE_SIZE / sizeof(struct page *));
>> -	pages = kmalloc_nolock(alloc_pages * sizeof(struct page *), __GFP_ACCOUNT, NUMA_NO_NODE);
>> -	if (!pages) {
>> -		bpf_map_memcg_exit(old_memcg, new_memcg);
>> -		return 0;
>> +
>> +	can_sleep = sleepable && !is_bpf_alloc_nonsleepable();
>> +	if (can_sleep) {
>> +		alloc_pages = page_cnt;
>> +		pages = kvcalloc(page_cnt, sizeof(struct page *), GFP_KERNEL_ACCOUNT);
>> +		if (!pages)
>> +			goto out_memcg;
>> +
>> +		ret = bpf_map_alloc_pages(&arena->map, node_id, page_cnt, pages);
>> +		if (ret)
>> +			goto out_free_array;
>> +		data.i = 0;
>> +	} else {
>> +		/* Cap allocation size so kmalloc_nolock() can succeed. */
>> +		alloc_pages = min(page_cnt, KMALLOC_MAX_CACHE_SIZE / sizeof(struct page *));
>> +		pages = kmalloc_nolock(alloc_pages * sizeof(struct page *), __GFP_ACCOUNT,
>> +				       NUMA_NO_NODE);
>> +		if (!pages)
>> +			goto out_memcg;
>>   	}
>> +
>>   	data.arena = arena;
>>   	data.pages = pages;
>>   
>>   	if (raw_res_spin_lock_irqsave(&arena->spinlock, flags))
>>   		goto out_free_pages;
>>   
>> -	if (uaddr) {
>> -		ret = is_range_tree_set(&arena->rt, pgoff, page_cnt);
>> -		if (ret)
>> -			goto out_unlock_free_pages;
>> -		ret = range_tree_clear(&arena->rt, pgoff, page_cnt);
>> -	} else {
>> -		ret = pgoff = range_tree_find(&arena->rt, page_cnt);
>> -		if (pgoff >= 0)
>> -			ret = range_tree_clear(&arena->rt, pgoff, page_cnt);
>> +	ret = arena_adjust_tree(arena, uaddr, page_cnt, &pgoff);
>> +	if (ret) {
>> +		raw_res_spin_unlock_irqrestore(&arena->spinlock, flags);
>> +		goto out_free_pages;
>>   	}
>> -	if (ret)
>> -		goto out_unlock_free_pages;
>>   
>>   	remaining = page_cnt;
>>   	uaddr32 = (u32)(arena->user_vm_start + pgoff * PAGE_SIZE);
>> @@ -773,12 +803,14 @@ static long arena_alloc_pages(struct bpf_arena *arena, long uaddr, long page_cnt
>>   	while (remaining) {
>>   		long this_batch = min(remaining, alloc_pages);
>>   
>> -		/* zeroing is needed, since alloc_pages_bulk() only fills in non-zero entries */
>> -		memset(pages, 0, this_batch * sizeof(struct page *));
>> +		if (!can_sleep) {
>> +			/* alloc_pages_bulk() only fills in non-zero entries. */
>> +			memset(pages, 0, this_batch * sizeof(struct page *));
>>   
>> -		ret = bpf_map_alloc_pages(&arena->map, node_id, this_batch, pages);
>> -		if (ret)
>> -			goto out;
>> +			ret = bpf_map_alloc_pages(&arena->map, node_id, this_batch, pages);
>> +			if (ret)
>> +				goto out_unmap;
>> +		}
>>   
>>   		/*
>>   		 * Earlier checks made sure that uaddr32 + page_cnt * PAGE_SIZE - 1
>> @@ -793,35 +825,50 @@ static long arena_alloc_pages(struct bpf_arena *arena, long uaddr, long page_cnt
>>   					  kern_vm_start + uaddr32 + (mapped << PAGE_SHIFT),
>>   					  this_batch << PAGE_SHIFT, apply_range_set_cb, &data);
>>   		if (ret) {
>> -			/* data.i pages were mapped, account them and free the remaining */
>> +			/* data.i pages were mapped, account them and free the remaining. */
>>   			mapped += data.i;
>> -			for (i = data.i; i < this_batch; i++)
>> -				free_pages_nolock(pages[i], 0);
>> -			goto out;
>> +			if (!can_sleep)
>> +				for (i = data.i; i < this_batch; i++)
>> +					free_pages_nolock(pages[i], 0);
>> +			goto out_unmap;
>>   		}
>>   
>>   		mapped += this_batch;
>>   		remaining -= this_batch;
>>   	}
>> +
>>   	flush_vmap_cache(kern_vm_start + uaddr32, mapped << PAGE_SHIFT);
>>   	raw_res_spin_unlock_irqrestore(&arena->spinlock, flags);
>> -	kfree_nolock(pages);
>> -	bpf_map_memcg_exit(old_memcg, new_memcg);
>> -	return clear_lo32(arena->user_vm_start) + uaddr32;
>> -out:
>> +
>> +	addr = clear_lo32(arena->user_vm_start) + uaddr32;
>> +	goto out_free_array;
>> +
>> +out_unmap:
>> +	if (can_sleep)
>> +		flush_vmap_cache(kern_vm_start + uaddr32, mapped << PAGE_SHIFT);
>>   	range_tree_set(&arena->rt, pgoff + mapped, page_cnt - mapped);
>>   	raw_res_spin_unlock_irqrestore(&arena->spinlock, flags);
>> -	if (mapped) {
>> -		flush_vmap_cache(kern_vm_start + uaddr32, mapped << PAGE_SHIFT);
>> -		arena_free_pages(arena, uaddr32, mapped, sleepable);
>> +	if (mapped || can_sleep) {
>> +		if (!can_sleep)
>> +			flush_vmap_cache(kern_vm_start + uaddr32, mapped << PAGE_SHIFT);
>> +		arena_free_pages(arena, uaddr32, mapped, can_sleep);
>>   	}
>> -	goto out_free_pages;
>> -out_unlock_free_pages:
>> -	raw_res_spin_unlock_irqrestore(&arena->spinlock, flags);
>> +
>>   out_free_pages:
>> -	kfree_nolock(pages);
>> +	if (can_sleep)
>> +		for (i = data.i; i < page_cnt; i++)
>> +			__free_page(pages[i]);
>> +
>> +out_free_array:
>> +	if (can_sleep)
>> +		kvfree(pages);
>> +	else
>> +		kfree_nolock(pages);
>> +
>> +out_memcg:
>>   	bpf_map_memcg_exit(old_memcg, new_memcg);
>> -	return 0;
>> +
>> +	return addr;
>>   }
>>   
>>   /*
>> diff --git a/kernel/bpf/syscall.c b/kernel/bpf/syscall.c
>> index 74496fd716d3..d193d84cc33e 100644
>> --- a/kernel/bpf/syscall.c
>> +++ b/kernel/bpf/syscall.c
>> @@ -596,15 +596,9 @@ static void bpf_map_release_memcg(struct bpf_map *map)
>>   }
>>   #endif
>>   
>> -static bool can_alloc_pages(void)
>> -{
>> -	return preempt_count() == 0 && !irqs_disabled() &&
>> -		!IS_ENABLED(CONFIG_PREEMPT_RT);
>> -}
>> -
>>   static struct page *__bpf_alloc_page(int nid)
>>   {
>> -	if (!can_alloc_pages())
>> +	if (is_bpf_alloc_nonsleepable())
>>   		return alloc_pages_nolock(__GFP_ACCOUNT, nid, 0);
>>   
>>   	return alloc_pages_node(nid,


  reply	other threads:[~2026-09-24 17:56 UTC|newest]

Thread overview: 9+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-24  5:36 [PATCH bpf v2 0/2] Make sleepable arena paths use sleepable alloc_pages Emil Tsalapatis
2026-09-24  5:36 ` [PATCH bpf v2 1/2] bpf: Add sleepable arena page allocation path Emil Tsalapatis
2026-09-24  6:29   ` bot+bpf-ci
2026-09-24 18:29     ` Emil Tsalapatis
2026-09-24 10:11   ` Jiayuan Chen
2026-09-24 17:56     ` Emil Tsalapatis [this message]
2026-09-24 21:33   ` Alexei Starovoitov
2026-09-24 22:03     ` Emil Tsalapatis
2026-09-24  5:36 ` [PATCH bpf v2 2/2] selftests/bpf: Test large allocations for both sleepable/nonsleepable arena users Emil Tsalapatis

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=DLNQO2KI4P26.3IJRH3BW8EQ13@etsalapatis.com \
    --to=emil@etsalapatis.com \
    --cc=andrii@kernel.org \
    --cc=ast@kernel.org \
    --cc=bpf@vger.kernel.org \
    --cc=daniel@iogearbox.net \
    --cc=eddyz87@gmail.com \
    --cc=jiayuan.chen@linux.dev \
    --cc=memxor@gmail.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox