From: David Stevens <stevensd@google.com>
To: Catalin Marinas <catalin.marinas@arm.com>,
Will Deacon <will@kernel.org>, Thomas Gleixner <tglx@kernel.org>,
Ingo Molnar <mingo@redhat.com>, Borislav Petkov <bp@alien8.de>,
Dave Hansen <dave.hansen@linux.intel.com>,
x86@kernel.org, "H . Peter Anvin" <hpa@zytor.com>,
Andrew Morton <akpm@linux-foundation.org>,
Dave Chinner <david@fromorbit.com>,
Qi Zheng <qi.zheng@linux.dev>,
Roman Gushchin <roman.gushchin@linux.dev>,
Muchun Song <muchun.song@linux.dev>,
Peter Zijlstra <peterz@infradead.org>,
Juri Lelli <juri.lelli@redhat.com>,
Vincent Guittot <vincent.guittot@linaro.org>,
Dietmar Eggemann <dietmar.eggemann@arm.com>,
Steven Rostedt <rostedt@goodmis.org>,
Ben Segall <bsegall@google.com>, Mel Gorman <mgorman@suse.de>,
Valentin Schneider <vschneid@redhat.com>,
K Prateek Nayak <kprateek.nayak@amd.com>,
Uladzislau Rezki <urezki@gmail.com>,
David Hildenbrand <david@kernel.org>,
Lorenzo Stoakes <ljs@kernel.org>,
"Liam R . Howlett" <liam@infradead.org>,
Vlastimil Babka <vbabka@kernel.org>,
Mike Rapoport <rppt@kernel.org>,
Suren Baghdasaryan <surenb@google.com>,
Michal Hocko <mhocko@suse.com>, Kees Cook <kees@kernel.org>,
Sebastian Andrzej Siewior <bigeasy@linutronix.de>,
Clark Williams <clrkwllms@kernel.org>,
suleiman@google.com
Cc: linux-kernel@vger.kernel.org,
linux-arm-kernel@lists.infradead.org, linux-mm@kvack.org,
linux-rt-devel@lists.linux.dev,
David Stevens <stevensd@google.com>
Subject: [RFC 05/10] fork: allocate reclaimable stacks with VM_SPARSE
Date: Thu, 27 Aug 2026 16:29:43 -0700 [thread overview]
Message-ID: <20260827232948.2520558-6-stevensd@google.com> (raw)
In-Reply-To: <20260827232948.2520558-1-stevensd@google.com>
Since the pages of reclaimable stacks will not always be populated, we
need to set VM_SPARSE on their vm_structs to avoid crashes when
vread_iter sees a partially reclaimed stack. Note that stacks will only
be partially reclaimed when they are associated with a live task, so the
stack management and caching code in fork.c won't actually ever see a
partially reclaimed stack.
Directly managing the vm_structs instead of going through vmalloc also
requires doing the freeing of the stack on a work queue instead of in an
RCU callback. Previously, we were relying on deferred vfree work to move
much of the cleanup work from softirq to process context.
The fact that vmap stack pages are included in vmalloc's vmstat is well
known. We should continue that to avoid changing procfs.
Signed-off-by: David Stevens <stevensd@google.com>
---
kernel/fork.c | 92 +++++++++++++++++++++++++++++++++++++++++++++++++++
1 file changed, 92 insertions(+)
diff --git a/kernel/fork.c b/kernel/fork.c
index 10bd76f6dd20..6acad0038b78 100644
--- a/kernel/fork.c
+++ b/kernel/fork.c
@@ -267,6 +267,97 @@ static bool try_release_thread_stack_to_cache(struct vm_struct *vm_area)
return false;
}
+#ifdef CONFIG_RECLAIMABLE_STACK
+static void free_vmap_stack(struct vm_struct *vm_area)
+{
+ int i;
+
+ remove_vm_area(vm_area->addr);
+
+ for (i = 0; i < vm_area->nr_pages; i++) {
+ mod_node_page_state(page_pgdat(vm_area->pages[i]), NR_VMALLOC, -1);
+ __free_page(vm_area->pages[i]);
+ }
+
+ kfree(vm_area->pages);
+ kfree(vm_area);
+}
+
+static struct vm_struct *alloc_vmap_stack(int node)
+{
+ struct vm_struct *vm_area;
+ int ret;
+
+ vm_area = get_vm_area_node(THREAD_SIZE, THREAD_ALIGN, VM_MAP | VM_SPARSE, node);
+ if (!vm_area)
+ return NULL;
+
+ vm_area->pages = kcalloc_node(THREAD_SIZE >> PAGE_SHIFT, sizeof(*vm_area->pages),
+ GFP_KERNEL | __GFP_ZERO, node);
+ if (!vm_area->pages)
+ goto alloc_failure;
+
+ while (vm_area->nr_pages < THREAD_SIZE >> PAGE_SHIFT) {
+ struct page *page;
+ gfp_t gfp = GFP_VMAP_STACK | __GFP_HIGHMEM;
+
+ if (node == NUMA_NO_NODE)
+ page = alloc_pages(gfp, 0);
+ else
+ page = alloc_pages_node(node, gfp, 0);
+
+ if (!page)
+ goto alloc_failure;
+
+ /*
+ * Non-reclaimable vmap stacks pages aren't charged against an
+ * memcg until account_kernel_stack(), but they are added to
+ * the node's NR_VMALLOC counter. Copy that behavior to avoid
+ * confusing userspace.
+ */
+ mod_node_page_state(page_pgdat(page), NR_VMALLOC, 1);
+ vm_area->pages[vm_area->nr_pages++] = page;
+ }
+
+ ret = vmap_pages_range((unsigned long)vm_area->addr,
+ (unsigned long)vm_area->addr + THREAD_SIZE,
+ PAGE_KERNEL, vm_area->pages, PAGE_SHIFT);
+ if (ret)
+ goto alloc_failure;
+
+ return vm_area;
+
+alloc_failure:
+ free_vmap_stack(vm_area);
+ return NULL;
+}
+
+struct vm_stack {
+ struct rcu_work work;
+ struct vm_struct *stack_vm_area;
+};
+
+static void thread_stack_free_work(struct work_struct *work)
+{
+ struct vm_stack *vm_stack = container_of(to_rcu_work(work), struct vm_stack, work);
+ struct vm_struct *vm_area = vm_stack->stack_vm_area;
+
+ if (try_release_thread_stack_to_cache(vm_stack->stack_vm_area))
+ return;
+
+ free_vmap_stack(vm_area);
+}
+
+static void thread_stack_delayed_free(struct task_struct *tsk)
+{
+ struct vm_stack *vm_stack = tsk->stack;
+
+ vm_stack->stack_vm_area = tsk->stack_vm_area;
+ INIT_RCU_WORK(&vm_stack->work, thread_stack_free_work);
+ queue_rcu_work(system_wq, &vm_stack->work);
+}
+
+#else /* !CONFIG_RECLAIMABLE_STACK */
static void free_vmap_stack(struct vm_struct *vm_area)
{
vfree(vm_area->addr);
@@ -305,6 +396,7 @@ static void thread_stack_delayed_free(struct task_struct *tsk)
vm_stack->stack_vm_area = tsk->stack_vm_area;
call_rcu(&vm_stack->rcu, thread_stack_free_rcu);
}
+#endif /* CONFIG_RECLAIMABLE_STACK */
static int free_vm_stack_cache(unsigned int cpu)
{
--
2.55.0.897.gb25b4bd76c-goog
next prev parent reply other threads:[~2026-08-27 23:32 UTC|newest]
Thread overview: 35+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-08-27 23:29 [RFC 00/10] Reclaimable kernel stacks David Stevens
2026-08-27 23:29 ` [RFC 01/10] Add !MEMCG memcg_list_lru_alloc implementation David Stevens
2026-08-27 23:29 ` [RFC 02/10] mm/vmalloc: Skip vmallocinfo NUMA stats for VM_SPARSE David Stevens
2026-08-27 23:29 ` [RFC 03/10] fork: refactor vmap stack alloc/free into helpers David Stevens
2026-08-27 23:29 ` [RFC 04/10] mm: vmalloc: support creating aligned vm areas David Stevens
2026-08-27 23:29 ` David Stevens [this message]
2026-08-27 23:29 ` [RFC 06/10] Reclaim memory from blocked kernel stacks David Stevens
2026-08-28 11:54 ` Peter Zijlstra
2026-08-28 12:01 ` Peter Zijlstra
2026-08-28 12:04 ` Peter Zijlstra
2026-08-29 0:18 ` David Stevens
2026-08-28 12:41 ` Peter Zijlstra
2026-08-28 12:57 ` Peter Zijlstra
2026-08-28 23:33 ` David Stevens
2026-08-28 13:36 ` Sebastian Andrzej Siewior
2026-08-28 13:59 ` Peter Zijlstra
2026-08-28 14:25 ` Peter Zijlstra
2026-08-28 15:58 ` Sebastian Andrzej Siewior
2026-08-28 15:10 ` Sebastian Andrzej Siewior
2026-08-28 19:08 ` Steven Rostedt
2026-08-28 19:13 ` Steven Rostedt
2026-08-28 19:17 ` Steven Rostedt
2026-08-28 20:50 ` David Stevens
2026-08-28 21:17 ` David Stevens
2026-08-27 23:29 ` [RFC 07/10] Reclaim stacks via a shrinker David Stevens
2026-08-27 23:29 ` [RFC 08/10] Set PF_RECLAIMABLE_STACK in various places David Stevens
2026-08-28 6:33 ` K Prateek Nayak
2026-08-27 23:29 ` [RFC 09/10] x86: Enable reclaimable stacks David Stevens
2026-08-27 23:29 ` [RFC 10/10] arm64: " David Stevens
2026-08-28 12:47 ` [RFC 00/10] Reclaimable kernel stacks Peter Zijlstra
2026-08-28 14:33 ` Steven Rostedt
2026-08-28 14:35 ` Peter Zijlstra
2026-08-28 14:45 ` Peter Zijlstra
2026-08-28 16:10 ` Steven Rostedt
2026-08-28 17:58 ` David Stevens
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260827232948.2520558-6-stevensd@google.com \
--to=stevensd@google.com \
--cc=akpm@linux-foundation.org \
--cc=bigeasy@linutronix.de \
--cc=bp@alien8.de \
--cc=bsegall@google.com \
--cc=catalin.marinas@arm.com \
--cc=clrkwllms@kernel.org \
--cc=dave.hansen@linux.intel.com \
--cc=david@fromorbit.com \
--cc=david@kernel.org \
--cc=dietmar.eggemann@arm.com \
--cc=hpa@zytor.com \
--cc=juri.lelli@redhat.com \
--cc=kees@kernel.org \
--cc=kprateek.nayak@amd.com \
--cc=liam@infradead.org \
--cc=linux-arm-kernel@lists.infradead.org \
--cc=linux-kernel@vger.kernel.org \
--cc=linux-mm@kvack.org \
--cc=linux-rt-devel@lists.linux.dev \
--cc=ljs@kernel.org \
--cc=mgorman@suse.de \
--cc=mhocko@suse.com \
--cc=mingo@redhat.com \
--cc=muchun.song@linux.dev \
--cc=peterz@infradead.org \
--cc=qi.zheng@linux.dev \
--cc=roman.gushchin@linux.dev \
--cc=rostedt@goodmis.org \
--cc=rppt@kernel.org \
--cc=suleiman@google.com \
--cc=surenb@google.com \
--cc=tglx@kernel.org \
--cc=urezki@gmail.com \
--cc=vbabka@kernel.org \
--cc=vincent.guittot@linaro.org \
--cc=vschneid@redhat.com \
--cc=will@kernel.org \
--cc=x86@kernel.org \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox