All of lore.kernel.org
 help / color / mirror / Atom feed
From: Mostafa Saleh <smostafa@google.com>
To: linux-doc@vger.kernel.org, linux-kernel@vger.kernel.org,
	 linux-arm-kernel@lists.infradead.org, linux-mm@kvack.org,
	 linux-hardening@vger.kernel.org, linux-rt-devel@lists.linux.dev
Cc: corbet@lwn.net, skhan@linuxfoundation.org, rdunlap@infradead.org,
	 catalin.marinas@arm.com, will@kernel.org, mark.rutland@arm.com,
	 akpm@linux-foundation.org, urezki@gmail.com, mingo@redhat.com,
	 peterz@infradead.org, juri.lelli@redhat.com,
	vincent.guittot@linaro.org,  dietmar.eggemann@arm.com,
	rostedt@goodmis.org, bsegall@google.com,  mgorman@suse.de,
	vschneid@redhat.com, kprateek.nayak@amd.com, kees@kernel.org,
	 david@kernel.org, ljs@kernel.org, liam@infradead.org,
	vbabka@kernel.org,  rppt@kernel.org, surenb@google.com,
	mhocko@suse.com, gustavoars@kernel.org,  bigeasy@linutronix.de,
	clrkwllms@kernel.org,  Mostafa Saleh <smostafa@google.com>,
	Pasha Tatashin <pasha.tatashin@soleen.com>,
	 Linus Walleij <linus.walleij@linaro.org>,
	David Stevens <stevensd@google.com>
Subject: [RFC PATCH 13/15] fork: Implement partial VMAP stack allocation
Date: Mon, 28 Sep 2026 17:41:20 +0000	[thread overview]
Message-ID: <20260928174122.3380703-14-smostafa@google.com> (raw)
In-Reply-To: <20260928174122.3380703-1-smostafa@google.com>

When CONFIG_ARCH_HAS_VARIABLE_STACK_SIZE is enabled, architectures can
allocate a VMAP stack that is smaller than the full THREAD_SIZE.

This introduces arch abstraction arch_vm_stack_pages() which defaults
to THREAD_SIZE / PAGE_SIZE.
During stack allocation, the kernel will map only the requested number
of pages at the top of the THREAD_SIZE virtual area (as the stack
grows down), leaving the bottom unmapped. This saves memory while
retaining the same virtual alignment and guard page overflow detection
characteristics.

Signed-off-by: Pasha Tatashin <pasha.tatashin@soleen.com>
[Rebased, used vm_area->nr_pages directly in one instance]
[Depends on !PREEMPT_RT]
Signed-off-by: Linus Walleij <linus.walleij@linaro.org>
[Fix races around accounting]
[Use GFP_ATOMIC when executing in the scheduler]
[Depend on INIT_STACK_ALL_* config]
[Fix bugs in some error paths and edge cases]
[Don't cache partially faulted stacks]
[Added out-var to tell if address is on target stack]
Signed-off-by: David Stevens <stevensd@google.com>
[Remove dynamic stack related code, and update commit message]
[Fix memory leak in error path of alloc_vmap_stack()] and use
 VM_UNINITIALIZED before installing the pages]
Signed-off-by: Mostafa Saleh <smostafa@google.com>
---
 include/linux/thread_info.h |  8 +++++
 kernel/fork.c               | 68 +++++++++++++++++++++++++++++++++++++
 2 files changed, 76 insertions(+)

diff --git a/include/linux/thread_info.h b/include/linux/thread_info.h
index 307b8390fc67..8f052ec82402 100644
--- a/include/linux/thread_info.h
+++ b/include/linux/thread_info.h
@@ -92,6 +92,14 @@ static inline long set_restart_fn(struct restart_block *restart,
 #define THREAD_ALIGN	THREAD_SIZE
 #endif
 
+/*
+ * Arch selecting CONFIG_ARCH_HAS_VARIABLE_STACK_SIZE should override this
+ * with kernel stack size.
+ */
+#ifndef arch_vm_stack_pages
+#define arch_vm_stack_pages()	(THREAD_SIZE / PAGE_SIZE)
+#endif
+
 #define THREADINFO_GFP		(GFP_KERNEL_ACCOUNT | __GFP_ZERO | __GFP_SKIP_KASAN)
 
 /*
diff --git a/kernel/fork.c b/kernel/fork.c
index ca8e68882316..66fbad3d4cbd 100644
--- a/kernel/fork.c
+++ b/kernel/fork.c
@@ -119,6 +119,8 @@
 
 /* For dup_mmap(). */
 #include "../mm/internal.h"
+/* For clear_vm_uninitialized_flag(). */
+#include "../mm/vmalloc.h"
 
 #include <trace/events/sched.h>
 
@@ -272,6 +274,71 @@ static bool try_release_thread_stack_to_cache(struct vm_struct *vm_area)
 	return false;
 }
 
+#ifdef CONFIG_ARCH_HAS_VARIABLE_STACK_SIZE
+static struct vm_struct *alloc_vmap_stack(int node)
+{
+	gfp_t gfp = GFP_VMAP_STACK;
+	unsigned long addr, end;
+	struct vm_struct *vm_area;
+	int err, i;
+
+	vm_area = get_vm_area_node(THREAD_SIZE, THREAD_ALIGN,
+				   VM_MAP | VM_UNINITIALIZED, node, gfp,
+				   __builtin_return_address(0));
+	if (!vm_area)
+		return NULL;
+
+	vm_area->pages = kmalloc_node(sizeof(void *) * (THREAD_SIZE >> PAGE_SHIFT),
+				      gfp, node);
+	if (!vm_area->pages)
+		goto cleanup_err;
+
+	for (i = 0; i < arch_vm_stack_pages(); i++) {
+		vm_area->pages[i] = alloc_pages_node(node, gfp, 0);
+		if (!vm_area->pages[i])
+			goto cleanup_err;
+		vm_area->nr_pages++;
+		mod_lruvec_page_state(vm_area->pages[i], NR_VMALLOC, 1);
+	}
+
+	end = (unsigned long)kasan_reset_tag(vm_area->addr) + THREAD_SIZE;
+	addr = end - (arch_vm_stack_pages() * PAGE_SIZE);
+
+	err = vmap_pages_range(addr, end, PAGE_KERNEL, vm_area->pages, PAGE_SHIFT);
+	if (err)
+		goto cleanup_err;
+
+	clear_vm_uninitialized_flag(vm_area);
+	return vm_area;
+
+cleanup_err:
+	remove_vm_area(vm_area->addr);
+	if (vm_area->pages) {
+		for (i = 0; i < vm_area->nr_pages; i++) {
+			mod_lruvec_page_state(vm_area->pages[i], NR_VMALLOC, -1);
+			__free_page(vm_area->pages[i]);
+		}
+		kfree(vm_area->pages);
+	}
+	kfree(vm_area);
+	return NULL;
+}
+
+static void free_vmap_stack(struct vm_struct *vm_area)
+{
+	int i;
+
+	remove_vm_area(vm_area->addr);
+
+	for (i = 0; i < vm_area->nr_pages; i++) {
+		mod_lruvec_page_state(vm_area->pages[i], NR_VMALLOC, -1);
+		__free_page(vm_area->pages[i]);
+	}
+
+	kfree(vm_area->pages);
+	kfree(vm_area);
+}
+#else
 static inline struct vm_struct *alloc_vmap_stack(int node)
 {
 	void *stack;
@@ -286,6 +353,7 @@ static inline void free_vmap_stack(struct vm_struct *vm_area)
 {
 	vfree(vm_area->addr);
 }
+#endif /* CONFIG_ARCH_HAS_VARIABLE_STACK_SIZE */
 
 static void thread_stack_free_work(struct work_struct *work)
 {
-- 
2.56.0.rc1.315.gc6ed9934b7-goog


  parent reply	other threads:[~2026-09-28 17:42 UTC|newest]

Thread overview: 18+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-28 17:41 [RFC PATCH 00/15] arm64: Set kernel stack size from cmdline Mostafa Saleh
2026-09-28 17:41 ` [RFC PATCH 01/15] fork: Remove assumption that vm_area->nr_pages equals to THREAD_SIZE Mostafa Saleh
2026-09-28 17:41 ` [RFC PATCH 02/15] fork: Don't assume fully populated stack during reuse Mostafa Saleh
2026-09-28 17:41 ` [RFC PATCH 03/15] fork: Move vm_stack to the beginning of the stack Mostafa Saleh
2026-09-28 17:41 ` [RFC PATCH 04/15] fork: Separate vmap stack allocation and free calls Mostafa Saleh
2026-09-28 17:41 ` [RFC PATCH 05/15] sched/task_stack: Add helpers for stack high/low Mostafa Saleh
2026-09-28 17:41 ` [RFC PATCH 06/15] exit: Don't assume the kernel stack size Mostafa Saleh
2026-09-28 17:41 ` [RFC PATCH 07/15] usercopy: " Mostafa Saleh
2026-09-28 17:41 ` [RFC PATCH 08/15] mm: kmemleak: " Mostafa Saleh
2026-09-28 17:41 ` [RFC PATCH 09/15] arm64: " Mostafa Saleh
2026-09-28 17:41 ` [RFC PATCH 10/15] mm/vmalloc: Add a get_vm_area_node() Mostafa Saleh
2026-09-28 17:41 ` [RFC PATCH 11/15] fork: Move vmap stack freeing to work queue Mostafa Saleh
2026-09-28 17:41 ` [RFC PATCH 12/15] sched/task_stack: Introduce ARCH_HAS_VARIABLE_STACK_SIZE Mostafa Saleh
2026-09-28 20:37   ` Randy Dunlap
2026-09-29 10:27     ` Mostafa Saleh
2026-09-28 17:41 ` Mostafa Saleh [this message]
2026-09-28 17:41 ` [RFC PATCH 14/15] arm64: mm: Relax kernel stack alignment Mostafa Saleh
2026-09-28 17:41 ` [RFC PATCH 15/15] arm64: mm: Set stack size from the kernel command line Mostafa Saleh

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260928174122.3380703-14-smostafa@google.com \
    --to=smostafa@google.com \
    --cc=akpm@linux-foundation.org \
    --cc=bigeasy@linutronix.de \
    --cc=bsegall@google.com \
    --cc=catalin.marinas@arm.com \
    --cc=clrkwllms@kernel.org \
    --cc=corbet@lwn.net \
    --cc=david@kernel.org \
    --cc=dietmar.eggemann@arm.com \
    --cc=gustavoars@kernel.org \
    --cc=juri.lelli@redhat.com \
    --cc=kees@kernel.org \
    --cc=kprateek.nayak@amd.com \
    --cc=liam@infradead.org \
    --cc=linus.walleij@linaro.org \
    --cc=linux-arm-kernel@lists.infradead.org \
    --cc=linux-doc@vger.kernel.org \
    --cc=linux-hardening@vger.kernel.org \
    --cc=linux-kernel@vger.kernel.org \
    --cc=linux-mm@kvack.org \
    --cc=linux-rt-devel@lists.linux.dev \
    --cc=ljs@kernel.org \
    --cc=mark.rutland@arm.com \
    --cc=mgorman@suse.de \
    --cc=mhocko@suse.com \
    --cc=mingo@redhat.com \
    --cc=pasha.tatashin@soleen.com \
    --cc=peterz@infradead.org \
    --cc=rdunlap@infradead.org \
    --cc=rostedt@goodmis.org \
    --cc=rppt@kernel.org \
    --cc=skhan@linuxfoundation.org \
    --cc=stevensd@google.com \
    --cc=surenb@google.com \
    --cc=urezki@gmail.com \
    --cc=vbabka@kernel.org \
    --cc=vincent.guittot@linaro.org \
    --cc=vschneid@redhat.com \
    --cc=will@kernel.org \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.