From: Baoquan He <baoquan.he@linux.dev>
To: linux-mm@kvack.org
Cc: chrisl@kernel.org, nphamcs@gmail.com, kasong@tencent.com,
baohua@kernel.org, youngjun.park@lge.com, hannes@cmpxchg.org,
yosry@kernel.org, david@kernel.org, shikemeng@huaweicloud.com,
chengming.zhou@linux.dev, linux-kernel@vger.kernel.org,
Baoquan He <baoquan.he@linux.dev>
Subject: [RFC PATCH v2 03/10] mm, swap: add xswap cluster grow via VM_SPARSE vmalloc
Date: Wed, 5 Aug 2026 15:53:26 +0800 [thread overview]
Message-ID: <20260805075336.3579395-4-baoquan.he@linux.dev> (raw)
In-Reply-To: <20260805075336.3579395-1-baoquan.he@linux.dev>
Implement dynamic cluster_info array growth for xswap devices using a
VM_SPARSE vmalloc area:
1. xswap_map_clusters(): Allocate physical pages and map them into
the pre-reserved VM_SPARSE KVA region via vm_area_map_pages().
2. xswap_unmap_clusters(): Unmap pages from the VM_SPARSE area via
vm_area_unmap_pages() (used by the error/teardown paths, shrink
comes later).
3. setup_swap_clusters_info() xswap path: Use get_vm_area(VM_SPARSE)
for the cluster_info array, lazily mapping only the initial chunk.
4. free_swap_cluster_info(): Refactor to take swap_info_struct*.
For xswap, unmap all clusters and free_vm_area().
5. swapoff: Remove snapshot locals; move p->max/p->cluster_info
clearing after free_swap_cluster_info().
The grow path runs in the swap allocation context which may have
PF_MEMALLOC set during reclaim, so avoid consuming emergency reserves
by using __GFP_HIGH | __GFP_NOMEMALLOC on alloc_page() and
kmalloc_array(), switching to vm_area_map_pages_gfp(), and wrapping
the entire allocation block with memalloc_noreclaim_save() to prevent
recursive reclaim from internal page table allocations.
Concurrent grow operations race on vm_area_map_pages(), triggering
WARN_ON(!pte_none) in the vmap page table walk. Add a per-device
mutex (xswap_lock) held across xswap_map_clusters and
xswap_unmap_clusters to serialize page table modifications. The
shrink path is already deferred to a workqueue so it does not contend
with itself.
Signed-off-by: Baoquan He <baoquan.he@linux.dev>
---
include/linux/swap.h | 1 +
mm/swapfile.c | 256 +++++++++++++++++++++++++++++++++++++++++--
2 files changed, 245 insertions(+), 12 deletions(-)
diff --git a/include/linux/swap.h b/include/linux/swap.h
index 970232f6359d..536b0e989c48 100644
--- a/include/linux/swap.h
+++ b/include/linux/swap.h
@@ -252,6 +252,7 @@ struct swap_info_struct {
struct vm_struct *cluster_vm; /* VM_SPARSE area for xswap dynamic cluster_info */
unsigned long nr_clusters; /* total cluster count for xswap */
unsigned long nr_clusters_mapped; /* currently mapped cluster count */
+ struct mutex xswap_lock; /* serialize map/unmap operations */
#endif
struct list_head free_clusters; /* free clusters list */
struct list_head full_clusters; /* full clusters list */
diff --git a/mm/swapfile.c b/mm/swapfile.c
index 08c49d5bea84..37c5dca153bc 100644
--- a/mm/swapfile.c
+++ b/mm/swapfile.c
@@ -49,6 +49,24 @@
#include "internal.h"
#include "swap.h"
+#ifdef CONFIG_XSWAP
+/*
+ * xswap: dynamically grow the cluster_info array via a VM_SPARSE area.
+ *
+ * XSWAP_GROW_CLUSTERS is the number of clusters to map in one grow
+ * operation. It is set to the number of cluster_info structs that
+ * fit in a single page (at least 16), so that the vmalloc page table
+ * overhead is proportional to the number of clusters mapped.
+ */
+#define XSWAP_GROW_CLUSTERS \
+ max_t(unsigned long, PAGE_SIZE / sizeof(struct swap_cluster_info), 16)
+
+static int xswap_map_clusters(struct swap_info_struct *si,
+ unsigned long start_idx, unsigned long nr);
+static void xswap_unmap_clusters(struct swap_info_struct *si,
+ unsigned long start_idx, unsigned long nr);
+#endif
+
static void swap_range_alloc(struct swap_info_struct *si,
unsigned int nr_entries);
static bool folio_swapcache_freeable(struct folio *folio);
@@ -3041,20 +3059,47 @@ static void wait_for_allocation(struct swap_info_struct *si)
BUG_ON(si->flags & SWP_WRITEOK);
+#ifdef CONFIG_XSWAP
+ /*
+ * xswap clusters beyond nr_clusters_mapped have been unmapped
+ * by the shrinker and their vmalloc pages are no longer
+ * accessible. Only iterate over currently mapped clusters.
+ */
+ if (si->flags & SWP_XSWAP)
+ end = min(end, READ_ONCE(si->nr_clusters_mapped) *
+ SWAPFILE_CLUSTER);
+#endif
+
for (offset = 0; offset < end; offset += SWAPFILE_CLUSTER) {
ci = swap_cluster_lock(si, offset);
swap_cluster_unlock(ci);
}
}
-static void free_swap_cluster_info(struct swap_cluster_info *cluster_info,
- unsigned long maxpages)
+static void free_swap_cluster_info(struct swap_info_struct *si)
{
+ struct swap_cluster_info *cluster_info = si->cluster_info;
+ unsigned long maxpages = si->max;
struct swap_cluster_info *ci;
- int i, nr_clusters = DIV_ROUND_UP(maxpages, SWAPFILE_CLUSTER);
+ int i, nr_clusters;
if (!cluster_info)
return;
+
+#ifdef CONFIG_XSWAP
+ if (si->flags & SWP_XSWAP) {
+ /* Unmap all mapped clusters and free the VM_SPARSE area */
+ if (si->nr_clusters_mapped > 0)
+ xswap_unmap_clusters(si, 0, si->nr_clusters_mapped);
+ free_vm_area(si->cluster_vm);
+ si->cluster_vm = NULL;
+ si->nr_clusters = 0;
+ si->nr_clusters_mapped = 0;
+ return;
+ }
+#endif
+
+ nr_clusters = DIV_ROUND_UP(maxpages, SWAPFILE_CLUSTER);
for (i = 0; i < nr_clusters; i++) {
ci = cluster_info + i;
/* Cluster with bad marks count will have a remaining table */
@@ -3093,11 +3138,9 @@ static void flush_percpu_swap_cluster(struct swap_info_struct *si)
SYSCALL_DEFINE1(swapoff, const char __user *, specialfile)
{
struct swap_info_struct *p = NULL;
- struct swap_cluster_info *cluster_info;
struct file *swap_file, *victim;
struct address_space *mapping;
struct inode *inode;
- unsigned int maxpages;
int err, found = 0;
if (!capable(CAP_SYS_ADMIN))
@@ -3189,10 +3232,6 @@ SYSCALL_DEFINE1(swapoff, const char __user *, specialfile)
swap_file = p->swap_file;
p->swap_file = NULL;
- maxpages = p->max;
- cluster_info = p->cluster_info;
- p->max = 0;
- p->cluster_info = NULL;
spin_unlock(&p->lock);
spin_unlock(&swap_lock);
arch_swap_invalidate_area(p->type);
@@ -3200,7 +3239,9 @@ SYSCALL_DEFINE1(swapoff, const char __user *, specialfile)
mutex_unlock(&swapon_mutex);
kfree(p->global_cluster);
p->global_cluster = NULL;
- free_swap_cluster_info(cluster_info, maxpages);
+ free_swap_cluster_info(p);
+ p->max = 0;
+ p->cluster_info = NULL;
inode = mapping->host;
@@ -3564,6 +3605,139 @@ static unsigned long read_swap_header(struct swap_info_struct *si,
return maxpages;
}
+#ifdef CONFIG_XSWAP
+static int xswap_map_clusters(struct swap_info_struct *si,
+ unsigned long start_idx, unsigned long nr)
+{
+ unsigned long start_addr = (unsigned long)si->cluster_info +
+ (size_t)start_idx * sizeof(struct swap_cluster_info);
+ unsigned long end_addr = start_addr + (size_t)nr * sizeof(struct swap_cluster_info);
+ /*
+ * vm_area_map_pages() requires that start and end be page-aligned.
+ * If start_addr falls within a page that was already mapped by a
+ * previous batch (grow path), round it up to skip the already-mapped
+ * partial page. Always round end_addr up so the vmap page table walk
+ * terminates correctly (the walk loop exits when addr == end, and addr
+ * advances by PAGE_SIZE each iteration).
+ */
+ unsigned long vm_start = PAGE_ALIGN(start_addr);
+ unsigned long vm_end = PAGE_ALIGN(end_addr);
+ unsigned int noreclaim_flags;
+ unsigned long npages;
+ struct page **pages;
+ unsigned long i;
+
+ mutex_lock(&si->xswap_lock);
+
+ if (vm_start >= vm_end) {
+ /* All requested clusters fall within already-mapped pages. */
+ for (i = start_idx; i < start_idx + nr; i++)
+ spin_lock_init(&si->cluster_info[i].lock);
+ WRITE_ONCE(si->nr_clusters_mapped, start_idx + nr);
+ mutex_unlock(&si->xswap_lock);
+ return 0;
+ }
+
+ npages = (vm_end - vm_start) >> PAGE_SHIFT;
+
+ /*
+ * Prevent recursive reclaim: vm_area_map_pages() internally
+ * allocates page tables with GFP_PGTABLE_KERNEL, which lacks
+ * __GFP_NOMEMALLOC. memalloc_noreclaim_save() ensures those
+ * allocations cannot recurse into swap by disabling __GFP_FS/IO.
+ */
+ noreclaim_flags = memalloc_noreclaim_save();
+
+ pages = kmalloc_array(npages, sizeof(*pages),
+ __GFP_HIGH | __GFP_NOMEMALLOC | GFP_KERNEL);
+ if (!pages) {
+ memalloc_noreclaim_restore(noreclaim_flags);
+ mutex_unlock(&si->xswap_lock);
+ return -ENOMEM;
+ }
+
+ for (i = 0; i < npages; i++) {
+ /*
+ * __GFP_ZERO is critical: cluster_info structs contain pointer
+ * fields (extend_table, zero_bitmap, memcg_table, table) that
+ * must start as NULL. Without zeroing, stale data from a
+ * previous user of the page would look like valid pointers.
+ */
+ pages[i] = alloc_page(__GFP_HIGH | __GFP_NOMEMALLOC |
+ GFP_KERNEL | __GFP_ZERO);
+ if (!pages[i])
+ goto fail;
+ }
+
+ if (vm_area_map_pages(si->cluster_vm, vm_start, vm_end, pages)) {
+ i = npages; /* free all pages on failure */
+ goto fail;
+ }
+
+ kfree(pages);
+ memalloc_noreclaim_restore(noreclaim_flags);
+
+ /* Initialize spinlocks for newly mapped clusters */
+ for (i = start_idx; i < start_idx + nr; i++)
+ spin_lock_init(&si->cluster_info[i].lock);
+
+ /*
+ * Pairs with READ_ONCE() in shrink/grow paths.
+ */
+ WRITE_ONCE(si->nr_clusters_mapped, start_idx + nr);
+ mutex_unlock(&si->xswap_lock);
+ return 0;
+
+fail:
+ while (i > 0) {
+ i--;
+ if (pages[i])
+ __free_page(pages[i]);
+ }
+ memalloc_noreclaim_restore(noreclaim_flags);
+ kfree(pages);
+ mutex_unlock(&si->xswap_lock);
+ return -ENOMEM;
+}
+
+static void xswap_unmap_clusters(struct swap_info_struct *si,
+ unsigned long start_idx, unsigned long nr)
+{
+ unsigned long start_addr = (unsigned long)si->cluster_info +
+ (size_t)start_idx * sizeof(struct swap_cluster_info);
+ unsigned long end_addr = start_addr + (size_t)nr * sizeof(struct swap_cluster_info);
+ /*
+ * Round to page boundaries: start up (skip partial page that may
+ * contain clusters still in use before start_idx), end up so the
+ * entire range is covered. vm_area_unmap_pages() operates on
+ * whole pages.
+ */
+ unsigned long vm_start = PAGE_ALIGN(start_addr);
+ unsigned long vm_end = PAGE_ALIGN(end_addr);
+
+ mutex_lock(&si->xswap_lock);
+
+ if (vm_start >= vm_end) {
+ mutex_unlock(&si->xswap_lock);
+ goto out;
+ }
+
+ vm_area_unmap_pages(si->cluster_vm, vm_start, vm_end);
+ /*
+ * vm_area_unmap_pages() only clears PTEs; it does not free the
+ * physical pages. Walk the page table to find and free them.
+ */
+ /* TODO: free backing pages via page table walk or tracking bitmap */
+ mutex_unlock(&si->xswap_lock);
+
+out:
+ /*
+ * Pairs with READ_ONCE() in shrink/grow paths.
+ */
+ WRITE_ONCE(si->nr_clusters_mapped, start_idx);
+}
+#endif /* CONFIG_XSWAP */
+
static int setup_swap_clusters_info(struct swap_info_struct *si,
union swap_header *swap_header,
unsigned long maxpages)
@@ -3573,6 +3747,64 @@ static int setup_swap_clusters_info(struct swap_info_struct *si,
int err = -ENOMEM;
unsigned long i;
+#ifdef CONFIG_XSWAP
+ if (si->flags & SWP_XSWAP) {
+ unsigned long size = PAGE_ALIGN(nr_clusters * sizeof(*cluster_info));
+ struct vm_struct *vm;
+
+ vm = get_vm_area(size, VM_SPARSE);
+ if (!vm)
+ goto err;
+
+ cluster_info = vm->addr;
+ si->cluster_vm = vm;
+ si->nr_clusters = nr_clusters;
+ si->cluster_info = cluster_info;
+
+ /* Map the initial chunk (at least cluster 0) */
+ if (xswap_map_clusters(si, 0, min_t(unsigned long,
+ XSWAP_GROW_CLUSTERS, nr_clusters)))
+ goto err_free_vm;
+
+ /* xswap: only cluster 0 slot 0 is bad */
+ err = swap_cluster_setup_bad_slot(si, cluster_info, 0, false);
+ if (err)
+ goto err_unmap;
+
+ INIT_LIST_HEAD(&si->free_clusters);
+ INIT_LIST_HEAD(&si->full_clusters);
+ INIT_LIST_HEAD(&si->discard_clusters);
+ for (i = 0; i < SWAP_NR_ORDERS; i++) {
+ INIT_LIST_HEAD(&si->nonfull_clusters[i]);
+ INIT_LIST_HEAD(&si->frag_clusters[i]);
+ }
+
+ /* Mark mapped clusters: cluster 0 has 1 bad slot, rest free */
+ for (i = 0; i < si->nr_clusters_mapped; i++) {
+ struct swap_cluster_info *ci = &cluster_info[i];
+
+ if (i == 0) {
+ ci->flags = CLUSTER_FLAG_NONFULL;
+ list_add_tail(&ci->list, &si->nonfull_clusters[0]);
+ } else {
+ ci->flags = CLUSTER_FLAG_FREE;
+ list_add_tail(&ci->list, &si->free_clusters);
+ }
+ }
+
+ mutex_init(&si->xswap_lock);
+ return 0;
+
+err_unmap:
+ xswap_unmap_clusters(si, 0, si->nr_clusters_mapped);
+err_free_vm:
+ free_vm_area(si->cluster_vm);
+ si->cluster_vm = NULL;
+ si->cluster_info = NULL;
+ return err;
+ }
+#endif /* CONFIG_XSWAP */
+
cluster_info = kvzalloc_objs(*cluster_info, nr_clusters);
if (!cluster_info)
goto err;
@@ -3640,7 +3872,7 @@ static int setup_swap_clusters_info(struct swap_info_struct *si,
si->cluster_info = cluster_info;
return 0;
err:
- free_swap_cluster_info(cluster_info, maxpages);
+ free_swap_cluster_info(si);
return err;
}
@@ -3859,7 +4091,7 @@ SYSCALL_DEFINE2(swapon, const char __user *, specialfile, int, swap_flags)
si->global_cluster = NULL;
inode = NULL;
destroy_swap_extents(si, swap_file);
- free_swap_cluster_info(si->cluster_info, si->max);
+ free_swap_cluster_info(si);
si->cluster_info = NULL;
/*
* Clear the SWP_USED flag after all resources are freed so
--
2.54.0
next prev parent reply other threads:[~2026-08-05 7:54 UTC|newest]
Thread overview: 15+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-08-05 7:53 [RFC PATCH v2 00/10] mm, swap: dynamic cluster management for xswap devices Baoquan He
2026-08-05 7:53 ` [RFC PATCH v2 01/10] mm: xswap support for zswap Baoquan He
2026-08-05 14:17 ` Johannes Weiner
2026-08-07 9:11 ` Baoquan He
2026-08-10 1:49 ` Youngjun Park
2026-08-10 7:25 ` Baoquan He
2026-08-05 7:53 ` [RFC PATCH v2 02/10] mm, swap: add CONFIG_XSWAP and xswap fields to swap_info_struct Baoquan He
2026-08-05 7:53 ` Baoquan He [this message]
2026-08-05 7:53 ` [RFC PATCH v2 04/10] mm, swap: add xswap grow trigger on cluster allocation Baoquan He
2026-08-05 7:53 ` [RFC PATCH v2 05/10] mm, swap: add xswap_try_shrink and shrink trigger on cluster free Baoquan He
2026-08-05 7:53 ` [RFC PATCH v2 06/10] mm, swap: free backing pages in xswap_unmap_clusters Baoquan He
2026-08-05 7:53 ` [RFC PATCH v2 07/10] mm, swap: add nr_free_tail for O(1) xswap shrink detection Baoquan He
2026-08-05 7:53 ` [RFC PATCH v2 08/10] mm, swap: add adjustable runtime ceiling (nr_clusters) for xswap Baoquan He
2026-08-05 7:53 ` [RFC PATCH v2 09/10] mm, swap: add debugfs knob for xswap per-device cluster limit Baoquan He
2026-08-05 7:53 ` [RFC PATCH v2 10/10] mm, swap: defer xswap shrink to workqueue to avoid lock recursion Baoquan He
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260805075336.3579395-4-baoquan.he@linux.dev \
--to=baoquan.he@linux.dev \
--cc=baohua@kernel.org \
--cc=chengming.zhou@linux.dev \
--cc=chrisl@kernel.org \
--cc=david@kernel.org \
--cc=hannes@cmpxchg.org \
--cc=kasong@tencent.com \
--cc=linux-kernel@vger.kernel.org \
--cc=linux-mm@kvack.org \
--cc=nphamcs@gmail.com \
--cc=shikemeng@huaweicloud.com \
--cc=yosry@kernel.org \
--cc=youngjun.park@lge.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.