From: Kiryl Shutsemau <kirill@shutemov.name>
To: Andrew Morton <akpm@linux-foundation.org>,
David Hildenbrand <david@kernel.org>,
Lorenzo Stoakes <ljs@kernel.org>
Cc: linux-mm@kvack.org, linux-kernel@vger.kernel.org,
kernel-team@meta.com, Zi Yan <ziy@nvidia.com>,
Baolin Wang <baolin.wang@linux.alibaba.com>,
"Liam R . Howlett" <liam@infradead.org>,
Nico Pache <nico.pache@linux.dev>,
Ryan Roberts <ryan.roberts@arm.com>, Dev Jain <dev.jain@arm.com>,
Barry Song <baohua@kernel.org>, Lance Yang <lance.yang@linux.dev>,
Usama Arif <usama.arif@linux.dev>,
Vlastimil Babka <vbabka@kernel.org>, Jann Horn <jannh@google.com>,
"Kiryl Shutsemau (Meta)" <kas@kernel.org>
Subject: [PATCH 10/12] mm/collapse: work out the orders a VMA allows once per VMA
Date: Fri, 4 Sep 2026 16:10:24 +0100 [thread overview]
Message-ID: <3ce85bb56f2bc60f91bf4e9645f247460f3c3f1e.1788533997.git.kas@kernel.org> (raw)
In-Reply-To: <cover.1788533997.git.kas@kernel.org>
From: "Kiryl Shutsemau (Meta)" <kas@kernel.org>
The scan asked collapse_possible_orders() for every PTE table, for an
answer that is a property of the VMA. Both callers walk a VMA a table at
a time, so let them work it out once and pass the mask in. It is only
good while the lock that produced it is held, so madvise_collapse() takes
it again after every collapse.
The mask is then sampled once per VMA rather than once per table. A thp
enabled knob written during a walk takes effect one VMA later, and cannot
widen a collapse: hugepage_vma_revalidate() tests the order again under
the lock the collapse retakes.
Assisted-by: Claude-Code:claude-opus-5
Signed-off-by: Kiryl Shutsemau (Meta) <kas@kernel.org>
---
mm/khugepaged.c | 32 ++++++++++++++++++--------------
1 file changed, 18 insertions(+), 14 deletions(-)
diff --git a/mm/khugepaged.c b/mm/khugepaged.c
index 40fcdd4f2712..f862abb1dbbd 100644
--- a/mm/khugepaged.c
+++ b/mm/khugepaged.c
@@ -1551,12 +1551,12 @@ static enum scan_result mthp_collapse(struct mm_struct *mm,
}
static enum scan_result collapse_scan_anon_pmd(struct vm_area_struct *vma,
- unsigned long start_addr, struct collapse_control *cc)
+ unsigned long start_addr, struct collapse_control *cc,
+ unsigned long enabled_orders)
{
const unsigned int max_ptes_shared = collapse_max_ptes_shared(cc, HPAGE_PMD_ORDER);
const unsigned int max_ptes_swap = collapse_max_ptes_swap(cc, HPAGE_PMD_ORDER);
unsigned int max_ptes_none = collapse_max_ptes_none(cc, vma, HPAGE_PMD_ORDER);
- enum tva_type tva_flags = cc->policy.tva_type;
struct mm_struct *mm = vma->vm_mm;
pmd_t *pmd;
pte_t *pte, *_pte, pteval;
@@ -1567,7 +1567,6 @@ static enum scan_result collapse_scan_anon_pmd(struct vm_area_struct *vma,
struct folio *folio = NULL;
unsigned long failed_pfn = -1;
unsigned long addr;
- unsigned long enabled_orders;
spinlock_t *ptl;
int node = NUMA_NO_NODE, unmapped = 0;
@@ -1581,8 +1580,6 @@ static enum scan_result collapse_scan_anon_pmd(struct vm_area_struct *vma,
collapse_scan_reset(cc);
- enabled_orders = collapse_possible_orders(vma, vma->vm_flags, tva_flags);
-
/*
* If PMD is the only enabled order, enforce max_ptes_none, otherwise
* scan all pages to populate the bitmap for mTHP collapse. The bitmap
@@ -2773,7 +2770,8 @@ static void collapse_control_release(struct collapse_control *cc)
}
static enum scan_result collapse_scan_pmd(struct vm_area_struct *vma,
- unsigned long addr, struct collapse_control *cc)
+ unsigned long addr, struct collapse_control *cc,
+ unsigned long orders)
{
mmap_assert_locked(vma->vm_mm);
/* Whatever the last scan found has to have been run by now */
@@ -2783,7 +2781,7 @@ static enum scan_result collapse_scan_pmd(struct vm_area_struct *vma,
}
if (vma_is_anonymous(vma))
- return collapse_scan_anon_pmd(vma, addr, cc);
+ return collapse_scan_anon_pmd(vma, addr, cc, orders);
/*
* A file collapse works on the page cache and never sees a VMA, so take
@@ -2877,15 +2875,17 @@ static void collapse_scan_mm_slot(unsigned int progress_max,
vma_iter_init(&vmi, mm, khugepaged_scan.address);
for_each_vma(vmi, vma) {
- unsigned long hstart, hend;
+ unsigned long hstart, hend, orders;
cond_resched();
if (unlikely(collapse_test_exit_or_disable(mm))) {
cc->progress++;
break;
}
- if (!collapse_possible_orders(vma, vma->vm_flags,
- TVA_KHUGEPAGED)) {
+ /* One mask for the whole VMA */
+ orders = collapse_possible_orders(vma, vma->vm_flags,
+ cc->policy.tva_type);
+ if (!orders) {
cc->progress++;
continue;
}
@@ -2914,7 +2914,7 @@ static void collapse_scan_mm_slot(unsigned int progress_max,
/* move to next address */
khugepaged_scan.address += HPAGE_PMD_SIZE;
- *result = collapse_scan_pmd(vma, addr, cc);
+ *result = collapse_scan_pmd(vma, addr, cc, orders);
/* Nothing to collapse here, and the lock is still ours */
if (*result != SCAN_SUCCEED) {
if (cc->progress >= progress_max)
@@ -3198,14 +3198,16 @@ int madvise_collapse(struct vm_area_struct *vma, unsigned long start,
{
struct collapse_control *cc;
struct mm_struct *mm = vma->vm_mm;
- unsigned long hstart, hend, addr;
+ unsigned long hstart, hend, addr, orders;
enum scan_result last_fail = SCAN_FAIL;
int thps = 0;
BUG_ON(vma->vm_start > start);
BUG_ON(vma->vm_end < end);
- if (!collapse_possible_orders(vma, vma->vm_flags, TVA_FORCED_COLLAPSE))
+ orders = collapse_possible_orders(vma, vma->vm_flags,
+ TVA_FORCED_COLLAPSE);
+ if (!orders)
return -EINVAL;
hstart = ALIGN(start, HPAGE_PMD_SIZE);
@@ -3243,9 +3245,11 @@ int madvise_collapse(struct vm_area_struct *vma, unsigned long start,
}
vma = found;
hend = min(hend, vma->vm_end & HPAGE_PMD_MASK);
+ orders = collapse_possible_orders(vma, vma->vm_flags,
+ cc->policy.tva_type);
}
- result = collapse_scan_pmd(vma, addr, cc);
+ result = collapse_scan_pmd(vma, addr, cc, orders);
/* Nothing to collapse here, and the lock is still ours */
if (result != SCAN_SUCCEED)
goto tally;
--
2.54.0
next prev parent reply other threads:[~2026-09-04 15:11 UTC|newest]
Thread overview: 37+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-04 15:10 [PATCH 00/12] mm/collapse: separate a collapse from its callers Kiryl Shutsemau
2026-09-04 15:10 ` [PATCH 01/12] mm/khugepaged: drop redundant mm_struct pin in madvise_collapse() Kiryl Shutsemau
2026-09-04 15:58 ` Zi Yan
2026-09-07 7:33 ` Baolin Wang
2026-09-04 15:10 ` [PATCH 02/12] mm/khugepaged: count collapses where khugepaged makes them Kiryl Shutsemau
2026-09-05 2:25 ` Zi Yan
2026-09-07 7:40 ` Baolin Wang
2026-09-04 15:10 ` [PATCH 03/12] mm/khugepaged: rename mthp_present_ptes bitmap to eligible_ptes Kiryl Shutsemau
2026-09-05 2:28 ` Zi Yan
2026-09-07 7:54 ` Baolin Wang
2026-09-07 10:35 ` Kiryl Shutsemau
2026-09-04 15:10 ` [PATCH 04/12] mm/collapse: add collapse.h for the collapse interface Kiryl Shutsemau
2026-09-05 2:36 ` Zi Yan
2026-09-07 10:41 ` Kiryl Shutsemau
2026-09-07 8:04 ` Baolin Wang
2026-09-04 15:10 ` [PATCH 05/12] mm/collapse: state what a collapse may do in the policy Kiryl Shutsemau
2026-09-05 2:44 ` Zi Yan
2026-09-07 10:49 ` Kiryl Shutsemau
2026-09-07 19:40 ` Zi Yan
2026-09-07 9:05 ` Baolin Wang
2026-09-07 10:56 ` Kiryl Shutsemau
2026-09-08 1:48 ` Baolin Wang
2026-09-04 15:10 ` [PATCH 06/12] mm/collapse: drop the collapse_possible() wrapper Kiryl Shutsemau
2026-09-05 2:45 ` Zi Yan
2026-09-07 8:28 ` Baolin Wang
2026-09-04 15:10 ` [PATCH 07/12] mm/collapse: name the per-table scan reset for what it resets Kiryl Shutsemau
2026-09-05 18:05 ` Zi Yan
2026-09-07 8:31 ` Baolin Wang
2026-09-04 15:10 ` [PATCH 08/12] mm/collapse: separate scanning a PTE table from collapsing it Kiryl Shutsemau
2026-09-06 2:30 ` Zi Yan
2026-09-07 11:34 ` Kiryl Shutsemau
2026-09-04 15:10 ` [PATCH 09/12] mm/collapse: open-code collapse_single_pmd() in its two callers Kiryl Shutsemau
2026-09-04 15:10 ` Kiryl Shutsemau [this message]
2026-09-04 15:10 ` [PATCH 11/12] mm/collapse: declare the collapse interface in collapse.h Kiryl Shutsemau
2026-09-04 15:10 ` [PATCH 12/12] mm/collapse: implement MADV_COLLAPSE in madvise.c Kiryl Shutsemau
2026-09-06 0:23 ` [PATCH 00/12] mm/collapse: separate a collapse from its callers Andrew Morton
2026-09-07 10:28 ` Kiryl Shutsemau
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=3ce85bb56f2bc60f91bf4e9645f247460f3c3f1e.1788533997.git.kas@kernel.org \
--to=kirill@shutemov.name \
--cc=akpm@linux-foundation.org \
--cc=baohua@kernel.org \
--cc=baolin.wang@linux.alibaba.com \
--cc=david@kernel.org \
--cc=dev.jain@arm.com \
--cc=jannh@google.com \
--cc=kas@kernel.org \
--cc=kernel-team@meta.com \
--cc=lance.yang@linux.dev \
--cc=liam@infradead.org \
--cc=linux-kernel@vger.kernel.org \
--cc=linux-mm@kvack.org \
--cc=ljs@kernel.org \
--cc=nico.pache@linux.dev \
--cc=ryan.roberts@arm.com \
--cc=usama.arif@linux.dev \
--cc=vbabka@kernel.org \
--cc=ziy@nvidia.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.