From: Gregory Price <gourry@gourry.net>
To: linux-mm@kvack.org
Cc: linux-kernel@vger.kernel.org, linux-kselftest@vger.kernel.org,
kernel-team@meta.com, akpm@linux-foundation.org,
liam@infradead.org, ljs@kernel.org, david@kernel.org,
vbabka@kernel.org, jannh@google.com, rppt@kernel.org,
surenb@google.com, mhocko@suse.com, shuah@kernel.org,
"Gregory Price (Meta)" <gourry@gourry.net>
Subject: [PATCH 07/10] mm/madvise: separate PTE-batch folio processing
Date: Tue, 22 Sep 2026 19:58:27 -0400 [thread overview]
Message-ID: <20260922235830.2350770-8-gourry@gourry.net> (raw)
In-Reply-To: <20260922235830.2350770-1-gourry@gourry.net>
The PTE loop combines page-table iteration with validation and
processing of each folio batch. This hides the batching rule for large
folios and the ownership transferred when a partial folio must be split.
Move one batch into a PTL-locked helper. A split candidate is returned
locked and referenced so the caller can release the PTE mapping before
calling split_folio().
No functional change intended.
Assisted-by: LLM
Signed-off-by: Gregory Price (Meta) <gourry@gourry.net>
---
mm/madvise.c | 132 +++++++++++++++++++++++++--------------------------
1 file changed, 65 insertions(+), 67 deletions(-)
diff --git a/mm/madvise.c b/mm/madvise.c
index 83b27258c9673..350b854ccc197 100644
--- a/mm/madvise.c
+++ b/mm/madvise.c
@@ -394,6 +394,54 @@ static bool madvise_lru_folio_is_filtered(struct folio *folio,
(pageout_anon_only && !folio_test_anon(folio));
}
+/* Return a split candidate locked and referenced for use after the PTL drop. */
+static struct folio *
+madvise_lru_pte_batch_locked(pte_t *pte, unsigned long addr,
+ unsigned long end, struct mm_walk *walk,
+ struct list_head *folio_list, bool pageout_anon_only, int *nr)
+{
+ const struct madvise_walk_private *private = walk->private;
+ struct vm_area_struct *vma = walk->vma;
+ struct folio *folio;
+ pte_t ptent;
+
+ *nr = 1;
+ ptent = ptep_get(pte);
+ if (pte_none(ptent) || !pte_present(ptent))
+ return NULL;
+
+ folio = vm_normal_folio(vma, addr, ptent);
+ if (!folio || folio_is_zone_device(folio))
+ return NULL;
+
+ /* Split PTE-mapped large folios before advising only part of them. */
+ if (folio_test_large(folio)) {
+ *nr = madvise_folio_pte_batch(addr, end, folio, pte, &ptent);
+ if (*nr < folio_nr_pages(folio)) {
+ if (madvise_lru_folio_is_filtered(folio, pageout_anon_only))
+ return NULL;
+ if (!folio_trylock(folio))
+ return NULL;
+ folio_get(folio);
+ return folio;
+ }
+ }
+
+ if (!folio_test_lru(folio) ||
+ folio_mapcount(folio) != folio_nr_pages(folio))
+ return NULL;
+ if (pageout_anon_only && !folio_test_anon(folio))
+ return NULL;
+
+ if (!private->pageout && pte_young(ptent)) {
+ clear_young_dirty_ptes(vma, addr, pte, *nr, CYDP_CLEAR_YOUNG);
+ tlb_remove_tlb_entries(private->tlb, pte, *nr, addr);
+ }
+
+ madvise_lru_folio(folio, private->pageout, folio_list);
+ return NULL;
+}
+
#ifdef CONFIG_TRANSPARENT_HUGEPAGE
static void madvise_cold_pmd(struct mmu_gather *tlb, struct vm_area_struct *vma,
pmd_t *pmd, unsigned long addr, pmd_t orig_pmd)
@@ -491,7 +539,7 @@ static int madvise_lru_pmd_entry(pmd_t *pmd, unsigned long addr,
bool pageout = private->pageout;
struct mm_struct *mm = tlb->mm;
struct vm_area_struct *vma = walk->vma;
- pte_t *start_pte, *pte, ptent;
+ pte_t *start_pte, *pte;
spinlock_t *ptl;
struct folio *folio = NULL;
LIST_HEAD(folio_list);
@@ -515,9 +563,6 @@ static int madvise_lru_pmd_entry(pmd_t *pmd, unsigned long addr,
flush_tlb_batched_pending(mm);
lazy_mmu_mode_enable();
for (; addr < end; pte += nr, addr += nr * PAGE_SIZE) {
- nr = 1;
- ptent = ptep_get(pte);
-
if (++batch_count == SWAP_CLUSTER_MAX) {
batch_count = 0;
if (need_resched()) {
@@ -528,71 +573,24 @@ static int madvise_lru_pmd_entry(pmd_t *pmd, unsigned long addr,
}
}
- if (pte_none(ptent))
- continue;
-
- if (!pte_present(ptent))
- continue;
-
- folio = vm_normal_folio(vma, addr, ptent);
- if (!folio || folio_is_zone_device(folio))
- continue;
-
- /*
- * If we encounter a large folio, only split it if it is not
- * fully mapped within the range we are operating on. Otherwise
- * leave it as is so that it can be swapped out whole. If we
- * fail to split a folio, leave it in place and advance to the
- * next pte in the range.
- */
- if (folio_test_large(folio)) {
- nr = madvise_folio_pte_batch(addr, end, folio, pte, &ptent);
- if (nr < folio_nr_pages(folio)) {
- int err;
-
- if (madvise_lru_folio_is_filtered(folio, pageout_anon_only))
- continue;
- if (!folio_trylock(folio))
- continue;
- folio_get(folio);
- lazy_mmu_mode_disable();
- pte_unmap_unlock(start_pte, ptl);
- start_pte = NULL;
- err = split_folio(folio);
- folio_unlock(folio);
- folio_put(folio);
- start_pte = pte =
- pte_offset_map_lock(mm, pmd, addr, &ptl);
- if (!start_pte)
- break;
- flush_tlb_batched_pending(mm);
- lazy_mmu_mode_enable();
- if (!err)
- nr = 0;
- continue;
- }
- }
-
- /*
- * Do not interfere with other mappings of this folio and
- * non-LRU folio. If we have a large folio at this point, we
- * know it is fully mapped so if its mapcount is the same as its
- * number of pages, it must be exclusive.
- */
- if (!folio_test_lru(folio) ||
- folio_mapcount(folio) != folio_nr_pages(folio))
- continue;
-
- if (pageout_anon_only && !folio_test_anon(folio))
+ folio = madvise_lru_pte_batch_locked(pte, addr, end, walk,
+ &folio_list, pageout_anon_only, &nr);
+ if (!folio)
continue;
- if (!pageout && pte_young(ptent)) {
- clear_young_dirty_ptes(vma, addr, pte, nr,
- CYDP_CLEAR_YOUNG);
- tlb_remove_tlb_entries(tlb, pte, nr, addr);
- }
-
- madvise_lru_folio(folio, pageout, &folio_list);
+ lazy_mmu_mode_disable();
+ pte_unmap_unlock(start_pte, ptl);
+ start_pte = NULL;
+ if (!split_folio(folio))
+ nr = 0;
+ folio_unlock(folio);
+ folio_put(folio);
+ start_pte = pte_offset_map_lock(mm, pmd, addr, &ptl);
+ if (!start_pte)
+ break;
+ pte = start_pte;
+ flush_tlb_batched_pending(mm);
+ lazy_mmu_mode_enable();
}
out:
--
2.53.0-Meta
next prev parent reply other threads:[~2026-09-22 23:59 UTC|newest]
Thread overview: 23+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-22 23:58 [PATCH 00/10] mm/madvise: refactor cold and pageout page table walks Gregory Price
2026-09-22 23:58 ` [PATCH 01/10] selftests/mm: exercise MADV_COLD and MADV_PAGEOUT Gregory Price
2026-09-23 14:26 ` Lorenzo Stoakes (ARM)
2026-09-23 14:44 ` Gregory Price
2026-09-23 14:46 ` Lorenzo Stoakes (ARM)
2026-09-24 11:32 ` David Hildenbrand (Arm)
2026-09-24 14:01 ` Gregory Price
2026-09-22 23:58 ` [PATCH 02/10] mm/madvise: name the shared LRU PMD callback Gregory Price
2026-09-23 14:44 ` Lorenzo Stoakes (ARM)
2026-09-22 23:58 ` [PATCH 03/10] mm/madvise: factor shared LRU folio handling Gregory Price
2026-09-23 16:00 ` Lorenzo Stoakes (ARM)
2026-09-22 23:58 ` [PATCH 04/10] mm/madvise: use the PMD softleaf validity helper Gregory Price
2026-09-23 16:02 ` Lorenzo Stoakes (ARM)
2026-09-22 23:58 ` [PATCH 05/10] mm/madvise: factor huge-PMD folio processing Gregory Price
2026-09-23 16:43 ` Lorenzo Stoakes (ARM)
2026-09-23 17:06 ` Gregory Price
2026-09-23 17:14 ` Lorenzo Stoakes (ARM)
2026-09-23 17:26 ` Gregory Price
2026-09-22 23:58 ` [PATCH 06/10] mm/madvise: separate huge PMDs from the PTE walk Gregory Price
2026-09-22 23:58 ` Gregory Price [this message]
2026-09-22 23:58 ` [PATCH 08/10] mm/madvise: separate the PTL-held PTE scan Gregory Price
2026-09-22 23:58 ` [PATCH 09/10] mm/madvise: make cold and pageout PTE lock ownership explicit Gregory Price
2026-09-22 23:58 ` [PATCH 10/10] mm/madvise: share cold and pageout walk setup Gregory Price
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260922235830.2350770-8-gourry@gourry.net \
--to=gourry@gourry.net \
--cc=akpm@linux-foundation.org \
--cc=david@kernel.org \
--cc=jannh@google.com \
--cc=kernel-team@meta.com \
--cc=liam@infradead.org \
--cc=linux-kernel@vger.kernel.org \
--cc=linux-kselftest@vger.kernel.org \
--cc=linux-mm@kvack.org \
--cc=ljs@kernel.org \
--cc=mhocko@suse.com \
--cc=rppt@kernel.org \
--cc=shuah@kernel.org \
--cc=surenb@google.com \
--cc=vbabka@kernel.org \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox