* [PATCH v2 1/6] mm/memcontrol: make lru_zone_size atomic and simplify sanity check
2026-08-24 15:26 [PATCH v2 0/6] mm/mglru: clean up folio counters and flag usage Kairui Song via B4 Relay
@ 2026-08-24 15:26 ` Kairui Song via B4 Relay
2026-08-25 1:48 ` Baoquan He
2026-08-24 15:26 ` [PATCH v2 2/6] mm/mglru: introduce helpers for manipulating gen and refs flags Kairui Song via B4 Relay
` (4 subsequent siblings)
5 siblings, 1 reply; 9+ messages in thread
From: Kairui Song via B4 Relay @ 2026-08-24 15:26 UTC (permalink / raw)
To: linux-mm
Cc: Andrew Morton, Barry Song, Axel Rasmussen, Yuanchu Xie, Wei Xu,
Baoquan He, Shakeel Butt, Johannes Weiner, Michal Hocko,
Roman Gushchin, Muchun Song, Chris Li, Baolin Wang, Ridong Chen,
David Hildenbrand, Lorenzo Stoakes, Liam R. Howlett,
Vlastimil Babka, Yu Zhao, Zi Yan, Qi Zheng, cgroups, linux-kernel,
Kairui Song, Kairui Song
From: Kairui Song <kasong@tencent.com>
commit ca707239e8a7 ("mm: update_lru_size warn and reset bad lru_size")
introduced a sanity check to catch memcg counter underflow, which was
more of a workaround for another bug: lru_zone_size is unsigned, so
underflow wraps it around and returns an enormously large number, then
the memcg shrinker loops almost forever as the calculated number of
folios to shrink is huge. That commit also checked if a zero value
matches the empty LRU list, so we have to hold the LRU lock, and
handle the positive and negative deltas separately.
But later commit b4536f0c829c ("mm, memcg: fix the active list aging
for lowmem requests when memcg is enabled") already removed the LRU
emptiness check, so handling the deltas separately is no longer
needed. And if we just turn it into an atomic long, underflow isn't a
big issue either, and can be checked at the reader side, which is
called much less frequently than the updater.
So let's turn the counter into an atomic long and check at the reader
side instead, which has a smaller overhead. The underflow correction
is removed: a massive leak of the LRU size counter would indicate
that something else has gone very wrong, and one should fix that
leaking site instead. Besides, the updater-side sanity check is
unlikely to catch the leaking site anyway: if a folio was removed
without updating the counter while other folios remain on the LRU,
the WARN only triggers much later, from a likely innocent callsite.
Reviewed-by: Ridong Chen <ridong.chen@linux.dev>
Signed-off-by: Kairui Song <kasong@tencent.com>
---
include/linux/memcontrol.h | 9 +++++++--
mm/memcontrol.c | 18 +-----------------
2 files changed, 8 insertions(+), 19 deletions(-)
diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h
index 215e2e87f42b..7b89d0cb5f6c 100644
--- a/include/linux/memcontrol.h
+++ b/include/linux/memcontrol.h
@@ -113,7 +113,7 @@ struct mem_cgroup_per_node {
/* Fields which get updated often at the end. */
struct lruvec lruvec;
CACHELINE_PADDING(_pad2_);
- unsigned long lru_zone_size[MAX_NR_ZONES][NR_LRU_LISTS];
+ atomic_long_t lru_zone_size[MAX_NR_ZONES][NR_LRU_LISTS];
struct mem_cgroup_reclaim_iter iter;
/*
@@ -902,10 +902,15 @@ static inline
unsigned long mem_cgroup_get_zone_lru_size(struct lruvec *lruvec,
enum lru_list lru, int zone_idx)
{
+ long val;
struct mem_cgroup_per_node *mz;
mz = container_of(lruvec, struct mem_cgroup_per_node, lruvec);
- return READ_ONCE(mz->lru_zone_size[zone_idx][lru]);
+ val = atomic_long_read(&mz->lru_zone_size[zone_idx][lru]);
+ if (WARN_ON_ONCE(val < 0))
+ return 0;
+
+ return val;
}
void __mem_cgroup_handle_over_high(gfp_t gfp_mask);
diff --git a/mm/memcontrol.c b/mm/memcontrol.c
index 11b85f4b6828..a7572ded56c9 100644
--- a/mm/memcontrol.c
+++ b/mm/memcontrol.c
@@ -1529,28 +1529,12 @@ void mem_cgroup_update_lru_size(struct lruvec *lruvec, enum lru_list lru,
int zid, long nr_pages)
{
struct mem_cgroup_per_node *mz;
- unsigned long *lru_size;
- long size;
if (mem_cgroup_disabled())
return;
mz = container_of(lruvec, struct mem_cgroup_per_node, lruvec);
- lru_size = &mz->lru_zone_size[zid][lru];
-
- if (nr_pages < 0)
- *lru_size += nr_pages;
-
- size = *lru_size;
- if (WARN_ONCE(size < 0,
- "%s(%p, %d, %ld): lru_size %ld\n",
- __func__, lruvec, lru, nr_pages, size)) {
- VM_BUG_ON(1);
- *lru_size = 0;
- }
-
- if (nr_pages > 0)
- *lru_size += nr_pages;
+ atomic_long_add(nr_pages, &mz->lru_zone_size[zid][lru]);
}
/**
--
2.55.0
^ permalink raw reply related [flat|nested] 9+ messages in thread* Re: [PATCH v2 1/6] mm/memcontrol: make lru_zone_size atomic and simplify sanity check
2026-08-24 15:26 ` [PATCH v2 1/6] mm/memcontrol: make lru_zone_size atomic and simplify sanity check Kairui Song via B4 Relay
@ 2026-08-25 1:48 ` Baoquan He
0 siblings, 0 replies; 9+ messages in thread
From: Baoquan He @ 2026-08-25 1:48 UTC (permalink / raw)
To: kasong
Cc: linux-mm, Andrew Morton, Barry Song, Axel Rasmussen, Yuanchu Xie,
Wei Xu, Shakeel Butt, Johannes Weiner, Michal Hocko,
Roman Gushchin, Muchun Song, Chris Li, Baolin Wang, Ridong Chen,
David Hildenbrand, Lorenzo Stoakes, Liam R. Howlett,
Vlastimil Babka, Yu Zhao, Zi Yan, Qi Zheng, cgroups, linux-kernel,
Kairui Song
On 08/24/26 at 11:26pm, Kairui Song via B4 Relay wrote:
> From: Kairui Song <kasong@tencent.com>
>
> commit ca707239e8a7 ("mm: update_lru_size warn and reset bad lru_size")
> introduced a sanity check to catch memcg counter underflow, which was
> more of a workaround for another bug: lru_zone_size is unsigned, so
> underflow wraps it around and returns an enormously large number, then
> the memcg shrinker loops almost forever as the calculated number of
> folios to shrink is huge. That commit also checked if a zero value
> matches the empty LRU list, so we have to hold the LRU lock, and
> handle the positive and negative deltas separately.
>
> But later commit b4536f0c829c ("mm, memcg: fix the active list aging
> for lowmem requests when memcg is enabled") already removed the LRU
> emptiness check, so handling the deltas separately is no longer
> needed. And if we just turn it into an atomic long, underflow isn't a
> big issue either, and can be checked at the reader side, which is
> called much less frequently than the updater.
>
> So let's turn the counter into an atomic long and check at the reader
> side instead, which has a smaller overhead. The underflow correction
> is removed: a massive leak of the LRU size counter would indicate
> that something else has gone very wrong, and one should fix that
> leaking site instead. Besides, the updater-side sanity check is
> unlikely to catch the leaking site anyway: if a folio was removed
> without updating the counter while other folios remain on the LRU,
> the WARN only triggers much later, from a likely innocent callsite.
>
> Reviewed-by: Ridong Chen <ridong.chen@linux.dev>
> Signed-off-by: Kairui Song <kasong@tencent.com>
> ---
> include/linux/memcontrol.h | 9 +++++++--
> mm/memcontrol.c | 18 +-----------------
> 2 files changed, 8 insertions(+), 19 deletions(-)
>
> diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h
> index 215e2e87f42b..7b89d0cb5f6c 100644
> --- a/include/linux/memcontrol.h
> +++ b/include/linux/memcontrol.h
> @@ -113,7 +113,7 @@ struct mem_cgroup_per_node {
> /* Fields which get updated often at the end. */
> struct lruvec lruvec;
> CACHELINE_PADDING(_pad2_);
> - unsigned long lru_zone_size[MAX_NR_ZONES][NR_LRU_LISTS];
> + atomic_long_t lru_zone_size[MAX_NR_ZONES][NR_LRU_LISTS];
> struct mem_cgroup_reclaim_iter iter;
>
> /*
> @@ -902,10 +902,15 @@ static inline
> unsigned long mem_cgroup_get_zone_lru_size(struct lruvec *lruvec,
> enum lru_list lru, int zone_idx)
> {
> + long val;
> struct mem_cgroup_per_node *mz;
>
> mz = container_of(lruvec, struct mem_cgroup_per_node, lruvec);
> - return READ_ONCE(mz->lru_zone_size[zone_idx][lru]);
> + val = atomic_long_read(&mz->lru_zone_size[zone_idx][lru]);
> + if (WARN_ON_ONCE(val < 0))
> + return 0;
> +
> + return val;
> }
>
> void __mem_cgroup_handle_over_high(gfp_t gfp_mask);
> diff --git a/mm/memcontrol.c b/mm/memcontrol.c
> index 11b85f4b6828..a7572ded56c9 100644
> --- a/mm/memcontrol.c
> +++ b/mm/memcontrol.c
> @@ -1529,28 +1529,12 @@ void mem_cgroup_update_lru_size(struct lruvec *lruvec, enum lru_list lru,
> int zid, long nr_pages)
> {
> struct mem_cgroup_per_node *mz;
> - unsigned long *lru_size;
> - long size;
>
> if (mem_cgroup_disabled())
> return;
>
> mz = container_of(lruvec, struct mem_cgroup_per_node, lruvec);
> - lru_size = &mz->lru_zone_size[zid][lru];
> -
> - if (nr_pages < 0)
> - *lru_size += nr_pages;
> -
> - size = *lru_size;
> - if (WARN_ONCE(size < 0,
> - "%s(%p, %d, %ld): lru_size %ld\n",
> - __func__, lruvec, lru, nr_pages, size)) {
> - VM_BUG_ON(1);
> - *lru_size = 0;
If it happened, the resetting to 0 is removed, will it cause anything
unexpected and different behaviour? mem_cgroup_get_zone_lru_size() just
read minus value and always WARN_ON_ONCE() if no new adding to this
counter.
> - }
> -
> - if (nr_pages > 0)
> - *lru_size += nr_pages;
> + atomic_long_add(nr_pages, &mz->lru_zone_size[zid][lru]);
> }
>
> /**
>
> --
> 2.55.0
>
>
^ permalink raw reply [flat|nested] 9+ messages in thread
* [PATCH v2 2/6] mm/mglru: introduce helpers for manipulating gen and refs flags
2026-08-24 15:26 [PATCH v2 0/6] mm/mglru: clean up folio counters and flag usage Kairui Song via B4 Relay
2026-08-24 15:26 ` [PATCH v2 1/6] mm/memcontrol: make lru_zone_size atomic and simplify sanity check Kairui Song via B4 Relay
@ 2026-08-24 15:26 ` Kairui Song via B4 Relay
2026-08-24 16:30 ` Kairui Song
2026-08-24 15:26 ` [PATCH v2 3/6] mm/migrate: copy all referenced state via folio_migrate_lru_refs Kairui Song via B4 Relay
` (3 subsequent siblings)
5 siblings, 1 reply; 9+ messages in thread
From: Kairui Song via B4 Relay @ 2026-08-24 15:26 UTC (permalink / raw)
To: linux-mm
Cc: Andrew Morton, Barry Song, Axel Rasmussen, Yuanchu Xie, Wei Xu,
Baoquan He, Shakeel Butt, Johannes Weiner, Michal Hocko,
Roman Gushchin, Muchun Song, Chris Li, Baolin Wang, Ridong Chen,
David Hildenbrand, Lorenzo Stoakes, Liam R. Howlett,
Vlastimil Babka, Yu Zhao, Zi Yan, Qi Zheng, cgroups, linux-kernel,
Kairui Song, Kairui Song
From: Kairui Song <kasong@tencent.com>
Instead of doing bit ops on folio->flags.f, introduce helpers for
adjusting a folio's refs and generation info, making the code easier
to debug and understand.
No functional change is intended: some combined atomic operations are
split into two, which only creates harmless transient states. There is
no measurable performance impact, and some paths even look slightly
better in the generated assembly.
Signed-off-by: Kairui Song <kasong@tencent.com>
---
include/linux/mm_inline.h | 76 ++++++++++++++++++++++++++++++++++++++++++-----
include/linux/mmzone.h | 1 +
mm/folio.c | 19 +++++++-----
mm/vmscan.c | 61 ++++++++++++++++++++-----------------
4 files changed, 114 insertions(+), 43 deletions(-)
diff --git a/include/linux/mm_inline.h b/include/linux/mm_inline.h
index 621c8653d8f7..edfaf2661812 100644
--- a/include/linux/mm_inline.h
+++ b/include/linux/mm_inline.h
@@ -142,10 +142,42 @@ static inline int lru_tier_from_refs(int refs, bool workingset)
return workingset ? MAX_NR_TIERS - 1 : order_base_2(refs);
}
-static inline int folio_lru_refs(const struct folio *folio)
+/**
+ * lru_gen_from_flags - Return the LRU generation number from folio flags.
+ * @flags: folio flags
+ *
+ * Returns: A number between 0 and (MAX_NR_GENS - 1), inclusive. Returns
+ * -1 if the flags indicate the folio is off the list (e.g., isolated).
+ */
+static inline int lru_gen_from_flags(unsigned long flags)
+{
+ int gen = ((flags & LRU_GEN_MASK) >> LRU_GEN_PGOFF);
+
+ BUILD_BUG_ON(LRU_GEN_MASK & LRU_REFS_MASK);
+ gen -= 1;
+ VM_WARN_ON_ONCE(gen != -1 && gen >= MAX_NR_GENS);
+ return gen;
+}
+
+/**
+ * lru_gen_set_flags - Set the LRU generation number to specified folio flags.
+ * @flags: pointer to the folio flags
+ * @gen: generation number, between 0 and (MAX_NR_GENS - 1), inclusive.
+ */
+static inline void lru_gen_set_flags(unsigned long *flags, int gen)
{
- unsigned long flags = READ_ONCE(folio->flags.f);
+ VM_WARN_ON_ONCE(gen >= MAX_NR_GENS || gen < 0);
+
+ *flags &= ~LRU_GEN_MASK;
+ *flags |= (gen + 1UL) << LRU_GEN_PGOFF;
+}
+/**
+ * lru_refs_from_flags - Return LRU referenced / access count from folio flags.
+ * @flags: folio flags
+ */
+static inline int lru_refs_from_flags(unsigned long flags)
+{
if (!(flags & BIT(PG_referenced)))
return 0;
/*
@@ -155,11 +187,40 @@ static inline int folio_lru_refs(const struct folio *folio)
return ((flags & LRU_REFS_MASK) >> LRU_REFS_PGOFF) + 1;
}
-static inline int folio_lru_gen(const struct folio *folio)
+/**
+ * lru_refs_set_flags - Set the LRU referenced / access count to specified folio flags.
+ * @flags: pointer to the folio flags
+ * @refs: referenced / access count number, between 0 and LRU_REFS_MAX, inclusive.
+ */
+static inline void lru_refs_set_flags(unsigned long *flags, unsigned int refs)
+{
+ VM_WARN_ON_ONCE(refs > LRU_REFS_MAX);
+ BUILD_BUG_ON(LRU_REFS_MAX != (LRU_REFS_MASK >> LRU_REFS_PGOFF) + 1);
+
+ *flags &= ~LRU_REFS_FLAGS;
+ if (!refs)
+ return;
+ *flags |= (BIT(PG_referenced) | ((refs - 1UL) << LRU_REFS_PGOFF));
+}
+
+static inline int folio_lru_refs(const struct folio *folio)
{
- unsigned long flags = READ_ONCE(folio->flags.f);
+ return lru_refs_from_flags(READ_ONCE(*const_folio_flags(folio, 0)));
+}
+
+static inline void folio_set_lru_refs(struct folio *folio, unsigned int refs)
+{
+ unsigned long new_flags, old_flags = READ_ONCE(*folio_flags(folio, 0));
+
+ do {
+ new_flags = old_flags;
+ lru_refs_set_flags(&new_flags, refs);
+ } while (!try_cmpxchg(folio_flags(folio, 0), &old_flags, new_flags));
+}
- return ((flags & LRU_GEN_MASK) >> LRU_GEN_PGOFF) - 1;
+static inline int folio_lru_gen(const struct folio *folio)
+{
+ return lru_gen_from_flags(READ_ONCE(*const_folio_flags(folio, 0)));
}
static inline bool lru_gen_is_active(const struct lruvec *lruvec, int gen)
@@ -270,7 +331,7 @@ static inline bool lru_gen_add_folio(struct lruvec *lruvec, struct folio *folio,
gen = lru_gen_from_seq(seq);
flags = (gen + 1UL) << LRU_GEN_PGOFF;
/* see the comment on MIN_NR_GENS about PG_active */
- set_mask_bits(&folio->flags.f, LRU_GEN_MASK | BIT(PG_active), flags);
+ set_mask_bits(folio_flags(folio, 0), LRU_GEN_MASK | BIT(PG_active), flags);
lru_gen_update_size(lruvec, folio, -1, gen);
/* for folio_rotate_reclaimable() */
@@ -295,7 +356,7 @@ static inline bool lru_gen_del_folio(struct lruvec *lruvec, struct folio *folio,
/* for folio_migrate_flags() */
flags = !reclaiming && lru_gen_is_active(lruvec, gen) ? BIT(PG_active) : 0;
- flags = set_mask_bits(&folio->flags.f, LRU_GEN_MASK, flags);
+ flags = set_mask_bits(folio_flags(folio, 0), LRU_GEN_MASK, flags);
gen = ((flags & LRU_GEN_MASK) >> LRU_GEN_PGOFF) - 1;
lru_gen_update_size(lruvec, folio, gen, -1);
@@ -339,7 +400,6 @@ static inline bool lru_gen_del_folio(struct lruvec *lruvec, struct folio *folio,
static inline void folio_migrate_refs(struct folio *new, const struct folio *old)
{
-
}
#endif /* CONFIG_LRU_GEN */
diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h
index 94f9c3ff5416..c9ecf370cd9f 100644
--- a/include/linux/mmzone.h
+++ b/include/linux/mmzone.h
@@ -497,6 +497,7 @@ enum lruvec_flags {
#define LRU_GEN_MASK ((BIT(LRU_GEN_WIDTH) - 1) << LRU_GEN_PGOFF)
#define LRU_REFS_MASK ((BIT(LRU_REFS_WIDTH) - 1) << LRU_REFS_PGOFF)
+#define LRU_REFS_MAX BIT(LRU_REFS_WIDTH)
/*
* For folios accessed multiple times through file descriptors,
diff --git a/mm/folio.c b/mm/folio.c
index 59c477120b9a..0adfe4f5ef72 100644
--- a/mm/folio.c
+++ b/mm/folio.c
@@ -353,26 +353,28 @@ static void __lru_cache_activate_folio(struct folio *folio)
static void lru_gen_inc_refs(struct folio *folio)
{
- unsigned long new_flags, old_flags = READ_ONCE(folio->flags.f);
+ unsigned long new_flags, old_flags = READ_ONCE(*folio_flags(folio, 0));
+ int refs;
if (folio_test_unevictable(folio))
return;
/* see the comment on LRU_REFS_FLAGS */
- if (!folio_test_referenced(folio)) {
- set_mask_bits(&folio->flags.f, LRU_REFS_MASK, BIT(PG_referenced));
+ if (!folio_lru_refs(folio)) {
+ folio_set_lru_refs(folio, 1);
return;
}
do {
- if ((old_flags & LRU_REFS_MASK) == LRU_REFS_MASK) {
+ new_flags = old_flags;
+ refs = lru_refs_from_flags(old_flags);
+ if (refs == LRU_REFS_MAX) {
if (!folio_test_workingset(folio))
folio_set_workingset(folio);
return;
}
-
- new_flags = old_flags + BIT(LRU_REFS_PGOFF);
- } while (!try_cmpxchg(&folio->flags.f, &old_flags, new_flags));
+ lru_refs_set_flags(&new_flags, refs + 1);
+ } while (!try_cmpxchg(folio_flags(folio, 0), &old_flags, new_flags));
}
static bool lru_gen_clear_refs(struct folio *folio)
@@ -384,7 +386,8 @@ static bool lru_gen_clear_refs(struct folio *folio)
if (gen < 0)
return true;
- set_mask_bits(&folio->flags.f, LRU_REFS_FLAGS | BIT(PG_workingset), 0);
+ folio_set_lru_refs(folio, 0);
+ folio_clear_workingset(folio);
rcu_read_lock();
seq = READ_ONCE(folio_lruvec(folio)->lrugen.min_seq[type]);
diff --git a/mm/vmscan.c b/mm/vmscan.c
index c1404a59523d..080132997d87 100644
--- a/mm/vmscan.c
+++ b/mm/vmscan.c
@@ -843,19 +843,22 @@ static bool lru_gen_set_refs(struct folio *folio, const vma_flags_t *vma_flags)
if (!folio_test_referenced(folio) && !folio_test_workingset(folio)) {
/* Activate file-backed executable folios after first usage. */
if (is_exec_file_folio(folio, vma_flags)) {
- set_mask_bits(&folio->flags.f, LRU_REFS_FLAGS, BIT(PG_workingset));
+ folio_set_lru_refs(folio, 0);
+ folio_set_workingset(folio);
return true;
}
- set_mask_bits(&folio->flags.f, LRU_REFS_MASK, BIT(PG_referenced));
+ folio_set_lru_refs(folio, 1);
return false;
}
/* Promote on second access */
- if (folio_lru_refs(folio) > 1)
- set_mask_bits(&folio->flags.f, LRU_REFS_FLAGS, BIT(PG_workingset));
- else
+ if (folio_lru_refs(folio) > 1) {
+ folio_set_lru_refs(folio, 0);
+ folio_set_workingset(folio);
+ } else {
folio_mark_accessed(folio);
+ }
return true;
}
#else
@@ -3266,11 +3269,10 @@ static bool positive_ctrl_err(struct ctrl_pos *sp, struct ctrl_pos *pv)
******************************************************************************/
/* promote pages accessed through page tables */
-static int folio_update_gen(struct folio *folio, int gen, const vma_flags_t *vma_flags)
+static int folio_update_gen(struct folio *folio, int new_gen, const vma_flags_t *vma_flags)
{
- unsigned long new_flags, old_flags = READ_ONCE(folio->flags.f);
-
- VM_WARN_ON_ONCE(gen >= MAX_NR_GENS);
+ unsigned long new_flags, old_flags = READ_ONCE(*folio_flags(folio, 0));
+ int old_gen;
/*
* See the comment on LRU_REFS_FLAGS, and activate file-backed
@@ -3279,20 +3281,24 @@ static int folio_update_gen(struct folio *folio, int gen, const vma_flags_t *vma
*/
if (!folio_test_referenced(folio) && !folio_test_workingset(folio) &&
!is_exec_file_folio(folio, vma_flags)) {
- set_mask_bits(&folio->flags.f, LRU_REFS_MASK, BIT(PG_referenced));
+ folio_set_lru_refs(folio, 1);
return -1;
}
do {
+ old_gen = lru_gen_from_flags(old_flags);
+ new_flags = old_flags;
+
/* lru_gen_del_folio() has isolated this page? */
- if (!(old_flags & LRU_GEN_MASK))
- return -1;
+ if (old_gen < 0)
+ break;
- new_flags = old_flags & ~(LRU_GEN_MASK | LRU_REFS_FLAGS);
- new_flags |= ((gen + 1UL) << LRU_GEN_PGOFF) | BIT(PG_workingset);
- } while (!try_cmpxchg(&folio->flags.f, &old_flags, new_flags));
+ lru_gen_set_flags(&new_flags, new_gen);
+ lru_refs_set_flags(&new_flags, 0);
+ new_flags |= BIT(PG_workingset);
+ } while (!try_cmpxchg(folio_flags(folio, 0), &old_flags, new_flags));
- return ((old_flags & LRU_GEN_MASK) >> LRU_GEN_PGOFF) - 1;
+ return old_gen;
}
/* protect pages accessed multiple times through file descriptors */
@@ -3301,21 +3307,20 @@ static int folio_inc_gen(struct lruvec *lruvec, struct folio *folio)
int type = folio_is_file_lru(folio);
struct lru_gen_folio *lrugen = &lruvec->lrugen;
int new_gen, old_gen = lru_gen_from_seq(lrugen->min_seq[type]);
- unsigned long new_flags, old_flags = READ_ONCE(folio->flags.f);
-
- VM_WARN_ON_ONCE_FOLIO(!(old_flags & LRU_GEN_MASK), folio);
+ unsigned long new_flags, old_flags = READ_ONCE(*folio_flags(folio, 0));
do {
- new_gen = ((old_flags & LRU_GEN_MASK) >> LRU_GEN_PGOFF) - 1;
+ new_gen = lru_gen_from_flags(old_flags);
+
/* folio_update_gen() has promoted this page? */
if (new_gen >= 0 && new_gen != old_gen)
return new_gen;
+ new_flags = old_flags;
new_gen = (old_gen + 1) % MAX_NR_GENS;
-
- new_flags = old_flags & ~(LRU_GEN_MASK | LRU_REFS_FLAGS);
- new_flags |= (new_gen + 1UL) << LRU_GEN_PGOFF;
- } while (!try_cmpxchg(&folio->flags.f, &old_flags, new_flags));
+ lru_gen_set_flags(&new_flags, new_gen);
+ lru_refs_set_flags(&new_flags, 0);
+ } while (!try_cmpxchg(folio_flags(folio, 0), &old_flags, new_flags));
lru_gen_update_size(lruvec, folio, old_gen, new_gen);
@@ -4711,7 +4716,7 @@ static bool isolate_folio(struct lruvec *lruvec, struct folio *folio, struct sca
/* see the comment on LRU_REFS_FLAGS */
if (!folio_test_referenced(folio))
- set_mask_bits(&folio->flags.f, LRU_REFS_MASK, 0);
+ folio_set_lru_refs(folio, 0);
success = lru_gen_del_folio(lruvec, folio, true);
VM_WARN_ON_ONCE_FOLIO(!success, folio);
@@ -4927,8 +4932,10 @@ static int evict_folios(unsigned long nr_to_scan, struct lruvec *lruvec,
}
/* don't add rejected folios to the oldest generation */
- if (lru_gen_folio_seq(lruvec, folio, false) == min_seq[type])
- set_mask_bits(&folio->flags.f, LRU_REFS_FLAGS, BIT(PG_active));
+ if (lru_gen_folio_seq(lruvec, folio, false) == min_seq[type]) {
+ folio_set_lru_refs(folio, 0);
+ folio_set_active(folio);
+ }
}
move_folios_to_lru(&list);
--
2.55.0
^ permalink raw reply related [flat|nested] 9+ messages in thread* Re: [PATCH v2 2/6] mm/mglru: introduce helpers for manipulating gen and refs flags
2026-08-24 15:26 ` [PATCH v2 2/6] mm/mglru: introduce helpers for manipulating gen and refs flags Kairui Song via B4 Relay
@ 2026-08-24 16:30 ` Kairui Song
0 siblings, 0 replies; 9+ messages in thread
From: Kairui Song @ 2026-08-24 16:30 UTC (permalink / raw)
To: kasong
Cc: linux-mm, Andrew Morton, Barry Song, Axel Rasmussen, Yuanchu Xie,
Wei Xu, Baoquan He, Shakeel Butt, Johannes Weiner, Michal Hocko,
Roman Gushchin, Muchun Song, Chris Li, Baolin Wang, Ridong Chen,
David Hildenbrand, Lorenzo Stoakes, Liam R. Howlett,
Vlastimil Babka, Yu Zhao, Zi Yan, Qi Zheng, cgroups, linux-kernel
Got this from sashiko (I consider it false negative since that is
extremely unlikely to be an issue):
--- Patch [2]: [PATCH v2 2/6] mm/mglru: introduce helpers for
manipulating gen and refs flags ---
commit f0369e7dfac3789b1433352b10e442443527f4cc
Author: Kairui Song <kasong@tencent.com>
mm/mglru: introduce helpers for manipulating gen and refs flags
This commit introduces helpers for adjusting a folio's refs and generation
info. It replaces bit operations on folio->flags.f, splitting some combined
operations under the assumption that the resulting transient states are
harmless.
> diff --git a/mm/vmscan.c b/mm/vmscan.c
> index c1404a59523d6..080132997d876 100644
> --- a/mm/vmscan.c
> +++ b/mm/vmscan.c
[ ... ]
> @@ -843,19 +843,22 @@ static bool lru_gen_set_refs(struct folio *folio, const vma_flags_t *vma_flags)
> if (!folio_test_referenced(folio) && !folio_test_workingset(folio)) {
> /* Activate file-backed executable folios after first usage. */
> if (is_exec_file_folio(folio, vma_flags)) {
> - set_mask_bits(&folio->flags.f, LRU_REFS_FLAGS, BIT(PG_workingset));
> + folio_set_lru_refs(folio, 0);
> + folio_set_workingset(folio);
The commit message mentions that splitting these atomic operations only
creates harmless transient states, but could this specific split break
lockless MGLRU protection?
Because lru_gen_set_refs() can be called locklessly (e.g., from
lru_gen_look_around() holding only the page table lock), this creates a
transient window where the folio has both its LRU references cleared
(refs = 0) and the PG_workingset flag not yet set.
Concurrently, evict_folios() runs under the LRU lock and evaluates folios
for eviction using sort_folio():
mm/vmscan.c:sort_folio() {
int refs = folio_lru_refs(folio);
bool workingset = folio_test_workingset(folio);
int tier = lru_tier_from_refs(refs, workingset);
...
if (tier > tier_idx || refs + workingset == BIT(LRU_REFS_WIDTH) + 1) {
...
}
If sort_folio() reads the folio's flags during this transient window,
it will see refs = 0 and workingset = false, causing tier to evaluate to 0.
Will this cause the MGLRU protection check (tier > tier_idx) to fail and
incorrectly isolate actively used executable file folios for eviction?
> return true;
> }
>
> - set_mask_bits(&folio->flags.f, LRU_REFS_MASK, BIT(PG_referenced));
> + folio_set_lru_refs(folio, 1);
> return false;
> }
>
> /* Promote on second access */
> - if (folio_lru_refs(folio) > 1)
> - set_mask_bits(&folio->flags.f, LRU_REFS_FLAGS, BIT(PG_workingset));
> - else
> + if (folio_lru_refs(folio) > 1) {
> + folio_set_lru_refs(folio, 0);
> + folio_set_workingset(folio);
Does this have the same transient state race with sort_folio() as the split
above, potentially causing actively used folios to be wrongly evicted on
their second access?
> + } else {
> folio_mark_accessed(folio);
> + }
> return true;
> }
===
In theory, yes. In practise, it's very unlikely to occur, and even if
it did, we are totally fine.
And I have often observed that folios stuck in highest tier due to
PG_workingset is actually causing worse reclaim. But if we are very
paranoid here we can do folio_set_workingset and then
folio_set_lru_refs. (This may cause a in theory, still very very
unlikely, aggressive protection. And in fact I think we are definitely
fine either way). I might update this if there is a v3.
^ permalink raw reply [flat|nested] 9+ messages in thread
* [PATCH v2 3/6] mm/migrate: copy all referenced state via folio_migrate_lru_refs
2026-08-24 15:26 [PATCH v2 0/6] mm/mglru: clean up folio counters and flag usage Kairui Song via B4 Relay
2026-08-24 15:26 ` [PATCH v2 1/6] mm/memcontrol: make lru_zone_size atomic and simplify sanity check Kairui Song via B4 Relay
2026-08-24 15:26 ` [PATCH v2 2/6] mm/mglru: introduce helpers for manipulating gen and refs flags Kairui Song via B4 Relay
@ 2026-08-24 15:26 ` Kairui Song via B4 Relay
2026-08-24 15:26 ` [PATCH v2 4/6] mm/mglru: move max_seq read into walk_update_folio Kairui Song via B4 Relay
` (2 subsequent siblings)
5 siblings, 0 replies; 9+ messages in thread
From: Kairui Song via B4 Relay @ 2026-08-24 15:26 UTC (permalink / raw)
To: linux-mm
Cc: Andrew Morton, Barry Song, Axel Rasmussen, Yuanchu Xie, Wei Xu,
Baoquan He, Shakeel Butt, Johannes Weiner, Michal Hocko,
Roman Gushchin, Muchun Song, Chris Li, Baolin Wang, Ridong Chen,
David Hildenbrand, Lorenzo Stoakes, Liam R. Howlett,
Vlastimil Babka, Yu Zhao, Zi Yan, Qi Zheng, cgroups, linux-kernel,
Kairui Song, Kairui Song
From: Kairui Song <kasong@tencent.com>
folio_migrate_flags() copies PG_referenced separately from the MGLRU
refs counter, which folio_migrate_refs() transfers. Yet under MGLRU,
PG_referenced and the refs counter bits together describe the
referenced status of a folio.
Consolidate the two: rename folio_migrate_refs() to
folio_migrate_lru_refs() and let it copy the complete referenced
status, i.e., the MGLRU refs count including PG_referenced, or just
PG_referenced for the active/inactive LRU. Drop the open-coded
PG_referenced copy so the referenced status is transferred in one
place. No behavior change is intended: under the active/inactive LRU
the extra bits are unused, so operating on them is a noop.
Reviewed-by: Baoquan He <baoquan.he@linux.dev>
Signed-off-by: Kairui Song <kasong@tencent.com>
---
include/linux/mm_inline.h | 20 +++++++++++++++-----
mm/migrate.c | 6 +++---
2 files changed, 18 insertions(+), 8 deletions(-)
diff --git a/include/linux/mm_inline.h b/include/linux/mm_inline.h
index edfaf2661812..7f91a89b5ba3 100644
--- a/include/linux/mm_inline.h
+++ b/include/linux/mm_inline.h
@@ -365,11 +365,19 @@ static inline bool lru_gen_del_folio(struct lruvec *lruvec, struct folio *folio,
return true;
}
-static inline void folio_migrate_refs(struct folio *new, const struct folio *old)
+/**
+ * folio_migrate_lru_refs - copy the reference state to a new folio
+ * @new: the destination folio
+ * @old: the source folio
+ *
+ * Transfer the reference state to @new during migration: the MGLRU
+ * refs count, including PG_referenced, or just PG_referenced for the
+ * active/inactive LRU.
+ */
+static inline void folio_migrate_lru_refs(struct folio *new, const struct folio *old)
{
- unsigned long refs = READ_ONCE(old->flags.f) & LRU_REFS_MASK;
-
- set_mask_bits(&new->flags.f, LRU_REFS_MASK, refs);
+ BUILD_BUG_ON(LRU_REFS_MASK & BIT(PG_referenced));
+ folio_set_lru_refs(new, folio_lru_refs(old));
}
#else /* !CONFIG_LRU_GEN */
@@ -398,8 +406,10 @@ static inline bool lru_gen_del_folio(struct lruvec *lruvec, struct folio *folio,
return false;
}
-static inline void folio_migrate_refs(struct folio *new, const struct folio *old)
+static inline void folio_migrate_lru_refs(struct folio *new, const struct folio *old)
{
+ if (folio_test_referenced(old))
+ folio_set_referenced(new);
}
#endif /* CONFIG_LRU_GEN */
diff --git a/mm/migrate.c b/mm/migrate.c
index 15b45832bcfa..96eda490ba43 100644
--- a/mm/migrate.c
+++ b/mm/migrate.c
@@ -776,8 +776,6 @@ void folio_migrate_flags(struct folio *newfolio, struct folio *folio)
{
int cpupid;
- if (folio_test_referenced(folio))
- folio_set_referenced(newfolio);
if (folio_test_uptodate(folio))
folio_mark_uptodate(newfolio);
if (folio_test_clear_active(folio)) {
@@ -807,7 +805,9 @@ void folio_migrate_flags(struct folio *newfolio, struct folio *folio)
if (folio_test_idle(folio))
folio_set_idle(newfolio);
- folio_migrate_refs(newfolio, folio);
+ /* Copy the reference state, including PG_referenced */
+ folio_migrate_lru_refs(newfolio, folio);
+
/*
* Copy NUMA information to the new page, to prevent over-eager
* future migrations of this same page.
--
2.55.0
^ permalink raw reply related [flat|nested] 9+ messages in thread* [PATCH v2 4/6] mm/mglru: move max_seq read into walk_update_folio
2026-08-24 15:26 [PATCH v2 0/6] mm/mglru: clean up folio counters and flag usage Kairui Song via B4 Relay
` (2 preceding siblings ...)
2026-08-24 15:26 ` [PATCH v2 3/6] mm/migrate: copy all referenced state via folio_migrate_lru_refs Kairui Song via B4 Relay
@ 2026-08-24 15:26 ` Kairui Song via B4 Relay
2026-08-24 15:26 ` [PATCH v2 5/6] mm/mglru: use explicit tier range in read_ctrl_pos() Kairui Song via B4 Relay
2026-08-24 15:26 ` [PATCH v2 6/6] mm/mglru: fix potential generation folio number leak Kairui Song via B4 Relay
5 siblings, 0 replies; 9+ messages in thread
From: Kairui Song via B4 Relay @ 2026-08-24 15:26 UTC (permalink / raw)
To: linux-mm
Cc: Andrew Morton, Barry Song, Axel Rasmussen, Yuanchu Xie, Wei Xu,
Baoquan He, Shakeel Butt, Johannes Weiner, Michal Hocko,
Roman Gushchin, Muchun Song, Chris Li, Baolin Wang, Ridong Chen,
David Hildenbrand, Lorenzo Stoakes, Liam R. Howlett,
Vlastimil Babka, Yu Zhao, Zi Yan, Qi Zheng, cgroups, linux-kernel,
Kairui Song, Kairui Song
From: Kairui Song <kasong@tencent.com>
walk_pte_range(), walk_pmd_range_locked(), and lru_gen_look_around()
each read lrugen->max_seq to compute the target generation used by
walk_update_folio(), then pass it as a parameter. Move the read into
walk_update_folio() itself so the callers no longer need to compute
or pass the value.
The max_seq read now happens once per folio update rather than once
per walk range, so folios always get promoted to the current youngest
generation.
Reviewed-by: Baoquan He <baoquan.he@linux.dev>
Reviewed-by: Baolin Wang <baolin.wang@linux.alibaba.com>
Reviewed-by: Ridong Chen <ridong.chen@linux.dev>
Signed-off-by: Kairui Song <kasong@tencent.com>
---
mm/vmscan.c | 29 ++++++++++++-----------------
1 file changed, 12 insertions(+), 17 deletions(-)
diff --git a/mm/vmscan.c b/mm/vmscan.c
index 080132997d87..557927b498fa 100644
--- a/mm/vmscan.c
+++ b/mm/vmscan.c
@@ -3517,13 +3517,15 @@ static bool suitable_to_scan(int total, int young)
}
static void walk_update_folio(struct lru_gen_mm_walk *walk, struct vm_area_struct *vma,
- struct folio *folio, int new_gen, bool dirty)
+ struct lruvec *lruvec, struct folio *folio, bool dirty)
{
- int old_gen;
+ int new_gen, old_gen;
if (!folio)
return;
+ new_gen = lru_gen_from_seq(READ_ONCE(lruvec->lrugen.max_seq));
+
if (dirty && !folio_test_dirty(folio) &&
!(folio_test_anon(folio) && folio_test_swapbacked(folio) &&
!folio_test_swapcache(folio)))
@@ -3554,8 +3556,6 @@ static bool walk_pte_range(pmd_t *pmd, unsigned long start, unsigned long end,
struct lru_gen_mm_walk *walk = args->private;
struct mem_cgroup *memcg = lruvec_memcg(walk->lruvec);
struct pglist_data *pgdat = lruvec_pgdat(walk->lruvec);
- DEFINE_MAX_SEQ(walk->lruvec);
- int gen = lru_gen_from_seq(max_seq);
unsigned int nr;
pmd_t pmdval;
@@ -3606,7 +3606,7 @@ static bool walk_pte_range(pmd_t *pmd, unsigned long start, unsigned long end,
continue;
if (last != folio) {
- walk_update_folio(walk, args->vma, last, gen, dirty);
+ walk_update_folio(walk, args->vma, walk->lruvec, last, dirty);
last = folio;
dirty = false;
@@ -3619,7 +3619,7 @@ static bool walk_pte_range(pmd_t *pmd, unsigned long start, unsigned long end,
walk->mm_stats[MM_LEAF_YOUNG] += nr;
}
- walk_update_folio(walk, args->vma, last, gen, dirty);
+ walk_update_folio(walk, args->vma, walk->lruvec, last, dirty);
last = NULL;
if (i < PTRS_PER_PTE && get_next_vma(PMD_MASK, PAGE_SIZE, args, &start, &end))
@@ -3642,8 +3642,6 @@ static void walk_pmd_range_locked(pud_t *pud, unsigned long addr, struct vm_area
struct lru_gen_mm_walk *walk = args->private;
struct mem_cgroup *memcg = lruvec_memcg(walk->lruvec);
struct pglist_data *pgdat = lruvec_pgdat(walk->lruvec);
- DEFINE_MAX_SEQ(walk->lruvec);
- int gen = lru_gen_from_seq(max_seq);
VM_WARN_ON_ONCE(pud_leaf(*pud));
@@ -3697,7 +3695,7 @@ static void walk_pmd_range_locked(pud_t *pud, unsigned long addr, struct vm_area
goto next;
if (last != folio) {
- walk_update_folio(walk, vma, last, gen, dirty);
+ walk_update_folio(walk, vma, walk->lruvec, last, dirty);
last = folio;
dirty = false;
@@ -3711,7 +3709,7 @@ static void walk_pmd_range_locked(pud_t *pud, unsigned long addr, struct vm_area
i = i > MIN_LRU_BATCH ? 0 : find_next_bit(bitmap, MIN_LRU_BATCH, i) + 1;
} while (i <= MIN_LRU_BATCH);
- walk_update_folio(walk, vma, last, gen, dirty);
+ walk_update_folio(walk, vma, walk->lruvec, last, dirty);
lazy_mmu_mode_disable();
spin_unlock(ptl);
@@ -4275,8 +4273,6 @@ bool lru_gen_look_around(struct page_vma_mapped_walk *pvmw, unsigned int nr)
struct pglist_data *pgdat = folio_pgdat(folio);
struct lruvec *lruvec;
struct lru_gen_mm_state *mm_state;
- unsigned long max_seq;
- int gen;
lockdep_assert_held(pvmw->ptl);
VM_WARN_ON_ONCE_FOLIO(folio_test_lru(folio), folio);
@@ -4313,8 +4309,6 @@ bool lru_gen_look_around(struct page_vma_mapped_walk *pvmw, unsigned int nr)
memcg = get_mem_cgroup_from_folio(folio);
lruvec = mem_cgroup_lruvec(memcg, pgdat);
- max_seq = READ_ONCE((lruvec)->lrugen.max_seq);
- gen = lru_gen_from_seq(max_seq);
mm_state = get_mm_state(lruvec);
lazy_mmu_mode_enable();
@@ -4346,7 +4340,7 @@ bool lru_gen_look_around(struct page_vma_mapped_walk *pvmw, unsigned int nr)
continue;
if (last != folio) {
- walk_update_folio(walk, vma, last, gen, dirty);
+ walk_update_folio(walk, vma, lruvec, last, dirty);
last = folio;
dirty = false;
@@ -4358,13 +4352,14 @@ bool lru_gen_look_around(struct page_vma_mapped_walk *pvmw, unsigned int nr)
young += nr;
}
- walk_update_folio(walk, vma, last, gen, dirty);
+ walk_update_folio(walk, vma, lruvec, last, dirty);
lazy_mmu_mode_disable();
/* feedback from rmap walkers to page table walkers */
if (mm_state && suitable_to_scan(i, young))
- update_bloom_filter(mm_state, max_seq, pvmw->pmd);
+ update_bloom_filter(mm_state, READ_ONCE(lruvec->lrugen.max_seq),
+ pvmw->pmd);
mem_cgroup_put(memcg);
--
2.55.0
^ permalink raw reply related [flat|nested] 9+ messages in thread* [PATCH v2 5/6] mm/mglru: use explicit tier range in read_ctrl_pos()
2026-08-24 15:26 [PATCH v2 0/6] mm/mglru: clean up folio counters and flag usage Kairui Song via B4 Relay
` (3 preceding siblings ...)
2026-08-24 15:26 ` [PATCH v2 4/6] mm/mglru: move max_seq read into walk_update_folio Kairui Song via B4 Relay
@ 2026-08-24 15:26 ` Kairui Song via B4 Relay
2026-08-24 15:26 ` [PATCH v2 6/6] mm/mglru: fix potential generation folio number leak Kairui Song via B4 Relay
5 siblings, 0 replies; 9+ messages in thread
From: Kairui Song via B4 Relay @ 2026-08-24 15:26 UTC (permalink / raw)
To: linux-mm
Cc: Andrew Morton, Barry Song, Axel Rasmussen, Yuanchu Xie, Wei Xu,
Baoquan He, Shakeel Butt, Johannes Weiner, Michal Hocko,
Roman Gushchin, Muchun Song, Chris Li, Baolin Wang, Ridong Chen,
David Hildenbrand, Lorenzo Stoakes, Liam R. Howlett,
Vlastimil Babka, Yu Zhao, Zi Yan, Qi Zheng, cgroups, linux-kernel,
Kairui Song, Kairui Song
From: Kairui Song <kasong@tencent.com>
read_ctrl_pos() encodes the tier range in a single "tier" parameter
via "tier % MAX_NR_TIERS" as the start and "min(tier, MAX_NR_TIERS-1)"
as the end. This is hard to follow, maintain, or extend. Tier values
0..3 select a single tier, while tier == MAX_NR_TIERS selects the
full range.
Replace it with explicit (tier_min, tier_max) parameters using a
closed [tier_min, tier_max] interval, and add LRU_TIER_MIN and
LRU_TIER_MAX for the tier bounds. The call sites now become
self-documenting:
- get_tier_idx: (LRU_TIER_MIN, LRU_TIER_MIN) for the first tier,
(tier, tier) for each subsequent tier
- get_type_to_scan: (LRU_TIER_MIN, LRU_TIER_MAX) for the full range
No functional change.
Reviewed-by: Baolin Wang <baolin.wang@linux.alibaba.com>
Reviewed-by: Baoquan He <baoquan.he@linux.dev>
Reviewed-by: Barry Song <baohua@kernel.org>
Signed-off-by: Kairui Song <kasong@tencent.com>
---
include/linux/mmzone.h | 2 ++
mm/vmscan.c | 18 ++++++++++--------
2 files changed, 12 insertions(+), 8 deletions(-)
diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h
index c9ecf370cd9f..9b27cf53bdbc 100644
--- a/include/linux/mmzone.h
+++ b/include/linux/mmzone.h
@@ -492,6 +492,8 @@ enum lruvec_flags {
* folio->flags, masked by LRU_REFS_MASK.
*/
#define MAX_NR_TIERS 4U
+#define LRU_TIER_MIN 0U
+#define LRU_TIER_MAX (MAX_NR_TIERS - 1)
#ifndef __GENERATING_BOUNDS_H
diff --git a/mm/vmscan.c b/mm/vmscan.c
index 557927b498fa..f8e291968342 100644
--- a/mm/vmscan.c
+++ b/mm/vmscan.c
@@ -3198,8 +3198,8 @@ struct ctrl_pos {
int gain;
};
-static void read_ctrl_pos(struct lruvec *lruvec, int type, int tier, int gain,
- struct ctrl_pos *pos)
+static void read_ctrl_pos(struct lruvec *lruvec, int type, int tier_min,
+ int tier_max, int gain, struct ctrl_pos *pos)
{
int i;
struct lru_gen_folio *lrugen = &lruvec->lrugen;
@@ -3208,7 +3208,7 @@ static void read_ctrl_pos(struct lruvec *lruvec, int type, int tier, int gain,
pos->gain = gain;
pos->refaulted = pos->total = 0;
- for (i = tier % MAX_NR_TIERS; i <= min(tier, MAX_NR_TIERS - 1); i++) {
+ for (i = tier_min; i <= tier_max; i++) {
pos->refaulted += lrugen->avg_refaulted[type][i] +
atomic_long_read(&lrugen->refaulted[hist][type][i]);
pos->total += lrugen->avg_total[type][i] +
@@ -4804,9 +4804,9 @@ static int get_tier_idx(struct lruvec *lruvec, int type)
* This value is chosen because any other tier would have at least twice
* as many refaults as the first tier.
*/
- read_ctrl_pos(lruvec, type, 0, 2, &sp);
- for (tier = 1; tier < MAX_NR_TIERS; tier++) {
- read_ctrl_pos(lruvec, type, tier, 3, &pv);
+ read_ctrl_pos(lruvec, type, LRU_TIER_MIN, LRU_TIER_MIN, 2, &sp);
+ for (tier = LRU_TIER_MIN + 1; tier <= LRU_TIER_MAX; tier++) {
+ read_ctrl_pos(lruvec, type, tier, tier, 3, &pv);
if (!positive_ctrl_err(&sp, &pv))
break;
}
@@ -4827,8 +4827,10 @@ static int get_type_to_scan(struct lruvec *lruvec, int swappiness)
* Compare the sum of all tiers of anon with that of file to determine
* which type to scan.
*/
- read_ctrl_pos(lruvec, LRU_GEN_ANON, MAX_NR_TIERS, swappiness, &sp);
- read_ctrl_pos(lruvec, LRU_GEN_FILE, MAX_NR_TIERS, MAX_SWAPPINESS - swappiness, &pv);
+ read_ctrl_pos(lruvec, LRU_GEN_ANON, LRU_TIER_MIN, LRU_TIER_MAX,
+ swappiness, &sp);
+ read_ctrl_pos(lruvec, LRU_GEN_FILE, LRU_TIER_MIN, LRU_TIER_MAX,
+ MAX_SWAPPINESS - swappiness, &pv);
return positive_ctrl_err(&sp, &pv);
}
--
2.55.0
^ permalink raw reply related [flat|nested] 9+ messages in thread* [PATCH v2 6/6] mm/mglru: fix potential generation folio number leak
2026-08-24 15:26 [PATCH v2 0/6] mm/mglru: clean up folio counters and flag usage Kairui Song via B4 Relay
` (4 preceding siblings ...)
2026-08-24 15:26 ` [PATCH v2 5/6] mm/mglru: use explicit tier range in read_ctrl_pos() Kairui Song via B4 Relay
@ 2026-08-24 15:26 ` Kairui Song via B4 Relay
5 siblings, 0 replies; 9+ messages in thread
From: Kairui Song via B4 Relay @ 2026-08-24 15:26 UTC (permalink / raw)
To: linux-mm
Cc: Andrew Morton, Barry Song, Axel Rasmussen, Yuanchu Xie, Wei Xu,
Baoquan He, Shakeel Butt, Johannes Weiner, Michal Hocko,
Roman Gushchin, Muchun Song, Chris Li, Baolin Wang, Ridong Chen,
David Hildenbrand, Lorenzo Stoakes, Liam R. Howlett,
Vlastimil Babka, Yu Zhao, Zi Yan, Qi Zheng, cgroups, linux-kernel,
Kairui Song, Kairui Song
From: Kairui Song <kasong@tencent.com>
Each generation of MGLRU accounts anon and file folio numbers
separately, and the page table walker updates each generation's
counters in batch once the walk is done. The walker promotes a
folio's generation with a cmpxchg on folio->flags, and
update_batch_size() then reads the live flags again to pick the
anon/file column to charge. The walk holds neither the lruvec lock nor
the folio lock, so the type can flip between the cmpxchg and that
read: the lazyfree path clears PG_swapbacked, and reclaim sets it back
on a dirty lazyfree folio. The batched delta pair is then recorded in
the wrong type column. Nothing reconciles it afterwards, permanently
skewing lrugen->nr_pages and the reclaim budgets derived from it.
Fix it by capturing the type from the flags snapshot the cmpxchg
linearized against: folio_update_gen() returns the type of the state
it transitioned from, and update_batch_size() accounts with that.
A folio's type only changes while it is off the LRU list, inside a
del/add pair under the lruvec lock, with the gen bits cleared in
between. The generation and PG_swapbacked sit in the same
folio->flags word, so the cmpxchg snapshot captures them together.
Let G be the generation that snapshot captured (old_gen) and G' the
one it wrote (new_gen); the CAS can land in only three places:
- before the del: the folio is anon at G; the batch records anon
G -> G', and the del later removes the folio from the anon
counters;
- between del and add: gen == -1, so folio_update_gen() returns -1
without touching the flags and no batch is recorded; the del/add
pair accounts for the move alone;
- after the add: the folio is file at the fresh generation the add
charged; the batch records file, that gen -> G', matching that
charge.
Unlike the drift of lazy promotions, which sort_folio() repairs under
the lruvec lock, the phantom deltas from before this fix land in a
column the folio never occupies again, so nothing ever repairs them.
Fixes: 018ee47f1489 ("mm: multi-gen LRU: exploit locality in rmap")
Signed-off-by: Kairui Song <kasong@tencent.com>
---
include/linux/mm_inline.h | 7 ++++++-
mm/vmscan.c | 13 +++++++------
2 files changed, 13 insertions(+), 7 deletions(-)
diff --git a/include/linux/mm_inline.h b/include/linux/mm_inline.h
index 7f91a89b5ba3..8cf82989ec26 100644
--- a/include/linux/mm_inline.h
+++ b/include/linux/mm_inline.h
@@ -10,6 +10,11 @@
#include <linux/userfaultfd_k.h>
#include <linux/leafops.h>
+static inline int folio_flags_is_file_lru(const unsigned long *flags)
+{
+ return !test_bit(PG_swapbacked, flags);
+}
+
/**
* folio_is_file_lru - Should the folio be on a file LRU or anon LRU?
* @folio: The folio to test.
@@ -27,7 +32,7 @@
*/
static inline int folio_is_file_lru(const struct folio *folio)
{
- return !folio_test_swapbacked(folio);
+ return folio_flags_is_file_lru(const_folio_flags(folio, 0));
}
static __always_inline void __update_lru_size(struct lruvec *lruvec,
diff --git a/mm/vmscan.c b/mm/vmscan.c
index f8e291968342..a1497ad61ad7 100644
--- a/mm/vmscan.c
+++ b/mm/vmscan.c
@@ -3269,7 +3269,8 @@ static bool positive_ctrl_err(struct ctrl_pos *sp, struct ctrl_pos *pv)
******************************************************************************/
/* promote pages accessed through page tables */
-static int folio_update_gen(struct folio *folio, int new_gen, const vma_flags_t *vma_flags)
+static int folio_update_gen(struct folio *folio, int new_gen, int *is_file,
+ const vma_flags_t *vma_flags)
{
unsigned long new_flags, old_flags = READ_ONCE(*folio_flags(folio, 0));
int old_gen;
@@ -3298,6 +3299,7 @@ static int folio_update_gen(struct folio *folio, int new_gen, const vma_flags_t
new_flags |= BIT(PG_workingset);
} while (!try_cmpxchg(folio_flags(folio, 0), &old_flags, new_flags));
+ *is_file = folio_flags_is_file_lru(&old_flags);
return old_gen;
}
@@ -3328,9 +3330,8 @@ static int folio_inc_gen(struct lruvec *lruvec, struct folio *folio)
}
static void update_batch_size(struct lru_gen_mm_walk *walk, struct folio *folio,
- int old_gen, int new_gen)
+ int old_gen, int new_gen, int type)
{
- int type = folio_is_file_lru(folio);
int zone = folio_zonenum(folio);
int delta = folio_nr_pages(folio);
@@ -3519,7 +3520,7 @@ static bool suitable_to_scan(int total, int young)
static void walk_update_folio(struct lru_gen_mm_walk *walk, struct vm_area_struct *vma,
struct lruvec *lruvec, struct folio *folio, bool dirty)
{
- int new_gen, old_gen;
+ int new_gen, old_gen, file;
if (!folio)
return;
@@ -3532,9 +3533,9 @@ static void walk_update_folio(struct lru_gen_mm_walk *walk, struct vm_area_struc
folio_mark_dirty(folio);
if (walk) {
- old_gen = folio_update_gen(folio, new_gen, &vma->flags);
+ old_gen = folio_update_gen(folio, new_gen, &file, &vma->flags);
if (old_gen >= 0 && old_gen != new_gen)
- update_batch_size(walk, folio, old_gen, new_gen);
+ update_batch_size(walk, folio, old_gen, new_gen, file);
} else if (lru_gen_set_refs(folio, &vma->flags)) {
old_gen = folio_lru_gen(folio);
if (old_gen >= 0 && old_gen != new_gen)
--
2.55.0
^ permalink raw reply related [flat|nested] 9+ messages in thread