Linux filesystem development
 help / color / mirror / Atom feed
From: Hugh Dickins <hughd@google.com>
To: Andrew Morton <akpm@linux-foundation.org>
Cc: Ackerley Tng <ackerleytng@google.com>,
	 Alexander Viro <viro@zeniv.linux.org.uk>,
	 Baolin Wang <baolin.wang@linux.alibaba.com>,
	 Barry Song <baohua@kernel.org>,
	Binbin Wu <binbin.wu@linux.intel.com>,
	 Christian Brauner <brauner@kernel.org>,
	Christoph Hellwig <hch@lst.de>,
	 Christoph Lameter <cl@gentwo.org>,
	 Claudio Imbrenda <imbrenda@linux.ibm.com>,
	 David Hildenbrand <david@kernel.org>,
	JP Kobryn <jp.kobryn@linux.dev>,  Jan Kara <jack@suse.cz>,
	Jens Axboe <axboe@kernel.dk>,
	 Johannes Weiner <hannes@cmpxchg.org>,
	Kairui Song <ryncsn@gmail.com>,  Kiryl Shutsemau <kas@kernel.org>,
	Lance Yang <lance.yang@linux.dev>,
	 Leonardo Bras <leobras.c@gmail.com>,
	Lorenzo Stoakes <ljs@kernel.org>,
	 Marcelo Tosatti <mtosatti@redhat.com>,
	 Matthew Wilcox <willy@infradead.org>,
	 Mel Gorman <mgorman@techsingularity.net>,
	 Miaohe Lin <linmiaohe@huawei.com>,
	Michal Hocko <mhocko@suse.com>,  Minchan Kim <minchan@kernel.org>,
	Muchun Song <muchun.song@linux.dev>,
	 Oscar Salvador <osalvador@suse.de>,
	Peter Zijlstra <peterz@infradead.org>,
	 Qi Zheng <qi.zheng@linux.dev>, Rik van Riel <riel@surriel.com>,
	 Sebastian Andrzej Siewior <bigeasy@linutronix.de>,
	 Shakeel Butt <shakeel.butt@linux.dev>,
	 Suren Baghdasaryan <surenb@google.com>,
	 Vlastimil Babka <vbabka@kernel.org>,
	 Yang Shi <yang@os.amperecomputing.com>,
	Yu Zhao <yuzhao@google.com>,  Zach O'Keefe <zokeefe@google.com>,
	Zi Yan <ziy@nvidia.com>,
	 linux-block@vger.kernel.org, linux-fsdevel@vger.kernel.org,
	 linux-kernel@vger.kernel.org, linux-mm@kvack.org
Subject: [PATCH 09/25] mm/fbatch: restore mlock+munlock batching, without extra ref
Date: Mon, 24 Aug 2026 07:14:06 -0700 (PDT)	[thread overview]
Message-ID: <32d663fb-f192-e0e5-114e-8a92240fb0ac@google.com> (raw)
In-Reply-To: <14a16945-529b-8bc0-ab38-3ea97e54e223@google.com>

Update mlock_folio(), munlock_folio() and their fbatch callouts and
helpers, to do folio_try_get()s at batch processing time, instead of
holding a folio reference all the while in mlock_fbatch: as in folio.c.

But more interesting is the use of mod_mlock_count(), using try_cmpxchg()
to update folio->mlock_count safely when possible (now when on lru_add
fbatch as well as when unevictable). While __mlock_folio() is as hard to
think about as before, __munlock_folio() simpler because munlock_folio()
can adjust mlock_count itself without clear_lru() or lruvec lock, and so
do the folio_test_clear_mlocked() immediately for itself (without which
unevictable_pgs_cleared was likely to appear high, when it should be 0
or low to indicate good mlock health).

__munlock_folio() is safe for use even when the unreferenced folio has
been freed and reused. It appears that __mlock_folio() could affect a
folio which has been freed and reused, but only if it is reused as an
mlocked folio, in which case its mlock_count is spuriously incremented
(but usually a spurious munlock decrement will follow).  How grave is
this? If unevictable_pgs_cleared remains low, not so bad.

I've gone back and forth on whether to move mlock_fbatch and these
functions into mm/folio.c: for now they stay here in mm/mlock.c.

Signed-off-by: Hugh Dickins <hughd@google.com>
---
 mm/mlock.c | 147 +++++++++++++++++++++++++++++++----------------------
 1 file changed, 86 insertions(+), 61 deletions(-)

diff --git a/mm/mlock.c b/mm/mlock.c
index 53d754e82ba2..1050010bbe0b 100644
--- a/mm/mlock.c
+++ b/mm/mlock.c
@@ -58,6 +58,20 @@ EXPORT_SYMBOL(can_do_mlock);
  * indicate the unevictable state.
  */
 
+static long mod_mlock_count(struct folio *folio, long incdec)
+{
+	long mlock_count = READ_ONCE(folio->mlock_count);
+
+	while (mlock_count & MLOCK_COUNT_0) {
+		if (mlock_count + incdec < MLOCK_COUNT_0)
+			return MLOCK_COUNT_0;
+		if (try_cmpxchg(&folio->mlock_count, &mlock_count,
+				mlock_count + incdec))
+			return mlock_count + incdec;
+	}
+	return 0;
+}
+
 static struct lruvec *__mlock_folio(struct folio *folio, struct lruvec *lruvec)
 {
 	/* There is nothing more we can do while it's off LRU */
@@ -65,6 +79,7 @@ static struct lruvec *__mlock_folio(struct folio *folio, struct lruvec *lruvec)
 		return lruvec;
 
 	lruvec = folio_lruvec_relock_irq(folio, lruvec);
+	lruvec_del_folio(lruvec, folio);
 
 	if (unlikely(folio_evictable(folio))) {
 		/*
@@ -73,92 +88,82 @@ static struct lruvec *__mlock_folio(struct folio *folio, struct lruvec *lruvec)
 		 * folio be unevictable?  I'm not sure, but move it now if so.
 		 */
 		if (folio_test_unevictable(folio)) {
-			lruvec_del_folio(lruvec, folio);
 			folio_clear_unevictable(folio);
-			lruvec_add_folio(lruvec, folio);
-
 			__count_vm_events(UNEVICTABLE_PGRESCUED,
 					  folio_nr_pages(folio));
 		}
 		goto out;
 	}
 
+	/*
+	 * Something to keep in mind when studying the arithmetic here:
+	 * we only come to __mlock_folio() when mlock_folio() could not
+	 * mod_mlock_count() itself; but by the time this is processed,
+	 * the folio may have already been munlocked, or another mlock
+	 * already marked it as unevictable and so mod_mlock_countable.
+	 * And don't forget that a folio may be unevictable for reasons
+	 * other than mlocked (hence the folio_evictable() check above).
+	 */
+
 	if (folio_test_unevictable(folio)) {
 		if (folio_test_mlocked(folio))
-			folio->mlock_count += MLOCK_COUNT_1;
+			mod_mlock_count(folio, MLOCK_COUNT_1);
 		goto out;
 	}
 
-	lruvec_del_folio(lruvec, folio);
 	folio_clear_active(folio);
 	folio_set_unevictable(folio);
-	folio->mlock_count = MLOCK_COUNT_0;
-	if (folio_test_mlocked(folio))
-		folio->mlock_count += MLOCK_COUNT_1;
-	lruvec_add_folio(lruvec, folio);
 	__count_vm_events(UNEVICTABLE_PGCULLED, folio_nr_pages(folio));
+
+	if (!folio_test_mlocked(folio))
+		folio->mlock_count = MLOCK_COUNT_0;
+	else if (!mod_mlock_count(folio, MLOCK_COUNT_1))
+		folio->mlock_count = MLOCK_COUNT_0 + MLOCK_COUNT_1;
 out:
+	lruvec_add_folio(lruvec, folio);
 	folio_set_lru(folio);
 	return lruvec;
 }
 
 static struct lruvec *__munlock_folio(struct folio *folio, struct lruvec *lruvec)
 {
-	int nr_pages = folio_nr_pages(folio);
-	bool isolated = false;
+	long nr_pages = folio_nr_pages(folio);
 
-	if (!folio_test_clear_lru(folio))
-		goto munlock;
-
-	isolated = true;
-	lruvec = folio_lruvec_relock_irq(folio, lruvec);
-
-	if (folio_test_unevictable(folio)) {
-		/* Then mlock_count is maintained, but might undercount */
-		if (folio->mlock_count > MLOCK_COUNT_0)
-			folio->mlock_count -= MLOCK_COUNT_1;
-		if (folio->mlock_count > MLOCK_COUNT_0)
-			goto out;
-	}
-	/* else assume that was the last mlock: reclaim will fix it if not */
-
-munlock:
-	if (folio_test_clear_mlocked(folio)) {
-		__zone_stat_mod_folio(folio, NR_MLOCK, -nr_pages);
-		if (isolated || !folio_test_unevictable(folio))
-			__count_vm_events(UNEVICTABLE_PGMUNLOCKED, nr_pages);
-		else
+	/* There is nothing more we can do while it's off LRU */
+	if (!folio_test_clear_lru(folio)) {
+		if (folio_test_unevictable(folio) && folio_evictable(folio))
 			__count_vm_events(UNEVICTABLE_PGSTRANDED, nr_pages);
+		/* But whoever puts it back on LRU should rescue it */
+		return lruvec;
 	}
 
-	/* folio_evictable() has to be checked *after* clearing Mlocked */
-	if (isolated && folio_test_unevictable(folio) && folio_evictable(folio)) {
-		lruvec_del_folio(lruvec, folio);
+	lruvec = folio_lruvec_relock_irq(folio, lruvec);
+	lruvec_del_folio(lruvec, folio);
+
+	if (folio_test_unevictable(folio) && folio_evictable(folio)) {
 		folio_clear_unevictable(folio);
-		lruvec_add_folio(lruvec, folio);
 		__count_vm_events(UNEVICTABLE_PGRESCUED, nr_pages);
 	}
-out:
-	if (isolated)
-		folio_set_lru(folio);
+
+	lruvec_add_folio(lruvec, folio);
+	folio_set_lru(folio);
 	return lruvec;
 }
 
 /*
- * Flags held in the low bits of a struct folio pointer on the mlock_fbatch.
+ * Flag held in the low bits of a struct folio pointer on the mlock_fbatch.
  */
-#define LRU_FOLIO 0x1
-static inline struct folio *mlock_lru(struct folio *folio)
+#define MLOCK_FLAG 0x1
+static inline struct folio *mlock_flagged(struct folio *folio)
 {
-	return (struct folio *)((unsigned long)folio + LRU_FOLIO);
+	return (struct folio *)((unsigned long)folio + MLOCK_FLAG);
 }
 
 /*
  * mlock_folio_batch() is derived from folio_batch_move_lru(): perhaps that can
  * make use of such folio pointer flags in future, but for now just keep it for
- * mlock.  We could use three separate folio batches instead, but one feels
- * better (munlocking a full folio batch does not need to drain mlocking folio
- * batches first).
+ * mlock.  We could use separate folio batches instead, but one feels better
+ * (munlocking a full folio batch does not need to drain mlocking batch first).
  */
 static void mlock_folio_batch(struct folio_batch *fbatch)
 {
@@ -169,10 +174,15 @@ static void mlock_folio_batch(struct folio_batch *fbatch)
 
 	for (i = 0; i < folio_batch_count(fbatch); i++) {
 		folio = fbatch->folios[i];
-		mlock = (unsigned long)folio & LRU_FOLIO;
+		mlock = (unsigned long)folio & MLOCK_FLAG;
 		folio = (struct folio *)((unsigned long)folio - mlock);
 		fbatch->folios[i] = folio;
 
+		if (!folio_try_get(folio)) {
+			fbatch->folios[i] = NULL;
+			continue;
+		}
+
 		if (mlock)
 			lruvec = __mlock_folio(folio, lruvec);
 		else
@@ -218,19 +228,25 @@ void mlock_folio(struct folio *folio)
 {
 	struct folio_batch *fbatch;
 
-	local_lock(&mlock_fbatch.lock);
-	fbatch = this_cpu_ptr(&mlock_fbatch.fbatch);
-
 	if (!folio_test_set_mlocked(folio)) {
-		int nr_pages = folio_nr_pages(folio);
+		long nr_pages = folio_nr_pages(folio);
 
 		zone_stat_mod_folio(folio, NR_MLOCK, nr_pages);
-		__count_vm_events(UNEVICTABLE_PGMLOCKED, nr_pages);
+		count_vm_events(UNEVICTABLE_PGMLOCKED, nr_pages);
 	}
 
-	folio_get(folio);
-	if (!folio_batch_add(fbatch, mlock_lru(folio)) ||
-	    true || /* XXX Temporarily disable mlock batching */
+	/*
+	 * No more to do if mlock_count is maintained: either the folio
+	 * is on an lru_add fbatch, and will be moved to unevictable in
+	 * due course, or it's already counted as unevictable: no need
+	 * for an mlock_fbatch entry below.
+	 */
+	if (mod_mlock_count(folio, MLOCK_COUNT_1))
+		return;
+
+	local_lock(&mlock_fbatch.lock);
+	fbatch = this_cpu_ptr(&mlock_fbatch.fbatch);
+	if (!folio_batch_add(fbatch, mlock_flagged(folio)) ||
 	    !folio_may_be_lru_cached(folio) || lru_cache_disabled())
 		mlock_folio_batch(fbatch);
 	local_unlock(&mlock_fbatch.lock);
@@ -244,15 +260,24 @@ void munlock_folio(struct folio *folio)
 {
 	struct folio_batch *fbatch;
 
+	/*
+	 * No more to do if mlock_count is maintained and still raised.
+	 * But if mlock_count is unmaintained, we might need to queue an
+	 * munlock fbatch entry, just to cancel an undequeued mlock entry?
+	 */
+	if (mod_mlock_count(folio, -MLOCK_COUNT_1) > MLOCK_COUNT_0)
+		return;
+
+	if (folio_test_clear_mlocked(folio)) {
+		long nr_pages = folio_nr_pages(folio);
+
+		zone_stat_mod_folio(folio, NR_MLOCK, -nr_pages);
+		count_vm_events(UNEVICTABLE_PGMUNLOCKED, nr_pages);
+	}
+
 	local_lock(&mlock_fbatch.lock);
 	fbatch = this_cpu_ptr(&mlock_fbatch.fbatch);
-	/*
-	 * folio_test_clear_mlocked(folio) must be left to __munlock_folio(),
-	 * which will check whether the folio is multiply mlocked.
-	 */
-	folio_get(folio);
 	if (!folio_batch_add(fbatch, folio) ||
-	    true || /* XXX Temporarily disable munlock batching */
 	    !folio_may_be_lru_cached(folio) || lru_cache_disabled())
 		mlock_folio_batch(fbatch);
 	local_unlock(&mlock_fbatch.lock);
-- 
2.51.0


  parent reply	other threads:[~2026-08-24 14:14 UTC|newest]

Thread overview: 28+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-08-24 13:49 [PATCH 00/25] mm/fbatch: drain lru_add_drain() and _all() Hugh Dickins
2026-08-24 13:52 ` [PATCH 01/25] mm/fbatch: remove !CONFIG_SMP special case of folio_activate() Hugh Dickins
2026-08-24 13:55 ` [PATCH 02/25] mm/fbatch: allow folios_put_refs() to skip xa_is_value() entries Hugh Dickins
2026-08-24 13:58 ` [PATCH 03/25] mm/fbatch: temporarily disable lazyfree and mlock+munlock batching Hugh Dickins
2026-08-24 14:01 ` [PATCH 04/25] mm/fbatch: lru bit set, no extra ref, while folio on per-cpu fbatch Hugh Dickins
2026-08-24 14:03 ` [PATCH 05/25] mm/fbatch: lru_add_del_folio()+folio_add_lru() after clear_lru() Hugh Dickins
2026-08-24 14:06 ` [PATCH 06/25] mm/fbatch: fbatch_drain_lazyfree(onstack fbatch) before ptl unlock Hugh Dickins
2026-08-24 14:09 ` [PATCH 07/25] mm/fbatch: LRU_NEXT_ACTIVATE bit to optimize folio_activate() Hugh Dickins
2026-08-24 14:11 ` [PATCH 08/25] mm/fbatch: replace mlock_new_folio() by __folio_add_lru(,mlockit) Hugh Dickins
2026-08-24 14:14 ` Hugh Dickins [this message]
2026-08-24 14:16 ` [PATCH 10/25] mm/fbatch: remove several uses of mlock_drain_local() Hugh Dickins
2026-08-24 14:18 ` [PATCH 11/25] mm/fbatch: remove migration's PAGE_WAS_MLOCKED lru_add_drain() Hugh Dickins
2026-08-24 14:20 ` [PATCH 12/25] mm/fbatch: remove percpu_pvec_drained and folios_put() Hugh Dickins
2026-08-24 14:23 ` [PATCH 13/25] mm/fbatch: no lru_add_drain() to collect_longterm_unpinnable_folios() Hugh Dickins
2026-08-24 14:55   ` [PATCH alt " Hugh Dickins
2026-08-24 18:42     ` David Hildenbrand (Arm)
2026-08-24 14:25 ` [PATCH 14/25] mm/fbatch: no lru_add_drain() nor _all() for memfd_wait_for_pins() Hugh Dickins
2026-08-24 14:27 ` [PATCH 15/25] mm/fbatch: remove shake_folio() shake_page() from memory-failure Hugh Dickins
2026-08-24 14:30 ` [PATCH 16/25] mm/fbatch: remove lru_cache_disable(() from NUMA folio migration Hugh Dickins
2026-08-24 14:32 ` [PATCH 17/25] mm/fbatch: no lru_cache_disable() in __alloc_contig_migrate_range() Hugh Dickins
2026-08-24 14:34 ` [PATCH 18/25] mm/fbatch: remove lru_add_drain() and _all() calls from various Hugh Dickins
2026-08-24 14:36 ` [PATCH 19/25] mm/fbatch: vm/stat_refresh include lru_add_drain() on each cpu Hugh Dickins
2026-08-24 14:39 ` [PATCH 20/25] s390/fbatch: no lru_add_drain_all() in s390_wiggle_split_folio() Hugh Dickins
2026-08-24 14:41 ` [PATCH 21/25] block/fbatch: no lru_add_drain_all() in invalidate_bdev() Hugh Dickins
2026-08-24 14:44 ` [PATCH 22/25] fs/fbatch: drop_caches invalidate_bh_lrus() not lru_add_drain_all() Hugh Dickins
2026-08-24 14:47 ` [PATCH 23/25] fs,mm/fbatch: use invalidate_bh_lrus() not invalidate_bh_lrus_cpu() Hugh Dickins
2026-08-24 14:49 ` [PATCH 24/25] fs,mm/fbatch: lru_cache_disable() keep off buffer_head lrus only Hugh Dickins
2026-08-24 14:51 ` [PATCH 25/25] mm/fbatch: move lru_add_drain_all() declaration to mm/internal.h Hugh Dickins

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=32d663fb-f192-e0e5-114e-8a92240fb0ac@google.com \
    --to=hughd@google.com \
    --cc=ackerleytng@google.com \
    --cc=akpm@linux-foundation.org \
    --cc=axboe@kernel.dk \
    --cc=baohua@kernel.org \
    --cc=baolin.wang@linux.alibaba.com \
    --cc=bigeasy@linutronix.de \
    --cc=binbin.wu@linux.intel.com \
    --cc=brauner@kernel.org \
    --cc=cl@gentwo.org \
    --cc=david@kernel.org \
    --cc=hannes@cmpxchg.org \
    --cc=hch@lst.de \
    --cc=imbrenda@linux.ibm.com \
    --cc=jack@suse.cz \
    --cc=jp.kobryn@linux.dev \
    --cc=kas@kernel.org \
    --cc=lance.yang@linux.dev \
    --cc=leobras.c@gmail.com \
    --cc=linmiaohe@huawei.com \
    --cc=linux-block@vger.kernel.org \
    --cc=linux-fsdevel@vger.kernel.org \
    --cc=linux-kernel@vger.kernel.org \
    --cc=linux-mm@kvack.org \
    --cc=ljs@kernel.org \
    --cc=mgorman@techsingularity.net \
    --cc=mhocko@suse.com \
    --cc=minchan@kernel.org \
    --cc=mtosatti@redhat.com \
    --cc=muchun.song@linux.dev \
    --cc=osalvador@suse.de \
    --cc=peterz@infradead.org \
    --cc=qi.zheng@linux.dev \
    --cc=riel@surriel.com \
    --cc=ryncsn@gmail.com \
    --cc=shakeel.butt@linux.dev \
    --cc=surenb@google.com \
    --cc=vbabka@kernel.org \
    --cc=viro@zeniv.linux.org.uk \
    --cc=willy@infradead.org \
    --cc=yang@os.amperecomputing.com \
    --cc=yuzhao@google.com \
    --cc=ziy@nvidia.com \
    --cc=zokeefe@google.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox