All of lore.kernel.org
 help / color / mirror / Atom feed
From: Hugh Dickins <hughd@google.com>
To: Andrew Morton <akpm@linux-foundation.org>
Cc: Ackerley Tng <ackerleytng@google.com>,
	 Alexander Viro <viro@zeniv.linux.org.uk>,
	 Baolin Wang <baolin.wang@linux.alibaba.com>,
	 Barry Song <baohua@kernel.org>,
	Binbin Wu <binbin.wu@linux.intel.com>,
	 Christian Brauner <brauner@kernel.org>,
	Christoph Hellwig <hch@lst.de>,
	 Christoph Lameter <cl@gentwo.org>,
	 Claudio Imbrenda <imbrenda@linux.ibm.com>,
	 David Hildenbrand <david@kernel.org>,
	JP Kobryn <jp.kobryn@linux.dev>,  Jan Kara <jack@suse.cz>,
	Jens Axboe <axboe@kernel.dk>,
	 Johannes Weiner <hannes@cmpxchg.org>,
	Kairui Song <ryncsn@gmail.com>,  Kiryl Shutsemau <kas@kernel.org>,
	Lance Yang <lance.yang@linux.dev>,
	 Leonardo Bras <leobras.c@gmail.com>,
	Lorenzo Stoakes <ljs@kernel.org>,
	 Marcelo Tosatti <mtosatti@redhat.com>,
	 Matthew Wilcox <willy@infradead.org>,
	 Mel Gorman <mgorman@techsingularity.net>,
	 Miaohe Lin <linmiaohe@huawei.com>,
	Michal Hocko <mhocko@suse.com>,  Minchan Kim <minchan@kernel.org>,
	Muchun Song <muchun.song@linux.dev>,
	 Oscar Salvador <osalvador@suse.de>,
	Peter Zijlstra <peterz@infradead.org>,
	 Qi Zheng <qi.zheng@linux.dev>, Rik van Riel <riel@surriel.com>,
	 Sebastian Andrzej Siewior <bigeasy@linutronix.de>,
	 Shakeel Butt <shakeel.butt@linux.dev>,
	 Suren Baghdasaryan <surenb@google.com>,
	 Vlastimil Babka <vbabka@kernel.org>,
	 Yang Shi <yang@os.amperecomputing.com>,
	Yu Zhao <yuzhao@google.com>,  Zach O'Keefe <zokeefe@google.com>,
	Zi Yan <ziy@nvidia.com>,
	 linux-block@vger.kernel.org, linux-fsdevel@vger.kernel.org,
	 linux-kernel@vger.kernel.org, linux-mm@kvack.org
Subject: [PATCH 09/25] mm/fbatch: restore mlock+munlock batching, without extra ref
Date: Mon, 24 Aug 2026 07:14:06 -0700 (PDT)	[thread overview]
Message-ID: <32d663fb-f192-e0e5-114e-8a92240fb0ac@google.com> (raw)
In-Reply-To: <14a16945-529b-8bc0-ab38-3ea97e54e223@google.com>

Update mlock_folio(), munlock_folio() and their fbatch callouts and
helpers, to do folio_try_get()s at batch processing time, instead of
holding a folio reference all the while in mlock_fbatch: as in folio.c.

But more interesting is the use of mod_mlock_count(), using try_cmpxchg()
to update folio->mlock_count safely when possible (now when on lru_add
fbatch as well as when unevictable). While __mlock_folio() is as hard to
think about as before, __munlock_folio() simpler because munlock_folio()
can adjust mlock_count itself without clear_lru() or lruvec lock, and so
do the folio_test_clear_mlocked() immediately for itself (without which
unevictable_pgs_cleared was likely to appear high, when it should be 0
or low to indicate good mlock health).

__munlock_folio() is safe for use even when the unreferenced folio has
been freed and reused. It appears that __mlock_folio() could affect a
folio which has been freed and reused, but only if it is reused as an
mlocked folio, in which case its mlock_count is spuriously incremented
(but usually a spurious munlock decrement will follow).  How grave is
this? If unevictable_pgs_cleared remains low, not so bad.

I've gone back and forth on whether to move mlock_fbatch and these
functions into mm/folio.c: for now they stay here in mm/mlock.c.

Signed-off-by: Hugh Dickins <hughd@google.com>
---
 mm/mlock.c | 147 +++++++++++++++++++++++++++++++----------------------
 1 file changed, 86 insertions(+), 61 deletions(-)

diff --git a/mm/mlock.c b/mm/mlock.c
index 53d754e82ba2..1050010bbe0b 100644
--- a/mm/mlock.c
+++ b/mm/mlock.c
@@ -58,6 +58,20 @@ EXPORT_SYMBOL(can_do_mlock);
  * indicate the unevictable state.
  */
 
+static long mod_mlock_count(struct folio *folio, long incdec)
+{
+	long mlock_count = READ_ONCE(folio->mlock_count);
+
+	while (mlock_count & MLOCK_COUNT_0) {
+		if (mlock_count + incdec < MLOCK_COUNT_0)
+			return MLOCK_COUNT_0;
+		if (try_cmpxchg(&folio->mlock_count, &mlock_count,
+				mlock_count + incdec))
+			return mlock_count + incdec;
+	}
+	return 0;
+}
+
 static struct lruvec *__mlock_folio(struct folio *folio, struct lruvec *lruvec)
 {
 	/* There is nothing more we can do while it's off LRU */
@@ -65,6 +79,7 @@ static struct lruvec *__mlock_folio(struct folio *folio, struct lruvec *lruvec)
 		return lruvec;
 
 	lruvec = folio_lruvec_relock_irq(folio, lruvec);
+	lruvec_del_folio(lruvec, folio);
 
 	if (unlikely(folio_evictable(folio))) {
 		/*
@@ -73,92 +88,82 @@ static struct lruvec *__mlock_folio(struct folio *folio, struct lruvec *lruvec)
 		 * folio be unevictable?  I'm not sure, but move it now if so.
 		 */
 		if (folio_test_unevictable(folio)) {
-			lruvec_del_folio(lruvec, folio);
 			folio_clear_unevictable(folio);
-			lruvec_add_folio(lruvec, folio);
-
 			__count_vm_events(UNEVICTABLE_PGRESCUED,
 					  folio_nr_pages(folio));
 		}
 		goto out;
 	}
 
+	/*
+	 * Something to keep in mind when studying the arithmetic here:
+	 * we only come to __mlock_folio() when mlock_folio() could not
+	 * mod_mlock_count() itself; but by the time this is processed,
+	 * the folio may have already been munlocked, or another mlock
+	 * already marked it as unevictable and so mod_mlock_countable.
+	 * And don't forget that a folio may be unevictable for reasons
+	 * other than mlocked (hence the folio_evictable() check above).
+	 */
+
 	if (folio_test_unevictable(folio)) {
 		if (folio_test_mlocked(folio))
-			folio->mlock_count += MLOCK_COUNT_1;
+			mod_mlock_count(folio, MLOCK_COUNT_1);
 		goto out;
 	}
 
-	lruvec_del_folio(lruvec, folio);
 	folio_clear_active(folio);
 	folio_set_unevictable(folio);
-	folio->mlock_count = MLOCK_COUNT_0;
-	if (folio_test_mlocked(folio))
-		folio->mlock_count += MLOCK_COUNT_1;
-	lruvec_add_folio(lruvec, folio);
 	__count_vm_events(UNEVICTABLE_PGCULLED, folio_nr_pages(folio));
+
+	if (!folio_test_mlocked(folio))
+		folio->mlock_count = MLOCK_COUNT_0;
+	else if (!mod_mlock_count(folio, MLOCK_COUNT_1))
+		folio->mlock_count = MLOCK_COUNT_0 + MLOCK_COUNT_1;
 out:
+	lruvec_add_folio(lruvec, folio);
 	folio_set_lru(folio);
 	return lruvec;
 }
 
 static struct lruvec *__munlock_folio(struct folio *folio, struct lruvec *lruvec)
 {
-	int nr_pages = folio_nr_pages(folio);
-	bool isolated = false;
+	long nr_pages = folio_nr_pages(folio);
 
-	if (!folio_test_clear_lru(folio))
-		goto munlock;
-
-	isolated = true;
-	lruvec = folio_lruvec_relock_irq(folio, lruvec);
-
-	if (folio_test_unevictable(folio)) {
-		/* Then mlock_count is maintained, but might undercount */
-		if (folio->mlock_count > MLOCK_COUNT_0)
-			folio->mlock_count -= MLOCK_COUNT_1;
-		if (folio->mlock_count > MLOCK_COUNT_0)
-			goto out;
-	}
-	/* else assume that was the last mlock: reclaim will fix it if not */
-
-munlock:
-	if (folio_test_clear_mlocked(folio)) {
-		__zone_stat_mod_folio(folio, NR_MLOCK, -nr_pages);
-		if (isolated || !folio_test_unevictable(folio))
-			__count_vm_events(UNEVICTABLE_PGMUNLOCKED, nr_pages);
-		else
+	/* There is nothing more we can do while it's off LRU */
+	if (!folio_test_clear_lru(folio)) {
+		if (folio_test_unevictable(folio) && folio_evictable(folio))
 			__count_vm_events(UNEVICTABLE_PGSTRANDED, nr_pages);
+		/* But whoever puts it back on LRU should rescue it */
+		return lruvec;
 	}
 
-	/* folio_evictable() has to be checked *after* clearing Mlocked */
-	if (isolated && folio_test_unevictable(folio) && folio_evictable(folio)) {
-		lruvec_del_folio(lruvec, folio);
+	lruvec = folio_lruvec_relock_irq(folio, lruvec);
+	lruvec_del_folio(lruvec, folio);
+
+	if (folio_test_unevictable(folio) && folio_evictable(folio)) {
 		folio_clear_unevictable(folio);
-		lruvec_add_folio(lruvec, folio);
 		__count_vm_events(UNEVICTABLE_PGRESCUED, nr_pages);
 	}
-out:
-	if (isolated)
-		folio_set_lru(folio);
+
+	lruvec_add_folio(lruvec, folio);
+	folio_set_lru(folio);
 	return lruvec;
 }
 
 /*
- * Flags held in the low bits of a struct folio pointer on the mlock_fbatch.
+ * Flag held in the low bits of a struct folio pointer on the mlock_fbatch.
  */
-#define LRU_FOLIO 0x1
-static inline struct folio *mlock_lru(struct folio *folio)
+#define MLOCK_FLAG 0x1
+static inline struct folio *mlock_flagged(struct folio *folio)
 {
-	return (struct folio *)((unsigned long)folio + LRU_FOLIO);
+	return (struct folio *)((unsigned long)folio + MLOCK_FLAG);
 }
 
 /*
  * mlock_folio_batch() is derived from folio_batch_move_lru(): perhaps that can
  * make use of such folio pointer flags in future, but for now just keep it for
- * mlock.  We could use three separate folio batches instead, but one feels
- * better (munlocking a full folio batch does not need to drain mlocking folio
- * batches first).
+ * mlock.  We could use separate folio batches instead, but one feels better
+ * (munlocking a full folio batch does not need to drain mlocking batch first).
  */
 static void mlock_folio_batch(struct folio_batch *fbatch)
 {
@@ -169,10 +174,15 @@ static void mlock_folio_batch(struct folio_batch *fbatch)
 
 	for (i = 0; i < folio_batch_count(fbatch); i++) {
 		folio = fbatch->folios[i];
-		mlock = (unsigned long)folio & LRU_FOLIO;
+		mlock = (unsigned long)folio & MLOCK_FLAG;
 		folio = (struct folio *)((unsigned long)folio - mlock);
 		fbatch->folios[i] = folio;
 
+		if (!folio_try_get(folio)) {
+			fbatch->folios[i] = NULL;
+			continue;
+		}
+
 		if (mlock)
 			lruvec = __mlock_folio(folio, lruvec);
 		else
@@ -218,19 +228,25 @@ void mlock_folio(struct folio *folio)
 {
 	struct folio_batch *fbatch;
 
-	local_lock(&mlock_fbatch.lock);
-	fbatch = this_cpu_ptr(&mlock_fbatch.fbatch);
-
 	if (!folio_test_set_mlocked(folio)) {
-		int nr_pages = folio_nr_pages(folio);
+		long nr_pages = folio_nr_pages(folio);
 
 		zone_stat_mod_folio(folio, NR_MLOCK, nr_pages);
-		__count_vm_events(UNEVICTABLE_PGMLOCKED, nr_pages);
+		count_vm_events(UNEVICTABLE_PGMLOCKED, nr_pages);
 	}
 
-	folio_get(folio);
-	if (!folio_batch_add(fbatch, mlock_lru(folio)) ||
-	    true || /* XXX Temporarily disable mlock batching */
+	/*
+	 * No more to do if mlock_count is maintained: either the folio
+	 * is on an lru_add fbatch, and will be moved to unevictable in
+	 * due course, or it's already counted as unevictable: no need
+	 * for an mlock_fbatch entry below.
+	 */
+	if (mod_mlock_count(folio, MLOCK_COUNT_1))
+		return;
+
+	local_lock(&mlock_fbatch.lock);
+	fbatch = this_cpu_ptr(&mlock_fbatch.fbatch);
+	if (!folio_batch_add(fbatch, mlock_flagged(folio)) ||
 	    !folio_may_be_lru_cached(folio) || lru_cache_disabled())
 		mlock_folio_batch(fbatch);
 	local_unlock(&mlock_fbatch.lock);
@@ -244,15 +260,24 @@ void munlock_folio(struct folio *folio)
 {
 	struct folio_batch *fbatch;
 
+	/*
+	 * No more to do if mlock_count is maintained and still raised.
+	 * But if mlock_count is unmaintained, we might need to queue an
+	 * munlock fbatch entry, just to cancel an undequeued mlock entry?
+	 */
+	if (mod_mlock_count(folio, -MLOCK_COUNT_1) > MLOCK_COUNT_0)
+		return;
+
+	if (folio_test_clear_mlocked(folio)) {
+		long nr_pages = folio_nr_pages(folio);
+
+		zone_stat_mod_folio(folio, NR_MLOCK, -nr_pages);
+		count_vm_events(UNEVICTABLE_PGMUNLOCKED, nr_pages);
+	}
+
 	local_lock(&mlock_fbatch.lock);
 	fbatch = this_cpu_ptr(&mlock_fbatch.fbatch);
-	/*
-	 * folio_test_clear_mlocked(folio) must be left to __munlock_folio(),
-	 * which will check whether the folio is multiply mlocked.
-	 */
-	folio_get(folio);
 	if (!folio_batch_add(fbatch, folio) ||
-	    true || /* XXX Temporarily disable munlock batching */
 	    !folio_may_be_lru_cached(folio) || lru_cache_disabled())
 		mlock_folio_batch(fbatch);
 	local_unlock(&mlock_fbatch.lock);
-- 
2.51.0


  parent reply	other threads:[~2026-08-24 14:14 UTC|newest]

Thread overview: 46+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-08-24 13:49 [PATCH 00/25] mm/fbatch: drain lru_add_drain() and _all() Hugh Dickins
2026-08-24 13:52 ` [PATCH 01/25] mm/fbatch: remove !CONFIG_SMP special case of folio_activate() Hugh Dickins
2026-08-27 17:23   ` David Hildenbrand (Arm)
2026-08-24 13:55 ` [PATCH 02/25] mm/fbatch: allow folios_put_refs() to skip xa_is_value() entries Hugh Dickins
2026-08-27 17:35   ` David Hildenbrand (Arm)
2026-08-24 13:58 ` [PATCH 03/25] mm/fbatch: temporarily disable lazyfree and mlock+munlock batching Hugh Dickins
2026-08-24 14:01 ` [PATCH 04/25] mm/fbatch: lru bit set, no extra ref, while folio on per-cpu fbatch Hugh Dickins
2026-08-27 11:46   ` Kiryl Shutsemau
2026-08-28  8:04     ` Hugh Dickins
2026-08-28 14:40       ` Matthew Wilcox
2026-08-28 22:20         ` Hugh Dickins
2026-08-24 14:03 ` [PATCH 05/25] mm/fbatch: lru_add_del_folio()+folio_add_lru() after clear_lru() Hugh Dickins
2026-08-24 14:06 ` [PATCH 06/25] mm/fbatch: fbatch_drain_lazyfree(onstack fbatch) before ptl unlock Hugh Dickins
2026-08-24 14:09 ` [PATCH 07/25] mm/fbatch: LRU_NEXT_ACTIVATE bit to optimize folio_activate() Hugh Dickins
2026-08-27 12:02   ` Kiryl Shutsemau
2026-08-28  8:40     ` Hugh Dickins
2026-08-24 14:11 ` [PATCH 08/25] mm/fbatch: replace mlock_new_folio() by __folio_add_lru(,mlockit) Hugh Dickins
2026-08-24 14:14 ` Hugh Dickins [this message]
2026-08-24 14:16 ` [PATCH 10/25] mm/fbatch: remove several uses of mlock_drain_local() Hugh Dickins
2026-08-24 14:18 ` [PATCH 11/25] mm/fbatch: remove migration's PAGE_WAS_MLOCKED lru_add_drain() Hugh Dickins
2026-08-24 14:20 ` [PATCH 12/25] mm/fbatch: remove percpu_pvec_drained and folios_put() Hugh Dickins
2026-08-24 14:23 ` [PATCH 13/25] mm/fbatch: no lru_add_drain() to collect_longterm_unpinnable_folios() Hugh Dickins
2026-08-24 14:55   ` [PATCH alt " Hugh Dickins
2026-08-24 18:42     ` David Hildenbrand (Arm)
2026-08-27  9:16       ` Hugh Dickins
2026-08-27  9:20         ` David Hildenbrand (Arm)
2026-08-24 14:25 ` [PATCH 14/25] mm/fbatch: no lru_add_drain() nor _all() for memfd_wait_for_pins() Hugh Dickins
2026-08-24 14:27 ` [PATCH 15/25] mm/fbatch: remove shake_folio() shake_page() from memory-failure Hugh Dickins
2026-08-24 14:30 ` [PATCH 16/25] mm/fbatch: remove lru_cache_disable(() from NUMA folio migration Hugh Dickins
2026-08-24 14:32 ` [PATCH 17/25] mm/fbatch: no lru_cache_disable() in __alloc_contig_migrate_range() Hugh Dickins
2026-08-24 14:34 ` [PATCH 18/25] mm/fbatch: remove lru_add_drain() and _all() calls from various Hugh Dickins
2026-08-24 14:36 ` [PATCH 19/25] mm/fbatch: vm/stat_refresh include lru_add_drain() on each cpu Hugh Dickins
2026-08-24 14:39 ` [PATCH 20/25] s390/fbatch: no lru_add_drain_all() in s390_wiggle_split_folio() Hugh Dickins
2026-08-26 13:57   ` Claudio Imbrenda
2026-08-27  8:49     ` Hugh Dickins
2026-08-27 12:55       ` Claudio Imbrenda
2026-08-27 20:26         ` David Hildenbrand (Arm)
2026-08-28  8:57         ` Hugh Dickins
2026-08-28 15:02           ` Claudio Imbrenda
2026-08-28 22:02             ` Hugh Dickins
2026-08-24 14:41 ` [PATCH 21/25] block/fbatch: no lru_add_drain_all() in invalidate_bdev() Hugh Dickins
2026-08-24 14:44 ` [PATCH 22/25] fs/fbatch: drop_caches invalidate_bh_lrus() not lru_add_drain_all() Hugh Dickins
2026-08-24 14:47 ` [PATCH 23/25] fs,mm/fbatch: use invalidate_bh_lrus() not invalidate_bh_lrus_cpu() Hugh Dickins
2026-08-24 14:49 ` [PATCH 24/25] fs,mm/fbatch: lru_cache_disable() keep off buffer_head lrus only Hugh Dickins
2026-08-24 14:51 ` [PATCH 25/25] mm/fbatch: move lru_add_drain_all() declaration to mm/internal.h Hugh Dickins
2026-08-27 17:16 ` [PATCH 00/25] mm/fbatch: drain lru_add_drain() and _all() David Hildenbrand (Arm)

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=32d663fb-f192-e0e5-114e-8a92240fb0ac@google.com \
    --to=hughd@google.com \
    --cc=ackerleytng@google.com \
    --cc=akpm@linux-foundation.org \
    --cc=axboe@kernel.dk \
    --cc=baohua@kernel.org \
    --cc=baolin.wang@linux.alibaba.com \
    --cc=bigeasy@linutronix.de \
    --cc=binbin.wu@linux.intel.com \
    --cc=brauner@kernel.org \
    --cc=cl@gentwo.org \
    --cc=david@kernel.org \
    --cc=hannes@cmpxchg.org \
    --cc=hch@lst.de \
    --cc=imbrenda@linux.ibm.com \
    --cc=jack@suse.cz \
    --cc=jp.kobryn@linux.dev \
    --cc=kas@kernel.org \
    --cc=lance.yang@linux.dev \
    --cc=leobras.c@gmail.com \
    --cc=linmiaohe@huawei.com \
    --cc=linux-block@vger.kernel.org \
    --cc=linux-fsdevel@vger.kernel.org \
    --cc=linux-kernel@vger.kernel.org \
    --cc=linux-mm@kvack.org \
    --cc=ljs@kernel.org \
    --cc=mgorman@techsingularity.net \
    --cc=mhocko@suse.com \
    --cc=minchan@kernel.org \
    --cc=mtosatti@redhat.com \
    --cc=muchun.song@linux.dev \
    --cc=osalvador@suse.de \
    --cc=peterz@infradead.org \
    --cc=qi.zheng@linux.dev \
    --cc=riel@surriel.com \
    --cc=ryncsn@gmail.com \
    --cc=shakeel.butt@linux.dev \
    --cc=surenb@google.com \
    --cc=vbabka@kernel.org \
    --cc=viro@zeniv.linux.org.uk \
    --cc=willy@infradead.org \
    --cc=yang@os.amperecomputing.com \
    --cc=yuzhao@google.com \
    --cc=ziy@nvidia.com \
    --cc=zokeefe@google.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.