Linux-mm Archive on lore.kernel.org
 help / color / mirror / Atom feed
From: Xueyuan Chen <xueyuan.chen21@gmail.com>
To: baohua@kernel.org
Cc: akpm@linux-foundation.org, linux-mm@kvack.org,
	axelrasmussen@google.com, baolin.wang@linux.alibaba.com,
	baoquan.he@linux.dev, chenridong@xiaomi.com, david@kernel.org,
	hannes@cmpxchg.org, kasong@tencent.com, lianux.mm@gmail.com,
	linux-kernel@vger.kernel.org, ljs@kernel.org,
	lyugaofei@xiaomi.com, mhocko@kernel.org, qi.zheng@linux.dev,
	shakeel.butt@linux.dev, stevensd@chromium.org,
	wangzicheng@honor.com, weixugc@google.com, yuanchu@google.com,
	zhangbo56@xiaomi.com
Subject: Re: [PATCH 0/6] mm/mglru: speed up inc_min_seq() and fix cold/hot inversions
Date: Thu, 27 Aug 2026 11:54:04 +0800	[thread overview]
Message-ID: <20260827035416.3012015-1-xueyuan.chen21@gmail.com> (raw)
In-Reply-To: <20260821102538.22642-1-baohua@kernel.org>


On Fri, Aug 21, 2026 at 06:25:32PM +0800, Barry Song (Xiaomi) wrote:
>This is an aging speedup series split out from the MGLRU swappiness
>series [1], with the inc_min_seq changes separated to make them
>easier to review.
>
>Currently, inc_min_seq performance is crucial to both the swappiness
>fix and proactive aging. There are two problems with it:
>
>1. It processes each folio one by one, while many operations can be
>batched or skipped. For example, a batch of folios can be moved together
>from the oldest generation to the second-oldest generation, and the
>associated counting can also be done in batches.
>
>2. It may cause potential cold/hot inversion by placing promoted folios
>(which have been scanned and found to have young PTEs) behind
>non-promoted folios. A similar inversion can also occur among
>non-promoted folios, as tail folios from the oldest generation are
>placed before head folios when moving them to the second-oldest
>generation.
>
>This series tries to batch operations as much as possible and fix the
>potential cold/hot inversion by keeping promoted folios ahead of
>non-promoted folios, while also preserving the order of non-promoted
>folios when moving them from the oldest generation to the second-oldest
>generation.
>
>Minor issue: inc_min_seq() also counts protected folios
>improperly, as promoted folios should be skipped, as in
>sort_folio().
>
>We need a stable workload with a stable number of folios to measure
>aging and evaluate the speedup in inc_min_seq(). So I asked ChatGPT
>to generate the microbenchmark below. It ages an LRU vec containing
>512 MB of memory 100 times:
>
> #define _GNU_SOURCE
> 
> #include <stdio.h>
> #include <stdlib.h>
> #include <string.h>
> #include <stdint.h>
> #include <unistd.h>
> #include <fcntl.h>
> #include <errno.h>
> #include <time.h>
> #include <sys/mman.h>
> 
> #define SIZE		(512UL * 1024 * 1024)
> #define LRU_GEN		"/sys/kernel/debug/lru_gen"
> #define TARGET_CGROUP	"/system.slice/agetest.scope"
> #define START_GEN	3
> #define END_GEN	103
> 
> static long long nsec_diff(const struct timespec *start,
> 			   const struct timespec *end)
> {
> 	return (end->tv_sec - start->tv_sec) * 1000000000LL +
> 	       (end->tv_nsec - start->tv_nsec);
> }
> 
> static int find_memcg_id(void)
> {
> 	FILE *fp;
> 	char line[4096];
> 	int memcg_id;
> 
> 	fp = fopen(LRU_GEN, "r");
> 	if (!fp) {
> 		perror("fopen lru_gen");
> 		return -1;
> 	}
> 
> 	while (fgets(line, sizeof(line), fp)) {
> 		char *p;
> 
> 		if (strncmp(line, "memcg ", 6))
> 			continue;
> 
> 		p = line + 6;
> 
> 		if (sscanf(p, "%d", &memcg_id) != 1)
> 			continue;
> 
> 		/*
> 		 * The memcg path follows the numeric ID.
> 		 */
> 		p = strchr(p, ' ');
> 		if (!p)
> 			continue;
> 
> 		if (strstr(p, TARGET_CGROUP)) {
> 			fclose(fp);
> 			return memcg_id;
> 		}
> 	}
> 
> 	fclose(fp);
> 
> 	fprintf(stderr, "Cannot find %s\n", TARGET_CGROUP);
> 	return -1;
> }
> 
> int main(void)
> {
> 	void *addr;
> 	int memcg_id;
> 	int fd;
> 	long long total_ns = 0;
> 
> 	/*
> 	 * mmap 512 MB and touch every page.
> 	 */
> 	addr = mmap(NULL, SIZE, PROT_READ | PROT_WRITE,
> 		    MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
> 	if (addr == MAP_FAILED) {
> 		perror("mmap");
> 		return 1;
> 	}
> 
> 	memset(addr, 0x55, SIZE);
> 
> 	printf("mmap: %p, size: %lu MB\n",
> 	       addr, SIZE / 1024 / 1024);
> 
> 	/*
> 	 * Find the memcg ID automatically.
> 	 */
> 	memcg_id = find_memcg_id();
> 	if (memcg_id < 0)
> 		return 1;
> 
> 	printf("memcg: %d (%s)\n", memcg_id, TARGET_CGROUP);
> 	printf("aging generation %d -> %d\n",
> 	       START_GEN, END_GEN);
> 
> 	fd = open(LRU_GEN, O_WRONLY);
> 	if (fd < 0) {
> 		perror("open lru_gen");
> 		return 1;
> 	}
> 
> 	for (int gen = START_GEN; gen <= END_GEN; gen++) {
> 		char buf[128];
> 		int len;
> 		struct timespec start, end;
> 		long long ns;
> 
> 		len = snprintf(buf, sizeof(buf),
> 			       "+ %d 0 %d\n", memcg_id, gen);
> 
> 		clock_gettime(CLOCK_MONOTONIC, &start);
> 
> 		if (write(fd, buf, len) != len) {
> 			perror("write lru_gen");
> 			close(fd);
> 			return 1;
> 		}
> 
> 		clock_gettime(CLOCK_MONOTONIC, &end);
> 
> 		ns = nsec_diff(&start, &end);
> 		total_ns += ns;
> 
> 		printf("gen %3d: %8.3f ms\n",
> 		       gen, ns / 1000000.0);
> 		fflush(stdout);
> 	}
> 
> 	close(fd);
> 
> 	printf("\nTotal:   %.3f ms\n",
> 	       total_ns / 1000000.0);
> 	printf("Average: %.3f ms\n",
> 	       total_ns / (double)(END_GEN - START_GEN + 1) /
> 	       1000000.0);
> 
> 	while (1)
> 		sleep(1);
> 
> 	return 0;
> }
>
>Run the above microbenchmark with:
>systemd-run --scope --unit=agetest -p MemoryMax=1024M ./agetest
>
>I’m seeing inc_min_seq() become significantly faster:
>
>W/o patch:
>
>Running scope as unit: agetest.scope
>mmap: 0x72c1b5a00000, size: 512 MB
>memcg: 12673 (/system.slice/agetest.scope)
>aging generation 3 -> 103
>gen   3:    7.433 ms
>gen   4:    0.949 ms
>gen   5:    2.535 ms
>gen   6:    5.043 ms
>gen   7:    5.041 ms
>gen   8:    5.027 ms
>...
>gen 100:    5.035 ms
>gen 101:    5.011 ms
>gen 102:    5.029 ms
>gen 103:    5.056 ms
>
>Total:   503.946 ms
>Average: 4.990 ms
>
>W/ patch:
>
>Running scope as unit: agetest.scope
>mmap: 0x7c4b1d200000, size: 512 MB
>memcg: 12893 (/system.slice/agetest.scope)
>aging generation 3 -> 103
>gen   3:    7.538 ms
>gen   4:    0.937 ms
>gen   5:    2.348 ms
>gen   6:    2.300 ms
>gen   7:    2.302 ms
>gen   8:    2.294 ms
>gen   9:    2.296 ms
>...
>gen 100:    2.292 ms
>gen 101:    2.307 ms
>gen 102:    2.293 ms
>gen 103:    2.293 ms
>
>Total:   235.718 ms
>Average: 2.334 ms
>
>The average aging time drops from 4.990 ms to 2.334 ms!

Hi Barry,

I tested this series on my arm64 machine (24 cores, 4K base pages)
and reproduced the improvement with the microbenchmark from the
cover letter.

THP=never (PTE):
  baseline: 7.644 ms
  patched:  2.964 ms (-61.2%)

THP=always (PMD):
  baseline: 0.0373 ms
  patched:  0.0292 ms (-21.8%)

The PTE-level gain is larger than your x86 numbers (-61.2% vs
-53.2%). I suspect the per-folio cost of inc_min_seq() is higher
on my arm64 machine (or on arm64 in general).

The gain is also larger at the PTE level than at the PMD level
(-61.2% vs -21.8%), matching the much higher folio count
(131072 vs 256).

Tested-by: Xueyuan Chen <xueyuan.chen21@gmail.com>

thanks,
Xueyuan

>[1] https://lore.kernel.org/linux-mm/20260812121658.69965-1-baohua@kernel.org/
>
>Barry Song (Xiaomi) (6):
>  mm/mglru: batch update lrugen->nr_pages in inc_min_seq()
>  mm/mglru: batch update lrugen->protected in inc_min_seq()
>  mm/mglru: enhance cold/hot inversion handling in inc_min_seq()
>  mm/mglru: exclude folios promoted by aging from protected in
>    inc_min_seq()
>  mm/mglru: move folios from oldest gen to second-oldest gen from head
>    to tail
>  mm/mglru: batch move folios to the second-oldest gen's LRU
>
> mm/vmscan.c | 98 +++++++++++++++++++++++++++++++++++++++++++----------
> 1 file changed, 80 insertions(+), 18 deletions(-)
>
>-- 
>2.34.1
>
>



      parent reply	other threads:[~2026-08-27  3:54 UTC|newest]

Thread overview: 31+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-08-21 10:25 [PATCH 0/6] mm/mglru: speed up inc_min_seq() and fix cold/hot inversions Barry Song (Xiaomi)
2026-08-21 10:25 ` [PATCH 1/6] mm/mglru: batch update lrugen->nr_pages in inc_min_seq() Barry Song (Xiaomi)
2026-08-22  1:42   ` Lian Wang (ProcessMission)
2026-08-25 21:38     ` Barry Song
2026-08-26  8:23   ` Baoquan He
2026-08-27  3:20   ` Kairui Song
2026-08-27 11:21     ` Barry Song
2026-08-27 11:30       ` Kairui Song
2026-08-21 10:25 ` [PATCH 2/6] mm/mglru: batch update lrugen->protected " Barry Song (Xiaomi)
2026-08-26  9:10   ` Baoquan He
2026-08-27  5:09     ` Barry Song
2026-08-27 12:14       ` Xueyuan Chen
2026-08-21 10:25 ` [PATCH 3/6] mm/mglru: enhance cold/hot inversion handling " Barry Song (Xiaomi)
2026-08-26  8:56   ` Baoquan He
2026-08-26 21:43     ` Barry Song
2026-08-27  0:46       ` Baoquan He
2026-08-27  1:24         ` Barry Song
2026-08-27  2:14           ` Baoquan He
2026-08-27  2:19             ` Baoquan He
2026-08-27  4:30             ` Kairui Song
2026-08-27  6:13               ` Baoquan He
2026-08-27  4:37   ` Kairui Song
2026-08-21 10:25 ` [PATCH 4/6] mm/mglru: exclude folios promoted by aging from protected " Barry Song (Xiaomi)
2026-08-26  8:57   ` Baoquan He
2026-08-21 10:25 ` [PATCH 5/6] mm/mglru: move folios from oldest gen to second-oldest gen from head to tail Barry Song (Xiaomi)
2026-08-22  5:45   ` Kairui Song
2026-08-25 21:32     ` Barry Song
2026-08-26  9:06   ` Baoquan He
2026-08-21 10:25 ` [PATCH 6/6] mm/mglru: batch move folios to the second-oldest gen's LRU Barry Song (Xiaomi)
2026-08-26  9:34   ` Baoquan He
2026-08-27  3:54 ` Xueyuan Chen [this message]

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260827035416.3012015-1-xueyuan.chen21@gmail.com \
    --to=xueyuan.chen21@gmail.com \
    --cc=akpm@linux-foundation.org \
    --cc=axelrasmussen@google.com \
    --cc=baohua@kernel.org \
    --cc=baolin.wang@linux.alibaba.com \
    --cc=baoquan.he@linux.dev \
    --cc=chenridong@xiaomi.com \
    --cc=david@kernel.org \
    --cc=hannes@cmpxchg.org \
    --cc=kasong@tencent.com \
    --cc=lianux.mm@gmail.com \
    --cc=linux-kernel@vger.kernel.org \
    --cc=linux-mm@kvack.org \
    --cc=ljs@kernel.org \
    --cc=lyugaofei@xiaomi.com \
    --cc=mhocko@kernel.org \
    --cc=qi.zheng@linux.dev \
    --cc=shakeel.butt@linux.dev \
    --cc=stevensd@chromium.org \
    --cc=wangzicheng@honor.com \
    --cc=weixugc@google.com \
    --cc=yuanchu@google.com \
    --cc=zhangbo56@xiaomi.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox