Linux-mm Archive on lore.kernel.org
 help / color / mirror / Atom feed
From: Rik van Riel <riel@surriel.com>
To: Andrew Morton <akpm@linux-foundation.org>
Cc: linux-kernel@vger.kernel.org, linux-mm@kvack.org,
	kernel-team@meta.com, Dave Hansen <dave.hansen@linux.intel.com>,
	Peter Zijlstra <peterz@infradead.org>,
	Suren Baghdasaryan <surenb@google.com>,
	Lorenzo Stoakes <ljs@kernel.org>,
	Vlastimil Babka <vbabka@kernel.org>,
	David Hildenbrand <david@kernel.org>,
	"Liam R. Howlett" <liam@infradead.org>,
	Mike Rapoport <rppt@kernel.org>, Michal Hocko <mhocko@suse.com>,
	Jason Gunthorpe <jgg@ziepe.ca>,
	John Hubbard <jhubbard@nvidia.com>, Peter Xu <peterx@redhat.com>,
	Matthew Wilcox <willy@infradead.org>,
	Usama Arif <usamaarif642@gmail.com>,
	Rik van Riel <riel@surriel.com>, Shuah Khan <shuah@kernel.org>,
	linux-kselftest@vger.kernel.org
Subject: [PATCH RFC v4 12/12] selftests/mm: add a slow-GUP content and COW test for mTHP
Date: Fri, 24 Jul 2026 18:29:34 -0400	[thread overview]
Message-ID: <20260724222934.1463812-13-riel@surriel.com> (raw)
In-Reply-To: <20260724222934.1463812-1-riel@surriel.com>

follow_page_mask() now batches a PTE-mapped large folio (mTHP) into one
contiguous run for the slow get_user_pages() path. A mis-batched run would
hand back the wrong pages or a stale COW copy, which the existing tests do
not catch: gup_test checks only pin/unpin integrity, and cow exercises COW
mostly at PMD size.

Add mthp_gup_cow_test. It forces 64kB-only mTHP, writes a per-page-distinct
pattern, then pins the region on the slow path (PIN_LONGTERM without
USE_FAST) and compares the bytes the kernel copies back from the pinned
pages against that pattern.

The test covers a read pin, a write pin, and COW after fork(): a child
write-pins to force per-page unshare and checks the copied contents, then
rewrites its copy while the parent verifies its own contents are intact.

It also reports how many pages sit in contiguous large-folio runs, so a
kernel without mTHP shows light coverage rather than passing vacuously.

Assisted-by: Claude:claude-opus-4.8
Signed-off-by: Rik van Riel <riel@surriel.com>
---
 tools/testing/selftests/mm/Makefile           |   1 +
 .../testing/selftests/mm/mthp_gup_cow_test.c  | 213 ++++++++++++++++++
 tools/testing/selftests/mm/run_vmtests.sh     |   1 +
 3 files changed, 215 insertions(+)
 create mode 100644 tools/testing/selftests/mm/mthp_gup_cow_test.c

diff --git a/tools/testing/selftests/mm/Makefile b/tools/testing/selftests/mm/Makefile
index e6df968f0971..6b917a4f73e0 100644
--- a/tools/testing/selftests/mm/Makefile
+++ b/tools/testing/selftests/mm/Makefile
@@ -83,6 +83,7 @@ TEST_GEN_FILES += mrelease_test
 TEST_GEN_FILES += mremap_dontunmap
 TEST_GEN_FILES += mremap_test
 TEST_GEN_FILES += mseal_test
+TEST_GEN_FILES += mthp_gup_cow_test
 TEST_GEN_FILES += on-fault-limit
 TEST_GEN_FILES += pagemap_ioctl
 TEST_GEN_FILES += pfnmap
diff --git a/tools/testing/selftests/mm/mthp_gup_cow_test.c b/tools/testing/selftests/mm/mthp_gup_cow_test.c
new file mode 100644
index 000000000000..52ee329d8641
--- /dev/null
+++ b/tools/testing/selftests/mm/mthp_gup_cow_test.c
@@ -0,0 +1,213 @@
+// SPDX-License-Identifier: GPL-2.0-only
+/*
+ * Verify that the slow GUP path (pin_user_pages -> follow_page_mask ->
+ * follow_pte_batch) returns the correct pages for a PTE-mapped large folio
+ * (mTHP), including that COW copies produce the right content.
+ *
+ * Uses the CONFIG_GUP_TEST PIN_LONGTERM interface: START pins a range on the
+ * slow path, READ copies the pinned pages' bytes back so we can compare them
+ * against the pattern we wrote.
+ */
+#define _GNU_SOURCE
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <stdint.h>
+#include <fcntl.h>
+#include <unistd.h>
+#include <errno.h>
+#include <sys/ioctl.h>
+#include <sys/mman.h>
+#include <sys/wait.h>
+#include <linux/types.h>
+
+#define ARRAY_SIZE(a)	(sizeof(a) / sizeof((a)[0]))
+
+#define GUP_DEV "/sys/kernel/debug/gup_test"
+
+#define PIN_LONGTERM_TEST_START	_IOW('g', 7, struct pin_longterm_test)
+#define PIN_LONGTERM_TEST_STOP	_IO('g', 8)
+#define PIN_LONGTERM_TEST_READ	_IOW('g', 9, __u64)
+#define USE_WRITE	1
+#define USE_FAST	2
+
+struct pin_longterm_test {
+	__u64 addr;
+	__u64 size;
+	__u32 flags;
+};
+
+#define ORDER_KB	64
+#define NR_FOLIOS	16
+#define REGION		((size_t)ORDER_KB * 1024 * NR_FOLIOS)
+
+static long PS;
+static int fails;
+static int tap;
+
+static void ok(int cond, const char *desc)
+{
+	printf("%s %d %s\n", cond ? "ok" : "not ok", ++tap, desc);
+	if (!cond)
+		fails++;
+}
+
+/* Deterministic, per-page-distinct pattern so any mis-order or leak shows. */
+static void fill(char *base, size_t sz, uint32_t salt)
+{
+	for (size_t off = 0; off < sz; off += PS) {
+		uint32_t k = off / PS;
+		uint64_t v = ((uint64_t)salt << 32) ^ (k * 0x9E3779B1u + 0x1234);
+
+		for (size_t i = 0; i < PS; i += sizeof(v))
+			memcpy(base + off + i, &v, sizeof(v));
+	}
+}
+
+static int wsysfs(const char *path, const char *val)
+{
+	int fd = open(path, O_WRONLY);
+
+	if (fd < 0)
+		return -1;
+	int r = write(fd, val, strlen(val));
+
+	close(fd);
+	return r < 0 ? -1 : 0;
+}
+
+/* Force sub-PMD 64kB mTHP only, so faults produce PTE-mapped large folios. */
+static void setup_mthp(void)
+{
+	const char *thp = "/sys/kernel/mm/transparent_hugepage";
+	char p[256];
+	static const int kb[] = { 16, 32, 64, 128, 256, 512, 1024, 2048 };
+
+	wsysfs("/sys/kernel/mm/transparent_hugepage/enabled", "never");
+	for (unsigned int i = 0; i < ARRAY_SIZE(kb); i++) {
+		snprintf(p, sizeof(p), "%s/hugepages-%dkB/enabled", thp, kb[i]);
+		wsysfs(p, kb[i] == ORDER_KB ? "always" : "never");
+	}
+}
+
+/* Count how many pages sit in a contiguous >=ORDER_KB PFN run (via pagemap). */
+static int count_large_pages(char *base, size_t sz)
+{
+	int pm = open("/proc/self/pagemap", O_RDONLY);
+	size_t n = sz / PS, large = 0;
+	uint64_t *pfn = calloc(n, sizeof(*pfn));
+
+	if (pm < 0)
+		return -1;
+	for (size_t k = 0; k < n; k++) {
+		uint64_t ent;
+		off_t idx = ((uintptr_t)base + k * PS) / PS * sizeof(ent);
+
+		if (pread(pm, &ent, sizeof(ent), idx) != sizeof(ent) ||
+		    !(ent & (1ULL << 63)))
+			pfn[k] = 0;
+		else
+			pfn[k] = ent & ((1ULL << 55) - 1);
+	}
+	close(pm);
+	for (size_t k = 0; k < n; k++)
+		if (k + 1 < n && pfn[k] && pfn[k + 1] == pfn[k] + 1)
+			large++;
+	free(pfn);
+	return large;
+}
+
+/* Pin @base..@sz on the slow path, read the pinned bytes back, compare to exp. */
+static int pin_verify(int fd, char *base, size_t sz, uint32_t wr, char *exp)
+{
+	struct pin_longterm_test a = {
+		.addr = (uintptr_t)base, .size = sz,
+		.flags = wr ? USE_WRITE : 0,		/* USE_FAST unset => slow */
+	};
+	char *got = mmap(NULL, sz, PROT_READ | PROT_WRITE,
+			 MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
+	__u64 ga = (uintptr_t)got;
+	int rc = -1;
+
+	if (got == MAP_FAILED)
+		return -1;
+	if (ioctl(fd, PIN_LONGTERM_TEST_START, &a)) {
+		fprintf(stderr, "START(%s) failed: %s\n",
+			wr ? "write" : "read", strerror(errno));
+		goto out;
+	}
+	if (ioctl(fd, PIN_LONGTERM_TEST_READ, &ga)) {
+		fprintf(stderr, "READ failed: %s\n", strerror(errno));
+		ioctl(fd, PIN_LONGTERM_TEST_STOP);
+		goto out;
+	}
+	ioctl(fd, PIN_LONGTERM_TEST_STOP);
+	rc = memcmp(got, exp, sz) ? 1 : 0;
+out:
+	munmap(got, sz);
+	return rc;
+}
+
+int main(void)
+{
+	PS = sysconf(_SC_PAGESIZE);
+	setup_mthp();
+
+	int fd = open(GUP_DEV, O_RDWR);
+
+	if (fd < 0) {
+		fprintf(stderr, "open %s: %s (CONFIG_GUP_TEST?)\n",
+			GUP_DEV, strerror(errno));
+		return 2;
+	}
+
+	char *r = mmap(NULL, REGION, PROT_READ | PROT_WRITE,
+		       MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
+	if (r == MAP_FAILED) {
+		perror("mmap");
+		return 2;
+	}
+	fill(r, REGION, 0xA1);				/* pattern P */
+	char *expP = malloc(REGION);
+
+	memcpy(expP, r, REGION);
+
+	int large = count_large_pages(r, REGION);
+
+	printf("# %d/%zu pages in contiguous large-folio runs\n",
+	       large, REGION / PS);
+	if (large < (int)(REGION / PS) / 4)
+		printf("# WARN: little mTHP backing; batch path lightly covered\n");
+
+	/* A: read pin over writable mTHP -> batches -> content must equal P. */
+	ok(pin_verify(fd, r, REGION, 0, expP) == 0,
+	   "slow read-pin of mTHP returns correct contents");
+
+	/* B: write pin -> FOLL_WRITE batch path -> content must equal P. */
+	ok(pin_verify(fd, r, REGION, 1, expP) == 0,
+	   "slow write-pin of mTHP returns correct contents");
+
+	/* C: COW isolation. Child write-pins (unshares) then rewrites; parent P. */
+	pid_t pid = fork();
+
+	if (pid == 0) {
+		int cfd = open(GUP_DEV, O_RDWR);
+		int a = pin_verify(cfd, r, REGION, 1, expP);	/* COW copy == P */
+
+		fill(r, REGION, 0xB2);			/* child writes Q */
+		_exit(a == 0 ? 0 : 1);
+	}
+	int st = 0;
+
+	waitpid(pid, &st, 0);
+	ok(WIFEXITED(st) && WEXITSTATUS(st) == 0,
+	   "child write-pin after COW returns correct (copied) contents");
+	ok(memcmp(r, expP, REGION) == 0,
+	   "parent contents intact after child COW writes");
+	/* D: parent read-pin again after the COW split still correct. */
+	ok(pin_verify(fd, r, REGION, 0, expP) == 0,
+	   "parent slow read-pin after COW still correct");
+
+	printf("# totals: pass:%d fail:%d\n", tap - fails, fails);
+	return fails ? 1 : 0;
+}
diff --git a/tools/testing/selftests/mm/run_vmtests.sh b/tools/testing/selftests/mm/run_vmtests.sh
index 8c296dedf047..b49a3eb0c205 100755
--- a/tools/testing/selftests/mm/run_vmtests.sh
+++ b/tools/testing/selftests/mm/run_vmtests.sh
@@ -289,6 +289,7 @@ fi
 # Dump pages 0, 19, and 4096, using pin_user_pages:
 CATEGORY="gup_test" run_test ./gup_test -ct -F 0x1 0 19 0x1000
 CATEGORY="gup_test" run_test ./gup_longterm
+CATEGORY="gup_test" run_test ./mthp_gup_cow_test
 
 CATEGORY="userfaultfd" run_test ./uffd-unit-tests
 uffd_stress_bin=./uffd-stress
-- 
2.53.0-Meta



      parent reply	other threads:[~2026-07-24 22:30 UTC|newest]

Thread overview: 13+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-07-24 22:29 [PATCH RFC v4 0/12] mm: use per-VMA lock in __access_remote_vm for improved monitoring reliability Rik van Riel
2026-07-24 22:29 ` [PATCH RFC v4 01/12] x86/mm: add untagged_addr_remote_unlocked() Rik van Riel
2026-07-24 22:29 ` [PATCH RFC v4 02/12] riscv/mm: " Rik van Riel
2026-07-24 22:29 ` [PATCH RFC v4 03/12] mm: rename get_user_page_vma_remote() to get_user_page_lookup_vma() Rik van Riel
2026-07-24 22:29 ` [PATCH RFC v4 04/12] mm/gup: let check_vma_flags() ignore selected VMA flags Rik van Riel
2026-07-24 22:29 ` [PATCH RFC v4 05/12] mm/gup: add get_user_page_vma() to fault in a page under a held lock Rik van Riel
2026-07-24 22:29 ` [PATCH RFC v4 06/12] mm: use per-VMA lock in __access_remote_vm() for single-VMA accesses Rik van Riel
2026-07-24 22:29 ` [PATCH RFC v4 07/12] mm: read remote strings under the per-VMA lock Rik van Riel
2026-07-24 22:29 ` [PATCH RFC v4 08/12] selftests/mm: cover /proc/pid/mem access to VM_PFNMAP memory Rik van Riel
2026-07-24 22:29 ` [PATCH RFC v4 09/12] mm/gup: build get_user_page_lookup_vma() on get_user_page_vma() Rik van Riel
2026-07-24 22:29 ` [PATCH RFC v4 10/12] mm/gup: pass an end address to follow_page_mask() and return a page count Rik van Riel
2026-07-24 22:29 ` [PATCH RFC v4 11/12] mm/gup: batch contiguous PTE-mapped large folios in follow_page_mask() Rik van Riel
2026-07-24 22:29 ` Rik van Riel [this message]

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260724222934.1463812-13-riel@surriel.com \
    --to=riel@surriel.com \
    --cc=akpm@linux-foundation.org \
    --cc=dave.hansen@linux.intel.com \
    --cc=david@kernel.org \
    --cc=jgg@ziepe.ca \
    --cc=jhubbard@nvidia.com \
    --cc=kernel-team@meta.com \
    --cc=liam@infradead.org \
    --cc=linux-kernel@vger.kernel.org \
    --cc=linux-kselftest@vger.kernel.org \
    --cc=linux-mm@kvack.org \
    --cc=ljs@kernel.org \
    --cc=mhocko@suse.com \
    --cc=peterx@redhat.com \
    --cc=peterz@infradead.org \
    --cc=rppt@kernel.org \
    --cc=shuah@kernel.org \
    --cc=surenb@google.com \
    --cc=usamaarif642@gmail.com \
    --cc=vbabka@kernel.org \
    --cc=willy@infradead.org \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox