Linux Test Project
 help / color / mirror / Atom feed
* [LTP] [RFC PATCH] getrusage03: account for percpu RSS counter batching
@ 2026-08-25 15:25 Nirmoy Das via ltp
  2026-08-25 17:34 ` [LTP] " linuxtestproject.agent
  2026-09-02 12:25 ` [LTP] [RFC PATCH] " Cyril Hrubis
  0 siblings, 2 replies; 12+ messages in thread
From: Nirmoy Das via ltp @ 2026-08-25 15:25 UTC (permalink / raw)
  To: ltp; +Cc: Nirmoy Das

getrusage03 allows ru_maxrss to differ from the expected value by a
fixed 20 MiB. That can be too small on large systems after Linux commit
f1a7941243c1 ("mm: convert mm's rss stats into percpu_counter").

The fast RSS read may miss percpu counter batches, and the batch size
scales with the number of online CPUs. The test child is single threaded,
but it can migrate while faulting memory, so allow two batches worth of
slack based on the online CPU count and page size. Keep the existing
20 MiB floor for smaller systems.

This is sent as RFC because it makes the test tolerate Linux's current
percpu counter batching behavior instead of treating the smaller
ru_maxrss value as a kernel failure.

Signed-off-by: Nirmoy Das <nirmoyd@nvidia.com>
---
 .../kernel/syscalls/getrusage/getrusage03.h     | 17 +++++++++++++++--
 1 file changed, 15 insertions(+), 2 deletions(-)

diff --git a/testcases/kernel/syscalls/getrusage/getrusage03.h b/testcases/kernel/syscalls/getrusage/getrusage03.h
index 58a98b430..9a7d8e8fa 100644
--- a/testcases/kernel/syscalls/getrusage/getrusage03.h
+++ b/testcases/kernel/syscalls/getrusage/getrusage03.h
@@ -7,9 +7,20 @@
 #define LTP_GETRUSAGE03_H
 
 #include <sched.h>
+#include "tst_cpu.h"
 #include "tst_test.h"
 
-#define DELTA_MAX 20480
+#define DELTA_MAX 20480L
+#define MAXRSS_BATCHES 2L
+
+static long maxrss_delta(void)
+{
+	long batch = MAX(32L, tst_ncpus() * 2);
+	long delta = MAXRSS_BATCHES * batch * SAFE_SYSCONF(_SC_PAGESIZE) / 1024;
+
+	/* Linux RSS counters can miss percpu counter batches after migration. */
+	return MAX(DELTA_MAX, delta);
+}
 
 static void force_context_switches(int iterations)
 {
@@ -40,7 +51,9 @@ static void consume_mb(int consume_nr)
 
 static int is_in_delta(long value)
 {
-	return (value >= -DELTA_MAX && value <= DELTA_MAX);
+	long delta = maxrss_delta();
+
+	return (value >= -delta && value <= delta);
 }
 
 #endif //LTP_GETRUSAGE03_H
-- 
2.43.0


-- 
Mailing list info: https://lists.linux.it/listinfo/ltp

^ permalink raw reply related	[flat|nested] 12+ messages in thread
* [LTP] [PATCH v3] getrusage03: account for percpu RSS counter batching
@ 2026-10-02 13:17 Nirmoy Das via ltp
  2026-10-02 13:40 ` [LTP] " linuxtestproject.agent
  0 siblings, 1 reply; 12+ messages in thread
From: Nirmoy Das via ltp @ 2026-10-02 13:17 UTC (permalink / raw)
  To: ltp; +Cc: Nirmoy Das

After f1a7941243c1 ("mm: convert mm's rss stats into percpu_counter"),
the fixed 20 MiB tolerance can be too small on large systems. Pinning
limits pending RSS updates to one CPU, but that CPU can still hold one
batch outside the fast RSS read.

Pin the test to its current CPU and allow one batch of lower-side slack
for the 100, 300 and 400 MiB allocation checks. Keep the existing upper
and inheritance tolerances, and skip if one batch is at least 100 MiB.

On 352 CPUs with 64 KiB pages, one batch is 44 MiB. Pinning alone
reported 270336 KiB for the 300 MiB check, outside the 20 MiB tolerance.
The updated test passed 10/10 runs on the 352-CPU Vera system.

Suggested-by: Cyril Hrubis <chrubis@suse.cz>
Co-developed-by: Jan Stancek <jstancek@redhat.com>
Signed-off-by: Jan Stancek <jstancek@redhat.com>
Assisted-by: ClaudeCode:claude-opus-5-5
Signed-off-by: Nirmoy Das <nirmoyd@nvidia.com>
---
Changes in v3:
- Use tst_ncpus() and pin to the current CPU (Cyril).
- Allocate the CPU mask dynamically and log the allowance.

v2: https://lore.kernel.org/ltp/20260902160113.1207205-1-nirmoyd@nvidia.com/

 .../kernel/syscalls/getrusage/getrusage03.c   | 87 +++++++++++++++----
 1 file changed, 71 insertions(+), 16 deletions(-)

diff --git a/testcases/kernel/syscalls/getrusage/getrusage03.c b/testcases/kernel/syscalls/getrusage/getrusage03.c
index a2cdd6158..6beae130d 100644
--- a/testcases/kernel/syscalls/getrusage/getrusage03.c
+++ b/testcases/kernel/syscalls/getrusage/getrusage03.c
@@ -13,22 +13,88 @@
  * this program.
  */
 
-#include <stdlib.h>
+#define _GNU_SOURCE
 #include <stdio.h>
+#include <stdlib.h>
 
 #include "tst_test.h"
+#include "lapi/cpuset.h"
+#include "lapi/sched.h"
 #include "getrusage03.h"
 
 #define TESTBIN "getrusage03_child"
 
 static struct rusage ru;
 static long maxrss_init;
+static long lower_allowance;
 
 static const char *const resource[] = {
 	TESTBIN,
 	NULL,
 };
 
+/*
+ * Pin the test to the CPU it runs on. Every forked or exec'd child inherits
+ * the affinity, so only that CPU can hold RSS counter updates outside the
+ * fast percpu_counter read.
+ */
+static void pin_to_current_cpu(void)
+{
+	unsigned int cpu;
+	cpu_set_t *set;
+	size_t size;
+
+	if (getcpu(&cpu, NULL))
+		tst_brk(TBROK | TERRNO, "getcpu() failed");
+
+	set = CPU_ALLOC(cpu + 1);
+	if (!set)
+		tst_brk(TBROK | TERRNO, "CPU_ALLOC() failed");
+
+	size = CPU_ALLOC_SIZE(cpu + 1);
+	CPU_ZERO_S(size, set);
+	CPU_SET_S(cpu, size, set);
+	if (sched_setaffinity(0, size, set))
+		tst_brk(TBROK | TERRNO, "sched_setaffinity() failed");
+
+	CPU_FREE(set);
+	tst_res(TINFO, "Pinned to CPU %u", cpu);
+}
+
+/*
+ * The pinned CPU can still hold up to one percpu_counter batch,
+ * max(32, 2 * online CPUs) pages, outside the fast RSS read. Count CPUs
+ * before pinning because some libc implementations count only the affinity
+ * mask.
+ */
+static void setup(void)
+{
+	long online_cpus = tst_ncpus();
+	long batch = MAX(32L, online_cpus * 2);
+	long page_size = SAFE_SYSCONF(_SC_PAGESIZE);
+	long batch_kib = batch * page_size / 1024;
+
+	lower_allowance = MAX(20 * 1024L, batch_kib);
+	if (lower_allowance >= 102400L)
+		tst_brk(TCONF, "Per-CPU RSS allowance is too large: %li KiB",
+			lower_allowance);
+
+	tst_res(TINFO, "%li online CPUs, lower RSS allowance %li KiB",
+		online_cpus, lower_allowance);
+
+	pin_to_current_cpu();
+}
+
+static void check_maxrss(long actual, long expected, const char *name,
+			 const char *size)
+{
+	if (actual >= expected - lower_allowance &&
+	    actual <= expected + DELTA_MAX)
+		tst_res(TPASS, "%s ~= %s", name, size);
+	else
+		tst_res(TFAIL, "%s = %li, expected %li", name, actual, expected);
+}
+
 static void inherit_fork1(void)
 {
 	SAFE_GETRUSAGE(RUSAGE_SELF, &ru);
@@ -51,11 +117,7 @@ static void inherit_fork2(void)
 {
 	SAFE_GETRUSAGE(RUSAGE_CHILDREN, &ru);
 
-	if (is_in_delta(ru.ru_maxrss - 102400))
-		tst_res(TPASS, "initial.children ~= 100MB");
-	else
-		tst_res(TFAIL, "initial.children = %li, expected %i",
-			ru.ru_maxrss, 102400);
+	check_maxrss(ru.ru_maxrss, 102400, "initial.children", "100MB");
 
 	if (!SAFE_FORK()) {
 		SAFE_GETRUSAGE(RUSAGE_CHILDREN, &ru);
@@ -78,11 +140,7 @@ static void grandchild_maxrss(void)
 	tst_reap_children();
 	SAFE_GETRUSAGE(RUSAGE_CHILDREN, &ru);
 
-	if (is_in_delta(ru.ru_maxrss - 307200))
-		tst_res(TPASS, "child.children ~= 300MB");
-	else
-		tst_res(TFAIL, "child.children = %li, expected %i",
-			ru.ru_maxrss, 307200);
+	check_maxrss(ru.ru_maxrss, 307200, "child.children", "300MB");
 }
 
 static void zombie(void)
@@ -106,11 +164,7 @@ static void zombie(void)
 
 	tst_reap_children();
 	SAFE_GETRUSAGE(RUSAGE_CHILDREN, &ru);
-	if (is_in_delta(ru.ru_maxrss - 409600))
-		tst_res(TPASS, "post_wait.children ~= 400MB");
-	else
-		tst_res(TFAIL, "post_wait.children = %li, expected %i",
-			ru.ru_maxrss, 409600);
+	check_maxrss(ru.ru_maxrss, 409600, "post_wait.children", "400MB");
 }
 
 static void sig_ign(void)
@@ -174,6 +228,7 @@ static void run(unsigned int i)
 static struct tst_test test = {
 	.forks_child = 1,
 	.child_needs_reinit = 1,
+	.setup = setup,
 	.resource_files = resource,
 	.min_mem_avail = 512,
 	.tags = (const struct tst_tag[]) {

base-commit: dca9f0d272b2923256852fc37abd222b533d91e0
-- 
2.43.0

-- 
Mailing list info: https://lists.linux.it/listinfo/ltp

^ permalink raw reply related	[flat|nested] 12+ messages in thread

end of thread, other threads:[~2026-10-02 13:41 UTC | newest]

Thread overview: 12+ messages (download: mbox.gz follow: Atom feed
-- links below jump to the message on this page --
2026-08-25 15:25 [LTP] [RFC PATCH] getrusage03: account for percpu RSS counter batching Nirmoy Das via ltp
2026-08-25 17:34 ` [LTP] " linuxtestproject.agent
2026-09-02 12:25 ` [LTP] [RFC PATCH] " Cyril Hrubis
2026-09-02 16:01   ` [LTP] [PATCH v2] " Nirmoy Das via ltp
2026-09-02 17:42     ` [LTP] " linuxtestproject.agent
2026-09-15 15:29     ` [LTP] [PATCH v2] " Cyril Hrubis
2026-09-16 15:58       ` Nirmoy Das via ltp
2026-09-17 12:36         ` Jan Stancek via ltp
2026-09-22 17:36           ` Petr Vorel
2026-10-01 18:55             ` Nirmoy Das via ltp
2026-09-24 11:26     ` Cyril Hrubis
  -- strict thread matches above, loose matches on Subject: below --
2026-10-02 13:17 [LTP] [PATCH v3] " Nirmoy Das via ltp
2026-10-02 13:40 ` [LTP] " linuxtestproject.agent

This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox