Linux Test Project
 help / color / mirror / Atom feed
* [LTP] [PATCH v3] getrusage03: account for percpu RSS counter batching
@ 2026-10-02 13:17 Nirmoy Das via ltp
  2026-10-02 13:40 ` [LTP] " linuxtestproject.agent
  0 siblings, 1 reply; 4+ messages in thread
From: Nirmoy Das via ltp @ 2026-10-02 13:17 UTC (permalink / raw)
  To: ltp; +Cc: Nirmoy Das

After f1a7941243c1 ("mm: convert mm's rss stats into percpu_counter"),
the fixed 20 MiB tolerance can be too small on large systems. Pinning
limits pending RSS updates to one CPU, but that CPU can still hold one
batch outside the fast RSS read.

Pin the test to its current CPU and allow one batch of lower-side slack
for the 100, 300 and 400 MiB allocation checks. Keep the existing upper
and inheritance tolerances, and skip if one batch is at least 100 MiB.

On 352 CPUs with 64 KiB pages, one batch is 44 MiB. Pinning alone
reported 270336 KiB for the 300 MiB check, outside the 20 MiB tolerance.
The updated test passed 10/10 runs on the 352-CPU Vera system.

Suggested-by: Cyril Hrubis <chrubis@suse.cz>
Co-developed-by: Jan Stancek <jstancek@redhat.com>
Signed-off-by: Jan Stancek <jstancek@redhat.com>
Assisted-by: ClaudeCode:claude-opus-5-5
Signed-off-by: Nirmoy Das <nirmoyd@nvidia.com>
---
Changes in v3:
- Use tst_ncpus() and pin to the current CPU (Cyril).
- Allocate the CPU mask dynamically and log the allowance.

v2: https://lore.kernel.org/ltp/20260902160113.1207205-1-nirmoyd@nvidia.com/

 .../kernel/syscalls/getrusage/getrusage03.c   | 87 +++++++++++++++----
 1 file changed, 71 insertions(+), 16 deletions(-)

diff --git a/testcases/kernel/syscalls/getrusage/getrusage03.c b/testcases/kernel/syscalls/getrusage/getrusage03.c
index a2cdd6158..6beae130d 100644
--- a/testcases/kernel/syscalls/getrusage/getrusage03.c
+++ b/testcases/kernel/syscalls/getrusage/getrusage03.c
@@ -13,22 +13,88 @@
  * this program.
  */
 
-#include <stdlib.h>
+#define _GNU_SOURCE
 #include <stdio.h>
+#include <stdlib.h>
 
 #include "tst_test.h"
+#include "lapi/cpuset.h"
+#include "lapi/sched.h"
 #include "getrusage03.h"
 
 #define TESTBIN "getrusage03_child"
 
 static struct rusage ru;
 static long maxrss_init;
+static long lower_allowance;
 
 static const char *const resource[] = {
 	TESTBIN,
 	NULL,
 };
 
+/*
+ * Pin the test to the CPU it runs on. Every forked or exec'd child inherits
+ * the affinity, so only that CPU can hold RSS counter updates outside the
+ * fast percpu_counter read.
+ */
+static void pin_to_current_cpu(void)
+{
+	unsigned int cpu;
+	cpu_set_t *set;
+	size_t size;
+
+	if (getcpu(&cpu, NULL))
+		tst_brk(TBROK | TERRNO, "getcpu() failed");
+
+	set = CPU_ALLOC(cpu + 1);
+	if (!set)
+		tst_brk(TBROK | TERRNO, "CPU_ALLOC() failed");
+
+	size = CPU_ALLOC_SIZE(cpu + 1);
+	CPU_ZERO_S(size, set);
+	CPU_SET_S(cpu, size, set);
+	if (sched_setaffinity(0, size, set))
+		tst_brk(TBROK | TERRNO, "sched_setaffinity() failed");
+
+	CPU_FREE(set);
+	tst_res(TINFO, "Pinned to CPU %u", cpu);
+}
+
+/*
+ * The pinned CPU can still hold up to one percpu_counter batch,
+ * max(32, 2 * online CPUs) pages, outside the fast RSS read. Count CPUs
+ * before pinning because some libc implementations count only the affinity
+ * mask.
+ */
+static void setup(void)
+{
+	long online_cpus = tst_ncpus();
+	long batch = MAX(32L, online_cpus * 2);
+	long page_size = SAFE_SYSCONF(_SC_PAGESIZE);
+	long batch_kib = batch * page_size / 1024;
+
+	lower_allowance = MAX(20 * 1024L, batch_kib);
+	if (lower_allowance >= 102400L)
+		tst_brk(TCONF, "Per-CPU RSS allowance is too large: %li KiB",
+			lower_allowance);
+
+	tst_res(TINFO, "%li online CPUs, lower RSS allowance %li KiB",
+		online_cpus, lower_allowance);
+
+	pin_to_current_cpu();
+}
+
+static void check_maxrss(long actual, long expected, const char *name,
+			 const char *size)
+{
+	if (actual >= expected - lower_allowance &&
+	    actual <= expected + DELTA_MAX)
+		tst_res(TPASS, "%s ~= %s", name, size);
+	else
+		tst_res(TFAIL, "%s = %li, expected %li", name, actual, expected);
+}
+
 static void inherit_fork1(void)
 {
 	SAFE_GETRUSAGE(RUSAGE_SELF, &ru);
@@ -51,11 +117,7 @@ static void inherit_fork2(void)
 {
 	SAFE_GETRUSAGE(RUSAGE_CHILDREN, &ru);
 
-	if (is_in_delta(ru.ru_maxrss - 102400))
-		tst_res(TPASS, "initial.children ~= 100MB");
-	else
-		tst_res(TFAIL, "initial.children = %li, expected %i",
-			ru.ru_maxrss, 102400);
+	check_maxrss(ru.ru_maxrss, 102400, "initial.children", "100MB");
 
 	if (!SAFE_FORK()) {
 		SAFE_GETRUSAGE(RUSAGE_CHILDREN, &ru);
@@ -78,11 +140,7 @@ static void grandchild_maxrss(void)
 	tst_reap_children();
 	SAFE_GETRUSAGE(RUSAGE_CHILDREN, &ru);
 
-	if (is_in_delta(ru.ru_maxrss - 307200))
-		tst_res(TPASS, "child.children ~= 300MB");
-	else
-		tst_res(TFAIL, "child.children = %li, expected %i",
-			ru.ru_maxrss, 307200);
+	check_maxrss(ru.ru_maxrss, 307200, "child.children", "300MB");
 }
 
 static void zombie(void)
@@ -106,11 +164,7 @@ static void zombie(void)
 
 	tst_reap_children();
 	SAFE_GETRUSAGE(RUSAGE_CHILDREN, &ru);
-	if (is_in_delta(ru.ru_maxrss - 409600))
-		tst_res(TPASS, "post_wait.children ~= 400MB");
-	else
-		tst_res(TFAIL, "post_wait.children = %li, expected %i",
-			ru.ru_maxrss, 409600);
+	check_maxrss(ru.ru_maxrss, 409600, "post_wait.children", "400MB");
 }
 
 static void sig_ign(void)
@@ -174,6 +228,7 @@ static void run(unsigned int i)
 static struct tst_test test = {
 	.forks_child = 1,
 	.child_needs_reinit = 1,
+	.setup = setup,
 	.resource_files = resource,
 	.min_mem_avail = 512,
 	.tags = (const struct tst_tag[]) {

base-commit: dca9f0d272b2923256852fc37abd222b533d91e0
-- 
2.43.0

-- 
Mailing list info: https://lists.linux.it/listinfo/ltp

^ permalink raw reply related	[flat|nested] 4+ messages in thread
* [LTP] [PATCH v2] getrusage03: account for percpu RSS counter batching
@ 2026-09-02 16:01 Nirmoy Das via ltp
  2026-09-02 17:42 ` [LTP] " linuxtestproject.agent
  0 siblings, 1 reply; 4+ messages in thread
From: Nirmoy Das via ltp @ 2026-09-02 16:01 UTC (permalink / raw)
  To: ltp

getrusage03 allows ru_maxrss to differ from the expected value by a
fixed 20 MiB. That can be too small on large systems after Linux commit
f1a7941243c1, which converted mm RSS stats into percpu_counter.

Pin the whole test to one CPU during setup, before it forks or execs any
children. Every descendant inherits the affinity, which removes task
migration as a multiplier. One CPU can still retain almost one batch of
the deliberate anonymous allocation outside the fast RSS read.

Use one batch only as lower-side slack for the 100, 300, and 400 MiB
allocation checks. Preserve the fixed 20 MiB tolerance for their upper
side and for all inheritance comparisons. Count online CPUs from
/proc/stat because some libc implementations report only the calling
task affinity. Skip the suite if one batch is at least the smallest
100 MiB allocation and would make that signal non-discriminating.

On a 352-CPU arm64 system with 64 KiB pages, a batch is 704 pages
(44 MiB). With only the pin and fixed 20 MiB tolerance, the 300 MiB
case returned 270336 KiB instead of 307200 KiB and failed 10/10 runs.
The one-batch lower allowance passed 10/10 runs.

Signed-off-by: Nirmoy Das <nirmoyd@nvidia.com>
---
Changes in v2:
- Pin the whole test once in setup so every fork and exec descendant
  inherits the singleton CPU affinity.
- Keep one batch only as lower-side slack for the deliberate allocation
  checks; pinning with the fixed 20 MiB tolerance failed 10/10 runs on
  the 352-CPU, 64 KiB-page system.
- Count kernel-wide online CPUs independently of task affinity and skip
  when one batch would make the smallest RSS signal non-discriminating.

 .../kernel/syscalls/getrusage/getrusage03.c   | 110 +++++++++++++++---
 1 file changed, 95 insertions(+), 15 deletions(-)

diff --git a/testcases/kernel/syscalls/getrusage/getrusage03.c b/testcases/kernel/syscalls/getrusage/getrusage03.c
index a2cdd6158..38a100576 100644
--- a/testcases/kernel/syscalls/getrusage/getrusage03.c
+++ b/testcases/kernel/syscalls/getrusage/getrusage03.c
@@ -13,9 +13,13 @@
  * this program.
  */
 
+#define _GNU_SOURCE
 #include <stdlib.h>
 #include <stdio.h>
 
+#include "lapi/cpuset.h"
+#include "tst_safe_stdio.h"
+#include "tst_cpu.h"
 #include "tst_test.h"
 #include "getrusage03.h"
 
@@ -23,12 +27,99 @@
 
 static struct rusage ru;
 static long maxrss_init;
+static long lower_allowance;
 
 static const char *const resource[] = {
 	TESTBIN,
 	NULL,
 };
 
+static long count_online_cpus(void)
+{
+	FILE *fp = SAFE_FOPEN("/proc/stat", "r");
+	char line[BUFSIZ];
+	long count = 0;
+
+	while (fgets(line, sizeof(line), fp)) {
+		if (line[0] == 'c' && line[1] == 'p' && line[2] == 'u' &&
+		    line[3] >= '0' && line[3] <= '9')
+			count++;
+	}
+
+	if (ferror(fp))
+		tst_brk(TBROK | TERRNO, "fgets(/proc/stat)");
+
+	SAFE_FCLOSE(fp);
+
+	if (!count)
+		tst_brk(TBROK, "No online CPUs found in /proc/stat");
+
+	return count;
+}
+
+static void pin_to_cpu(void)
+{
+	long ncpus = tst_ncpus_max();
+	size_t size = CPU_ALLOC_SIZE(ncpus);
+	cpu_set_t *mask = CPU_ALLOC(ncpus);
+	int cpu = -1;
+
+	if (!mask)
+		tst_brk(TBROK | TERRNO, "CPU_ALLOC()");
+
+	CPU_ZERO_S(size, mask);
+	if (sched_getaffinity(0, size, mask) < 0) {
+		CPU_FREE(mask);
+		tst_brk(TBROK | TERRNO, "sched_getaffinity()");
+	}
+
+	for (long i = 0; i < ncpus; i++) {
+		if (CPU_ISSET_S((int)i, size, mask)) {
+			cpu = (int)i;
+			break;
+		}
+	}
+
+	if (cpu < 0) {
+		CPU_FREE(mask);
+		tst_brk(TBROK, "sched_getaffinity() returned an empty CPU mask");
+	}
+
+	CPU_ZERO_S(size, mask);
+	CPU_SET_S(cpu, size, mask);
+	if (sched_setaffinity(0, size, mask) < 0) {
+		CPU_FREE(mask);
+		tst_brk(TBROK | TERRNO, "sched_setaffinity()");
+	}
+
+	CPU_FREE(mask);
+}
+
+static void setup(void)
+{
+	long online_cpus = count_online_cpus();
+	long batch = MAX(32L, online_cpus * 2);
+	long page_size = SAFE_SYSCONF(_SC_PAGESIZE);
+	long batch_kib = batch * page_size / 1024;
+
+	lower_allowance = MAX(20 * 1024L, batch_kib);
+	if (lower_allowance >= 102400L)
+		tst_brk(TCONF, "Per-CPU RSS allowance is too large: %li KiB",
+			lower_allowance);
+
+	pin_to_cpu();
+}
+
+static void check_maxrss(long actual, long expected, const char *name,
+			 const char *size)
+{
+	if (actual >= expected - lower_allowance &&
+	    actual <= expected + DELTA_MAX)
+		tst_res(TPASS, "%s ~= %s", name, size);
+	else
+		tst_res(TFAIL, "%s = %li, expected %li", name, actual, expected);
+}
+
 static void inherit_fork1(void)
 {
 	SAFE_GETRUSAGE(RUSAGE_SELF, &ru);
@@ -51,11 +142,7 @@ static void inherit_fork2(void)
 {
 	SAFE_GETRUSAGE(RUSAGE_CHILDREN, &ru);
 
-	if (is_in_delta(ru.ru_maxrss - 102400))
-		tst_res(TPASS, "initial.children ~= 100MB");
-	else
-		tst_res(TFAIL, "initial.children = %li, expected %i",
-			ru.ru_maxrss, 102400);
+	check_maxrss(ru.ru_maxrss, 102400, "initial.children", "100MB");
 
 	if (!SAFE_FORK()) {
 		SAFE_GETRUSAGE(RUSAGE_CHILDREN, &ru);
@@ -78,11 +165,7 @@ static void grandchild_maxrss(void)
 	tst_reap_children();
 	SAFE_GETRUSAGE(RUSAGE_CHILDREN, &ru);
 
-	if (is_in_delta(ru.ru_maxrss - 307200))
-		tst_res(TPASS, "child.children ~= 300MB");
-	else
-		tst_res(TFAIL, "child.children = %li, expected %i",
-			ru.ru_maxrss, 307200);
+	check_maxrss(ru.ru_maxrss, 307200, "child.children", "300MB");
 }
 
 static void zombie(void)
@@ -106,11 +189,7 @@ static void zombie(void)
 
 	tst_reap_children();
 	SAFE_GETRUSAGE(RUSAGE_CHILDREN, &ru);
-	if (is_in_delta(ru.ru_maxrss - 409600))
-		tst_res(TPASS, "post_wait.children ~= 400MB");
-	else
-		tst_res(TFAIL, "post_wait.children = %li, expected %i",
-			ru.ru_maxrss, 409600);
+	check_maxrss(ru.ru_maxrss, 409600, "post_wait.children", "400MB");
 }
 
 static void sig_ign(void)
@@ -174,6 +253,7 @@ static void run(unsigned int i)
 static struct tst_test test = {
 	.forks_child = 1,
 	.child_needs_reinit = 1,
+	.setup = setup,
 	.resource_files = resource,
 	.min_mem_avail = 512,
 	.tags = (const struct tst_tag[]) {

base-commit: 12724413534a6d4160ff9694ba6f09daa4ccb6bd
-- 
2.43.0


-- 
Mailing list info: https://lists.linux.it/listinfo/ltp

^ permalink raw reply related	[flat|nested] 4+ messages in thread
* [LTP] [RFC PATCH] getrusage03: account for percpu RSS counter batching
@ 2026-08-25 15:25 Nirmoy Das via ltp
  2026-08-25 17:34 ` [LTP] " linuxtestproject.agent
  0 siblings, 1 reply; 4+ messages in thread
From: Nirmoy Das via ltp @ 2026-08-25 15:25 UTC (permalink / raw)
  To: ltp; +Cc: Nirmoy Das

getrusage03 allows ru_maxrss to differ from the expected value by a
fixed 20 MiB. That can be too small on large systems after Linux commit
f1a7941243c1 ("mm: convert mm's rss stats into percpu_counter").

The fast RSS read may miss percpu counter batches, and the batch size
scales with the number of online CPUs. The test child is single threaded,
but it can migrate while faulting memory, so allow two batches worth of
slack based on the online CPU count and page size. Keep the existing
20 MiB floor for smaller systems.

This is sent as RFC because it makes the test tolerate Linux's current
percpu counter batching behavior instead of treating the smaller
ru_maxrss value as a kernel failure.

Signed-off-by: Nirmoy Das <nirmoyd@nvidia.com>
---
 .../kernel/syscalls/getrusage/getrusage03.h     | 17 +++++++++++++++--
 1 file changed, 15 insertions(+), 2 deletions(-)

diff --git a/testcases/kernel/syscalls/getrusage/getrusage03.h b/testcases/kernel/syscalls/getrusage/getrusage03.h
index 58a98b430..9a7d8e8fa 100644
--- a/testcases/kernel/syscalls/getrusage/getrusage03.h
+++ b/testcases/kernel/syscalls/getrusage/getrusage03.h
@@ -7,9 +7,20 @@
 #define LTP_GETRUSAGE03_H
 
 #include <sched.h>
+#include "tst_cpu.h"
 #include "tst_test.h"
 
-#define DELTA_MAX 20480
+#define DELTA_MAX 20480L
+#define MAXRSS_BATCHES 2L
+
+static long maxrss_delta(void)
+{
+	long batch = MAX(32L, tst_ncpus() * 2);
+	long delta = MAXRSS_BATCHES * batch * SAFE_SYSCONF(_SC_PAGESIZE) / 1024;
+
+	/* Linux RSS counters can miss percpu counter batches after migration. */
+	return MAX(DELTA_MAX, delta);
+}
 
 static void force_context_switches(int iterations)
 {
@@ -40,7 +51,9 @@ static void consume_mb(int consume_nr)
 
 static int is_in_delta(long value)
 {
-	return (value >= -DELTA_MAX && value <= DELTA_MAX);
+	long delta = maxrss_delta();
+
+	return (value >= -delta && value <= delta);
 }
 
 #endif //LTP_GETRUSAGE03_H
-- 
2.43.0


-- 
Mailing list info: https://lists.linux.it/listinfo/ltp

^ permalink raw reply related	[flat|nested] 4+ messages in thread

end of thread, other threads:[~2026-10-02 13:41 UTC | newest]

Thread overview: 4+ messages (download: mbox.gz follow: Atom feed
-- links below jump to the message on this page --
2026-10-02 13:17 [LTP] [PATCH v3] getrusage03: account for percpu RSS counter batching Nirmoy Das via ltp
2026-10-02 13:40 ` [LTP] " linuxtestproject.agent
  -- strict thread matches above, loose matches on Subject: below --
2026-09-02 16:01 [LTP] [PATCH v2] " Nirmoy Das via ltp
2026-09-02 17:42 ` [LTP] " linuxtestproject.agent
2026-08-25 15:25 [LTP] [RFC PATCH] " Nirmoy Das via ltp
2026-08-25 17:34 ` [LTP] " linuxtestproject.agent

This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox