All of lore.kernel.org
 help / color / mirror / Atom feed
From: Arnaldo Carvalho de Melo <acme@kernel.org>
To: Namhyung Kim <namhyung@kernel.org>
Cc: Ingo Molnar <mingo@kernel.org>,
	Thomas Gleixner <tglx@linutronix.de>,
	James Clark <james.clark@linaro.org>,
	Jiri Olsa <jolsa@kernel.org>, Ian Rogers <irogers@google.com>,
	Adrian Hunter <adrian.hunter@intel.com>,
	Clark Williams <williams@redhat.com>,
	linux-kernel@vger.kernel.org, linux-perf-users@vger.kernel.org,
	Arnaldo Carvalho de Melo <acme@redhat.com>
Subject: [PATCH 4/4] perf test: Add false_sharing workload exhibiting cross-CPU false sharing
Date: Mon, 28 Sep 2026 18:22:50 +0200	[thread overview]
Message-ID: <20260928162250.2413383-5-acme@kernel.org> (raw)
In-Reply-To: <20260928162250.2413383-1-acme@kernel.org>

From: Arnaldo Carvalho de Melo <acme@redhat.com>

Add a 'perf test -w false_sharing' workload that hammers one shared
struct from several CPUs, shaped as a TCP connection: a read-mostly
identity (five-tuple) shares a cacheline with per-packet rx counters
(the false-sharing line), a second line has packet-path private tx and
congestion control counters, and a third the connection config.

The packet path runs in the main thread and up to four lookup threads
hash the five-tuple and pull the config, reading one volatile shared
instance directly so the accesses are PC-relative and resolvable by the
data type profiler.

Assisted-by: LLM
Signed-off-by: Arnaldo Carvalho de Melo <acme@redhat.com>
---
 tools/perf/tests/builtin-test.c               |   1 +
 tools/perf/tests/shell/data_type_profiling.sh |   9 +-
 tools/perf/tests/tests.h                      |   1 +
 tools/perf/tests/workloads/Build              |   2 +
 tools/perf/tests/workloads/false_sharing.c    | 236 ++++++++++++++++++
 5 files changed, 247 insertions(+), 2 deletions(-)
 create mode 100644 tools/perf/tests/workloads/false_sharing.c

diff --git a/tools/perf/tests/builtin-test.c b/tools/perf/tests/builtin-test.c
index d2f594921e25bda9..98134b1c74cf80f4 100644
--- a/tools/perf/tests/builtin-test.c
+++ b/tools/perf/tests/builtin-test.c
@@ -175,6 +175,7 @@ static struct test_workload *workloads[] = {
 	&workload__context_switch_loop,
 	&workload__deterministic,
 	&workload__callchain,
+	&workload__false_sharing,
 
 #ifdef HAVE_RUST_SUPPORT
 	&workload__code_with_type,
diff --git a/tools/perf/tests/shell/data_type_profiling.sh b/tools/perf/tests/shell/data_type_profiling.sh
index a916c410274aa888..57203e0e5f857540 100755
--- a/tools/perf/tests/shell/data_type_profiling.sh
+++ b/tools/perf/tests/shell/data_type_profiling.sh
@@ -8,8 +8,8 @@ set -e
 # data type profiling manifestation
 
 # Values in testtypes and testprogs should match
-testtypes=("# data-type: struct Buf" "# data-type: struct buf")
-testprogs=("perf test -w code_with_type" "perf test -w datasym")
+testtypes=("# data-type: struct Buf" "# data-type: struct buf" "# data-type: struct net_conn")
+testprogs=("perf test -w code_with_type" "perf test -w datasym" "perf test -w false_sharing")
 
 err=0
 perfdata=$(mktemp /tmp/__perf_test.perf.data.XXXXX)
@@ -59,6 +59,9 @@ test_basic_annotate() {
 
     "xC")
     index=1 ;;
+
+    "xFS")
+    index=2 ;;
   esac
 
   # Under 'set -e' a bare failing command aborts the script through the EXIT
@@ -114,6 +117,8 @@ test_basic_annotate Basic Rust
 test_basic_annotate Pipe Rust
 test_basic_annotate Basic C
 test_basic_annotate Pipe C
+test_basic_annotate Basic FS
+test_basic_annotate Pipe FS
 
 cleanup
 exit $err
diff --git a/tools/perf/tests/tests.h b/tools/perf/tests/tests.h
index 9c96f33483d14356..f379620069d34f3a 100644
--- a/tools/perf/tests/tests.h
+++ b/tools/perf/tests/tests.h
@@ -250,6 +250,7 @@ DECLARE_WORKLOAD(jitdump);
 DECLARE_WORKLOAD(context_switch_loop);
 DECLARE_WORKLOAD(deterministic);
 DECLARE_WORKLOAD(callchain);
+DECLARE_WORKLOAD(false_sharing);
 
 #ifdef HAVE_RUST_SUPPORT
 DECLARE_WORKLOAD(code_with_type);
diff --git a/tools/perf/tests/workloads/Build b/tools/perf/tests/workloads/Build
index 048e371eb63e3164..18fceddccb56c013 100644
--- a/tools/perf/tests/workloads/Build
+++ b/tools/perf/tests/workloads/Build
@@ -14,6 +14,7 @@ perf-test-y += jitdump.o
 perf-test-y += context_switch_loop.o
 perf-test-y += deterministic.o
 perf-test-y += callchain.o
+perf-test-y += false_sharing.o
 
 ifeq ($(CONFIG_RUST_SUPPORT),y)
     perf-test-y += code_with_type.o
@@ -29,3 +30,4 @@ CFLAGS_inlineloop.o       = -g -O2
 CFLAGS_deterministic.o    = -g -O0 -fno-inline -U_FORTIFY_SOURCE
 CFLAGS_named_threads.o    = -g -O0 -fno-inline -U_FORTIFY_SOURCE
 CFLAGS_callchain.o        = -g -O0 -fno-inline -U_FORTIFY_SOURCE
+CFLAGS_false_sharing.o    = -g -O0 -fno-inline -U_FORTIFY_SOURCE
diff --git a/tools/perf/tests/workloads/false_sharing.c b/tools/perf/tests/workloads/false_sharing.c
new file mode 100644
index 0000000000000000..6fccb81e3ed1cb9a
--- /dev/null
+++ b/tools/perf/tests/workloads/false_sharing.c
@@ -0,0 +1,236 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * False-sharing demo for data type profiling, shaped as a TCP
+ * connection: a read-mostly identity shares a cacheline with per-packet
+ * rx counters (the false-sharing line), a second line has packet-path
+ * private tx and congestion control counters, and a third the connection
+ * config.
+ *
+ * With a 'perf mem record' of this workload the profiler shows accesses
+ * to different members of the same cacheline, which the CTF stream lets
+ * pahole flag as false sharing on line 0.
+ */
+#include <pthread.h>
+#include <sched.h>
+#include <stdint.h>
+#include <stdlib.h>
+#include <stdio.h>
+#include <signal.h>
+#include <unistd.h>
+#include <linux/compiler.h>
+#include "../tests.h"
+
+struct net_conn {
+	/* cacheline 0: identity (read-mostly) + rx counters (per packet) */
+	uint32_t	saddr;		/*   0 */
+	uint32_t	daddr;		/*   4 */
+	uint16_t	sport;		/*   8 */
+	uint16_t	dport;		/*  10 */
+	uint8_t		state;		/*  12: 1 == ESTABLISHED */
+	uint8_t		protocol;	/*  13: 6 == TCP */
+	uint16_t	__pad0;		/*  14 */
+	uint64_t	bytes_rx;	/*  16: every packet */
+	uint64_t	packets_rx;	/*  24: every packet */
+	uint32_t	rx_queue;	/*  32: backlog depth, fluctuates */
+	uint8_t		__pad1[24];	/*  36..59 */
+	uint32_t	last_ack;	/*  60: written per ACK */
+	/* cacheline 1: tx + congestion control (packet-path private) */
+	uint64_t	bytes_tx;	/*  64: every packet */
+	uint64_t	packets_tx;	/*  72: every packet */
+	uint32_t	cwnd;		/*  80: on every ACK */
+	uint32_t	ssthresh;	/*  84: on loss */
+	uint32_t	rtt_us;		/*  88: on every ACK */
+	uint32_t	retrans;	/*  92: on timeout */
+	uint32_t	__pad2[8];	/*  96..127 */
+	/* cacheline 2: config, set at setup, read by everybody */
+	uint16_t	mss;		/* 128 */
+	uint8_t		snd_wscale;	/* 130 */
+	uint8_t		rcv_wscale;	/* 131 */
+	uint32_t	keepalive_int;	/* 132 */
+	uint32_t	mark;		/* 136: firewall mark */
+	uint32_t	priority;	/* 140: traffic class */
+	uint32_t	__pad3[12];	/* 144..191 */
+} __attribute__((aligned(64)));
+
+/* Volatile so every iteration really loads and stores. */
+static volatile struct net_conn conn;
+/* Keeps the reader checksums alive after the threads join. */
+static volatile unsigned long fs_sink;
+
+static volatile sig_atomic_t done;
+
+struct fs_reader {
+	pthread_t	thread;
+	int		cpu;
+	unsigned long	sum;
+	char		__pad[64 - sizeof(pthread_t) - sizeof(int) - sizeof(unsigned long)];
+} __attribute__((aligned(64)));
+
+static void sighandler(int sig __maybe_unused)
+{
+	done = 1;
+}
+
+static void pin_to_cpu(int cpu)
+{
+	cpu_set_t set;
+
+	CPU_ZERO(&set);
+	CPU_SET(cpu, &set);
+	/* Best effort: in a restricted cpuset this fails and the thread runs unpinned. */
+	pthread_setaffinity_np(pthread_self(), sizeof(set), &set);
+}
+
+/*
+ * Connection lookup, as a load balancer or 'ss' scrape would do it: the
+ * reads go straight to the global, this file is built -O0 and a local
+ * pointer would be reloaded from a stack slot, a form the data type
+ * resolver does not track.
+ */
+static void *reader_fn(void *arg)
+{
+	struct fs_reader *r = arg;
+	unsigned long sum = 0;
+
+	pthread_setname_np(pthread_self(), "fs-reader");
+	pin_to_cpu(r->cpu);
+
+	while (!done) {
+		sum += conn.saddr + conn.daddr + conn.sport + conn.dport +
+		       conn.state + conn.protocol;
+		sum += conn.mss + conn.snd_wscale + conn.rcv_wscale +
+		       conn.keepalive_int + conn.mark + conn.priority;
+	}
+	r->sum = sum;
+	return NULL;
+}
+
+static int false_sharing(int argc, const char **argv)
+{
+	double sec = 2.0;
+	int nreaders = 0, nr_allowed = 0, err = 1;
+	int *allowed = NULL, nallowed = 0, ncpus;
+	cpu_set_t set;
+	struct fs_reader *readers = NULL;
+	int i, writer_cpu;
+	unsigned long n = 0;
+
+	pthread_setname_np(pthread_self(), "fs-writer");
+	if (argc > 0)
+		sec = atof(argv[0]);
+	if (!(sec > 0.0)) {
+		fprintf(stderr, "Error: seconds (%f) must be > 0\n", sec);
+		return 1;
+	}
+	if (argc > 1)
+		nreaders = atoi(argv[1]);
+
+	/*
+	 * A connection that just got established: identity and config fixed
+	 * from here on, counters at zero.
+	 */
+	conn.saddr = 0x0a000001;	/* 10.0.0.1 */
+	conn.daddr = 0x0a000002;	/* 10.0.0.2 */
+	conn.sport = 54321;
+	conn.dport = 443;
+	conn.state = 1;			/* ESTABLISHED */
+	conn.protocol = 6;		/* TCP */
+	conn.mss = 1448;
+	conn.snd_wscale = 7;
+	conn.rcv_wscale = 7;
+	conn.keepalive_int = 7200;
+	conn.cwnd = 10;
+	conn.ssthresh = 65535;
+	conn.rtt_us = 50;
+
+	/* Pin against the allowed set, restricted cpusets still spread the threads. */
+	ncpus = sysconf(_SC_NPROCESSORS_CONF);
+	if (sched_getaffinity(0, sizeof(set), &set) == 0) {
+		for (i = 0; i < ncpus && i < CPU_SETSIZE; i++) {
+			if (!CPU_ISSET(i, &set))
+				continue;
+			nr_allowed++;
+		}
+		allowed = malloc(nr_allowed * sizeof(int));
+		if (allowed == NULL) {
+			fprintf(stderr, "Error: malloc failed for CPU list\n");
+			return 1;
+		}
+		for (i = 0; i < ncpus && i < CPU_SETSIZE; i++) {
+			if (CPU_ISSET(i, &set))
+				allowed[nallowed++] = i;
+		}
+	}
+	if (nreaders <= 0) {
+		/* By default leave one CPU for the packet path, up to 4 readers. */
+		nreaders = nallowed > 1 ? nallowed - 1 : 1;
+		if (nreaders > 4)
+			nreaders = 4;
+	}
+
+	signal(SIGINT, sighandler);
+	signal(SIGALRM, sighandler);
+
+	readers = calloc(nreaders, sizeof(*readers));
+	if (readers == NULL) {
+		fprintf(stderr, "Error: calloc failed for %d readers\n", nreaders);
+		goto out;
+	}
+	for (i = 0; i < nreaders; i++) {
+		int cpu = nallowed > 1 ? allowed[(i + 1) % nallowed] : -1;
+
+		readers[i].cpu = cpu;
+		if (pthread_create(&readers[i].thread, NULL, reader_fn, &readers[i])) {
+			fprintf(stderr, "Error: failed to create reader %d\n", i);
+			done = 1; // Ensure started threads terminate.
+			nreaders = i;
+			goto out_join;
+		}
+	}
+	writer_cpu = nallowed > 0 ? allowed[0] : -1;
+	if (nallowed == 1)
+		fprintf(stderr, "Warning: single CPU allowed, no cross-CPU traffic expected\n");
+	if (writer_cpu >= 0)
+		pin_to_cpu(writer_cpu);
+
+	/*
+	 * The packet path: receive, acknowledge, transmit, repeat; every 64th
+	 * packet simulates a loss so ssthresh/retrans get sampled too.
+	 */
+	if (sec < 1.0) {
+		useconds_t usecs = (useconds_t)(sec * 1000000.0);
+
+		ualarm(usecs > 0 ? usecs : 1, 0);
+	} else
+		alarm((unsigned int)sec);
+	while (!done) {
+		conn.bytes_rx += conn.mss;
+		conn.packets_rx++;
+		conn.rx_queue = (uint32_t)(n & 0x3f);
+		conn.last_ack = (uint32_t)n;
+		conn.bytes_tx += conn.mss;
+		conn.packets_tx++;
+		conn.cwnd = 10 + (n & 15);
+		conn.rtt_us = 50 + (n & 7);
+		if ((n & 63) == 0) {
+			conn.ssthresh = conn.cwnd / 2;
+			conn.retrans++;
+		}
+		n++;
+	}
+	err = 0;
+out_join:
+	for (i = 0; i < nreaders; i++) {
+		if (readers[i].thread) {
+			pthread_join(readers[i].thread, /*retval=*/NULL);
+			fs_sink += readers[i].sum;
+		}
+	}
+	fs_sink += (unsigned long)(conn.bytes_rx + conn.bytes_tx + n);
+	free(readers);
+out:
+	free(allowed);
+	return err;
+}
+
+DEFINE_WORKLOAD(false_sharing);
-- 
2.55.0


  parent reply	other threads:[~2026-09-28 16:23 UTC|newest]

Thread overview: 11+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-28 16:22 [PATCH 0/4 v1] perf tools: Add progress diagnostics and a false-sharing workload Arnaldo Carvalho de Melo
2026-09-28 16:22 ` [PATCH 1/4] perf config: Move perf_config__set_variable() to util/config.c Arnaldo Carvalho de Melo
2026-09-28 16:38   ` sashiko-bot
2026-09-28 16:22 ` [PATCH 2/4] perf report: Add --progress option Arnaldo Carvalho de Melo
2026-09-28 16:31   ` sashiko-bot
2026-09-28 16:22 ` [PATCH 3/4] perf scripts: Add perf-stuck, to tell where a running perf is stuck Arnaldo Carvalho de Melo
2026-09-28 16:37   ` sashiko-bot
2026-09-28 16:22 ` Arnaldo Carvalho de Melo [this message]
2026-09-28 16:31   ` [PATCH 4/4] perf test: Add false_sharing workload exhibiting cross-CPU false sharing sashiko-bot
  -- strict thread matches above, loose matches on Subject: below --
2026-09-28 22:06 [PATCH v3 0/4] perf tools: Add progress diagnostics and a false-sharing workload Arnaldo Carvalho de Melo
2026-09-28 22:06 ` [PATCH 4/4] perf test: Add false_sharing workload exhibiting cross-CPU false sharing Arnaldo Carvalho de Melo
2026-09-28 22:14   ` sashiko-bot

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260928162250.2413383-5-acme@kernel.org \
    --to=acme@kernel.org \
    --cc=acme@redhat.com \
    --cc=adrian.hunter@intel.com \
    --cc=irogers@google.com \
    --cc=james.clark@linaro.org \
    --cc=jolsa@kernel.org \
    --cc=linux-kernel@vger.kernel.org \
    --cc=linux-perf-users@vger.kernel.org \
    --cc=mingo@kernel.org \
    --cc=namhyung@kernel.org \
    --cc=tglx@linutronix.de \
    --cc=williams@redhat.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.