From: Puranjay Mohan <puranjay@kernel.org>
To: "Lai Jiangshan" <jiangshanlai@gmail.com>,
"Paul E. McKenney" <paulmck@kernel.org>,
"Josh Triplett" <josh@joshtriplett.org>,
"Onur Özkan" <work@onurozkan.dev>,
"Frederic Weisbecker" <frederic@kernel.org>,
"Neeraj Upadhyay" <neeraj.upadhyay@kernel.org>,
"Joel Fernandes" <joelagnelf@nvidia.com>,
"Boqun Feng" <boqun@kernel.org>,
"Uladzislau Rezki" <urezki@gmail.com>,
"Davidlohr Bueso" <dave@stgolabs.net>,
"Andrii Nakryiko" <andrii@kernel.org>,
"Eduard Zingerman" <eddyz87@gmail.com>,
"Alexei Starovoitov" <ast@kernel.org>,
"Daniel Borkmann" <daniel@iogearbox.net>,
"Kumar Kartikeya Dwivedi" <memxor@gmail.com>
Cc: Puranjay Mohan <puranjay@kernel.org>,
Steven Rostedt <rostedt@goodmis.org>,
Mathieu Desnoyers <mathieu.desnoyers@efficios.com>,
Zqiang <qiang.zhang@linux.dev>,
Martin KaFai Lau <martin.lau@linux.dev>,
Song Liu <song@kernel.org>,
Yonghong Song <yonghong.song@linux.dev>,
Jiri Olsa <jolsa@kernel.org>,
Emil Tsalapatis <emil@etsalapatis.com>,
Matt Fleming <mfleming@cloudflare.com>,
"Harry Yoo (Oracle)" <harry@kernel.org>,
linux-kernel@vger.kernel.org, rcu@vger.kernel.org,
bpf@vger.kernel.org, linux-rt-devel@lists.linux.dev
Subject: [PATCH v3 5/6] rcutorture: Exercise ->call() from NMI context
Date: Wed, 5 Aug 2026 05:23:42 -0700 [thread overview]
Message-ID: <20260805122346.269445-6-puranjay@kernel.org> (raw)
In-Reply-To: <20260805122346.269445-1-puranjay@kernel.org>
call_rcu() and call_srcu() are now safe to invoke from NMI, but
rcutorture never does, leaving the deferral path untested.
Add an ->nmi_capable flag to rcu_torture_ops. For flavors that set it,
arm a per-CPU hardware perf counter whose overflow handler submits a
callback via ->call(). The handler acts only when in_nmi(), so only a
genuine NMI exercises the deferral path. One preallocated callback is
kept in flight (guarded by an atomic) to avoid allocating in NMI.
Report the count issued from NMI ("nmi-calls:") and the count invoked
("nmi-cbs:"). rcu_torture_cleanup() disables the counters and then calls
cb_barrier(), which drains every deferred callback, so the two counts must
then match; a mismatch means a lost callback and fails the test. This
relies on srcu_barrier()/rcu_barrier() flushing deferred callbacks, as
added earlier in the series.
Set ->nmi_capable on the NMI-safe flavors: rcu, srcu, srcud, and
tasks-tracing (call_srcu() under the hood). Tasks and Tasks Rude are left
alone, as call_rcu_tasks_generic() is not yet NMI-safe.
Enabled by default; the nmi_calls parameter disables it, which helps rule
NMI handling in or out when triaging a failure. Requires
CONFIG_PERF_EVENTS and a hardware PMU, else silently skipped.
Signed-off-by: Puranjay Mohan <puranjay@kernel.org>
---
kernel/rcu/rcutorture.c | 154 +++++++++++++++++++++++++++++++++++++++-
1 file changed, 152 insertions(+), 2 deletions(-)
diff --git a/kernel/rcu/rcutorture.c b/kernel/rcu/rcutorture.c
index 39426a8718fe9..715c4c6c51b90 100644
--- a/kernel/rcu/rcutorture.c
+++ b/kernel/rcu/rcutorture.c
@@ -48,6 +48,7 @@
#include <linux/tick.h>
#include <linux/rcupdate_trace.h>
#include <linux/nmi.h>
+#include <linux/perf_event.h>
#include "rcu.h"
@@ -115,6 +116,7 @@ torture_param(int, leakpointer, 0, "Leak pointer dereferences from readers");
torture_param(int, n_barrier_cbs, 0, "# of callbacks/kthreads for barrier testing");
torture_param(int, n_up_down, 32, "# of concurrent up/down hrtimer-based RCU readers");
torture_param(int, nfakewriters, 4, "Number of RCU fake writer threads");
+torture_param(bool, nmi_calls, true, "Exercise ->call() from NMI on nmi_capable flavors");
torture_param(int, nreaders, -1, "Number of RCU reader threads");
torture_param(bool, nwriters, 1, "Number of RCU writer threads (0 or 1)");
torture_param(int, object_debug, 0, "Enable debug-object double call_rcu() testing");
@@ -216,6 +218,8 @@ static long n_rcu_torture_boost_failure;
static long n_rcu_torture_boosts;
static atomic_long_t n_rcu_torture_timers;
static atomic_long_t n_rcu_torture_irqs;
+static atomic_long_t n_rcu_torture_nmi_call;
+static atomic_long_t n_rcu_torture_nmi_cb;
static long n_barrier_attempts;
static long n_barrier_successes; /* did rcu_barrier test succeed? */
static unsigned long n_read_exits;
@@ -433,6 +437,7 @@ struct rcu_torture_ops {
bool (*is_task_rcu_boosted)(void);
long cbflood_max;
int irq_capable;
+ int nmi_capable;
int can_boost;
int extendables;
int slow_gps;
@@ -648,6 +653,7 @@ static struct rcu_torture_ops rcu_ops = {
.extendables = RCUTORTURE_MAX_EXTEND,
.debug_objects = 1,
.start_poll_irqsoff = 1,
+ .nmi_capable = 1,
.name = "rcu"
};
@@ -942,6 +948,7 @@ static struct rcu_torture_ops srcu_ops = {
.debug_objects = 1,
.have_up_down = IS_ENABLED(CONFIG_TINY_SRCU)
? 0 : SRCU_READ_FLAVOR_NORMAL | SRCU_READ_FLAVOR_FAST_UPDOWN,
+ .nmi_capable = 1,
.name = "srcu"
};
@@ -1005,6 +1012,7 @@ static struct rcu_torture_ops srcud_ops = {
.debug_objects = 1,
.have_up_down = IS_ENABLED(CONFIG_TINY_SRCU)
? 0 : SRCU_READ_FLAVOR_NORMAL | SRCU_READ_FLAVOR_FAST_UPDOWN,
+ .nmi_capable = 1,
.name = "srcud"
};
@@ -1269,6 +1277,7 @@ static struct rcu_torture_ops tasks_tracing_ops = {
.cbflood_max = 50000,
.irq_capable = 1,
.slow_gps = 1,
+ .nmi_capable = 1,
.name = "tasks-tracing"
};
@@ -2659,12 +2668,131 @@ static bool rcu_torture_one_read(struct torture_random_state *trsp, long myid)
static DEFINE_TORTURE_RANDOM_PERCPU(rcu_torture_timer_rand);
+/*
+ * Exercise ->call() from NMI context for flavors that set ->nmi_capable. A
+ * per-CPU hardware perf counter overflows into an NMI, and its handler submits
+ * one preallocated callback via ->call(). One callback is in flight at a time
+ * (guarded by an atomic) to avoid allocating in NMI. This mirrors how a BPF
+ * program reaches ->call() from NMI.
+ */
+#ifdef CONFIG_PERF_EVENTS
+static struct perf_event_attr rcu_torture_nmi_attr = {
+ .type = PERF_TYPE_HARDWARE,
+ .config = PERF_COUNT_HW_CPU_CYCLES,
+ .size = sizeof(struct perf_event_attr),
+ .pinned = 1,
+ .disabled = 1,
+ .freq = 1,
+ .sample_freq = 100,
+};
+
+/* One in-flight callback per CPU; ->inuse is released by the callback. */
+struct rcu_torture_nmi_cbs {
+ struct rcu_head rh;
+ atomic_t inuse;
+};
+
+static struct perf_event **rcu_torture_nmi_events;
+static int rcu_torture_nmi_hp_state;
+static DEFINE_PER_CPU(struct rcu_torture_nmi_cbs, rcu_torture_nmi_cbs);
+
+static void rcu_torture_nmi_cb(struct rcu_head *rhp)
+{
+ struct rcu_torture_nmi_cbs *rtncp = container_of(rhp, struct rcu_torture_nmi_cbs, rh);
+
+ atomic_long_inc(&n_rcu_torture_nmi_cb);
+ atomic_set(&rtncp->inuse, 0);
+}
+
+static void rcu_torture_nmi_overflow(struct perf_event *event,
+ struct perf_sample_data *data,
+ struct pt_regs *regs)
+{
+ struct rcu_torture_nmi_cbs *rtncp = this_cpu_ptr(&rcu_torture_nmi_cbs);
+
+ if (!in_nmi())
+ return;
+ if (cur_ops->call && !atomic_xchg(&rtncp->inuse, 1)) {
+ atomic_long_inc(&n_rcu_torture_nmi_call);
+ cur_ops->call(&rtncp->rh, rcu_torture_nmi_cb);
+ }
+}
+
+static int rcu_torture_nmi_online(unsigned int cpu)
+{
+ struct perf_event *event;
+
+ event = perf_event_create_kernel_counter(&rcu_torture_nmi_attr, cpu, NULL,
+ rcu_torture_nmi_overflow, NULL);
+ if (IS_ERR(event))
+ return 0;
+ rcu_torture_nmi_events[cpu] = event;
+ perf_event_enable(event);
+ return 0;
+}
+
+static int rcu_torture_nmi_offline(unsigned int cpu)
+{
+ struct perf_event *event = rcu_torture_nmi_events[cpu];
+
+ if (event) {
+ rcu_torture_nmi_events[cpu] = NULL;
+ perf_event_disable(event);
+ perf_event_release_kernel(event);
+ }
+ return 0;
+}
+
+/*
+ * Drive the counters from CPU-hotplug callbacks so that coverage survives the
+ * onoff testing that most scenarios run.
+ */
+static void rcu_torture_nmi_init(void)
+{
+ int ret;
+
+ if (!nmi_calls || !cur_ops->nmi_capable || !cur_ops->call)
+ return;
+ rcu_torture_nmi_events = kcalloc(nr_cpu_ids, sizeof(*rcu_torture_nmi_events),
+ GFP_KERNEL);
+ if (!rcu_torture_nmi_events)
+ return;
+ ret = cpuhp_setup_state(CPUHP_AP_ONLINE_DYN, "rcutorture/nmi:online",
+ rcu_torture_nmi_online, rcu_torture_nmi_offline);
+ if (ret < 0) {
+ kfree(rcu_torture_nmi_events);
+ rcu_torture_nmi_events = NULL;
+ return;
+ }
+ rcu_torture_nmi_hp_state = ret;
+}
+
+static void rcu_torture_nmi_cleanup(void)
+{
+ if (!rcu_torture_nmi_events)
+ return;
+ if (rcu_torture_nmi_hp_state > 0) {
+ cpuhp_remove_state(rcu_torture_nmi_hp_state);
+ rcu_torture_nmi_hp_state = 0;
+ }
+ kfree(rcu_torture_nmi_events);
+ rcu_torture_nmi_events = NULL;
+ if (!atomic_long_read(&n_rcu_torture_nmi_call))
+ pr_alert("%s: nmi_calls set but no ->call() issued from NMI: not supported here.\n",
+ __func__);
+}
+#else /* #ifdef CONFIG_PERF_EVENTS */
+static void rcu_torture_nmi_init(void) { }
+static void rcu_torture_nmi_cleanup(void) { }
+#endif /* #else #ifdef CONFIG_PERF_EVENTS */
+
/*
* RCU torture reader from timer handler. Dereferences rcu_torture_current,
* incrementing the corresponding element of the pipeline array. The
* counter in the element should never be greater than 1, otherwise, the
* RCU implementation is broken.
*/
+
static void rcu_torture_timer(struct timer_list *unused)
{
WARN_ON_ONCE(!in_serving_softirq());
@@ -3047,6 +3175,9 @@ rcu_torture_stats_print(void)
data_race(n_barrier_attempts),
data_race(n_rcu_torture_barrier_error));
pr_cont("read-exits: %ld ", data_race(n_read_exits)); // Statistic.
+ pr_cont("nmi-calls: %ld nmi-cbs: %ld ",
+ atomic_long_read(&n_rcu_torture_nmi_call),
+ atomic_long_read(&n_rcu_torture_nmi_cb));
pr_cont("nocb-toggles: %ld:%ld ",
atomic_long_read(&n_nocb_offload), atomic_long_read(&n_nocb_deoffload));
pr_cont("gpwraps: %ld\n", n_gpwraps);
@@ -3195,7 +3326,7 @@ rcu_torture_print_module_parms(struct rcu_torture_ops *cur_ops, const char *tag)
"read_exit_delay=%d read_exit_burst=%d "
"reader_flavor=%x "
"nocbs_nthreads=%d nocbs_toggle=%d "
- "test_nmis=%d "
+ "test_nmis=%d nmi_calls=%d "
"preempt_duration=%d preempt_interval=%d n_up_down=%d\n",
torture_type, tag, nrealreaders, nwriters, nrealfakewriters,
stat_interval, verbose, test_no_idle_hz, shuffle_interval,
@@ -3209,7 +3340,7 @@ rcu_torture_print_module_parms(struct rcu_torture_ops *cur_ops, const char *tag)
read_exit_delay, read_exit_burst,
reader_flavor,
nocbs_nthreads, nocbs_toggle,
- test_nmis,
+ test_nmis, nmi_calls,
preempt_duration, preempt_interval, n_up_down);
}
@@ -4283,6 +4414,7 @@ rcu_torture_cleanup(void)
int i;
if (torture_cleanup_begin()) {
+ rcu_torture_nmi_cleanup();
if (cur_ops->cb_barrier != NULL) {
pr_info("%s: Invoking %pS().\n", __func__, cur_ops->cb_barrier);
cur_ops->cb_barrier();
@@ -4325,6 +4457,8 @@ rcu_torture_cleanup(void)
kfree(reader_tasks);
reader_tasks = NULL;
}
+ /* Disable the perf counters (and thus the NMI ->call() firing) now. */
+ rcu_torture_nmi_cleanup();
kfree(rcu_torture_reader_mbchk);
rcu_torture_reader_mbchk = NULL;
@@ -4354,6 +4488,20 @@ rcu_torture_cleanup(void)
pr_info("%s: Invoking %pS().\n", __func__, cur_ops->cb_barrier);
cur_ops->cb_barrier();
}
+
+ /*
+ * cb_barrier() above drained every deferred NMI ->call() callback, so the
+ * count issued from NMI must equal the count invoked; a mismatch means a
+ * callback was lost.
+ */
+ if (atomic_long_read(&n_rcu_torture_nmi_call) !=
+ atomic_long_read(&n_rcu_torture_nmi_cb)) {
+ pr_alert("%s: NMI ->call() lost a callback: issued %ld invoked %ld\n",
+ __func__, atomic_long_read(&n_rcu_torture_nmi_call),
+ atomic_long_read(&n_rcu_torture_nmi_cb));
+ atomic_inc(&n_rcu_torture_error);
+ }
+
if (cur_ops->cleanup != NULL)
cur_ops->cleanup();
@@ -4786,6 +4934,8 @@ rcu_torture_init(void)
firsterr = -ENOMEM;
goto unwind;
}
+ /* Arm the per-CPU perf counters that drive ->call() from NMI. */
+ rcu_torture_nmi_init();
for (i = 0; i < nrealreaders; i++) {
rcu_torture_reader_mbchk[i].rtc_chkrdr = -1;
firsterr = torture_create_kthread(rcu_torture_reader, (void *)i,
--
2.53.0-Meta
next prev parent reply other threads:[~2026-08-05 12:24 UTC|newest]
Thread overview: 13+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-08-05 12:23 [PATCH v3 0/6] rcu,srcu: Make call_rcu()/call_srcu() safe from any context Puranjay Mohan
2026-08-05 12:23 ` [PATCH v3 1/6] rcu: Make call_rcu() safe to call " Puranjay Mohan
2026-08-05 12:36 ` sashiko-bot
2026-08-05 16:49 ` Paul E. McKenney
2026-08-05 12:23 ` [PATCH v3 2/6] rcu: Make Tiny " Puranjay Mohan
2026-08-05 12:37 ` sashiko-bot
2026-08-05 12:23 ` [PATCH v3 3/6] srcu: Make call_srcu() " Puranjay Mohan
2026-08-05 12:35 ` sashiko-bot
2026-08-06 14:04 ` Zqiang
2026-08-06 14:08 ` Puranjay Mohan
2026-08-05 12:23 ` [PATCH v3 4/6] srcu: Make Tiny " Puranjay Mohan
2026-08-05 12:23 ` Puranjay Mohan [this message]
2026-08-05 12:23 ` [PATCH v3 6/6] selftests/bpf: Add a call_srcu() re-entry reproducer Puranjay Mohan
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260805122346.269445-6-puranjay@kernel.org \
--to=puranjay@kernel.org \
--cc=andrii@kernel.org \
--cc=ast@kernel.org \
--cc=boqun@kernel.org \
--cc=bpf@vger.kernel.org \
--cc=daniel@iogearbox.net \
--cc=dave@stgolabs.net \
--cc=eddyz87@gmail.com \
--cc=emil@etsalapatis.com \
--cc=frederic@kernel.org \
--cc=harry@kernel.org \
--cc=jiangshanlai@gmail.com \
--cc=joelagnelf@nvidia.com \
--cc=jolsa@kernel.org \
--cc=josh@joshtriplett.org \
--cc=linux-kernel@vger.kernel.org \
--cc=linux-rt-devel@lists.linux.dev \
--cc=martin.lau@linux.dev \
--cc=mathieu.desnoyers@efficios.com \
--cc=memxor@gmail.com \
--cc=mfleming@cloudflare.com \
--cc=neeraj.upadhyay@kernel.org \
--cc=paulmck@kernel.org \
--cc=qiang.zhang@linux.dev \
--cc=rcu@vger.kernel.org \
--cc=rostedt@goodmis.org \
--cc=song@kernel.org \
--cc=urezki@gmail.com \
--cc=work@onurozkan.dev \
--cc=yonghong.song@linux.dev \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.