* [PATCH v3 1/6] rcu: Make call_rcu() safe to call from any context
2026-08-05 12:23 [PATCH v3 0/6] rcu,srcu: Make call_rcu()/call_srcu() safe from any context Puranjay Mohan
@ 2026-08-05 12:23 ` Puranjay Mohan
2026-08-05 12:36 ` sashiko-bot
2026-08-05 12:23 ` [PATCH v3 2/6] rcu: Make Tiny " Puranjay Mohan
` (4 subsequent siblings)
5 siblings, 1 reply; 12+ messages in thread
From: Puranjay Mohan @ 2026-08-05 12:23 UTC (permalink / raw)
To: Lai Jiangshan, Paul E. McKenney, Josh Triplett, Onur Özkan,
Frederic Weisbecker, Neeraj Upadhyay, Joel Fernandes, Boqun Feng,
Uladzislau Rezki, Davidlohr Bueso, Andrii Nakryiko,
Eduard Zingerman, Alexei Starovoitov, Daniel Borkmann,
Kumar Kartikeya Dwivedi
Cc: Puranjay Mohan, Steven Rostedt, Mathieu Desnoyers, Zqiang,
Martin KaFai Lau, Song Liu, Yonghong Song, Jiri Olsa,
Emil Tsalapatis, Matt Fleming, Harry Yoo (Oracle), linux-kernel,
rcu, bpf, linux-rt-devel
RCU's per-CPU callback list is only touched with interrupts disabled:
the enqueue runs under local_irq_save() (and the nocb locks when
offloaded), as do callback invocation and grace-period work. A
call_rcu() that arrives with interrupts already disabled, whether from an
NMI or from instrumentation that re-enters RCU, can interrupt one of
those and corrupt the list or deadlock.
Handle it by deferring: stage the callback on a per-CPU llist and raise
an irq_work that re-issues it once interrupts are on. The re-issue goes
straight to the enqueue so it cannot defer again. Callers that only hold
interrupts off are deferred too, which is harmless. Skip the gate while
the scheduler is down (RCU_SCHEDULER_INACTIVE), since irq_work is not
usable that early and rcu_init() already calls call_rcu().
rcu_barrier() flushes deferred callbacks before it scans the lists by
draining every CPU's ->defer_head itself, and rcutree_migrate_callbacks()
drains an outgoing CPU's. ->defer_lock is held across llist_del_all() and
the re-issue, so these drainers serialize: a drain that finds the list
empty knows any re-issue in flight has already reached a callback list.
A deferred callback is re-issued without the lazy hint: it has already
waited for the drain, so batching it further would only add latency.
The re-issue runs with interrupts disabled, so instrumentation on the
enqueue path can re-enter call_rcu(), stage another callback, and
re-raise the irq_work, livelocking the drain. A per-CPU flag guards it:
a call_rcu() that tries to defer while the drain is running, and is not
from an NMI, is dropped with WARN_ONCE() instead of re-queued. Dropping
leaks the callback, but the alternative is an unbounded loop.
The irq_work is IRQ_WORK_INIT_HARD so the re-issue stays prompt on
PREEMPT_RT, where a non-HARD irq_work runs in a kthread that can be
delayed under load. A hidden CONFIG_RCU_DEFER gates the code and its
IRQ_WORK dependency; without it call_rcu() enqueues directly as before.
Under CONFIG_PROVE_RCU, warn if the direct path is reached from an NMI.
Suggested-by: Paul E. McKenney <paulmck@kernel.org>
Signed-off-by: Puranjay Mohan <puranjay@kernel.org>
---
kernel/rcu/Kconfig | 6 ++
kernel/rcu/rcu.h | 15 +++++
kernel/rcu/tree.c | 139 +++++++++++++++++++++++++++++++++++++++++----
kernel/rcu/tree.h | 6 ++
4 files changed, 156 insertions(+), 10 deletions(-)
diff --git a/kernel/rcu/Kconfig b/kernel/rcu/Kconfig
index f15da8038d0ba..1a5fb3156c062 100644
--- a/kernel/rcu/Kconfig
+++ b/kernel/rcu/Kconfig
@@ -175,6 +175,12 @@ config RCU_STALL_COMMON
config RCU_NEED_SEGCBLIST
def_bool ( TREE_RCU || TREE_SRCU || TASKS_RCU_GENERIC )
+# The deferral (and the IRQ_WORK it uses) is only needed where call_rcu() /
+# call_srcu() can be invoked while a callback-list operation is in flight.
+config RCU_DEFER
+ def_bool HAVE_NMI || KPROBES || FUNCTION_TRACER || TRACEPOINTS
+ select IRQ_WORK
+
config RCU_FANOUT
int "Tree-based hierarchical RCU fanout value"
range 2 64 if 64BIT
diff --git a/kernel/rcu/rcu.h b/kernel/rcu/rcu.h
index 39a9f6fa9a7b2..fd075d91b80cf 100644
--- a/kernel/rcu/rcu.h
+++ b/kernel/rcu/rcu.h
@@ -572,6 +572,21 @@ static inline void tasks_cblist_init_generic(void) { }
#define RCU_SCHEDULER_INIT 1
#define RCU_SCHEDULER_RUNNING 2
+/*
+ * Defer a call_rcu()/call_srcu() callback rather than enqueue it now? Defer
+ * whenever interrupts are disabled, since a callback-list operation may be in
+ * flight on this CPU. Not before the scheduler is up, though: that covers
+ * early boot, where irq_work is not yet usable and rcu_init() itself already
+ * calls call_rcu().
+ */
+static inline bool should_rcu_defer(void)
+{
+ if (!IS_ENABLED(CONFIG_RCU_DEFER))
+ return false;
+
+ return irqs_disabled() && rcu_scheduler_active != RCU_SCHEDULER_INACTIVE;
+}
+
enum rcutorture_type {
RCU_FLAVOR,
RCU_TASKS_FLAVOR,
diff --git a/kernel/rcu/tree.c b/kernel/rcu/tree.c
index 21b6ce1dffb63..6a83408974c2b 100644
--- a/kernel/rcu/tree.c
+++ b/kernel/rcu/tree.c
@@ -24,6 +24,7 @@
#include <linux/smp.h>
#include <linux/rcupdate_wait.h>
#include <linux/interrupt.h>
+#include <linux/llist.h>
#include <linux/sched.h>
#include <linux/sched/debug.h>
#include <linux/nmi.h>
@@ -3148,21 +3149,25 @@ static void check_cb_ovld(struct rcu_data *rdp)
raw_spin_unlock_rcu_node(rnp);
}
-static void
-__call_rcu_common(struct rcu_head *head, rcu_callback_t func, bool lazy_in)
+/*
+ * The callback list is only accessed with interrupts disabled, so a call_rcu()
+ * that arrives with interrupts off (see should_rcu_defer()) stages the callback
+ * on a per-CPU llist that an irq_work re-issues once interrupts are on.
+ */
+static void rcu_defer_drain(struct irq_work *iw);
+
+/*
+ * Enqueue @head on this CPU's rcu_segcblist. Also called by rcu_defer_drain()
+ * to re-issue a deferred callback, so it must not re-check the deferral
+ * condition. Either caller may have interrupts already disabled.
+ */
+static void rcu_do_enqueue(struct rcu_head *head, rcu_callback_t func, bool lazy_in)
{
static atomic_t doublefrees;
unsigned long flags;
bool lazy;
struct rcu_data *rdp;
- /* Misaligned rcu_head! */
- WARN_ON_ONCE((unsigned long)head & (sizeof(void *) - 1));
-
- /* Avoid NULL dereference if callback is NULL. */
- if (WARN_ON_ONCE(!func))
- return;
-
if (debug_rcu_head_queue(head)) {
/*
* Probable double call_rcu(), so leak the callback.
@@ -3206,6 +3211,104 @@ __call_rcu_common(struct rcu_head *head, rcu_callback_t func, bool lazy_in)
local_irq_restore(flags);
}
+/*
+ * Re-issue deferred callbacks straight to the enqueue so they cannot defer
+ * again. ->defer_lock serializes the drainers: this CPU's irq_work,
+ * rcu_defer_flush() and rcutree_migrate_callbacks().
+ */
+static void __rcu_defer_drain(struct rcu_data *rdp, bool guard)
+{
+ struct llist_node *node, *next;
+ unsigned long flags;
+
+ if (!IS_ENABLED(CONFIG_RCU_DEFER))
+ return;
+
+ raw_spin_lock_irqsave(&rdp->defer_lock, flags);
+ if (guard)
+ WRITE_ONCE(rdp->defer_draining, true);
+ llist_for_each_safe(node, next, llist_del_all(&rdp->defer_head)) {
+ struct rcu_head *head = (struct rcu_head *)node;
+
+ head->next = NULL;
+ rcu_do_enqueue(head, head->func, false);
+ }
+ if (guard)
+ WRITE_ONCE(rdp->defer_draining, false);
+ raw_spin_unlock_irqrestore(&rdp->defer_lock, flags);
+}
+
+/*
+ * Only the irq_work drain can be re-fed by its own re-issue, so only it sets
+ * ->defer_draining. A direct drain re-issues onto this CPU, and anything
+ * staged during it is picked up by that CPU's own irq_work.
+ */
+static void rcu_defer_drain(struct irq_work *iw)
+{
+ __rcu_defer_drain(container_of(iw, struct rcu_data, defer_work), true);
+}
+
+/* Stage @head for this CPU's irq_work when call_rcu() cannot enqueue now. */
+static void call_rcu_defer(struct rcu_head *head, rcu_callback_t func)
+{
+ struct rcu_data *rdp = this_cpu_ptr(&rcu_data);
+
+ /*
+ * Instrumentation on the enqueue path can re-enter here from inside
+ * rcu_defer_drain(). Re-queuing would livelock the drain, so drop the
+ * callback; an NMI is one-shot and cannot loop, so let it through.
+ */
+ if (READ_ONCE(rdp->defer_draining) && !in_nmi()) {
+ WARN_ONCE(IS_ENABLED(CONFIG_PROVE_RCU),
+ "call_rcu() re-entered during callback drain; leaking callback\n");
+ return;
+ }
+ head->func = func;
+ if (llist_add((struct llist_node *)head, &rdp->defer_head))
+ irq_work_queue(&rdp->defer_work);
+}
+
+/*
+ * Register pending deferred callbacks into the callback lists so a following
+ * rcu_barrier() waits for them. This runs before rcu_barrier() scans the
+ * lists.
+ */
+static void rcu_defer_flush(void)
+{
+ int cpu;
+
+ if (!IS_ENABLED(CONFIG_RCU_DEFER))
+ return;
+
+ for_each_possible_cpu(cpu) {
+ struct rcu_data *rdp = per_cpu_ptr(&rcu_data, cpu);
+
+ if (!llist_empty(&rdp->defer_head))
+ __rcu_defer_drain(rdp, false);
+ }
+}
+
+static void
+__call_rcu_common(struct rcu_head *head, rcu_callback_t func, bool lazy_in)
+{
+ /* Misaligned rcu_head! */
+ WARN_ON_ONCE((unsigned long)head & (sizeof(void *) - 1));
+
+ /* Avoid NULL dereference if callback is NULL. */
+ if (WARN_ON_ONCE(!func))
+ return;
+
+ if (should_rcu_defer()) {
+ call_rcu_defer(head, func);
+ return;
+ }
+
+ /* An NMI reaching here entered with irqs enabled, so the enqueue can race. */
+ WARN_ON_ONCE(IS_ENABLED(CONFIG_PROVE_RCU) && in_nmi());
+
+ rcu_do_enqueue(head, func, lazy_in);
+}
+
#ifdef CONFIG_RCU_LAZY
static bool enable_rcu_lazy __read_mostly = !IS_ENABLED(CONFIG_RCU_LAZY_DEFAULT_OFF);
module_param(enable_rcu_lazy, bool, 0444);
@@ -3896,8 +3999,12 @@ void rcu_barrier(void)
unsigned long flags;
unsigned long gseq;
struct rcu_data *rdp;
- unsigned long s = rcu_seq_snap(&rcu_state.barrier_sequence);
+ unsigned long s;
+ /* Register any deferred callbacks before snapshotting the sequence. */
+ rcu_defer_flush();
+
+ s = rcu_seq_snap(&rcu_state.barrier_sequence);
rcu_barrier_trace(TPS("Begin"), -1, s);
/* Take mutex to serialize concurrent rcu_barrier() requests. */
@@ -4231,6 +4338,10 @@ rcu_boot_init_percpu_data(int cpu)
rdp->rcu_onl_gp_state = RCU_GP_CLEANED;
rdp->last_sched_clock = jiffies;
rdp->cpu = cpu;
+ init_llist_head(&rdp->defer_head);
+ raw_spin_lock_init(&rdp->defer_lock);
+ /* Hard irq_work so the re-issue runs promptly. */
+ rdp->defer_work = IRQ_WORK_INIT_HARD(rcu_defer_drain);
rcu_boot_init_nocb_percpu_data(rdp);
}
@@ -4528,6 +4639,14 @@ void rcutree_migrate_callbacks(int cpu)
struct rcu_data *rdp = per_cpu_ptr(&rcu_data, cpu);
bool needwake;
+ /*
+ * Callbacks the outgoing CPU deferred late in the offline path (past the
+ * point its irq_work can run) sit on ->defer_head, which the ->cblist
+ * migration below does not cover. Drain them here, before the early
+ * returns; the re-issue lands on this CPU.
+ */
+ __rcu_defer_drain(rdp, false);
+
if (rcu_rdp_is_offloaded(rdp))
return;
diff --git a/kernel/rcu/tree.h b/kernel/rcu/tree.h
index eedfa43059e80..b7cac7a13b4f2 100644
--- a/kernel/rcu/tree.h
+++ b/kernel/rcu/tree.h
@@ -229,6 +229,12 @@ struct rcu_data {
struct rcu_head barrier_head;
int exp_watching_snap; /* Double-check need for IPI. */
+ /* Deferral of an NMI/reentrant call_rcu(); see __call_rcu_common(). */
+ struct llist_head defer_head;
+ struct irq_work defer_work;
+ raw_spinlock_t defer_lock;
+ bool defer_draining;
+
/* 5) Callback offloading. */
#ifdef CONFIG_RCU_NOCB_CPU
struct swait_queue_head nocb_cb_wq; /* For nocb kthreads to sleep on. */
--
2.53.0-Meta
^ permalink raw reply related [flat|nested] 12+ messages in thread* Re: [PATCH v3 1/6] rcu: Make call_rcu() safe to call from any context
2026-08-05 12:23 ` [PATCH v3 1/6] rcu: Make call_rcu() safe to call " Puranjay Mohan
@ 2026-08-05 12:36 ` sashiko-bot
0 siblings, 0 replies; 12+ messages in thread
From: sashiko-bot @ 2026-08-05 12:36 UTC (permalink / raw)
To: Puranjay Mohan; +Cc: linux-rt-devel, bpf, linux-kernel
Thank you for your contribution! Sashiko AI review found 3 potential issue(s) to consider:
- [Critical] A lockless check in rcu_defer_flush() causes rcu_barrier() to miss in-flight deferred callbacks, violating the core rcu_barrier() guarantee.
- [High] __rcu_defer_drain() creates an unbounded non-interruptible critical section, breaking PREEMPT_RT constraints and risking hard lockups.
- [Medium] KASAN stack trace recording for deferred callbacks records the asynchronous irq_work context instead of the actual call_rcu() caller.
--
--- Patch [1]: [PATCH v3 1/6] rcu: Make call_rcu() safe to call from any context ---
commit 2b52f34260090add38843ff0920806c5e56bf18c
Author: Puranjay Mohan <puranjay@kernel.org>
Subject: rcu: Make call_rcu() safe to call from any context
This commit makes call_rcu() safe to call from any context by deferring
callbacks to a per-CPU llist when interrupts are disabled. An irq_work
then re-issues these callbacks once interrupts are re-enabled, preventing
deadlocks or list corruption.
> diff --git a/kernel/rcu/tree.c b/kernel/rcu/tree.c
> index 96848fc1f02b8..6fd7331aebc58 100644
> --- a/kernel/rcu/tree.c
> +++ b/kernel/rcu/tree.c
[ ... ]
> @@ -3206,6 +3211,104 @@ __call_rcu_common(struct rcu_head *head, rcu_callback_t func, bool lazy_in)
> local_irq_restore(flags);
> }
>
> +/*
> + * Re-issue deferred callbacks straight to the enqueue so they cannot defer
> + * again. ->defer_lock serializes the drainers: this CPU's irq_work,
> + * rcu_defer_flush() and rcutree_migrate_callbacks().
> + */
> +static void __rcu_defer_drain(struct rcu_data *rdp, bool guard)
> +{
> + struct llist_node *node, *next;
> + unsigned long flags;
> +
> + if (!IS_ENABLED(CONFIG_RCU_DEFER))
> + return;
> +
> + raw_spin_lock_irqsave(&rdp->defer_lock, flags);
> + if (guard)
> + WRITE_ONCE(rdp->defer_draining, true);
> + llist_for_each_safe(node, next, llist_del_all(&rdp->defer_head)) {
> + struct rcu_head *head = (struct rcu_head *)node;
> +
> + head->next = NULL;
> + rcu_do_enqueue(head, head->func, false);
> + }
[Severity: High]
Could holding this raw spinlock across an unbounded number of enqueues cause
latency spikes or hard lockups, especially on PREEMPT_RT?
If a bulk teardown path calls call_rcu() in a loop with interrupts disabled,
thousands of callbacks could be staged. When __rcu_defer_drain() runs, it
loops over all of them in a single uninterrupted pass. If another CPU
executes rcu_barrier() and calls __rcu_defer_drain(), it would also spin on
this raw lock with interrupts disabled for the entire duration of the drain.
> + if (guard)
> + WRITE_ONCE(rdp->defer_draining, false);
> + raw_spin_unlock_irqrestore(&rdp->defer_lock, flags);
> +}
[ ... ]
> +/* Stage @head for this CPU's irq_work when call_rcu() cannot enqueue now. */
> +static void call_rcu_defer(struct rcu_head *head, rcu_callback_t func)
> +{
> + struct rcu_data *rdp = this_cpu_ptr(&rcu_data);
[ ... ]
> + head->func = func;
> + if (llist_add((struct llist_node *)head, &rdp->defer_head))
> + irq_work_queue(&rdp->defer_work);
> +}
[Severity: Medium]
Will this delay KASAN stack trace recording to the irq_work context?
Because call_rcu_defer() stages the callback without recording the stack trace,
the trace gets recorded later by kasan_record_aux_stack(head) inside
rcu_do_enqueue(). This means any use-after-free reports might point to
the irq_work instead of the original call_rcu() caller, making debugging
more difficult.
> +/*
> + * Register pending deferred callbacks into the callback lists so a following
> + * rcu_barrier() waits for them. This runs before rcu_barrier() scans the
> + * lists.
> + */
> +static void rcu_defer_flush(void)
> +{
> + int cpu;
> +
> + if (!IS_ENABLED(CONFIG_RCU_DEFER))
> + return;
> +
> + for_each_possible_cpu(cpu) {
> + struct rcu_data *rdp = per_cpu_ptr(&rcu_data, cpu);
> +
> + if (!llist_empty(&rdp->defer_head))
> + __rcu_defer_drain(rdp, false);
> + }
> +}
[Severity: Critical]
Can this lockless check cause rcu_barrier() to miss in-flight callbacks?
If CPU A's irq_work detaches the list via llist_del_all() in
__rcu_defer_drain(), defer_head becomes empty immediately. If CPU B
executes rcu_defer_flush() concurrently, the lockless llist_empty()
check will return true and bypass acquiring defer_lock.
Since CPU A has detached the callbacks but hasn't yet enqueued them
into cblist, CPU B might skip __rcu_defer_drain(), check a cblist that
does not yet contain the detached callbacks, and allow rcu_barrier()
to return prematurely while objects are still in use.
--
Sashiko AI review · https://sashiko.dev/#/patchset/20260805122346.269445-1-puranjay@kernel.org?part=1
^ permalink raw reply [flat|nested] 12+ messages in thread
* [PATCH v3 2/6] rcu: Make Tiny call_rcu() safe to call from any context
2026-08-05 12:23 [PATCH v3 0/6] rcu,srcu: Make call_rcu()/call_srcu() safe from any context Puranjay Mohan
2026-08-05 12:23 ` [PATCH v3 1/6] rcu: Make call_rcu() safe to call " Puranjay Mohan
@ 2026-08-05 12:23 ` Puranjay Mohan
2026-08-05 12:37 ` sashiko-bot
2026-08-05 12:23 ` [PATCH v3 3/6] srcu: Make call_srcu() " Puranjay Mohan
` (3 subsequent siblings)
5 siblings, 1 reply; 12+ messages in thread
From: Puranjay Mohan @ 2026-08-05 12:23 UTC (permalink / raw)
To: Lai Jiangshan, Paul E. McKenney, Josh Triplett, Onur Özkan,
Frederic Weisbecker, Neeraj Upadhyay, Joel Fernandes, Boqun Feng,
Uladzislau Rezki, Davidlohr Bueso, Andrii Nakryiko,
Eduard Zingerman, Alexei Starovoitov, Daniel Borkmann,
Kumar Kartikeya Dwivedi
Cc: Puranjay Mohan, Steven Rostedt, Mathieu Desnoyers, Zqiang,
Martin KaFai Lau, Song Liu, Yonghong Song, Jiri Olsa,
Emil Tsalapatis, Matt Fleming, Harry Yoo (Oracle), linux-kernel,
rcu, bpf, linux-rt-devel
Give Tiny call_rcu() the same treatment as Tree RCU. When interrupts are
disabled and the scheduler is up, stage the callback on a lockless list
that an irq_work re-issues later. One global list and irq_work suffice
since Tiny RCU is uniprocessor, and there is no CPU-offline drain.
The re-issue runs with interrupts disabled and can be re-entered by
instrumentation, so a draining flag drops a deferring call_rcu() seen
mid-drain (unless from an NMI), as in Tree RCU. Gated by CONFIG_RCU_DEFER.
Suggested-by: Paul E. McKenney <paulmck@kernel.org>
Signed-off-by: Puranjay Mohan <puranjay@kernel.org>
---
kernel/rcu/tiny.c | 117 ++++++++++++++++++++++++++++++++++++++--------
1 file changed, 98 insertions(+), 19 deletions(-)
diff --git a/kernel/rcu/tiny.c b/kernel/rcu/tiny.c
index dccccd6be9411..baffe660043f5 100644
--- a/kernel/rcu/tiny.c
+++ b/kernel/rcu/tiny.c
@@ -11,6 +11,8 @@
*/
#include <linux/completion.h>
#include <linux/interrupt.h>
+#include <linux/irq_work.h>
+#include <linux/llist.h>
#include <linux/notifier.h>
#include <linux/rcupdate_wait.h>
#include <linux/kernel.h>
@@ -42,8 +44,99 @@ static struct rcu_ctrlblk rcu_ctrlblk = {
.gp_seq = 0 - 300UL,
};
+/*
+ * The callback list is only accessed with interrupts disabled, so a call_rcu()
+ * that arrives with interrupts off stages the callback on a lockless list that
+ * an irq_work re-issues later. One global list and irq_work suffice, as Tiny
+ * RCU is uniprocessor.
+ */
+static void rcu_defer_drain(struct irq_work *iw);
+static LLIST_HEAD(rcu_defer_list);
+static struct irq_work rcu_defer_iw = IRQ_WORK_INIT_HARD(rcu_defer_drain);
+static bool rcu_defer_draining;
+
+/*
+ * Enqueue @head on the callback list. Also called by rcu_defer_drain() to
+ * re-issue a deferred callback, so it must not re-check the deferral condition.
+ */
+static void rcu_do_enqueue(struct rcu_head *head, rcu_callback_t func)
+{
+ static atomic_t doublefrees;
+ unsigned long flags;
+
+ if (debug_rcu_head_queue(head)) {
+ if (atomic_inc_return(&doublefrees) < 4) {
+ pr_err("%s(): Double-freed CB %p->%pS()!!! ", __func__, head, head->func);
+ mem_dump_obj(head);
+ }
+ return;
+ }
+
+ head->func = func;
+ head->next = NULL;
+
+ local_irq_save(flags);
+ *rcu_ctrlblk.curtail = head;
+ rcu_ctrlblk.curtail = &head->next;
+ local_irq_restore(flags);
+
+ if (unlikely(is_idle_task(current))) {
+ /* force scheduling for rcu_qs() */
+ resched_cpu(0);
+ }
+}
+
+static void __rcu_defer_drain(bool guard)
+{
+ struct llist_node *node, *next;
+ unsigned long flags;
+
+ /* Callbacks are unordered, so drain in llist order without reversing. */
+ local_irq_save(flags);
+ if (guard)
+ WRITE_ONCE(rcu_defer_draining, true);
+ llist_for_each_safe(node, next, llist_del_all(&rcu_defer_list)) {
+ struct rcu_head *head = (struct rcu_head *)node;
+
+ head->next = NULL;
+ rcu_do_enqueue(head, head->func);
+ }
+ if (guard)
+ WRITE_ONCE(rcu_defer_draining, false);
+ local_irq_restore(flags);
+}
+
+/* Only the irq_work drain can be re-fed by its own re-issue; see Tree RCU. */
+static void rcu_defer_drain(struct irq_work *iw)
+{
+ __rcu_defer_drain(true);
+}
+
+static void call_rcu_defer(struct rcu_head *head, rcu_callback_t func)
+{
+ /* A re-entrant call_rcu() during the drain would livelock it; drop it. */
+ if (READ_ONCE(rcu_defer_draining) && !in_nmi()) {
+ WARN_ONCE(IS_ENABLED(CONFIG_PROVE_RCU),
+ "call_rcu() re-entered during callback drain; leaking callback\n");
+ return;
+ }
+ head->func = func;
+ if (llist_add((struct llist_node *)head, &rcu_defer_list))
+ irq_work_queue(&rcu_defer_iw);
+}
+
+/* Register any deferred callbacks so a following rcu_barrier() waits for them. */
+static void rcu_defer_flush(void)
+{
+ if (!IS_ENABLED(CONFIG_RCU_DEFER))
+ return;
+ __rcu_defer_drain(false);
+}
+
void rcu_barrier(void)
{
+ /* Register any deferred callbacks first. */
+ rcu_defer_flush();
wait_rcu_gp(call_rcu_hurry);
}
EXPORT_SYMBOL(rcu_barrier);
@@ -157,29 +250,15 @@ EXPORT_SYMBOL_GPL(synchronize_rcu);
*/
void call_rcu(struct rcu_head *head, rcu_callback_t func)
{
- static atomic_t doublefrees;
- unsigned long flags;
-
- if (debug_rcu_head_queue(head)) {
- if (atomic_inc_return(&doublefrees) < 4) {
- pr_err("%s(): Double-freed CB %p->%pS()!!! ", __func__, head, head->func);
- mem_dump_obj(head);
- }
+ if (should_rcu_defer()) {
+ call_rcu_defer(head, func);
return;
}
- head->func = func;
- head->next = NULL;
-
- local_irq_save(flags);
- *rcu_ctrlblk.curtail = head;
- rcu_ctrlblk.curtail = &head->next;
- local_irq_restore(flags);
+ /* An NMI reaching here entered with irqs enabled, so the enqueue can race. */
+ WARN_ON_ONCE(IS_ENABLED(CONFIG_PROVE_RCU) && in_nmi());
- if (unlikely(is_idle_task(current))) {
- /* force scheduling for rcu_qs() */
- resched_cpu(0);
- }
+ rcu_do_enqueue(head, func);
}
EXPORT_SYMBOL_GPL(call_rcu);
--
2.53.0-Meta
^ permalink raw reply related [flat|nested] 12+ messages in thread* Re: [PATCH v3 2/6] rcu: Make Tiny call_rcu() safe to call from any context
2026-08-05 12:23 ` [PATCH v3 2/6] rcu: Make Tiny " Puranjay Mohan
@ 2026-08-05 12:37 ` sashiko-bot
0 siblings, 0 replies; 12+ messages in thread
From: sashiko-bot @ 2026-08-05 12:37 UTC (permalink / raw)
To: Puranjay Mohan; +Cc: linux-rt-devel, bpf, linux-kernel
Thank you for your contribution! Sashiko AI review found 2 potential issue(s) to consider:
- [High] Bypass of debug validation causes fatal RCU list corruption on double-free.
- [High] O(N) runqueue spinlock acquisitions cause severe interrupt latency during callback drain.
--
--- Patch [2]: [PATCH v3 2/6] rcu: Make Tiny call_rcu() safe to call from any context ---
commit 695767be85ec82f2457b846af53d8afe8fb742d1
Author: Puranjay Mohan <puranjay@kernel.org>
rcu: Make Tiny call_rcu() safe to call from any context
This commit updates Tiny RCU's call_rcu() to defer callbacks when called
with interrupts disabled. It uses a lockless list and an irq_work to
re-issue them later, similar to Tree RCU.
> diff --git a/kernel/rcu/tiny.c b/kernel/rcu/tiny.c
> index dccccd6be9411..baffe660043f5 100644
> --- a/kernel/rcu/tiny.c
> +++ b/kernel/rcu/tiny.c
[ ... ]
> +static void rcu_do_enqueue(struct rcu_head *head, rcu_callback_t func)
> +{
> + static atomic_t doublefrees;
> + unsigned long flags;
> +
> + if (debug_rcu_head_queue(head)) {
> + if (atomic_inc_return(&doublefrees) < 4) {
> + pr_err("%s(): Double-freed CB %p->%pS()!!! ", __func__, head, head->func);
> + mem_dump_obj(head);
> + }
> + return;
> + }
> +
> + head->func = func;
> + head->next = NULL;
> +
> + local_irq_save(flags);
> + *rcu_ctrlblk.curtail = head;
> + rcu_ctrlblk.curtail = &head->next;
> + local_irq_restore(flags);
> +
> + if (unlikely(is_idle_task(current))) {
> + /* force scheduling for rcu_qs() */
> + resched_cpu(0);
> + }
> +}
> +
> +static void __rcu_defer_drain(bool guard)
> +{
> + struct llist_node *node, *next;
> + unsigned long flags;
> +
> + /* Callbacks are unordered, so drain in llist order without reversing. */
> + local_irq_save(flags);
> + if (guard)
> + WRITE_ONCE(rcu_defer_draining, true);
> + llist_for_each_safe(node, next, llist_del_all(&rcu_defer_list)) {
> + struct rcu_head *head = (struct rcu_head *)node;
> +
> + head->next = NULL;
> + rcu_do_enqueue(head, head->func);
[Severity: High]
Does this code cause severe interrupt latency during callback drain?
If a batch of RCU callbacks is deferred and the CPU subsequently goes idle,
__rcu_defer_drain() loops over the entire batch in hardirq context.
For each callback, rcu_do_enqueue() will call resched_cpu(0) because
current is the idle task. Since resched_cpu() unconditionally takes the
runqueue spinlock, this results in an O(N) locking sequence in hardirq
context, which could lead to watchdog soft lockups.
> + }
> + if (guard)
> + WRITE_ONCE(rcu_defer_draining, false);
> + local_irq_restore(flags);
> +}
[ ... ]
> @@ -157,29 +250,15 @@ EXPORT_SYMBOL_GPL(synchronize_rcu);
> */
> void call_rcu(struct rcu_head *head, rcu_callback_t func)
> {
> - static atomic_t doublefrees;
> - unsigned long flags;
> -
> - if (debug_rcu_head_queue(head)) {
> - if (atomic_inc_return(&doublefrees) < 4) {
> - pr_err("%s(): Double-freed CB %p->%pS()!!! ", __func__, head, head->func);
> - mem_dump_obj(head);
> - }
> + if (should_rcu_defer()) {
> + call_rcu_defer(head, func);
[Severity: High]
Does moving the debug validation after the deferral bypass it during a
double-free?
If a subsystem double-frees an RCU callback while interrupts are disabled,
it will take this deferral path instead of checking debug_rcu_head_queue().
In call_rcu_defer():
if (llist_add((struct llist_node *)head, &rcu_defer_list))
If the callback is already in the list, llist_add() modifies its next
pointer, creating a cycle in the lockless list. Later, when the
irq_work drains the list, llist_for_each_safe() in __rcu_defer_drain()
will loop infinitely, causing a system hard lockup in hardirq context.
> return;
> }
>
> - head->func = func;
> - head->next = NULL;
> -
> - local_irq_save(flags);
> - *rcu_ctrlblk.curtail = head;
> - rcu_ctrlblk.curtail = &head->next;
> - local_irq_restore(flags);
> + /* An NMI reaching here entered with irqs enabled, so the enqueue can race. */
> + WARN_ON_ONCE(IS_ENABLED(CONFIG_PROVE_RCU) && in_nmi());
>
> - if (unlikely(is_idle_task(current))) {
> - /* force scheduling for rcu_qs() */
> - resched_cpu(0);
> - }
> + rcu_do_enqueue(head, func);
> }
--
Sashiko AI review · https://sashiko.dev/#/patchset/20260805122346.269445-1-puranjay@kernel.org?part=2
^ permalink raw reply [flat|nested] 12+ messages in thread
* [PATCH v3 3/6] srcu: Make call_srcu() safe to call from any context
2026-08-05 12:23 [PATCH v3 0/6] rcu,srcu: Make call_rcu()/call_srcu() safe from any context Puranjay Mohan
2026-08-05 12:23 ` [PATCH v3 1/6] rcu: Make call_rcu() safe to call " Puranjay Mohan
2026-08-05 12:23 ` [PATCH v3 2/6] rcu: Make Tiny " Puranjay Mohan
@ 2026-08-05 12:23 ` Puranjay Mohan
2026-08-05 12:35 ` sashiko-bot
2026-08-06 14:04 ` Zqiang
2026-08-05 12:23 ` [PATCH v3 4/6] srcu: Make Tiny " Puranjay Mohan
` (2 subsequent siblings)
5 siblings, 2 replies; 12+ messages in thread
From: Puranjay Mohan @ 2026-08-05 12:23 UTC (permalink / raw)
To: Lai Jiangshan, Paul E. McKenney, Josh Triplett, Onur Özkan,
Frederic Weisbecker, Neeraj Upadhyay, Joel Fernandes, Boqun Feng,
Uladzislau Rezki, Davidlohr Bueso, Andrii Nakryiko,
Eduard Zingerman, Alexei Starovoitov, Daniel Borkmann,
Kumar Kartikeya Dwivedi
Cc: Puranjay Mohan, Steven Rostedt, Mathieu Desnoyers, Zqiang,
Martin KaFai Lau, Song Liu, Yonghong Song, Jiri Olsa,
Emil Tsalapatis, Matt Fleming, Harry Yoo (Oracle), linux-kernel,
rcu, bpf, linux-rt-devel
call_srcu() has the same constraint as call_rcu(): its callback list and
locks are only touched with interrupts disabled. srcu_gp_start_if_needed()
enqueues under raw_spin_lock_irqsave() and may walk the srcu_node tree, as
do callback invocation and grace-period work. A call_srcu() with
interrupts disabled can race a list operation in flight on this CPU and
corrupt the list or deadlock. call_rcu_tasks_trace() is call_srcu() under
the hood, so a sleepable BPF program freeing an object can reach this.
Defer as call_rcu() does: stage the callback on the srcu_data's
->defer_cbs, chain that srcu_data onto a per-CPU list, and raise a per-CPU
irq_work that re-issues it straight to the enqueue helper, never back
through __call_srcu(). The common path is unchanged and keeps interrupts
enabled across srcu_gp_start_if_needed().
The irq_work is per-CPU rather than per-srcu_struct and statically
initialized, so deferral never runs check_init_srcu_struct(); it is
IRQ_WORK_INIT_HARD as for call_rcu(). srcu_barrier() and
cleanup_srcu_struct() flush it first, and rcutree_migrate_callbacks()
calls srcu_offline_drain() for an outgoing CPU. ->lock is held across the
drain so the drainers serialize.
As in call_rcu(), the re-issue runs with interrupts disabled and can be
re-entered by instrumentation, so a flag on the srcu_data being drained
drops a deferring call_srcu() seen mid-drain (unless from an NMI). Staging
records only the callback, so an expedited request is remembered per
srcu_data in ->defer_exp and the whole batch is re-issued expedited rather
than silently downgraded to a normal grace period. A dropped callback can
also strand state its caller associated with it, not just the callback
itself.
Gated by CONFIG_RCU_DEFER. Under CONFIG_PROVE_RCU, warn if the direct
path is reached from an NMI.
Suggested-by: Paul E. McKenney <paulmck@kernel.org>
Signed-off-by: Puranjay Mohan <puranjay@kernel.org>
---
include/linux/srcutree.h | 5 ++
kernel/rcu/rcu.h | 3 +
kernel/rcu/srcutree.c | 171 ++++++++++++++++++++++++++++++++++++++-
kernel/rcu/tree.c | 2 +
4 files changed, 177 insertions(+), 4 deletions(-)
diff --git a/include/linux/srcutree.h b/include/linux/srcutree.h
index 75e54e4f963fa..09a9c8f4a6d24 100644
--- a/include/linux/srcutree.h
+++ b/include/linux/srcutree.h
@@ -13,6 +13,8 @@
#include <linux/rcu_node_tree.h>
#include <linux/completion.h>
+#include <linux/irq_work_types.h>
+#include <linux/llist.h>
struct srcu_node;
struct srcu_struct;
@@ -41,6 +43,9 @@ struct srcu_data {
bool srcu_cblist_invoking; /* Invoking these CBs? */
struct timer_list delay_work; /* Delay for CB invoking */
struct work_struct work; /* Context for CB invoking. */
+ struct llist_head defer_cbs; /* Callbacks deferred on re-entry. */
+ struct llist_node defer_link; /* Links onto the per-CPU deferral drain list */
+ bool defer_exp; /* A deferred callback asked to expedite. */
struct rcu_head srcu_barrier_head; /* For srcu_barrier() use. */
struct rcu_head srcu_ec_head; /* For srcu_expedite_current() use. */
int srcu_ec_state; /* State for srcu_expedite_current(). */
diff --git a/kernel/rcu/rcu.h b/kernel/rcu/rcu.h
index fd075d91b80cf..84d74cd5a351c 100644
--- a/kernel/rcu/rcu.h
+++ b/kernel/rcu/rcu.h
@@ -587,6 +587,9 @@ static inline bool should_rcu_defer(void)
return irqs_disabled() && rcu_scheduler_active != RCU_SCHEDULER_INACTIVE;
}
+/* Drain an outgoing CPU's deferred SRCU callbacks; see rcutree_migrate_callbacks(). */
+void srcu_offline_drain(int cpu);
+
enum rcutorture_type {
RCU_FLAVOR,
RCU_TASKS_FLAVOR,
diff --git a/kernel/rcu/srcutree.c b/kernel/rcu/srcutree.c
index 304112674e8a2..35fface51d50b 100644
--- a/kernel/rcu/srcutree.c
+++ b/kernel/rcu/srcutree.c
@@ -20,6 +20,7 @@
#include <linux/percpu.h>
#include <linux/preempt.h>
#include <linux/irq_work.h>
+#include <linux/llist.h>
#include <linux/rcupdate_wait.h>
#include <linux/sched.h>
#include <linux/smp.h>
@@ -79,6 +80,45 @@ static void process_srcu(struct work_struct *work);
static void srcu_irq_work(struct irq_work *work);
static void srcu_delay_timer(struct timer_list *t);
+struct srcu_defer;
+static void srcu_defer_drain(struct irq_work *iw);
+static void __srcu_defer_drain(struct srcu_defer *sndp, bool guard);
+
+/*
+ * Per-CPU call_srcu() deferral state, shared by every srcu_struct. A deferred
+ * callback is staged on its srcu_data's ->defer_cbs; that srcu_data is chained
+ * via ->defer_link onto ->list, which the irq_work walks.
+ */
+struct srcu_defer {
+ struct llist_head list;
+ struct irq_work iw;
+ raw_spinlock_t lock;
+ bool draining;
+};
+
+static DEFINE_PER_CPU(struct srcu_defer, srcu_defer) = {
+ .lock = __RAW_SPIN_LOCK_UNLOCKED(srcu_defer.lock),
+ .iw = IRQ_WORK_INIT_HARD(srcu_defer_drain),
+};
+
+/*
+ * Flush pending deferred callbacks so a following srcu_barrier() waits for them.
+ */
+static void srcu_defer_flush(void)
+{
+ int cpu;
+
+ if (!IS_ENABLED(CONFIG_RCU_DEFER))
+ return;
+
+ for_each_possible_cpu(cpu) {
+ struct srcu_defer *sndp = &per_cpu(srcu_defer, cpu);
+
+ if (!llist_empty(&sndp->list))
+ __srcu_defer_drain(sndp, false);
+ }
+}
+
/*
* Initialize SRCU per-CPU data. Note that statically allocated
* srcu_struct structures might already have srcu_read_lock() and
@@ -107,6 +147,11 @@ static void init_srcu_struct_data(struct srcu_struct *ssp)
sdp->cpu = cpu;
INIT_WORK(&sdp->work, srcu_invoke_callbacks);
timer_setup(&sdp->delay_work, srcu_delay_timer, 0);
+ /*
+ * ->defer_cbs, ->defer_link and ->defer_exp are valid when zeroed
+ * and are not reinitialized here, lest we clobber callbacks a
+ * reentrant call_srcu() already staged. See __call_srcu().
+ */
sdp->ssp = ssp;
}
}
@@ -695,7 +740,12 @@ void cleanup_srcu_struct(struct srcu_struct *ssp)
return; /* Just leak it! */
if (WARN_ON(srcu_readers_active(ssp)))
return; /* Just leak it! */
- /* Wait for irq_work to finish first as it may queue a new work. */
+ /*
+ * Drain deferred callbacks before syncing ->irq_work: re-issuing one can
+ * start a grace period and re-queue ->irq_work, which then schedules
+ * ->work, so both must be waited out after the drain.
+ */
+ srcu_defer_flush();
irq_work_sync(&sup->irq_work);
flush_delayed_work(&sup->work);
for_each_possible_cpu(cpu) {
@@ -1410,8 +1460,8 @@ static unsigned long srcu_gp_start_if_needed(struct srcu_struct *ssp,
* srcu_read_lock(), and srcu_read_unlock() that are all passed the same
* srcu_struct structure.
*/
-static void __call_srcu(struct srcu_struct *ssp, struct rcu_head *rhp,
- rcu_callback_t func, bool do_norm)
+static void srcu_do_enqueue(struct srcu_struct *ssp, struct rcu_head *rhp,
+ rcu_callback_t func, bool do_norm)
{
if (debug_rcu_head_queue(rhp)) {
/* Probable double call_srcu(), so leak the callback. */
@@ -1423,6 +1473,111 @@ static void __call_srcu(struct srcu_struct *ssp, struct rcu_head *rhp,
(void)srcu_gp_start_if_needed(ssp, rhp, do_norm);
}
+/*
+ * The srcu_cblist and srcu_node tree are only accessed with interrupts disabled
+ * (srcu_gp_start_if_needed() enqueues under raw_spin_lock_irqsave() and may walk
+ * the tree). Like call_rcu(), __call_srcu() defers when interrupts are already
+ * disabled, so a re-entrant call_srcu() -- e.g. call_rcu_tasks_trace() from a
+ * BPF program -- cannot corrupt the list or deadlock.
+ */
+static void __call_srcu(struct srcu_struct *ssp, struct rcu_head *rhp,
+ rcu_callback_t func, bool do_norm)
+{
+ if (should_rcu_defer()) {
+ struct srcu_defer *sndp = this_cpu_ptr(&srcu_defer);
+ struct srcu_data *sdp;
+
+ /*
+ * Instrumentation on the enqueue path can re-enter here from
+ * inside srcu_defer_drain(). Re-queuing would livelock the
+ * drain, so drop the callback; an NMI cannot loop, so let it in.
+ */
+ if (READ_ONCE(sndp->draining) && !in_nmi()) {
+ WARN_ONCE(IS_ENABLED(CONFIG_PROVE_RCU),
+ "call_srcu() re-entered during callback drain; leaking callback\n");
+ return;
+ }
+ sdp = this_cpu_ptr(ssp->sda);
+ rhp->func = func;
+ if (!do_norm)
+ WRITE_ONCE(sdp->defer_exp, true);
+ if (llist_add((struct llist_node *)rhp, &sdp->defer_cbs)) {
+ /*
+ * Chain this srcu_data for the drain. ->ssp must be
+ * published here: deferral skips check_init_srcu_struct(),
+ * so on a never-initialized static srcu_struct the
+ * statically zeroed ->sda still has a NULL ->ssp.
+ */
+ sdp->ssp = ssp;
+ if (llist_add(&sdp->defer_link, &sndp->list))
+ irq_work_queue(&sndp->iw);
+ }
+ return;
+ }
+
+ /* An NMI reaching here entered with irqs enabled, so the enqueue can race. */
+ WARN_ON_ONCE(IS_ENABLED(CONFIG_PROVE_RCU) && in_nmi());
+
+ srcu_do_enqueue(ssp, rhp, func, do_norm);
+}
+
+/*
+ * Re-issue deferred callbacks straight to srcu_do_enqueue() so they cannot defer
+ * again. ->lock serializes the drainers: the irq_work, srcu_defer_flush() and
+ * srcu_offline_drain().
+ */
+static void __srcu_defer_drain(struct srcu_defer *sndp, bool guard)
+{
+ struct llist_node *snode, *snext;
+ unsigned long flags;
+
+ raw_spin_lock_irqsave(&sndp->lock, flags);
+ if (guard)
+ WRITE_ONCE(sndp->draining, true);
+ llist_for_each_safe(snode, snext, llist_del_all(&sndp->list)) {
+ struct srcu_data *sdp = container_of(snode, struct srcu_data, defer_link);
+ struct srcu_struct *ssp = sdp->ssp;
+ struct llist_node *cnode, *cnext;
+ bool do_norm;
+
+ cnode = llist_del_all(&sdp->defer_cbs);
+ do_norm = !READ_ONCE(sdp->defer_exp);
+ if (!do_norm)
+ WRITE_ONCE(sdp->defer_exp, false);
+ llist_for_each_safe(cnode, cnext, cnode) {
+ struct rcu_head *rhp = (struct rcu_head *)cnode;
+
+ rhp->next = NULL;
+ srcu_do_enqueue(ssp, rhp, rhp->func, do_norm);
+ }
+ }
+ if (guard)
+ WRITE_ONCE(sndp->draining, false);
+ raw_spin_unlock_irqrestore(&sndp->lock, flags);
+}
+
+/*
+ * Only the irq_work drain can be re-fed by its own re-issue, so only it sets
+ * ->draining. A direct drain re-issues onto this CPU, and anything staged
+ * during it is picked up by that CPU's own irq_work.
+ */
+static void srcu_defer_drain(struct irq_work *iw)
+{
+ __srcu_defer_drain(container_of(iw, struct srcu_defer, iw), true);
+}
+
+/*
+ * Drain @cpu's deferred call_srcu() callbacks from rcutree_migrate_callbacks()
+ * once @cpu is dead. One pass covers every srcu_struct, and the re-issue lands
+ * on the current CPU.
+ */
+void srcu_offline_drain(int cpu)
+{
+ if (!IS_ENABLED(CONFIG_RCU_DEFER))
+ return;
+ __srcu_defer_drain(&per_cpu(srcu_defer, cpu), false);
+}
+
/**
* call_srcu() - Queue a callback for invocation after an SRCU grace period
* @ssp: srcu_struct in queue the callback
@@ -1677,9 +1832,17 @@ void srcu_barrier(struct srcu_struct *ssp)
{
int cpu;
int idx;
- unsigned long s = rcu_seq_snap(&ssp->srcu_sup->srcu_barrier_seq);
+ unsigned long s;
check_init_srcu_struct(ssp);
+
+ /*
+ * Register any deferred callbacks before snapshotting the sequence. The
+ * shared irq_work may also drain other srcu_structs', which is harmless.
+ */
+ srcu_defer_flush();
+
+ s = rcu_seq_snap(&ssp->srcu_sup->srcu_barrier_seq);
mutex_lock(&ssp->srcu_sup->srcu_barrier_mutex);
if (rcu_seq_done(&ssp->srcu_sup->srcu_barrier_seq, s)) {
smp_mb(); /* Force ordering following return. */
diff --git a/kernel/rcu/tree.c b/kernel/rcu/tree.c
index 6a83408974c2b..227790a87b02f 100644
--- a/kernel/rcu/tree.c
+++ b/kernel/rcu/tree.c
@@ -4646,6 +4646,8 @@ void rcutree_migrate_callbacks(int cpu)
* returns; the re-issue lands on this CPU.
*/
__rcu_defer_drain(rdp, false);
+ /* Likewise for the outgoing CPU's deferred call_srcu() callbacks. */
+ srcu_offline_drain(cpu);
if (rcu_rdp_is_offloaded(rdp))
return;
--
2.53.0-Meta
^ permalink raw reply related [flat|nested] 12+ messages in thread* Re: [PATCH v3 3/6] srcu: Make call_srcu() safe to call from any context
2026-08-05 12:23 ` [PATCH v3 3/6] srcu: Make call_srcu() " Puranjay Mohan
@ 2026-08-05 12:35 ` sashiko-bot
2026-08-06 14:04 ` Zqiang
1 sibling, 0 replies; 12+ messages in thread
From: sashiko-bot @ 2026-08-05 12:35 UTC (permalink / raw)
To: Puranjay Mohan; +Cc: linux-rt-devel, bpf, linux-kernel
Thank you for your contribution! Sashiko AI review found 2 potential issue(s) to consider:
- [Critical] A TOCTOU race in `srcu_defer_flush()` locklessly checking `llist_empty()` can cause `srcu_barrier()` to return prematurely, leading to a Use-After-Free.
- [Medium] Expedited SRCU callbacks can be silently downgraded to normal callbacks due to a non-atomic update race on `sdp->defer_exp` between `__call_srcu()` and `__srcu_defer_drain()`.
--
--- Patch [3]: [PATCH v3 3/6] srcu: Make call_srcu() safe to call from any context ---
commit 137e7f3c08249a4bd260b0c0d6f275423685df28
Author: Puranjay Mohan <puranjay@kernel.org>
srcu: Make call_srcu() safe to call from any context
call_srcu() has the same constraint as call_rcu(): its callback list and
locks are only touched with interrupts disabled. Defer as call_rcu()
does: stage the callback on the srcu_data's ->defer_cbs, chain that
srcu_data onto a per-CPU list, and raise a per-CPU irq_work that
re-issues it straight to the enqueue helper, never back through
__call_srcu().
> diff --git a/kernel/rcu/srcutree.c b/kernel/rcu/srcutree.c
> index 304112674e8a2..35fface51d50b 100644
> --- a/kernel/rcu/srcutree.c
> +++ b/kernel/rcu/srcutree.c
[ ... ]
> @@ -79,6 +80,45 @@ static void process_srcu(struct work_struct *work);
> static void srcu_irq_work(struct irq_work *work);
> static void srcu_delay_timer(struct timer_list *t);
>
> +struct srcu_defer;
> +static void srcu_defer_drain(struct irq_work *iw);
> +static void __srcu_defer_drain(struct srcu_defer *sndp, bool guard);
> +
[ ... ]
> +/*
> + * Flush pending deferred callbacks so a following srcu_barrier() waits for them.
> + */
> +static void srcu_defer_flush(void)
> +{
> + int cpu;
> +
> + if (!IS_ENABLED(CONFIG_RCU_DEFER))
> + return;
> +
> + for_each_possible_cpu(cpu) {
> + struct srcu_defer *sndp = &per_cpu(srcu_defer, cpu);
> +
> + if (!llist_empty(&sndp->list))
> + __srcu_defer_drain(sndp, false);
[Severity: Critical]
Can this lockless list check lead to a use-after-free during cleanup?
If an asynchronous irq_work is concurrently executing
__srcu_defer_drain() and has already detached the list:
CPU1 (irq_work)
__srcu_defer_drain()
raw_spin_lock_irqsave(&sndp->lock, flags);
llist_del_all(&sndp->list); // list is now empty
CPU2 (cleanup_srcu_struct or srcu_barrier)
srcu_defer_flush()
llist_empty(&sndp->list) // returns true, returns immediately
Then CPU2 could return prematurely from srcu_defer_flush(), and
cleanup_srcu_struct() might free the srcu_struct while CPU1 is still
iterating through the detached callbacks and accessing sdp->ssp.
> + }
> +}
[ ... ]
> @@ -1423,6 +1473,111 @@ static void __call_srcu(struct srcu_struct *ssp, struct rcu_head *rhp,
[ ... ]
> +static void __call_srcu(struct srcu_struct *ssp, struct rcu_head *rhp,
> + rcu_callback_t func, bool do_norm)
> +{
> + if (should_rcu_defer()) {
> + struct srcu_defer *sndp = this_cpu_ptr(&srcu_defer);
> + struct srcu_data *sdp;
[ ... ]
> + sdp = this_cpu_ptr(ssp->sda);
> + rhp->func = func;
> + if (!do_norm)
> + WRITE_ONCE(sdp->defer_exp, true);
> + if (llist_add((struct llist_node *)rhp, &sdp->defer_cbs)) {
[Severity: Medium]
Could this flag be overwritten by a concurrent __srcu_defer_drain()?
If __call_srcu() sets sdp->defer_exp to true just as another CPU is
draining the list (e.g., from srcu_defer_flush()), can the drainer's
delayed write to sdp->defer_exp clear the flag we just set here?
CPU1
__srcu_defer_drain()
cnode = llist_del_all(&sdp->defer_cbs);
CPU2
__call_srcu()
if (!do_norm)
WRITE_ONCE(sdp->defer_exp, true);
llist_add(rhp, &sdp->defer_cbs);
CPU1
__srcu_defer_drain()
do_norm = !READ_ONCE(sdp->defer_exp);
if (!do_norm)
WRITE_ONCE(sdp->defer_exp, false); // Overwrites CPU2's write
This would cause CPU2's newly enqueued expedited callback to lose its
expedited status.
> + /*
> + * Chain this srcu_data for the drain. ->ssp must be
> + * published here: deferral skips check_init_srcu_struct(),
> + * so on a never-initialized static srcu_struct the
> + * statically zeroed ->sda still has a NULL ->ssp.
> + */
> + sdp->ssp = ssp;
> + if (llist_add(&sdp->defer_link, &sndp->list))
> + irq_work_queue(&sndp->iw);
> + }
> + return;
> + }
--
Sashiko AI review · https://sashiko.dev/#/patchset/20260805122346.269445-1-puranjay@kernel.org?part=3
^ permalink raw reply [flat|nested] 12+ messages in thread* Re: [PATCH v3 3/6] srcu: Make call_srcu() safe to call from any context
2026-08-05 12:23 ` [PATCH v3 3/6] srcu: Make call_srcu() " Puranjay Mohan
2026-08-05 12:35 ` sashiko-bot
@ 2026-08-06 14:04 ` Zqiang
2026-08-06 14:08 ` Puranjay Mohan
1 sibling, 1 reply; 12+ messages in thread
From: Zqiang @ 2026-08-06 14:04 UTC (permalink / raw)
To: Puranjay Mohan, Lai Jiangshan, Paul E. McKenney, Josh Triplett,
Onur Özkan, Frederic Weisbecker, Neeraj Upadhyay,
Joel Fernandes, Boqun Feng, Uladzislau Rezki, Davidlohr Bueso,
Andrii Nakryiko, Eduard Zingerman, Alexei Starovoitov,
Daniel Borkmann, Kumar Kartikeya Dwivedi
Cc: Puranjay Mohan, Steven Rostedt, Mathieu Desnoyers,
Martin KaFai Lau, Song Liu, Yonghong Song, Jiri Olsa,
Emil Tsalapatis, Matt Fleming, Harry Yoo (Oracle), linux-kernel,
rcu, bpf, linux-rt-devel
>
> call_srcu() has the same constraint as call_rcu(): its callback list and
> locks are only touched with interrupts disabled. srcu_gp_start_if_needed()
> enqueues under raw_spin_lock_irqsave() and may walk the srcu_node tree, as
> do callback invocation and grace-period work. A call_srcu() with
> interrupts disabled can race a list operation in flight on this CPU and
> corrupt the list or deadlock. call_rcu_tasks_trace() is call_srcu() under
> the hood, so a sleepable BPF program freeing an object can reach this.
>
> Defer as call_rcu() does: stage the callback on the srcu_data's
> ->defer_cbs, chain that srcu_data onto a per-CPU list, and raise a per-CPU
> irq_work that re-issues it straight to the enqueue helper, never back
> through __call_srcu(). The common path is unchanged and keeps interrupts
> enabled across srcu_gp_start_if_needed().
>
> The irq_work is per-CPU rather than per-srcu_struct and statically
> initialized, so deferral never runs check_init_srcu_struct(); it is
> IRQ_WORK_INIT_HARD as for call_rcu(). srcu_barrier() and
> cleanup_srcu_struct() flush it first, and rcutree_migrate_callbacks()
> calls srcu_offline_drain() for an outgoing CPU. ->lock is held across the
> drain so the drainers serialize.
>
> As in call_rcu(), the re-issue runs with interrupts disabled and can be
> re-entered by instrumentation, so a flag on the srcu_data being drained
> drops a deferring call_srcu() seen mid-drain (unless from an NMI). Staging
> records only the callback, so an expedited request is remembered per
> srcu_data in ->defer_exp and the whole batch is re-issued expedited rather
> than silently downgraded to a normal grace period. A dropped callback can
> also strand state its caller associated with it, not just the callback
> itself.
>
> Gated by CONFIG_RCU_DEFER. Under CONFIG_PROVE_RCU, warn if the direct
> path is reached from an NMI.
>
> Suggested-by: Paul E. McKenney <paulmck@kernel.org>
> Signed-off-by: Puranjay Mohan <puranjay@kernel.org>
> ---
> include/linux/srcutree.h | 5 ++
> kernel/rcu/rcu.h | 3 +
> kernel/rcu/srcutree.c | 171 ++++++++++++++++++++++++++++++++++++++-
> kernel/rcu/tree.c | 2 +
> 4 files changed, 177 insertions(+), 4 deletions(-)
>
> diff --git a/include/linux/srcutree.h b/include/linux/srcutree.h
> index 75e54e4f963fa..09a9c8f4a6d24 100644
> --- a/include/linux/srcutree.h
> +++ b/include/linux/srcutree.h
> @@ -13,6 +13,8 @@
>
> #include <linux/rcu_node_tree.h>
> #include <linux/completion.h>
> +#include <linux/irq_work_types.h>
> +#include <linux/llist.h>
>
> struct srcu_node;
> struct srcu_struct;
> @@ -41,6 +43,9 @@ struct srcu_data {
> bool srcu_cblist_invoking; /* Invoking these CBs? */
> struct timer_list delay_work; /* Delay for CB invoking */
> struct work_struct work; /* Context for CB invoking. */
> + struct llist_head defer_cbs; /* Callbacks deferred on re-entry. */
> + struct llist_node defer_link; /* Links onto the per-CPU deferral drain list */
> + bool defer_exp; /* A deferred callback asked to expedite. */
> struct rcu_head srcu_barrier_head; /* For srcu_barrier() use. */
> struct rcu_head srcu_ec_head; /* For srcu_expedite_current() use. */
> int srcu_ec_state; /* State for srcu_expedite_current(). */
> diff --git a/kernel/rcu/rcu.h b/kernel/rcu/rcu.h
> index fd075d91b80cf..84d74cd5a351c 100644
> --- a/kernel/rcu/rcu.h
> +++ b/kernel/rcu/rcu.h
> @@ -587,6 +587,9 @@ static inline bool should_rcu_defer(void)
> return irqs_disabled() && rcu_scheduler_active != RCU_SCHEDULER_INACTIVE;
> }
>
> +/* Drain an outgoing CPU's deferred SRCU callbacks; see rcutree_migrate_callbacks(). */
> +void srcu_offline_drain(int cpu);
> +
> enum rcutorture_type {
> RCU_FLAVOR,
> RCU_TASKS_FLAVOR,
> diff --git a/kernel/rcu/srcutree.c b/kernel/rcu/srcutree.c
> index 304112674e8a2..35fface51d50b 100644
> --- a/kernel/rcu/srcutree.c
> +++ b/kernel/rcu/srcutree.c
> @@ -20,6 +20,7 @@
> #include <linux/percpu.h>
> #include <linux/preempt.h>
> #include <linux/irq_work.h>
> +#include <linux/llist.h>
> #include <linux/rcupdate_wait.h>
> #include <linux/sched.h>
> #include <linux/smp.h>
> @@ -79,6 +80,45 @@ static void process_srcu(struct work_struct *work);
> static void srcu_irq_work(struct irq_work *work);
> static void srcu_delay_timer(struct timer_list *t);
>
> +struct srcu_defer;
> +static void srcu_defer_drain(struct irq_work *iw);
> +static void __srcu_defer_drain(struct srcu_defer *sndp, bool guard);
> +
> +/*
> + * Per-CPU call_srcu() deferral state, shared by every srcu_struct. A deferred
> + * callback is staged on its srcu_data's ->defer_cbs; that srcu_data is chained
> + * via ->defer_link onto ->list, which the irq_work walks.
> + */
> +struct srcu_defer {
> + struct llist_head list;
> + struct irq_work iw;
> + raw_spinlock_t lock;
> + bool draining;
> +};
> +
> +static DEFINE_PER_CPU(struct srcu_defer, srcu_defer) = {
> + .lock = __RAW_SPIN_LOCK_UNLOCKED(srcu_defer.lock),
> + .iw = IRQ_WORK_INIT_HARD(srcu_defer_drain),
> +};
> +
> +/*
> + * Flush pending deferred callbacks so a following srcu_barrier() waits for them.
> + */
> +static void srcu_defer_flush(void)
> +{
> + int cpu;
> +
> + if (!IS_ENABLED(CONFIG_RCU_DEFER))
> + return;
> +
> + for_each_possible_cpu(cpu) {
> + struct srcu_defer *sndp = &per_cpu(srcu_defer, cpu);
> +
> + if (!llist_empty(&sndp->list))
> + __srcu_defer_drain(sndp, false);
> + }
> +}
> +
> /*
> * Initialize SRCU per-CPU data. Note that statically allocated
> * srcu_struct structures might already have srcu_read_lock() and
> @@ -107,6 +147,11 @@ static void init_srcu_struct_data(struct srcu_struct *ssp)
> sdp->cpu = cpu;
> INIT_WORK(&sdp->work, srcu_invoke_callbacks);
> timer_setup(&sdp->delay_work, srcu_delay_timer, 0);
> + /*
> + * ->defer_cbs, ->defer_link and ->defer_exp are valid when zeroed
> + * and are not reinitialized here, lest we clobber callbacks a
> + * reentrant call_srcu() already staged. See __call_srcu().
> + */
> sdp->ssp = ssp;
> }
> }
> @@ -695,7 +740,12 @@ void cleanup_srcu_struct(struct srcu_struct *ssp)
> return; /* Just leak it! */
> if (WARN_ON(srcu_readers_active(ssp)))
> return; /* Just leak it! */
> - /* Wait for irq_work to finish first as it may queue a new work. */
> + /*
> + * Drain deferred callbacks before syncing ->irq_work: re-issuing one can
> + * start a grace period and re-queue ->irq_work, which then schedules
> + * ->work, so both must be waited out after the drain.
> + */
> + srcu_defer_flush();
> irq_work_sync(&sup->irq_work);
> flush_delayed_work(&sup->work);
> for_each_possible_cpu(cpu) {
> @@ -1410,8 +1460,8 @@ static unsigned long srcu_gp_start_if_needed(struct srcu_struct *ssp,
> * srcu_read_lock(), and srcu_read_unlock() that are all passed the same
> * srcu_struct structure.
> */
> -static void __call_srcu(struct srcu_struct *ssp, struct rcu_head *rhp,
> - rcu_callback_t func, bool do_norm)
> +static void srcu_do_enqueue(struct srcu_struct *ssp, struct rcu_head *rhp,
> + rcu_callback_t func, bool do_norm)
> {
> if (debug_rcu_head_queue(rhp)) {
> /* Probable double call_srcu(), so leak the callback. */
> @@ -1423,6 +1473,111 @@ static void __call_srcu(struct srcu_struct *ssp, struct rcu_head *rhp,
> (void)srcu_gp_start_if_needed(ssp, rhp, do_norm);
> }
>
> +/*
> + * The srcu_cblist and srcu_node tree are only accessed with interrupts disabled
> + * (srcu_gp_start_if_needed() enqueues under raw_spin_lock_irqsave() and may walk
> + * the tree). Like call_rcu(), __call_srcu() defers when interrupts are already
> + * disabled, so a re-entrant call_srcu() -- e.g. call_rcu_tasks_trace() from a
> + * BPF program -- cannot corrupt the list or deadlock.
> + */
> +static void __call_srcu(struct srcu_struct *ssp, struct rcu_head *rhp,
> + rcu_callback_t func, bool do_norm)
> +{
> + if (should_rcu_defer()) {
> + struct srcu_defer *sndp = this_cpu_ptr(&srcu_defer);
> + struct srcu_data *sdp;
> +
> + /*
> + * Instrumentation on the enqueue path can re-enter here from
> + * inside srcu_defer_drain(). Re-queuing would livelock the
> + * drain, so drop the callback; an NMI cannot loop, so let it in.
> + */
> + if (READ_ONCE(sndp->draining) && !in_nmi()) {
> + WARN_ONCE(IS_ENABLED(CONFIG_PROVE_RCU),
> + "call_srcu() re-entered during callback drain; leaking callback\n");
> + return;
> + }
> + sdp = this_cpu_ptr(ssp->sda);
> + rhp->func = func;
> + if (!do_norm)
> + WRITE_ONCE(sdp->defer_exp, true);
> + if (llist_add((struct llist_node *)rhp, &sdp->defer_cbs)) {
> + /*
> + * Chain this srcu_data for the drain. ->ssp must be
> + * published here: deferral skips check_init_srcu_struct(),
> + * so on a never-initialized static srcu_struct the
> + * statically zeroed ->sda still has a NULL ->ssp.
> + */
> + sdp->ssp = ssp;
> + if (llist_add(&sdp->defer_link, &sndp->list))
> + irq_work_queue(&sndp->iw);
> + }
> + return;
> + }
> +
> + /* An NMI reaching here entered with irqs enabled, so the enqueue can race. */
> + WARN_ON_ONCE(IS_ENABLED(CONFIG_PROVE_RCU) && in_nmi());
> +
> + srcu_do_enqueue(ssp, rhp, func, do_norm);
> +}
> +
> +/*
> + * Re-issue deferred callbacks straight to srcu_do_enqueue() so they cannot defer
> + * again. ->lock serializes the drainers: the irq_work, srcu_defer_flush() and
> + * srcu_offline_drain().
> + */
> +static void __srcu_defer_drain(struct srcu_defer *sndp, bool guard)
> +{
> + struct llist_node *snode, *snext;
> + unsigned long flags;
> +
> + raw_spin_lock_irqsave(&sndp->lock, flags);
> + if (guard)
> + WRITE_ONCE(sndp->draining, true);
> + llist_for_each_safe(snode, snext, llist_del_all(&sndp->list)) {
> + struct srcu_data *sdp = container_of(snode, struct srcu_data, defer_link);
> + struct srcu_struct *ssp = sdp->ssp;
> + struct llist_node *cnode, *cnext;
> + bool do_norm;
> +
> + cnode = llist_del_all(&sdp->defer_cbs);
> + do_norm = !READ_ONCE(sdp->defer_exp);
> + if (!do_norm)
> + WRITE_ONCE(sdp->defer_exp, false);
> + llist_for_each_safe(cnode, cnext, cnode) {
> + struct rcu_head *rhp = (struct rcu_head *)cnode;
> +
> + rhp->next = NULL;
> + srcu_do_enqueue(ssp, rhp, rhp->func, do_norm);
> + }
> + }
> + if (guard)
> + WRITE_ONCE(sndp->draining, false);
> + raw_spin_unlock_irqrestore(&sndp->lock, flags);
> +}
> +
> +/*
> + * Only the irq_work drain can be re-fed by its own re-issue, so only it sets
> + * ->draining. A direct drain re-issues onto this CPU, and anything staged
> + * during it is picked up by that CPU's own irq_work.
> + */
> +static void srcu_defer_drain(struct irq_work *iw)
> +{
> + __srcu_defer_drain(container_of(iw, struct srcu_defer, iw), true);
> +}
> +
> +/*
> + * Drain @cpu's deferred call_srcu() callbacks from rcutree_migrate_callbacks()
> + * once @cpu is dead. One pass covers every srcu_struct, and the re-issue lands
> + * on the current CPU.
> + */
> +void srcu_offline_drain(int cpu)
> +{
> + if (!IS_ENABLED(CONFIG_RCU_DEFER))
> + return;
> + __srcu_defer_drain(&per_cpu(srcu_defer, cpu), false);
> +}
> +
> /**
> * call_srcu() - Queue a callback for invocation after an SRCU grace period
> * @ssp: srcu_struct in queue the callback
> @@ -1677,9 +1832,17 @@ void srcu_barrier(struct srcu_struct *ssp)
> {
> int cpu;
> int idx;
> - unsigned long s = rcu_seq_snap(&ssp->srcu_sup->srcu_barrier_seq);
> + unsigned long s;
>
> check_init_srcu_struct(ssp);
> +
> + /*
> + * Register any deferred callbacks before snapshotting the sequence. The
> + * shared irq_work may also drain other srcu_structs', which is harmless.
> + */
> + srcu_defer_flush();
If there are two ssp1 and ssp2 concurrent call srcu_barrier(),
and assuming there are only two CPUs, CPU0->sndp0 and CPU1->sndp1.
srcu_barrier(&ssp1)
->srcu_defer_flush()
->llist_empty(&sndp0->list) is not empty
->__srcu_defer_drain(sndp0)
->raw_spinlock()
->llist_for_each_safe(llist_del_all(&sndp0->list))
srcu_barrier(&ssp2)
->srcu_defer_flush()
->llist_empty(&sndp0->list) is empty
...
->srcu_barrier_one_cpu()
->sdp0 = per_cpu_ptr(ssp2->sda, 0)
->rcu_segcblist_entrain(&sdp0->srcu_cblist, ...)
//the sdp0->srcu_cblist is empty, return false.
return;
->srcu_do_enqueue(ssp2, rhp, rhp->func, do_norm);
// the srcu_barrier(&ssp2) has already return,
// miss waiting to current queue ssp2's callback
// to complete.
Thanks
Zqiang
> +
> + s = rcu_seq_snap(&ssp->srcu_sup->srcu_barrier_seq);
> mutex_lock(&ssp->srcu_sup->srcu_barrier_mutex);
> if (rcu_seq_done(&ssp->srcu_sup->srcu_barrier_seq, s)) {
> smp_mb(); /* Force ordering following return. */
> diff --git a/kernel/rcu/tree.c b/kernel/rcu/tree.c
> index 6a83408974c2b..227790a87b02f 100644
> --- a/kernel/rcu/tree.c
> +++ b/kernel/rcu/tree.c
> @@ -4646,6 +4646,8 @@ void rcutree_migrate_callbacks(int cpu)
> * returns; the re-issue lands on this CPU.
> */
> __rcu_defer_drain(rdp, false);
> + /* Likewise for the outgoing CPU's deferred call_srcu() callbacks. */
> + srcu_offline_drain(cpu);
>
> if (rcu_rdp_is_offloaded(rdp))
> return;
> --
> 2.53.0-Meta
>
^ permalink raw reply [flat|nested] 12+ messages in thread* Re: [PATCH v3 3/6] srcu: Make call_srcu() safe to call from any context
2026-08-06 14:04 ` Zqiang
@ 2026-08-06 14:08 ` Puranjay Mohan
0 siblings, 0 replies; 12+ messages in thread
From: Puranjay Mohan @ 2026-08-06 14:08 UTC (permalink / raw)
To: Zqiang
Cc: Lai Jiangshan, Paul E. McKenney, Josh Triplett, Onur Özkan,
Frederic Weisbecker, Neeraj Upadhyay, Joel Fernandes, Boqun Feng,
Uladzislau Rezki, Davidlohr Bueso, Andrii Nakryiko,
Eduard Zingerman, Alexei Starovoitov, Daniel Borkmann,
Kumar Kartikeya Dwivedi, Steven Rostedt, Mathieu Desnoyers,
Martin KaFai Lau, Song Liu, Yonghong Song, Jiri Olsa,
Emil Tsalapatis, Matt Fleming, Harry Yoo (Oracle), linux-kernel,
rcu, bpf, linux-rt-devel
On Thu, Aug 6, 2026 at 3:04 PM Zqiang <qiang.zhang@linux.dev> wrote:
>
> >
> > call_srcu() has the same constraint as call_rcu(): its callback list and
> > locks are only touched with interrupts disabled. srcu_gp_start_if_needed()
> > enqueues under raw_spin_lock_irqsave() and may walk the srcu_node tree, as
> > do callback invocation and grace-period work. A call_srcu() with
> > interrupts disabled can race a list operation in flight on this CPU and
> > corrupt the list or deadlock. call_rcu_tasks_trace() is call_srcu() under
> > the hood, so a sleepable BPF program freeing an object can reach this.
> >
> > Defer as call_rcu() does: stage the callback on the srcu_data's
> > ->defer_cbs, chain that srcu_data onto a per-CPU list, and raise a per-CPU
> > irq_work that re-issues it straight to the enqueue helper, never back
> > through __call_srcu(). The common path is unchanged and keeps interrupts
> > enabled across srcu_gp_start_if_needed().
> >
> > The irq_work is per-CPU rather than per-srcu_struct and statically
> > initialized, so deferral never runs check_init_srcu_struct(); it is
> > IRQ_WORK_INIT_HARD as for call_rcu(). srcu_barrier() and
> > cleanup_srcu_struct() flush it first, and rcutree_migrate_callbacks()
> > calls srcu_offline_drain() for an outgoing CPU. ->lock is held across the
> > drain so the drainers serialize.
> >
> > As in call_rcu(), the re-issue runs with interrupts disabled and can be
> > re-entered by instrumentation, so a flag on the srcu_data being drained
> > drops a deferring call_srcu() seen mid-drain (unless from an NMI). Staging
> > records only the callback, so an expedited request is remembered per
> > srcu_data in ->defer_exp and the whole batch is re-issued expedited rather
> > than silently downgraded to a normal grace period. A dropped callback can
> > also strand state its caller associated with it, not just the callback
> > itself.
> >
> > Gated by CONFIG_RCU_DEFER. Under CONFIG_PROVE_RCU, warn if the direct
> > path is reached from an NMI.
> >
> > Suggested-by: Paul E. McKenney <paulmck@kernel.org>
> > Signed-off-by: Puranjay Mohan <puranjay@kernel.org>
> > ---
> > include/linux/srcutree.h | 5 ++
> > kernel/rcu/rcu.h | 3 +
> > kernel/rcu/srcutree.c | 171 ++++++++++++++++++++++++++++++++++++++-
> > kernel/rcu/tree.c | 2 +
> > 4 files changed, 177 insertions(+), 4 deletions(-)
> >
> > diff --git a/include/linux/srcutree.h b/include/linux/srcutree.h
> > index 75e54e4f963fa..09a9c8f4a6d24 100644
> > --- a/include/linux/srcutree.h
> > +++ b/include/linux/srcutree.h
> > @@ -13,6 +13,8 @@
> >
> > #include <linux/rcu_node_tree.h>
> > #include <linux/completion.h>
> > +#include <linux/irq_work_types.h>
> > +#include <linux/llist.h>
> >
> > struct srcu_node;
> > struct srcu_struct;
> > @@ -41,6 +43,9 @@ struct srcu_data {
> > bool srcu_cblist_invoking; /* Invoking these CBs? */
> > struct timer_list delay_work; /* Delay for CB invoking */
> > struct work_struct work; /* Context for CB invoking. */
> > + struct llist_head defer_cbs; /* Callbacks deferred on re-entry. */
> > + struct llist_node defer_link; /* Links onto the per-CPU deferral drain list */
> > + bool defer_exp; /* A deferred callback asked to expedite. */
> > struct rcu_head srcu_barrier_head; /* For srcu_barrier() use. */
> > struct rcu_head srcu_ec_head; /* For srcu_expedite_current() use. */
> > int srcu_ec_state; /* State for srcu_expedite_current(). */
> > diff --git a/kernel/rcu/rcu.h b/kernel/rcu/rcu.h
> > index fd075d91b80cf..84d74cd5a351c 100644
> > --- a/kernel/rcu/rcu.h
> > +++ b/kernel/rcu/rcu.h
> > @@ -587,6 +587,9 @@ static inline bool should_rcu_defer(void)
> > return irqs_disabled() && rcu_scheduler_active != RCU_SCHEDULER_INACTIVE;
> > }
> >
> > +/* Drain an outgoing CPU's deferred SRCU callbacks; see rcutree_migrate_callbacks(). */
> > +void srcu_offline_drain(int cpu);
> > +
> > enum rcutorture_type {
> > RCU_FLAVOR,
> > RCU_TASKS_FLAVOR,
> > diff --git a/kernel/rcu/srcutree.c b/kernel/rcu/srcutree.c
> > index 304112674e8a2..35fface51d50b 100644
> > --- a/kernel/rcu/srcutree.c
> > +++ b/kernel/rcu/srcutree.c
> > @@ -20,6 +20,7 @@
> > #include <linux/percpu.h>
> > #include <linux/preempt.h>
> > #include <linux/irq_work.h>
> > +#include <linux/llist.h>
> > #include <linux/rcupdate_wait.h>
> > #include <linux/sched.h>
> > #include <linux/smp.h>
> > @@ -79,6 +80,45 @@ static void process_srcu(struct work_struct *work);
> > static void srcu_irq_work(struct irq_work *work);
> > static void srcu_delay_timer(struct timer_list *t);
> >
> > +struct srcu_defer;
> > +static void srcu_defer_drain(struct irq_work *iw);
> > +static void __srcu_defer_drain(struct srcu_defer *sndp, bool guard);
> > +
> > +/*
> > + * Per-CPU call_srcu() deferral state, shared by every srcu_struct. A deferred
> > + * callback is staged on its srcu_data's ->defer_cbs; that srcu_data is chained
> > + * via ->defer_link onto ->list, which the irq_work walks.
> > + */
> > +struct srcu_defer {
> > + struct llist_head list;
> > + struct irq_work iw;
> > + raw_spinlock_t lock;
> > + bool draining;
> > +};
> > +
> > +static DEFINE_PER_CPU(struct srcu_defer, srcu_defer) = {
> > + .lock = __RAW_SPIN_LOCK_UNLOCKED(srcu_defer.lock),
> > + .iw = IRQ_WORK_INIT_HARD(srcu_defer_drain),
> > +};
> > +
> > +/*
> > + * Flush pending deferred callbacks so a following srcu_barrier() waits for them.
> > + */
> > +static void srcu_defer_flush(void)
> > +{
> > + int cpu;
> > +
> > + if (!IS_ENABLED(CONFIG_RCU_DEFER))
> > + return;
> > +
> > + for_each_possible_cpu(cpu) {
> > + struct srcu_defer *sndp = &per_cpu(srcu_defer, cpu);
> > +
> > + if (!llist_empty(&sndp->list))
> > + __srcu_defer_drain(sndp, false);
> > + }
> > +}
> > +
> > /*
> > * Initialize SRCU per-CPU data. Note that statically allocated
> > * srcu_struct structures might already have srcu_read_lock() and
> > @@ -107,6 +147,11 @@ static void init_srcu_struct_data(struct srcu_struct *ssp)
> > sdp->cpu = cpu;
> > INIT_WORK(&sdp->work, srcu_invoke_callbacks);
> > timer_setup(&sdp->delay_work, srcu_delay_timer, 0);
> > + /*
> > + * ->defer_cbs, ->defer_link and ->defer_exp are valid when zeroed
> > + * and are not reinitialized here, lest we clobber callbacks a
> > + * reentrant call_srcu() already staged. See __call_srcu().
> > + */
> > sdp->ssp = ssp;
> > }
> > }
> > @@ -695,7 +740,12 @@ void cleanup_srcu_struct(struct srcu_struct *ssp)
> > return; /* Just leak it! */
> > if (WARN_ON(srcu_readers_active(ssp)))
> > return; /* Just leak it! */
> > - /* Wait for irq_work to finish first as it may queue a new work. */
> > + /*
> > + * Drain deferred callbacks before syncing ->irq_work: re-issuing one can
> > + * start a grace period and re-queue ->irq_work, which then schedules
> > + * ->work, so both must be waited out after the drain.
> > + */
> > + srcu_defer_flush();
> > irq_work_sync(&sup->irq_work);
> > flush_delayed_work(&sup->work);
> > for_each_possible_cpu(cpu) {
> > @@ -1410,8 +1460,8 @@ static unsigned long srcu_gp_start_if_needed(struct srcu_struct *ssp,
> > * srcu_read_lock(), and srcu_read_unlock() that are all passed the same
> > * srcu_struct structure.
> > */
> > -static void __call_srcu(struct srcu_struct *ssp, struct rcu_head *rhp,
> > - rcu_callback_t func, bool do_norm)
> > +static void srcu_do_enqueue(struct srcu_struct *ssp, struct rcu_head *rhp,
> > + rcu_callback_t func, bool do_norm)
> > {
> > if (debug_rcu_head_queue(rhp)) {
> > /* Probable double call_srcu(), so leak the callback. */
> > @@ -1423,6 +1473,111 @@ static void __call_srcu(struct srcu_struct *ssp, struct rcu_head *rhp,
> > (void)srcu_gp_start_if_needed(ssp, rhp, do_norm);
> > }
> >
> > +/*
> > + * The srcu_cblist and srcu_node tree are only accessed with interrupts disabled
> > + * (srcu_gp_start_if_needed() enqueues under raw_spin_lock_irqsave() and may walk
> > + * the tree). Like call_rcu(), __call_srcu() defers when interrupts are already
> > + * disabled, so a re-entrant call_srcu() -- e.g. call_rcu_tasks_trace() from a
> > + * BPF program -- cannot corrupt the list or deadlock.
> > + */
> > +static void __call_srcu(struct srcu_struct *ssp, struct rcu_head *rhp,
> > + rcu_callback_t func, bool do_norm)
> > +{
> > + if (should_rcu_defer()) {
> > + struct srcu_defer *sndp = this_cpu_ptr(&srcu_defer);
> > + struct srcu_data *sdp;
> > +
> > + /*
> > + * Instrumentation on the enqueue path can re-enter here from
> > + * inside srcu_defer_drain(). Re-queuing would livelock the
> > + * drain, so drop the callback; an NMI cannot loop, so let it in.
> > + */
> > + if (READ_ONCE(sndp->draining) && !in_nmi()) {
> > + WARN_ONCE(IS_ENABLED(CONFIG_PROVE_RCU),
> > + "call_srcu() re-entered during callback drain; leaking callback\n");
> > + return;
> > + }
> > + sdp = this_cpu_ptr(ssp->sda);
> > + rhp->func = func;
> > + if (!do_norm)
> > + WRITE_ONCE(sdp->defer_exp, true);
> > + if (llist_add((struct llist_node *)rhp, &sdp->defer_cbs)) {
> > + /*
> > + * Chain this srcu_data for the drain. ->ssp must be
> > + * published here: deferral skips check_init_srcu_struct(),
> > + * so on a never-initialized static srcu_struct the
> > + * statically zeroed ->sda still has a NULL ->ssp.
> > + */
> > + sdp->ssp = ssp;
> > + if (llist_add(&sdp->defer_link, &sndp->list))
> > + irq_work_queue(&sndp->iw);
> > + }
> > + return;
> > + }
> > +
> > + /* An NMI reaching here entered with irqs enabled, so the enqueue can race. */
> > + WARN_ON_ONCE(IS_ENABLED(CONFIG_PROVE_RCU) && in_nmi());
> > +
> > + srcu_do_enqueue(ssp, rhp, func, do_norm);
> > +}
> > +
> > +/*
> > + * Re-issue deferred callbacks straight to srcu_do_enqueue() so they cannot defer
> > + * again. ->lock serializes the drainers: the irq_work, srcu_defer_flush() and
> > + * srcu_offline_drain().
> > + */
> > +static void __srcu_defer_drain(struct srcu_defer *sndp, bool guard)
> > +{
> > + struct llist_node *snode, *snext;
> > + unsigned long flags;
> > +
> > + raw_spin_lock_irqsave(&sndp->lock, flags);
> > + if (guard)
> > + WRITE_ONCE(sndp->draining, true);
> > + llist_for_each_safe(snode, snext, llist_del_all(&sndp->list)) {
> > + struct srcu_data *sdp = container_of(snode, struct srcu_data, defer_link);
> > + struct srcu_struct *ssp = sdp->ssp;
> > + struct llist_node *cnode, *cnext;
> > + bool do_norm;
> > +
> > + cnode = llist_del_all(&sdp->defer_cbs);
> > + do_norm = !READ_ONCE(sdp->defer_exp);
> > + if (!do_norm)
> > + WRITE_ONCE(sdp->defer_exp, false);
> > + llist_for_each_safe(cnode, cnext, cnode) {
> > + struct rcu_head *rhp = (struct rcu_head *)cnode;
> > +
> > + rhp->next = NULL;
> > + srcu_do_enqueue(ssp, rhp, rhp->func, do_norm);
> > + }
> > + }
> > + if (guard)
> > + WRITE_ONCE(sndp->draining, false);
> > + raw_spin_unlock_irqrestore(&sndp->lock, flags);
> > +}
> > +
> > +/*
> > + * Only the irq_work drain can be re-fed by its own re-issue, so only it sets
> > + * ->draining. A direct drain re-issues onto this CPU, and anything staged
> > + * during it is picked up by that CPU's own irq_work.
> > + */
> > +static void srcu_defer_drain(struct irq_work *iw)
> > +{
> > + __srcu_defer_drain(container_of(iw, struct srcu_defer, iw), true);
> > +}
> > +
> > +/*
> > + * Drain @cpu's deferred call_srcu() callbacks from rcutree_migrate_callbacks()
> > + * once @cpu is dead. One pass covers every srcu_struct, and the re-issue lands
> > + * on the current CPU.
> > + */
> > +void srcu_offline_drain(int cpu)
> > +{
> > + if (!IS_ENABLED(CONFIG_RCU_DEFER))
> > + return;
> > + __srcu_defer_drain(&per_cpu(srcu_defer, cpu), false);
> > +}
> > +
> > /**
> > * call_srcu() - Queue a callback for invocation after an SRCU grace period
> > * @ssp: srcu_struct in queue the callback
> > @@ -1677,9 +1832,17 @@ void srcu_barrier(struct srcu_struct *ssp)
> > {
> > int cpu;
> > int idx;
> > - unsigned long s = rcu_seq_snap(&ssp->srcu_sup->srcu_barrier_seq);
> > + unsigned long s;
> >
> > check_init_srcu_struct(ssp);
> > +
> > + /*
> > + * Register any deferred callbacks before snapshotting the sequence. The
> > + * shared irq_work may also drain other srcu_structs', which is harmless.
> > + */
> > + srcu_defer_flush();
>
> If there are two ssp1 and ssp2 concurrent call srcu_barrier(),
> and assuming there are only two CPUs, CPU0->sndp0 and CPU1->sndp1.
>
>
> srcu_barrier(&ssp1)
> ->srcu_defer_flush()
> ->llist_empty(&sndp0->list) is not empty
> ->__srcu_defer_drain(sndp0)
> ->raw_spinlock()
> ->llist_for_each_safe(llist_del_all(&sndp0->list))
>
> srcu_barrier(&ssp2)
> ->srcu_defer_flush()
> ->llist_empty(&sndp0->list) is empty
> ...
> ->srcu_barrier_one_cpu()
> ->sdp0 = per_cpu_ptr(ssp2->sda, 0)
> ->rcu_segcblist_entrain(&sdp0->srcu_cblist, ...)
> //the sdp0->srcu_cblist is empty, return false.
> return;
>
> ->srcu_do_enqueue(ssp2, rhp, rhp->func, do_norm);
> // the srcu_barrier(&ssp2) has already return,
> // miss waiting to current queue ssp2's callback
> // to complete.
>
>
>
>
> Thanks
> Zqiang
Hi Zqiang,
Thanks for the feedback. I already realized this and fixed it for the
next version which I will post soon.
Thanks,
Puranjay
^ permalink raw reply [flat|nested] 12+ messages in thread
* [PATCH v3 4/6] srcu: Make Tiny call_srcu() safe to call from any context
2026-08-05 12:23 [PATCH v3 0/6] rcu,srcu: Make call_rcu()/call_srcu() safe from any context Puranjay Mohan
` (2 preceding siblings ...)
2026-08-05 12:23 ` [PATCH v3 3/6] srcu: Make call_srcu() " Puranjay Mohan
@ 2026-08-05 12:23 ` Puranjay Mohan
2026-08-05 12:23 ` [PATCH v3 5/6] rcutorture: Exercise ->call() from NMI context Puranjay Mohan
2026-08-05 12:23 ` [PATCH v3 6/6] selftests/bpf: Add a call_srcu() re-entry reproducer Puranjay Mohan
5 siblings, 0 replies; 12+ messages in thread
From: Puranjay Mohan @ 2026-08-05 12:23 UTC (permalink / raw)
To: Lai Jiangshan, Paul E. McKenney, Josh Triplett, Onur Özkan,
Frederic Weisbecker, Neeraj Upadhyay, Joel Fernandes, Boqun Feng,
Uladzislau Rezki, Davidlohr Bueso, Andrii Nakryiko,
Eduard Zingerman, Alexei Starovoitov, Daniel Borkmann,
Kumar Kartikeya Dwivedi
Cc: Puranjay Mohan, Steven Rostedt, Mathieu Desnoyers, Zqiang,
Martin KaFai Lau, Song Liu, Yonghong Song, Jiri Olsa,
Emil Tsalapatis, Matt Fleming, Harry Yoo (Oracle), linux-kernel,
rcu, bpf, linux-rt-devel
Give Tiny call_srcu() the same treatment as Tree SRCU. When interrupts
are disabled and the scheduler is up, stage the callback on the
srcu_struct's lockless list for an irq_work to re-issue later. Tiny SRCU
is uniprocessor, so there is no CPU-offline drain. A draining flag drops
a deferring call_srcu() that re-enters mid-drain (unless from an NMI), as
in Tree SRCU.
srcu_barrier() (now out of line) and cleanup_srcu_struct() drain the
deferred list first, so a deferred callback is re-issued onto the callback
list and invoked by the grace-period work that cleanup_srcu_struct()
flushes, rather than stranded on a soon-to-be-freed srcu_struct. cleanup
also syncs ->defer_iw, since that irq_work is embedded in the srcu_struct
the caller is about to free.
Gated by CONFIG_RCU_DEFER, like Tree SRCU.
Suggested-by: Paul E. McKenney <paulmck@kernel.org>
Signed-off-by: Puranjay Mohan <puranjay@kernel.org>
---
include/linux/srcutiny.h | 12 ++++--
kernel/rcu/srcutiny.c | 79 ++++++++++++++++++++++++++++++++++++++--
2 files changed, 83 insertions(+), 8 deletions(-)
diff --git a/include/linux/srcutiny.h b/include/linux/srcutiny.h
index fbcf13bc12d15..a0e54d9182baa 100644
--- a/include/linux/srcutiny.h
+++ b/include/linux/srcutiny.h
@@ -12,6 +12,7 @@
#define _LINUX_SRCU_TINY_H
#include <linux/irq_work_types.h>
+#include <linux/llist.h>
#include <linux/swait.h>
struct srcu_struct {
@@ -26,6 +27,8 @@ struct srcu_struct {
struct rcu_head **srcu_cb_tail; /* Pending callbacks: Tail. */
struct work_struct srcu_work; /* For driving grace periods. */
struct irq_work srcu_irq_work; /* Defer schedule_work() to irq work. */
+ struct llist_head defer_cbs; /* Callbacks deferred on re-entry. */
+ struct irq_work defer_iw; /* Registers defer_cbs later. */
#ifdef CONFIG_DEBUG_LOCK_ALLOC
struct lockdep_map dep_map;
#endif /* #ifdef CONFIG_DEBUG_LOCK_ALLOC */
@@ -33,6 +36,7 @@ struct srcu_struct {
void srcu_drive_gp(struct work_struct *wp);
void srcu_tiny_irq_work(struct irq_work *irq_work);
+void srcu_defer_drain(struct irq_work *irq_work);
#define __SRCU_STRUCT_INIT(name, __ignored, ___ignored, ____ignored) \
{ \
@@ -40,6 +44,9 @@ void srcu_tiny_irq_work(struct irq_work *irq_work);
.srcu_cb_tail = &name.srcu_cb_head, \
.srcu_work = __WORK_INITIALIZER(name.srcu_work, srcu_drive_gp), \
.srcu_irq_work = { .func = srcu_tiny_irq_work }, \
+ .defer_cbs = LLIST_HEAD_INIT(name.defer_cbs), \
+ .defer_iw = { .node = { .u_flags = IRQ_WORK_HARD_IRQ }, \
+ .func = srcu_defer_drain }, \
__SRCU_DEP_MAP_INIT(name) \
}
@@ -131,10 +138,7 @@ static inline void synchronize_srcu_expedited(struct srcu_struct *ssp)
synchronize_srcu(ssp);
}
-static inline void srcu_barrier(struct srcu_struct *ssp)
-{
- synchronize_srcu(ssp);
-}
+void srcu_barrier(struct srcu_struct *ssp);
static inline void srcu_expedite_current(struct srcu_struct *ssp) { }
#define srcu_check_read_flavor(ssp, read_flavor) do { } while (0)
diff --git a/kernel/rcu/srcutiny.c b/kernel/rcu/srcutiny.c
index f9c498ae75df2..ba36067909fd0 100644
--- a/kernel/rcu/srcutiny.c
+++ b/kernel/rcu/srcutiny.c
@@ -10,6 +10,7 @@
#include <linux/export.h>
#include <linux/irq_work.h>
+#include <linux/llist.h>
#include <linux/mutex.h>
#include <linux/preempt.h>
#include <linux/rcupdate_wait.h>
@@ -29,6 +30,8 @@ extern int rcu_scheduler_active;
static LIST_HEAD(srcu_boot_list);
static bool srcu_init_done;
+static void __srcu_defer_drain(struct srcu_struct *ssp, bool guard);
+
static int init_srcu_struct_fields(struct srcu_struct *ssp)
{
ssp->srcu_lock_nesting[0] = 0;
@@ -43,6 +46,8 @@ static int init_srcu_struct_fields(struct srcu_struct *ssp)
INIT_WORK(&ssp->srcu_work, srcu_drive_gp);
INIT_LIST_HEAD(&ssp->srcu_work.entry);
init_irq_work(&ssp->srcu_irq_work, srcu_tiny_irq_work);
+ init_llist_head(&ssp->defer_cbs);
+ ssp->defer_iw = IRQ_WORK_INIT_HARD(srcu_defer_drain);
return 0;
}
@@ -86,6 +91,11 @@ EXPORT_SYMBOL_GPL(init_srcu_struct_generic);
void cleanup_srcu_struct(struct srcu_struct *ssp)
{
WARN_ON(srcu_readers_active(ssp));
+ /* Re-issue any deferred callbacks, then wait out ->defer_iw before it is freed. */
+ if (IS_ENABLED(CONFIG_RCU_DEFER)) {
+ __srcu_defer_drain(ssp, false);
+ irq_work_sync(&ssp->defer_iw);
+ }
irq_work_sync(&ssp->srcu_irq_work);
flush_work(&ssp->srcu_work);
WARN_ON(ssp->srcu_gp_running);
@@ -215,11 +225,11 @@ static void srcu_gp_start_if_needed(struct srcu_struct *ssp)
}
/*
- * Enqueue an SRCU callback on the specified srcu_struct structure,
- * initiating grace-period processing if it is not already running.
+ * Enqueue @rhp on the callback list. Also called by srcu_defer_drain() to
+ * re-issue a deferred callback, so it must not re-check the deferral condition.
*/
-void call_srcu(struct srcu_struct *ssp, struct rcu_head *rhp,
- rcu_callback_t func)
+static void srcu_do_enqueue(struct srcu_struct *ssp, struct rcu_head *rhp,
+ rcu_callback_t func)
{
unsigned long flags;
@@ -233,6 +243,58 @@ void call_srcu(struct srcu_struct *ssp, struct rcu_head *rhp,
srcu_gp_start_if_needed(ssp);
preempt_enable();
}
+
+/* Set while srcu_defer_drain() re-issues, to catch a re-entrant call_srcu(). */
+static bool srcu_defer_draining;
+
+static void __srcu_defer_drain(struct srcu_struct *ssp, bool guard)
+{
+ struct llist_node *node, *next;
+ unsigned long flags;
+
+ /* Callbacks are unordered, so drain in llist order without reversing. */
+ local_irq_save(flags);
+ if (guard)
+ WRITE_ONCE(srcu_defer_draining, true);
+ llist_for_each_safe(node, next, llist_del_all(&ssp->defer_cbs)) {
+ struct rcu_head *rhp = (struct rcu_head *)node;
+
+ rhp->next = NULL;
+ srcu_do_enqueue(ssp, rhp, rhp->func);
+ }
+ if (guard)
+ WRITE_ONCE(srcu_defer_draining, false);
+ local_irq_restore(flags);
+}
+
+/* Only the irq_work drain can be re-fed by its own re-issue; see Tree SRCU. */
+void srcu_defer_drain(struct irq_work *iw)
+{
+ __srcu_defer_drain(container_of(iw, struct srcu_struct, defer_iw), true);
+}
+EXPORT_SYMBOL_GPL(srcu_defer_drain);
+
+void call_srcu(struct srcu_struct *ssp, struct rcu_head *rhp,
+ rcu_callback_t func)
+{
+ if (should_rcu_defer()) {
+ /* A re-entrant call_srcu() during the drain would livelock it. */
+ if (READ_ONCE(srcu_defer_draining) && !in_nmi()) {
+ WARN_ONCE(IS_ENABLED(CONFIG_PROVE_RCU),
+ "call_srcu() re-entered during callback drain; leaking callback\n");
+ return;
+ }
+ rhp->func = func;
+ if (llist_add((struct llist_node *)rhp, &ssp->defer_cbs))
+ irq_work_queue(&ssp->defer_iw);
+ return;
+ }
+
+ /* An NMI reaching here entered with irqs enabled, so the enqueue can race. */
+ WARN_ON_ONCE(IS_ENABLED(CONFIG_PROVE_RCU) && in_nmi());
+
+ srcu_do_enqueue(ssp, rhp, func);
+}
EXPORT_SYMBOL_GPL(call_srcu);
/*
@@ -262,6 +324,15 @@ void synchronize_srcu(struct srcu_struct *ssp)
}
EXPORT_SYMBOL_GPL(synchronize_srcu);
+/* Register any deferred callbacks, then wait for all in-flight ones. */
+void srcu_barrier(struct srcu_struct *ssp)
+{
+ if (IS_ENABLED(CONFIG_RCU_DEFER))
+ __srcu_defer_drain(ssp, false);
+ synchronize_srcu(ssp);
+}
+EXPORT_SYMBOL_GPL(srcu_barrier);
+
/*
* get_state_synchronize_srcu - Provide an end-of-grace-period cookie
*/
--
2.53.0-Meta
^ permalink raw reply related [flat|nested] 12+ messages in thread* [PATCH v3 5/6] rcutorture: Exercise ->call() from NMI context
2026-08-05 12:23 [PATCH v3 0/6] rcu,srcu: Make call_rcu()/call_srcu() safe from any context Puranjay Mohan
` (3 preceding siblings ...)
2026-08-05 12:23 ` [PATCH v3 4/6] srcu: Make Tiny " Puranjay Mohan
@ 2026-08-05 12:23 ` Puranjay Mohan
2026-08-05 12:23 ` [PATCH v3 6/6] selftests/bpf: Add a call_srcu() re-entry reproducer Puranjay Mohan
5 siblings, 0 replies; 12+ messages in thread
From: Puranjay Mohan @ 2026-08-05 12:23 UTC (permalink / raw)
To: Lai Jiangshan, Paul E. McKenney, Josh Triplett, Onur Özkan,
Frederic Weisbecker, Neeraj Upadhyay, Joel Fernandes, Boqun Feng,
Uladzislau Rezki, Davidlohr Bueso, Andrii Nakryiko,
Eduard Zingerman, Alexei Starovoitov, Daniel Borkmann,
Kumar Kartikeya Dwivedi
Cc: Puranjay Mohan, Steven Rostedt, Mathieu Desnoyers, Zqiang,
Martin KaFai Lau, Song Liu, Yonghong Song, Jiri Olsa,
Emil Tsalapatis, Matt Fleming, Harry Yoo (Oracle), linux-kernel,
rcu, bpf, linux-rt-devel
call_rcu() and call_srcu() are now safe to invoke from NMI, but
rcutorture never does, leaving the deferral path untested.
Add an ->nmi_capable flag to rcu_torture_ops. For flavors that set it,
arm a per-CPU hardware perf counter whose overflow handler submits a
callback via ->call(). The handler acts only when in_nmi(), so only a
genuine NMI exercises the deferral path. One preallocated callback is
kept in flight (guarded by an atomic) to avoid allocating in NMI.
Report the count issued from NMI ("nmi-calls:") and the count invoked
("nmi-cbs:"). rcu_torture_cleanup() disables the counters and then calls
cb_barrier(), which drains every deferred callback, so the two counts must
then match; a mismatch means a lost callback and fails the test. This
relies on srcu_barrier()/rcu_barrier() flushing deferred callbacks, as
added earlier in the series.
Set ->nmi_capable on the NMI-safe flavors: rcu, srcu, srcud, and
tasks-tracing (call_srcu() under the hood). Tasks and Tasks Rude are left
alone, as call_rcu_tasks_generic() is not yet NMI-safe.
Enabled by default; the nmi_calls parameter disables it, which helps rule
NMI handling in or out when triaging a failure. Requires
CONFIG_PERF_EVENTS and a hardware PMU, else silently skipped.
Signed-off-by: Puranjay Mohan <puranjay@kernel.org>
---
kernel/rcu/rcutorture.c | 154 +++++++++++++++++++++++++++++++++++++++-
1 file changed, 152 insertions(+), 2 deletions(-)
diff --git a/kernel/rcu/rcutorture.c b/kernel/rcu/rcutorture.c
index 39426a8718fe9..715c4c6c51b90 100644
--- a/kernel/rcu/rcutorture.c
+++ b/kernel/rcu/rcutorture.c
@@ -48,6 +48,7 @@
#include <linux/tick.h>
#include <linux/rcupdate_trace.h>
#include <linux/nmi.h>
+#include <linux/perf_event.h>
#include "rcu.h"
@@ -115,6 +116,7 @@ torture_param(int, leakpointer, 0, "Leak pointer dereferences from readers");
torture_param(int, n_barrier_cbs, 0, "# of callbacks/kthreads for barrier testing");
torture_param(int, n_up_down, 32, "# of concurrent up/down hrtimer-based RCU readers");
torture_param(int, nfakewriters, 4, "Number of RCU fake writer threads");
+torture_param(bool, nmi_calls, true, "Exercise ->call() from NMI on nmi_capable flavors");
torture_param(int, nreaders, -1, "Number of RCU reader threads");
torture_param(bool, nwriters, 1, "Number of RCU writer threads (0 or 1)");
torture_param(int, object_debug, 0, "Enable debug-object double call_rcu() testing");
@@ -216,6 +218,8 @@ static long n_rcu_torture_boost_failure;
static long n_rcu_torture_boosts;
static atomic_long_t n_rcu_torture_timers;
static atomic_long_t n_rcu_torture_irqs;
+static atomic_long_t n_rcu_torture_nmi_call;
+static atomic_long_t n_rcu_torture_nmi_cb;
static long n_barrier_attempts;
static long n_barrier_successes; /* did rcu_barrier test succeed? */
static unsigned long n_read_exits;
@@ -433,6 +437,7 @@ struct rcu_torture_ops {
bool (*is_task_rcu_boosted)(void);
long cbflood_max;
int irq_capable;
+ int nmi_capable;
int can_boost;
int extendables;
int slow_gps;
@@ -648,6 +653,7 @@ static struct rcu_torture_ops rcu_ops = {
.extendables = RCUTORTURE_MAX_EXTEND,
.debug_objects = 1,
.start_poll_irqsoff = 1,
+ .nmi_capable = 1,
.name = "rcu"
};
@@ -942,6 +948,7 @@ static struct rcu_torture_ops srcu_ops = {
.debug_objects = 1,
.have_up_down = IS_ENABLED(CONFIG_TINY_SRCU)
? 0 : SRCU_READ_FLAVOR_NORMAL | SRCU_READ_FLAVOR_FAST_UPDOWN,
+ .nmi_capable = 1,
.name = "srcu"
};
@@ -1005,6 +1012,7 @@ static struct rcu_torture_ops srcud_ops = {
.debug_objects = 1,
.have_up_down = IS_ENABLED(CONFIG_TINY_SRCU)
? 0 : SRCU_READ_FLAVOR_NORMAL | SRCU_READ_FLAVOR_FAST_UPDOWN,
+ .nmi_capable = 1,
.name = "srcud"
};
@@ -1269,6 +1277,7 @@ static struct rcu_torture_ops tasks_tracing_ops = {
.cbflood_max = 50000,
.irq_capable = 1,
.slow_gps = 1,
+ .nmi_capable = 1,
.name = "tasks-tracing"
};
@@ -2659,12 +2668,131 @@ static bool rcu_torture_one_read(struct torture_random_state *trsp, long myid)
static DEFINE_TORTURE_RANDOM_PERCPU(rcu_torture_timer_rand);
+/*
+ * Exercise ->call() from NMI context for flavors that set ->nmi_capable. A
+ * per-CPU hardware perf counter overflows into an NMI, and its handler submits
+ * one preallocated callback via ->call(). One callback is in flight at a time
+ * (guarded by an atomic) to avoid allocating in NMI. This mirrors how a BPF
+ * program reaches ->call() from NMI.
+ */
+#ifdef CONFIG_PERF_EVENTS
+static struct perf_event_attr rcu_torture_nmi_attr = {
+ .type = PERF_TYPE_HARDWARE,
+ .config = PERF_COUNT_HW_CPU_CYCLES,
+ .size = sizeof(struct perf_event_attr),
+ .pinned = 1,
+ .disabled = 1,
+ .freq = 1,
+ .sample_freq = 100,
+};
+
+/* One in-flight callback per CPU; ->inuse is released by the callback. */
+struct rcu_torture_nmi_cbs {
+ struct rcu_head rh;
+ atomic_t inuse;
+};
+
+static struct perf_event **rcu_torture_nmi_events;
+static int rcu_torture_nmi_hp_state;
+static DEFINE_PER_CPU(struct rcu_torture_nmi_cbs, rcu_torture_nmi_cbs);
+
+static void rcu_torture_nmi_cb(struct rcu_head *rhp)
+{
+ struct rcu_torture_nmi_cbs *rtncp = container_of(rhp, struct rcu_torture_nmi_cbs, rh);
+
+ atomic_long_inc(&n_rcu_torture_nmi_cb);
+ atomic_set(&rtncp->inuse, 0);
+}
+
+static void rcu_torture_nmi_overflow(struct perf_event *event,
+ struct perf_sample_data *data,
+ struct pt_regs *regs)
+{
+ struct rcu_torture_nmi_cbs *rtncp = this_cpu_ptr(&rcu_torture_nmi_cbs);
+
+ if (!in_nmi())
+ return;
+ if (cur_ops->call && !atomic_xchg(&rtncp->inuse, 1)) {
+ atomic_long_inc(&n_rcu_torture_nmi_call);
+ cur_ops->call(&rtncp->rh, rcu_torture_nmi_cb);
+ }
+}
+
+static int rcu_torture_nmi_online(unsigned int cpu)
+{
+ struct perf_event *event;
+
+ event = perf_event_create_kernel_counter(&rcu_torture_nmi_attr, cpu, NULL,
+ rcu_torture_nmi_overflow, NULL);
+ if (IS_ERR(event))
+ return 0;
+ rcu_torture_nmi_events[cpu] = event;
+ perf_event_enable(event);
+ return 0;
+}
+
+static int rcu_torture_nmi_offline(unsigned int cpu)
+{
+ struct perf_event *event = rcu_torture_nmi_events[cpu];
+
+ if (event) {
+ rcu_torture_nmi_events[cpu] = NULL;
+ perf_event_disable(event);
+ perf_event_release_kernel(event);
+ }
+ return 0;
+}
+
+/*
+ * Drive the counters from CPU-hotplug callbacks so that coverage survives the
+ * onoff testing that most scenarios run.
+ */
+static void rcu_torture_nmi_init(void)
+{
+ int ret;
+
+ if (!nmi_calls || !cur_ops->nmi_capable || !cur_ops->call)
+ return;
+ rcu_torture_nmi_events = kcalloc(nr_cpu_ids, sizeof(*rcu_torture_nmi_events),
+ GFP_KERNEL);
+ if (!rcu_torture_nmi_events)
+ return;
+ ret = cpuhp_setup_state(CPUHP_AP_ONLINE_DYN, "rcutorture/nmi:online",
+ rcu_torture_nmi_online, rcu_torture_nmi_offline);
+ if (ret < 0) {
+ kfree(rcu_torture_nmi_events);
+ rcu_torture_nmi_events = NULL;
+ return;
+ }
+ rcu_torture_nmi_hp_state = ret;
+}
+
+static void rcu_torture_nmi_cleanup(void)
+{
+ if (!rcu_torture_nmi_events)
+ return;
+ if (rcu_torture_nmi_hp_state > 0) {
+ cpuhp_remove_state(rcu_torture_nmi_hp_state);
+ rcu_torture_nmi_hp_state = 0;
+ }
+ kfree(rcu_torture_nmi_events);
+ rcu_torture_nmi_events = NULL;
+ if (!atomic_long_read(&n_rcu_torture_nmi_call))
+ pr_alert("%s: nmi_calls set but no ->call() issued from NMI: not supported here.\n",
+ __func__);
+}
+#else /* #ifdef CONFIG_PERF_EVENTS */
+static void rcu_torture_nmi_init(void) { }
+static void rcu_torture_nmi_cleanup(void) { }
+#endif /* #else #ifdef CONFIG_PERF_EVENTS */
+
/*
* RCU torture reader from timer handler. Dereferences rcu_torture_current,
* incrementing the corresponding element of the pipeline array. The
* counter in the element should never be greater than 1, otherwise, the
* RCU implementation is broken.
*/
+
static void rcu_torture_timer(struct timer_list *unused)
{
WARN_ON_ONCE(!in_serving_softirq());
@@ -3047,6 +3175,9 @@ rcu_torture_stats_print(void)
data_race(n_barrier_attempts),
data_race(n_rcu_torture_barrier_error));
pr_cont("read-exits: %ld ", data_race(n_read_exits)); // Statistic.
+ pr_cont("nmi-calls: %ld nmi-cbs: %ld ",
+ atomic_long_read(&n_rcu_torture_nmi_call),
+ atomic_long_read(&n_rcu_torture_nmi_cb));
pr_cont("nocb-toggles: %ld:%ld ",
atomic_long_read(&n_nocb_offload), atomic_long_read(&n_nocb_deoffload));
pr_cont("gpwraps: %ld\n", n_gpwraps);
@@ -3195,7 +3326,7 @@ rcu_torture_print_module_parms(struct rcu_torture_ops *cur_ops, const char *tag)
"read_exit_delay=%d read_exit_burst=%d "
"reader_flavor=%x "
"nocbs_nthreads=%d nocbs_toggle=%d "
- "test_nmis=%d "
+ "test_nmis=%d nmi_calls=%d "
"preempt_duration=%d preempt_interval=%d n_up_down=%d\n",
torture_type, tag, nrealreaders, nwriters, nrealfakewriters,
stat_interval, verbose, test_no_idle_hz, shuffle_interval,
@@ -3209,7 +3340,7 @@ rcu_torture_print_module_parms(struct rcu_torture_ops *cur_ops, const char *tag)
read_exit_delay, read_exit_burst,
reader_flavor,
nocbs_nthreads, nocbs_toggle,
- test_nmis,
+ test_nmis, nmi_calls,
preempt_duration, preempt_interval, n_up_down);
}
@@ -4283,6 +4414,7 @@ rcu_torture_cleanup(void)
int i;
if (torture_cleanup_begin()) {
+ rcu_torture_nmi_cleanup();
if (cur_ops->cb_barrier != NULL) {
pr_info("%s: Invoking %pS().\n", __func__, cur_ops->cb_barrier);
cur_ops->cb_barrier();
@@ -4325,6 +4457,8 @@ rcu_torture_cleanup(void)
kfree(reader_tasks);
reader_tasks = NULL;
}
+ /* Disable the perf counters (and thus the NMI ->call() firing) now. */
+ rcu_torture_nmi_cleanup();
kfree(rcu_torture_reader_mbchk);
rcu_torture_reader_mbchk = NULL;
@@ -4354,6 +4488,20 @@ rcu_torture_cleanup(void)
pr_info("%s: Invoking %pS().\n", __func__, cur_ops->cb_barrier);
cur_ops->cb_barrier();
}
+
+ /*
+ * cb_barrier() above drained every deferred NMI ->call() callback, so the
+ * count issued from NMI must equal the count invoked; a mismatch means a
+ * callback was lost.
+ */
+ if (atomic_long_read(&n_rcu_torture_nmi_call) !=
+ atomic_long_read(&n_rcu_torture_nmi_cb)) {
+ pr_alert("%s: NMI ->call() lost a callback: issued %ld invoked %ld\n",
+ __func__, atomic_long_read(&n_rcu_torture_nmi_call),
+ atomic_long_read(&n_rcu_torture_nmi_cb));
+ atomic_inc(&n_rcu_torture_error);
+ }
+
if (cur_ops->cleanup != NULL)
cur_ops->cleanup();
@@ -4786,6 +4934,8 @@ rcu_torture_init(void)
firsterr = -ENOMEM;
goto unwind;
}
+ /* Arm the per-CPU perf counters that drive ->call() from NMI. */
+ rcu_torture_nmi_init();
for (i = 0; i < nrealreaders; i++) {
rcu_torture_reader_mbchk[i].rtc_chkrdr = -1;
firsterr = torture_create_kthread(rcu_torture_reader, (void *)i,
--
2.53.0-Meta
^ permalink raw reply related [flat|nested] 12+ messages in thread* [PATCH v3 6/6] selftests/bpf: Add a call_srcu() re-entry reproducer
2026-08-05 12:23 [PATCH v3 0/6] rcu,srcu: Make call_rcu()/call_srcu() safe from any context Puranjay Mohan
` (4 preceding siblings ...)
2026-08-05 12:23 ` [PATCH v3 5/6] rcutorture: Exercise ->call() from NMI context Puranjay Mohan
@ 2026-08-05 12:23 ` Puranjay Mohan
5 siblings, 0 replies; 12+ messages in thread
From: Puranjay Mohan @ 2026-08-05 12:23 UTC (permalink / raw)
To: Lai Jiangshan, Paul E. McKenney, Josh Triplett, Onur Özkan,
Frederic Weisbecker, Neeraj Upadhyay, Joel Fernandes, Boqun Feng,
Uladzislau Rezki, Davidlohr Bueso, Andrii Nakryiko,
Eduard Zingerman, Alexei Starovoitov, Daniel Borkmann,
Kumar Kartikeya Dwivedi
Cc: Puranjay Mohan, Steven Rostedt, Mathieu Desnoyers, Zqiang,
Martin KaFai Lau, Song Liu, Yonghong Song, Jiri Olsa,
Emil Tsalapatis, Matt Fleming, Harry Yoo (Oracle), linux-kernel,
rcu, bpf, linux-rt-devel
Re-enter call_srcu() from a BPF program to exercise its any-context
safety (and thus call_rcu_tasks_trace(), which is call_srcu() on
rcu_tasks_trace_srcu_struct).
An fentry program on rcu_segcblist_enqueue() fires mid-enqueue: that
function is reached from srcu_gp_start_if_needed() with the srcu_data
->lock held. The program does a task-storage delete, whose only deferred
work is call_rcu_tasks_trace(), re-entering the enqueue on the same CPU.
The triggering thread is pinned to one CPU and matched by TID, so the
program fires only for the test's own delete.
Without the fix the nested call re-takes the same sdp lock and
self-deadlocks; with it the nested __call_srcu() sees interrupts disabled
and defers via irq_work, so the delete returns and the test passes. Since
it can hang an unfixed kernel, run it only against a kernel carrying the
fix.
Signed-off-by: Puranjay Mohan <puranjay@kernel.org>
Acked-by: Kumar Kartikeya Dwivedi <memxor@gmail.com>
---
.../selftests/bpf/prog_tests/rcu_reentry.c | 95 +++++++++++++++++++
.../testing/selftests/bpf/progs/rcu_reentry.c | 51 ++++++++++
2 files changed, 146 insertions(+)
create mode 100644 tools/testing/selftests/bpf/prog_tests/rcu_reentry.c
create mode 100644 tools/testing/selftests/bpf/progs/rcu_reentry.c
diff --git a/tools/testing/selftests/bpf/prog_tests/rcu_reentry.c b/tools/testing/selftests/bpf/prog_tests/rcu_reentry.c
new file mode 100644
index 0000000000000..fa813d1d492b6
--- /dev/null
+++ b/tools/testing/selftests/bpf/prog_tests/rcu_reentry.c
@@ -0,0 +1,95 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Exercise re-entry into call_srcu() from BPF; see progs/rcu_reentry.c.
+ *
+ * On a kernel without the call_srcu() any-context fix the nested call
+ * self-deadlocks on the srcu_data lock, so this hangs rather than fails.
+ */
+#define _GNU_SOURCE
+#include <sched.h>
+#include <sys/syscall.h>
+#include <test_progs.h>
+#include "rcu_reentry.skel.h"
+
+static int sys_pidfd_open(pid_t pid, unsigned int flags)
+{
+ return syscall(__NR_pidfd_open, pid, flags);
+}
+
+/* Tiny RCU builds have no rcu_segcblist_enqueue() to attach to. */
+static bool have_attach_target(void)
+{
+ char buf[256];
+ bool found = false;
+ FILE *f;
+
+ f = fopen("/proc/kallsyms", "r");
+ if (!f)
+ return true; /* cannot tell; let the attach decide */
+ while (fgets(buf, sizeof(buf), f)) {
+ if (strstr(buf, " rcu_segcblist_enqueue\n")) {
+ found = true;
+ break;
+ }
+ }
+ fclose(f);
+ return found;
+}
+
+void test_rcu_reentry(void)
+{
+ struct rcu_reentry *skel;
+ int err, pidfd = -1, map_fd;
+ cpu_set_t set, old_set;
+ bool affinity_saved;
+ __u64 val = 1;
+
+ if (!have_attach_target()) {
+ test__skip();
+ return;
+ }
+
+ skel = rcu_reentry__open_and_load();
+ if (!ASSERT_OK_PTR(skel, "skel_open_and_load"))
+ return;
+
+ err = rcu_reentry__attach(skel);
+ if (!ASSERT_OK(err, "skel_attach"))
+ goto out;
+
+ /* Keep the re-entry on a single CPU. */
+ affinity_saved = !sched_getaffinity(0, sizeof(old_set), &old_set);
+ CPU_ZERO(&set);
+ CPU_SET(0, &set);
+ if (!ASSERT_OK(sched_setaffinity(0, sizeof(set), &set), "setaffinity"))
+ goto out;
+
+ pidfd = sys_pidfd_open(getpid(), 0);
+ if (!ASSERT_GE(pidfd, 0, "pidfd_open"))
+ goto restore;
+ map_fd = bpf_map__fd(skel->maps.task_stg);
+ err = bpf_map_update_elem(map_fd, &pidfd, &val, BPF_NOEXIST);
+ if (!ASSERT_OK(err, "boot_create"))
+ goto restore;
+
+ /* Arm the handler for this thread, then trigger call_rcu_tasks_trace(). */
+ skel->bss->target_pid = syscall(__NR_gettid);
+ err = bpf_map_delete_elem(map_fd, &pidfd);
+ if (!ASSERT_OK(err, "boot_delete"))
+ goto restore;
+
+ /* Only Tree SRCU enqueues via rcu_segcblist_enqueue(); skip elsewhere. */
+ if (!skel->bss->hits) {
+ test__skip();
+ goto restore;
+ }
+ ASSERT_EQ(skel->bss->get_errs, 0, "nested_storage_get");
+ ASSERT_EQ(skel->bss->del_errs, 0, "nested_storage_delete");
+restore:
+ if (affinity_saved)
+ sched_setaffinity(0, sizeof(old_set), &old_set);
+out:
+ if (pidfd >= 0)
+ close(pidfd);
+ rcu_reentry__destroy(skel);
+}
diff --git a/tools/testing/selftests/bpf/progs/rcu_reentry.c b/tools/testing/selftests/bpf/progs/rcu_reentry.c
new file mode 100644
index 0000000000000..47a36f704cf3e
--- /dev/null
+++ b/tools/testing/selftests/bpf/progs/rcu_reentry.c
@@ -0,0 +1,51 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Re-enter call_srcu() from a BPF program. fentry on rcu_segcblist_enqueue()
+ * fires inside call_srcu()'s enqueue (reached from srcu_gp_start_if_needed()
+ * with the srcu_data ->lock held); the handler then calls call_rcu_tasks_trace()
+ * -- itself call_srcu() on rcu_tasks_trace_srcu_struct -- re-entering the same
+ * srcu_data on the same CPU.
+ */
+#include "vmlinux.h"
+#include <bpf/bpf_helpers.h>
+#include <bpf/bpf_tracing.h>
+
+char _license[] SEC("license") = "GPL";
+
+struct {
+ __uint(type, BPF_MAP_TYPE_TASK_STORAGE);
+ __uint(map_flags, BPF_F_NO_PREALLOC);
+ __type(key, int);
+ __type(value, __u64);
+} task_stg SEC(".maps");
+
+int target_pid;
+int hits;
+int get_errs;
+int del_errs;
+int done;
+
+SEC("fentry/rcu_segcblist_enqueue")
+int BPF_PROG(reenter)
+{
+ struct task_struct *cur;
+
+ if (done || !target_pid)
+ return 0;
+
+ cur = bpf_get_current_task_btf();
+ if (cur->pid != target_pid)
+ return 0;
+
+ /* Issue the nested call exactly once, so the test is deterministic. */
+ done = 1;
+ __sync_fetch_and_add(&hits, 1);
+
+ /* Re-enter via a task-storage delete, which calls call_rcu_tasks_trace(). */
+ if (!bpf_task_storage_get(&task_stg, cur, 0, BPF_LOCAL_STORAGE_GET_F_CREATE))
+ __sync_fetch_and_add(&get_errs, 1);
+ else if (bpf_task_storage_delete(&task_stg, cur))
+ __sync_fetch_and_add(&del_errs, 1);
+
+ return 0;
+}
--
2.53.0-Meta
^ permalink raw reply related [flat|nested] 12+ messages in thread