From: Puranjay Mohan <puranjay@kernel.org>
To: "Lai Jiangshan" <jiangshanlai@gmail.com>,
"Paul E. McKenney" <paulmck@kernel.org>,
"Josh Triplett" <josh@joshtriplett.org>,
"Onur Özkan" <work@onurozkan.dev>,
"Frederic Weisbecker" <frederic@kernel.org>,
"Neeraj Upadhyay" <neeraj.upadhyay@kernel.org>,
"Joel Fernandes" <joelagnelf@nvidia.com>,
"Boqun Feng" <boqun@kernel.org>,
"Uladzislau Rezki" <urezki@gmail.com>,
"Davidlohr Bueso" <dave@stgolabs.net>,
"Andrii Nakryiko" <andrii@kernel.org>,
"Eduard Zingerman" <eddyz87@gmail.com>,
"Alexei Starovoitov" <ast@kernel.org>,
"Daniel Borkmann" <daniel@iogearbox.net>,
"Kumar Kartikeya Dwivedi" <memxor@gmail.com>
Cc: Puranjay Mohan <puranjay@kernel.org>,
Steven Rostedt <rostedt@goodmis.org>,
Mathieu Desnoyers <mathieu.desnoyers@efficios.com>,
Zqiang <qiang.zhang@linux.dev>,
Martin KaFai Lau <martin.lau@linux.dev>,
Song Liu <song@kernel.org>,
Yonghong Song <yonghong.song@linux.dev>,
Jiri Olsa <jolsa@kernel.org>,
Emil Tsalapatis <emil@etsalapatis.com>,
Matt Fleming <mfleming@cloudflare.com>,
"Harry Yoo (Oracle)" <harry@kernel.org>,
linux-kernel@vger.kernel.org, rcu@vger.kernel.org,
bpf@vger.kernel.org, linux-rt-devel@lists.linux.dev
Subject: [PATCH v4 2/6] rcu: Make Tiny call_rcu() safe to call from any context
Date: Mon, 10 Aug 2026 05:27:51 -0700 [thread overview]
Message-ID: <20260810122758.183765-3-puranjay@kernel.org> (raw)
In-Reply-To: <20260810122758.183765-1-puranjay@kernel.org>
Give Tiny call_rcu() the same treatment as Tree RCU. When interrupts are
disabled and the scheduler is up, stage the callback on a lockless list
that an irq_work re-issues later. One global list and irq_work suffice
since Tiny RCU is uniprocessor, and there is no CPU-offline drain.
The re-issue runs with interrupts disabled and can be re-entered by
instrumentation, so a draining flag drops a deferring call_rcu() seen
mid-drain (unless from an NMI), as in Tree RCU. Gated by CONFIG_RCU_DEFER,
though the deferral state is unconditional.
Interrupts stay off for the whole batch, but the re-issue is a tail append
with no locks. TINY_RCU implies !SMP, where arch_irq_work_has_interrupt()
is false, so the drain always waits for the tick and a batch is whatever
one tick's worth of interrupts-disabled call_rcu()s staged. As in Tree
RCU the drain clears ->next before re-issuing, which bounds a node
self-linked by a double call_rcu(): rcu_do_enqueue()'s duplicate path
returns without clearing it. A longer cycle is not bounded; a double
call_rcu() stays undefined.
The idle-task reschedule moves out of the enqueue helper so that a drain
does it once for the batch rather than once per callback, which would
otherwise take the runqueue lock N times with interrupts disabled.
Suggested-by: Paul E. McKenney <paulmck@kernel.org>
Signed-off-by: Puranjay Mohan <puranjay@kernel.org>
---
kernel/rcu/tiny.c | 127 +++++++++++++++++++++++++++++++++++++---------
1 file changed, 104 insertions(+), 23 deletions(-)
diff --git a/kernel/rcu/tiny.c b/kernel/rcu/tiny.c
index dccccd6be9411..656b6a682e31a 100644
--- a/kernel/rcu/tiny.c
+++ b/kernel/rcu/tiny.c
@@ -11,6 +11,8 @@
*/
#include <linux/completion.h>
#include <linux/interrupt.h>
+#include <linux/irq_work.h>
+#include <linux/llist.h>
#include <linux/notifier.h>
#include <linux/rcupdate_wait.h>
#include <linux/kernel.h>
@@ -42,8 +44,100 @@ static struct rcu_ctrlblk rcu_ctrlblk = {
.gp_seq = 0 - 300UL,
};
+/*
+ * The callback list is only accessed with interrupts disabled, so a call_rcu()
+ * that arrives with interrupts off stages the callback on a lockless list that
+ * an irq_work re-issues later. One global list and irq_work suffice, as Tiny
+ * RCU is uniprocessor.
+ */
+static void rcu_defer_drain(struct irq_work *iw);
+static LLIST_HEAD(rcu_defer_list);
+static struct irq_work rcu_defer_iw = IRQ_WORK_INIT_HARD(rcu_defer_drain);
+static bool rcu_defer_draining;
+
+/*
+ * Also called by __rcu_defer_drain() to re-issue a deferred callback, so it
+ * must not re-check the deferral condition.
+ */
+static void rcu_do_enqueue(struct rcu_head *head, rcu_callback_t func)
+{
+ static atomic_t doublefrees;
+ unsigned long flags;
+
+ if (debug_rcu_head_queue(head)) {
+ if (atomic_inc_return(&doublefrees) < 4) {
+ pr_err("%s(): Double-freed CB %p->%pS()!!! ", __func__, head, head->func);
+ mem_dump_obj(head);
+ }
+ return;
+ }
+
+ head->func = func;
+ head->next = NULL;
+
+ local_irq_save(flags);
+ *rcu_ctrlblk.curtail = head;
+ rcu_ctrlblk.curtail = &head->next;
+ local_irq_restore(flags);
+}
+
+/* Force scheduling for rcu_qs() when enqueuing from the idle task. */
+static void rcu_resched_if_idle(void)
+{
+ if (unlikely(is_idle_task(current)))
+ resched_cpu(0);
+}
+
+static void __rcu_defer_drain(void)
+{
+ struct llist_node *node, *next;
+ bool drained = false;
+ unsigned long flags;
+
+ if (!IS_ENABLED(CONFIG_RCU_DEFER))
+ return;
+
+ /* Re-issued newest-first; nothing depends on call_rcu() ordering. */
+ local_irq_save(flags);
+ llist_for_each_safe(node, next, llist_del_all(&rcu_defer_list)) {
+ struct rcu_head *head = (struct rcu_head *)node;
+
+ /* Bounds a node self-linked by a double call_rcu(). */
+ head->next = NULL;
+ rcu_do_enqueue(head, head->func);
+ drained = true;
+ }
+ local_irq_restore(flags);
+
+ if (drained)
+ rcu_resched_if_idle();
+}
+
+/* Only the irq_work drain can be re-fed by its own re-issue; see Tree RCU. */
+static void rcu_defer_drain(struct irq_work *iw)
+{
+ WRITE_ONCE(rcu_defer_draining, true);
+ __rcu_defer_drain();
+ WRITE_ONCE(rcu_defer_draining, false);
+}
+
+static void call_rcu_defer(struct rcu_head *head, rcu_callback_t func)
+{
+ /* A re-entrant call_rcu() during the drain would livelock it; drop it. */
+ if (READ_ONCE(rcu_defer_draining) && !in_nmi()) {
+ WARN_ONCE(IS_ENABLED(CONFIG_PROVE_RCU),
+ "call_rcu() re-entered during callback drain; leaking callback\n");
+ return;
+ }
+ head->func = func;
+ if (llist_add((struct llist_node *)head, &rcu_defer_list))
+ irq_work_queue(&rcu_defer_iw);
+}
+
void rcu_barrier(void)
{
+ /* Register any deferred callbacks so the wait below covers them. */
+ __rcu_defer_drain();
wait_rcu_gp(call_rcu_hurry);
}
EXPORT_SYMBOL(rcu_barrier);
@@ -157,29 +251,19 @@ EXPORT_SYMBOL_GPL(synchronize_rcu);
*/
void call_rcu(struct rcu_head *head, rcu_callback_t func)
{
- static atomic_t doublefrees;
- unsigned long flags;
-
- if (debug_rcu_head_queue(head)) {
- if (atomic_inc_return(&doublefrees) < 4) {
- pr_err("%s(): Double-freed CB %p->%pS()!!! ", __func__, head, head->func);
- mem_dump_obj(head);
- }
+ if (should_rcu_defer()) {
+ call_rcu_defer(head, func);
return;
}
- head->func = func;
- head->next = NULL;
+ /*
+ * Only reachable from an NMI when deferral is off: before the scheduler
+ * is up, or with CONFIG_RCU_DEFER=n. The enqueue can then race.
+ */
+ WARN_ON_ONCE(IS_ENABLED(CONFIG_PROVE_RCU) && in_nmi());
- local_irq_save(flags);
- *rcu_ctrlblk.curtail = head;
- rcu_ctrlblk.curtail = &head->next;
- local_irq_restore(flags);
-
- if (unlikely(is_idle_task(current))) {
- /* force scheduling for rcu_qs() */
- resched_cpu(0);
- }
+ rcu_do_enqueue(head, func);
+ rcu_resched_if_idle();
}
EXPORT_SYMBOL_GPL(call_rcu);
@@ -211,10 +295,7 @@ unsigned long start_poll_synchronize_rcu(void)
{
unsigned long gp_seq = get_state_synchronize_rcu();
- if (unlikely(is_idle_task(current))) {
- /* force scheduling for rcu_qs() */
- resched_cpu(0);
- }
+ rcu_resched_if_idle();
return gp_seq;
}
EXPORT_SYMBOL_GPL(start_poll_synchronize_rcu);
--
2.53.0-Meta
next prev parent reply other threads:[~2026-08-10 12:28 UTC|newest]
Thread overview: 12+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-08-10 12:27 [PATCH v4 0/6] rcu,srcu: Make call_rcu()/call_srcu() safe from any context Puranjay Mohan
2026-08-10 12:27 ` [PATCH v4 1/6] rcu: Make call_rcu() safe to call " Puranjay Mohan
2026-08-10 12:43 ` sashiko-bot
2026-08-10 12:48 ` Puranjay Mohan
2026-08-10 12:27 ` Puranjay Mohan [this message]
2026-08-10 12:42 ` [PATCH v4 2/6] rcu: Make Tiny " sashiko-bot
2026-08-10 12:45 ` Puranjay Mohan
2026-08-10 12:27 ` [PATCH v4 3/6] srcu: Make call_srcu() " Puranjay Mohan
2026-08-10 12:27 ` [PATCH v4 4/6] srcu: Make Tiny " Puranjay Mohan
2026-08-10 12:27 ` [PATCH v4 5/6] rcutorture: Exercise ->call() from NMI context Puranjay Mohan
2026-08-10 12:27 ` [PATCH v4 6/6] selftests/bpf: Add a call_srcu() re-entry reproducer Puranjay Mohan
2026-08-12 0:10 ` [PATCH v4 0/6] rcu,srcu: Make call_rcu()/call_srcu() safe from any context Paul E. McKenney
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260810122758.183765-3-puranjay@kernel.org \
--to=puranjay@kernel.org \
--cc=andrii@kernel.org \
--cc=ast@kernel.org \
--cc=boqun@kernel.org \
--cc=bpf@vger.kernel.org \
--cc=daniel@iogearbox.net \
--cc=dave@stgolabs.net \
--cc=eddyz87@gmail.com \
--cc=emil@etsalapatis.com \
--cc=frederic@kernel.org \
--cc=harry@kernel.org \
--cc=jiangshanlai@gmail.com \
--cc=joelagnelf@nvidia.com \
--cc=jolsa@kernel.org \
--cc=josh@joshtriplett.org \
--cc=linux-kernel@vger.kernel.org \
--cc=linux-rt-devel@lists.linux.dev \
--cc=martin.lau@linux.dev \
--cc=mathieu.desnoyers@efficios.com \
--cc=memxor@gmail.com \
--cc=mfleming@cloudflare.com \
--cc=neeraj.upadhyay@kernel.org \
--cc=paulmck@kernel.org \
--cc=qiang.zhang@linux.dev \
--cc=rcu@vger.kernel.org \
--cc=rostedt@goodmis.org \
--cc=song@kernel.org \
--cc=urezki@gmail.com \
--cc=work@onurozkan.dev \
--cc=yonghong.song@linux.dev \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.