All of lore.kernel.org
 help / color / mirror / Atom feed
From: Aaron Tomlin <atomlin@atomlin.com>
To: mingo@redhat.com, peterz@infradead.org, juri.lelli@redhat.com,
	vincent.guittot@linaro.org
Cc: dietmar.eggemann@arm.com, rostedt@goodmis.org,
	bsegall@google.com, mgorman@suse.de, vschneid@redhat.com,
	kprateek.nayak@amd.com, zhanxusheng1024@gmail.com,
	neelx@suse.com, atomlin@atomlin.com, chjohnst@mail.com,
	mproche@mail.com, sean@ashe.io, steve@abita.co,
	rishil1999@outlook.com, linux-kernel@vger.kernel.org
Subject: [PATCH v6 1/6] sched: Annotate rq->rd with __rcu and update lockless readers
Date: Tue, 25 Aug 2026 14:46:32 -0400	[thread overview]
Message-ID: <20260825184637.888364-2-atomlin@atomlin.com> (raw)
In-Reply-To: <20260825184637.888364-1-atomlin@atomlin.com>

The root_domain pointer rd field in struct rq is updated dynamically
using RCU, and its memory reclamation is deferred via call_rcu() in
rq_attach_root(). However, struct rq's rd field was missing the __rcu
compiler annotation, and several lockless readers across the scheduler
subsystem accessed rq->rd directly without using RCU dereference
primitives.

Add the __rcu annotation to struct rq's rd field in kernel/sched/sched.h.
Update lockless readers across kernel/sched/ to use rcu_dereference(),
rcu_dereference_sched() or rcu_access_pointer() appropriately. This
ensures proper data-dependency barriers on all architectures, enables
Sparse static analysis validation, and documents RCU read-side ownership
contracts.

Signed-off-by: Aaron Tomlin <atomlin@atomlin.com>
---
 kernel/sched/core.c     | 16 ++++++++++------
 kernel/sched/deadline.c |  8 ++++----
 kernel/sched/fair.c     | 29 +++++++++++++++--------------
 kernel/sched/sched.h    |  2 +-
 4 files changed, 30 insertions(+), 25 deletions(-)

diff --git a/kernel/sched/core.c b/kernel/sched/core.c
index 2e7cde033a31..8c81f7c0cea0 100644
--- a/kernel/sched/core.c
+++ b/kernel/sched/core.c
@@ -8577,10 +8577,12 @@ void set_rq_offline(struct rq *rq)
 static inline void sched_set_rq_online(struct rq *rq, int cpu)
 {
 	struct rq_flags rf;
+	struct root_domain *rd;
 
 	rq_lock_irqsave(rq, &rf);
-	if (rq->rd) {
-		BUG_ON(!cpumask_test_cpu(cpu, rq->rd->span));
+	rd = rcu_dereference_protected(rq->rd, lockdep_is_held(&rq->__lock));
+	if (rd) {
+		BUG_ON(!cpumask_test_cpu(cpu, rd->span));
 		set_rq_online(rq);
 	}
 	rq_unlock_irqrestore(rq, &rf);
@@ -8589,10 +8591,12 @@ static inline void sched_set_rq_online(struct rq *rq, int cpu)
 static inline void sched_set_rq_offline(struct rq *rq, int cpu)
 {
 	struct rq_flags rf;
+	struct root_domain *rd;
 
 	rq_lock_irqsave(rq, &rf);
-	if (rq->rd) {
-		BUG_ON(!cpumask_test_cpu(cpu, rq->rd->span));
+	rd = rcu_dereference_protected(rq->rd, lockdep_is_held(&rq->__lock));
+	if (rd) {
+		BUG_ON(!cpumask_test_cpu(cpu, rd->span));
 		set_rq_offline(rq);
 	}
 	rq_unlock_irqrestore(rq, &rf);
@@ -9009,8 +9013,8 @@ void __init sched_init(void)
 #endif
 		rq->next_class = &idle_sched_class;
 
-		rq->sd = NULL;
-		rq->rd = NULL;
+		RCU_INIT_POINTER(rq->sd, NULL);
+		RCU_INIT_POINTER(rq->rd, NULL);
 		rq->cpu_capacity = SCHED_CAPACITY_SCALE;
 		rq->balance_callback = &balance_push_callback;
 		rq->active_balance = 0;
diff --git a/kernel/sched/deadline.c b/kernel/sched/deadline.c
index 857dbe3519a8..4137f8bbcef5 100644
--- a/kernel/sched/deadline.c
+++ b/kernel/sched/deadline.c
@@ -122,12 +122,12 @@ static inline struct dl_bw *dl_bw_of(int i)
 {
 	RCU_LOCKDEP_WARN(!rcu_read_lock_sched_held(),
 			 "sched RCU must be held");
-	return &cpu_rq(i)->rd->dl_bw;
+	return &rcu_dereference_sched(cpu_rq(i)->rd)->dl_bw;
 }
 
 static inline int dl_bw_cpus(int i)
 {
-	struct root_domain *rd = cpu_rq(i)->rd;
+	struct root_domain *rd = rcu_dereference_sched(cpu_rq(i)->rd);
 
 	RCU_LOCKDEP_WARN(!rcu_read_lock_sched_held(),
 			 "sched RCU must be held");
@@ -159,13 +159,13 @@ static inline unsigned long dl_bw_capacity(int i)
 		RCU_LOCKDEP_WARN(!rcu_read_lock_sched_held(),
 				 "sched RCU must be held");
 
-		return __dl_bw_capacity(cpu_rq(i)->rd->span);
+		return __dl_bw_capacity(rcu_dereference_sched(cpu_rq(i)->rd)->span);
 	}
 }
 
 bool dl_bw_visited(int cpu, u64 cookie)
 {
-	struct root_domain *rd = cpu_rq(cpu)->rd;
+	struct root_domain *rd = rcu_dereference_sched(cpu_rq(cpu)->rd);
 
 	if (rd->visit_cookie == cookie)
 		return true;
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
index f79fcba4afec..d55c21811306 100644
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -7865,13 +7865,10 @@ static inline void set_rd_overutilized(struct root_domain *rd, bool flag)
 
 static inline void check_update_overutilized_status(struct rq *rq)
 {
-	/*
-	 * overutilized field is used for load balancing decisions only
-	 * if energy aware scheduler is being used
-	 */
+	struct root_domain *rd = rcu_dereference(rq->rd);
 
-	if (!is_rd_overutilized(rq->rd) && cpu_overutilized(rq->cpu))
-		set_rd_overutilized(rq->rd, 1);
+	if (rd && !is_rd_overutilized(rd) && cpu_overutilized(rq->cpu))
+		set_rd_overutilized(rd, 1);
 }
 
 /* Runqueue only has SCHED_IDLE tasks enqueued */
@@ -9500,7 +9497,7 @@ static int find_energy_efficient_cpu(struct task_struct *p, int prev_cpu)
 	unsigned long prev_delta = ULONG_MAX, best_delta = ULONG_MAX;
 	unsigned long p_util_min = uclamp_is_used() ? uclamp_eff_value(p, UCLAMP_MIN) : 0;
 	unsigned long p_util_max = uclamp_is_used() ? uclamp_eff_value(p, UCLAMP_MAX) : 1024;
-	struct root_domain *rd = this_rq()->rd;
+	struct root_domain *rd = rcu_dereference(this_rq()->rd);
 	int cpu, best_energy_cpu, target = -1;
 	int prev_fits = -1, best_fits = -1;
 	unsigned long best_actual_cap = 0;
@@ -9704,7 +9701,7 @@ select_task_rq_fair(struct task_struct *p, int prev_cpu, int wake_flags)
 		    cpumask_test_cpu(cpu, p->cpus_ptr))
 			return cpu;
 
-		if (!is_rd_overutilized(this_rq()->rd)) {
+		if (!is_rd_overutilized(rcu_dereference(this_rq()->rd))) {
 			new_cpu = find_energy_efficient_cpu(p, prev_cpu);
 			if (new_cpu >= 0)
 				return new_cpu;
@@ -12690,13 +12687,15 @@ static inline void update_sd_lb_stats(struct lb_env *env, struct sd_lb_stats *sd
 		env->fbq_type = fbq_classify_group(&sds->busiest_stat);
 
 	if (!env->sd->parent) {
+		struct root_domain *rd = rcu_dereference(env->dst_rq->rd);
+
 		/* update overload indicator if we are at root domain */
-		set_rd_overloaded(env->dst_rq->rd, sg_overloaded);
+		set_rd_overloaded(rd, sg_overloaded);
 
 		/* Update over-utilization (tipping point, U >= 0) indicator */
-		set_rd_overutilized(env->dst_rq->rd, sg_overutilized);
+		set_rd_overutilized(rd, sg_overutilized);
 	} else if (sg_overutilized) {
-		set_rd_overutilized(env->dst_rq->rd, sg_overutilized);
+		set_rd_overutilized(rcu_dereference(env->dst_rq->rd), sg_overutilized);
 	}
 
 	update_idle_cpu_scan(env, sum_util);
@@ -12942,8 +12941,10 @@ static struct sched_group *sched_balance_find_src_group(struct lb_env *env)
 	if (busiest->group_type == group_misfit_task)
 		goto force_balance;
 
-	if (!is_rd_overutilized(env->dst_rq->rd) &&
-	    rcu_dereference_all(env->dst_rq->rd->pd))
+	struct root_domain *rd = rcu_dereference(env->dst_rq->rd);
+
+	if (rd && !is_rd_overutilized(rd) &&
+	    rcu_dereference_all(rd->pd))
 		goto out_balanced;
 
 	/* ASYM feature bypasses nice load balance check */
@@ -14573,7 +14574,7 @@ static int sched_balance_newidle(struct rq *this_rq, struct rq_flags *rf)
 	if (!sd)
 		goto out;
 
-	if (!get_rd_overloaded(this_rq->rd) ||
+	if (!get_rd_overloaded(rcu_dereference(this_rq->rd)) ||
 	    this_rq->avg_idle < sd->max_newidle_lb_cost) {
 
 		update_next_balance(sd, &next_balance);
diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h
index 26ae13c86b69..13a437032855 100644
--- a/kernel/sched/sched.h
+++ b/kernel/sched/sched.h
@@ -1256,7 +1256,7 @@ struct rq {
 	int membarrier_state;
 #endif
 
-	struct root_domain		*rd;
+	struct root_domain __rcu	*rd;
 	struct sched_domain __rcu	*sd;
 
 	struct balance_callback *balance_callback;
-- 
2.55.0


  reply	other threads:[~2026-08-25 18:46 UTC|newest]

Thread overview: 8+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-08-25 18:46 [PATCH v6 0/6] sched/debug: Introduce per-CPU debugfs files Aaron Tomlin
2026-08-25 18:46 ` Aaron Tomlin [this message]
2026-08-25 21:48   ` [PATCH v6 1/6] sched: Annotate rq->rd with __rcu and update lockless readers Aaron Tomlin
2026-08-25 18:46 ` [PATCH v6 2/6] sched/debug: Protect lockless rq->rd access in print_dl_rq() Aaron Tomlin
2026-08-25 18:46 ` [PATCH v6 3/6] sched/debug: Protect lockless rq->curr access in print_cpu() Aaron Tomlin
2026-08-25 18:46 ` [PATCH v6 4/6] sched/debug: Protect p->mm access in sched_show_numa() Aaron Tomlin
2026-08-25 18:46 ` [PATCH v6 5/6] sched/fair: Use list_for_each_entry_rcu() in print_cfs_stats() Aaron Tomlin
2026-08-25 18:46 ` [PATCH v6 6/6] sched/debug: Introduce per-CPU debugfs files Aaron Tomlin

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260825184637.888364-2-atomlin@atomlin.com \
    --to=atomlin@atomlin.com \
    --cc=bsegall@google.com \
    --cc=chjohnst@mail.com \
    --cc=dietmar.eggemann@arm.com \
    --cc=juri.lelli@redhat.com \
    --cc=kprateek.nayak@amd.com \
    --cc=linux-kernel@vger.kernel.org \
    --cc=mgorman@suse.de \
    --cc=mingo@redhat.com \
    --cc=mproche@mail.com \
    --cc=neelx@suse.com \
    --cc=peterz@infradead.org \
    --cc=rishil1999@outlook.com \
    --cc=rostedt@goodmis.org \
    --cc=sean@ashe.io \
    --cc=steve@abita.co \
    --cc=vincent.guittot@linaro.org \
    --cc=vschneid@redhat.com \
    --cc=zhanxusheng1024@gmail.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.