The Linux Kernel Mailing List
 help / color / mirror / Atom feed
From: Lu Wang <wanglu.priv@gmail.com>
To: yu.c.chen@intel.com, tim.c.chen@linux.intel.com
Cc: peterz@infradead.org, mingo@redhat.com, juri.lelli@redhat.com,
	vincent.guittot@linaro.org, dietmar.eggemann@arm.com,
	rostedt@goodmis.org, bsegall@google.com, mgorman@suse.de,
	vschneid@redhat.com, kprateek.nayak@amd.com,
	linux-kernel@vger.kernel.org, chen.yu@linux.dev,
	Lu Wang <wanglu.priv@gmail.com>
Subject: [PATCH v2] sched/cache: honor migrate_llc_task semantics in active load balance
Date: Sun,  9 Aug 2026 18:53:43 +0800	[thread overview]
Message-ID: <20260809105343.1189051-1-wanglu.priv@gmail.com> (raw)
In-Reply-To: <20260801122252.2476258-1-wanglu.priv@gmail.com>

A passive load-balance pass marks group_llc_balance as migrate_llc_task
and queues active balance when it cannot move a task. The CPU stopper
callback constructs a fresh lb_env, so select the stopper callback when
queueing active balance to preserve the migration semantics across the
asynchronous boundary.

For CAS-directed active balance, reject a candidate whose preferred LLC
does not match the destination LLC. This keeps the fallback from moving
a task away from its preferred LLC.

Fixes: e4c9a4cb244a (\"sched/cache: Add migrate_llc_task migration type for cache-aware balancing\")
Suggested-by: \"Chen, Yu C\" <yu.c.chen@intel.com>
Signed-off-by: Lu Wang <wanglu.priv@gmail.com>
---
Changes in v2:
 - Select the stopper callback at kick time to preserve
   migrate_llc_task semantics in active load balance without passing
   migration_type across the stopper, which affects delayed-dequeue tasks.
 - Link to V1: https://lore.kernel.org/all/20260801122252.2476258-1-wanglu.priv@gmail.com/

 kernel/sched/fair.c | 56 ++++++++++++++++++++++++++++++++++++++++-----
 1 file changed, 50 insertions(+), 6 deletions(-)

diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
index d78467ec6..37a470c41 100644
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -10254,6 +10254,7 @@ enum migration_type {
 #define LBF_SOME_PINNED	0x08
 #define LBF_ACTIVE_LB	0x10
 #define LBF_LLC_PINNED	0x20
+#define LBF_ACTIVE_LB_LLC	0x40
 
 struct lb_env {
 	struct sched_domain	*sd;
@@ -10645,6 +10646,20 @@ alb_break_llc(struct lb_env *env)
 	return false;
 }
 
+/*
+ * Returns true if p's preferred LLC does not match the destination CPU
+ * under migrate_llc_task semantics. Passive LB passes migrate_llc_task
+ * in migration_type, while active LB carries it in LBF_ACTIVE_LB_LLC.
+ */
+static inline bool
+migrate_llc_task_wrong_dst(struct task_struct *p, struct lb_env *env)
+{
+	return sched_cache_enabled() &&
+	       (env->migration_type == migrate_llc_task ||
+		env->flags & LBF_ACTIVE_LB_LLC) &&
+	       READ_ONCE(p->preferred_llc) != llc_id(env->dst_cpu);
+}
+
 /*
  * Check if migrating task p from env->src_cpu to
  * env->dst_cpu breaks LLC localiy.
@@ -10673,8 +10688,7 @@ static bool migrate_degrades_llc(struct task_struct *p, struct lb_env *env)
 	 * run on env->dst_cpu, skip the tasks do not prefer
 	 * env->dst_cpu, and find the one that prefers.
 	 */
-	if (env->migration_type == migrate_llc_task &&
-	    READ_ONCE(p->preferred_llc) != llc_id(env->dst_cpu))
+	if (migrate_llc_task_wrong_dst(p, env))
 		return true;
 
 	if (can_migrate_llc_task(env->src_cpu,
@@ -10697,6 +10711,12 @@ alb_break_llc(struct lb_env *env)
 	return false;
 }
 
+static inline bool
+migrate_llc_task_wrong_dst(struct task_struct *p, struct lb_env *env)
+{
+	return false;
+}
+
 static inline bool
 migrate_degrades_llc(struct task_struct *p, struct lb_env *env)
 {
@@ -10796,7 +10816,7 @@ int can_migrate_task(struct task_struct *p, struct lb_env *env)
 	 * 4) too many balance attempts have failed.
 	 */
 	if (env->flags & LBF_ACTIVE_LB)
-		return 1;
+		return !migrate_llc_task_wrong_dst(p, env);
 
 	degrades = migrate_degrades_locality(p, env);
 	if (!degrades) {
@@ -13156,6 +13176,20 @@ static int need_active_balance(struct lb_env *env)
 }
 
 static int active_load_balance_cpu_stop(void *data);
+static int active_load_balance_llc_cpu_stop(void *data);
+
+/*
+ * migration_type is checked elsewhere to decide migration policy, so
+ * it shouldn't be repurposed just to flag an LLC-directed active
+ * balance across the stopper. Pick the callback here instead.
+ */
+static inline cpu_stop_fn_t alb_stop_fn(struct lb_env *env)
+{
+	if (env->migration_type == migrate_llc_task)
+		return active_load_balance_llc_cpu_stop;
+
+	return active_load_balance_cpu_stop;
+}
 
 static int should_we_balance(struct lb_env *env)
 {
@@ -13492,7 +13526,7 @@ static int sched_balance_rq(int this_cpu, struct rq *this_rq,
 			raw_spin_rq_unlock_irqrestore(busiest, flags);
 			if (active_balance) {
 				stop_one_cpu_nowait(cpu_of(busiest),
-					active_load_balance_cpu_stop, busiest,
+					alb_stop_fn(&env), busiest,
 					&busiest->active_balance_work);
 			}
 			preempt_enable();
@@ -13603,7 +13637,7 @@ update_next_balance(struct sched_domain *sd, unsigned long *next_balance)
  * least 1 task to be running on each physical CPU where possible, and
  * avoids physical / logical imbalances.
  */
-static int active_load_balance_cpu_stop(void *data)
+static int __active_load_balance_cpu_stop(void *data, unsigned int lb_flags)
 {
 	struct rq *busiest_rq = data;
 	int busiest_cpu = cpu_of(busiest_rq);
@@ -13653,7 +13687,7 @@ static int active_load_balance_cpu_stop(void *data)
 			.src_cpu	= busiest_rq->cpu,
 			.src_rq		= busiest_rq,
 			.idle		= CPU_IDLE,
-			.flags		= LBF_ACTIVE_LB,
+			.flags		= LBF_ACTIVE_LB | lb_flags,
 		};
 
 		schedstat_inc(sd->alb_count);
@@ -13681,6 +13715,16 @@ static int active_load_balance_cpu_stop(void *data)
 	return 0;
 }
 
+static int active_load_balance_cpu_stop(void *data)
+{
+	return __active_load_balance_cpu_stop(data, 0);
+}
+
+static int active_load_balance_llc_cpu_stop(void *data)
+{
+	return __active_load_balance_cpu_stop(data, LBF_ACTIVE_LB_LLC);
+}
+
 /*
  * Scale the max sched_balance_rq interval with the number of CPUs in the system.
  * This trades load-balance latency on larger machines for less cross talk.
-- 
2.43.0


      parent reply	other threads:[~2026-08-09 10:54 UTC|newest]

Thread overview: 17+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-08-01 12:22 [PATCH] sched/cache: honor migrate_llc_task semantics in active load balance Lu Wang
2026-08-03  4:20 ` Chen, Yu C
2026-08-03 10:02   ` Lu Wang
2026-08-04  0:10     ` Tim Chen
2026-08-04  8:17       ` Chen, Yu C
2026-08-04 15:07         ` Lu Wang
2026-08-04 19:42           ` Tim Chen
2026-08-05  2:38             ` wanglu15
2026-08-05 16:04               ` Tim Chen
2026-08-05 16:43                 ` Chen, Yu C
2026-08-06 16:21                   ` Tim Chen
2026-08-04  8:30       ` Lu Wang
2026-08-06 15:35 ` Chen, Yu C
2026-08-06 17:22   ` Tim Chen
2026-08-07  6:48     ` Chen, Yu C
2026-08-07  9:58       ` Lu Wang
2026-08-09 10:53 ` Lu Wang [this message]

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260809105343.1189051-1-wanglu.priv@gmail.com \
    --to=wanglu.priv@gmail.com \
    --cc=bsegall@google.com \
    --cc=chen.yu@linux.dev \
    --cc=dietmar.eggemann@arm.com \
    --cc=juri.lelli@redhat.com \
    --cc=kprateek.nayak@amd.com \
    --cc=linux-kernel@vger.kernel.org \
    --cc=mgorman@suse.de \
    --cc=mingo@redhat.com \
    --cc=peterz@infradead.org \
    --cc=rostedt@goodmis.org \
    --cc=tim.c.chen@linux.intel.com \
    --cc=vincent.guittot@linaro.org \
    --cc=vschneid@redhat.com \
    --cc=yu.c.chen@intel.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox