All of lore.kernel.org
 help / color / mirror / Atom feed
From: Lu Wang <wanglu.priv@gmail.com>
To: yu.c.chen@intel.com, tim.c.chen@linux.intel.com
Cc: peterz@infradead.org, mingo@redhat.com, juri.lelli@redhat.com,
	vincent.guittot@linaro.org, dietmar.eggemann@arm.com,
	rostedt@goodmis.org, bsegall@google.com, mgorman@suse.de,
	vschneid@redhat.com, kprateek.nayak@amd.com,
	linux-kernel@vger.kernel.org, chen.yu@linux.dev,
	Lu Wang <wanglu.priv@gmail.com>
Subject: [PATCH v2] sched/cache: honor migrate_llc_task semantics in active load balance
Date: Sun,  9 Aug 2026 18:53:43 +0800	[thread overview]
Message-ID: <20260809105343.1189051-1-wanglu.priv@gmail.com> (raw)
In-Reply-To: <20260801122252.2476258-1-wanglu.priv@gmail.com>

A passive load-balance pass marks group_llc_balance as migrate_llc_task
and queues active balance when it cannot move a task. The CPU stopper
callback constructs a fresh lb_env, so select the stopper callback when
queueing active balance to preserve the migration semantics across the
asynchronous boundary.

For CAS-directed active balance, reject a candidate whose preferred LLC
does not match the destination LLC. This keeps the fallback from moving
a task away from its preferred LLC.

Fixes: e4c9a4cb244a (\"sched/cache: Add migrate_llc_task migration type for cache-aware balancing\")
Suggested-by: \"Chen, Yu C\" <yu.c.chen@intel.com>
Signed-off-by: Lu Wang <wanglu.priv@gmail.com>
---
Changes in v2:
 - Select the stopper callback at kick time to preserve
   migrate_llc_task semantics in active load balance without passing
   migration_type across the stopper, which affects delayed-dequeue tasks.
 - Link to V1: https://lore.kernel.org/all/20260801122252.2476258-1-wanglu.priv@gmail.com/

 kernel/sched/fair.c | 56 ++++++++++++++++++++++++++++++++++++++++-----
 1 file changed, 50 insertions(+), 6 deletions(-)

diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
index d78467ec6..37a470c41 100644
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -10254,6 +10254,7 @@ enum migration_type {
 #define LBF_SOME_PINNED	0x08
 #define LBF_ACTIVE_LB	0x10
 #define LBF_LLC_PINNED	0x20
+#define LBF_ACTIVE_LB_LLC	0x40
 
 struct lb_env {
 	struct sched_domain	*sd;
@@ -10645,6 +10646,20 @@ alb_break_llc(struct lb_env *env)
 	return false;
 }
 
+/*
+ * Returns true if p's preferred LLC does not match the destination CPU
+ * under migrate_llc_task semantics. Passive LB passes migrate_llc_task
+ * in migration_type, while active LB carries it in LBF_ACTIVE_LB_LLC.
+ */
+static inline bool
+migrate_llc_task_wrong_dst(struct task_struct *p, struct lb_env *env)
+{
+	return sched_cache_enabled() &&
+	       (env->migration_type == migrate_llc_task ||
+		env->flags & LBF_ACTIVE_LB_LLC) &&
+	       READ_ONCE(p->preferred_llc) != llc_id(env->dst_cpu);
+}
+
 /*
  * Check if migrating task p from env->src_cpu to
  * env->dst_cpu breaks LLC localiy.
@@ -10673,8 +10688,7 @@ static bool migrate_degrades_llc(struct task_struct *p, struct lb_env *env)
 	 * run on env->dst_cpu, skip the tasks do not prefer
 	 * env->dst_cpu, and find the one that prefers.
 	 */
-	if (env->migration_type == migrate_llc_task &&
-	    READ_ONCE(p->preferred_llc) != llc_id(env->dst_cpu))
+	if (migrate_llc_task_wrong_dst(p, env))
 		return true;
 
 	if (can_migrate_llc_task(env->src_cpu,
@@ -10697,6 +10711,12 @@ alb_break_llc(struct lb_env *env)
 	return false;
 }
 
+static inline bool
+migrate_llc_task_wrong_dst(struct task_struct *p, struct lb_env *env)
+{
+	return false;
+}
+
 static inline bool
 migrate_degrades_llc(struct task_struct *p, struct lb_env *env)
 {
@@ -10796,7 +10816,7 @@ int can_migrate_task(struct task_struct *p, struct lb_env *env)
 	 * 4) too many balance attempts have failed.
 	 */
 	if (env->flags & LBF_ACTIVE_LB)
-		return 1;
+		return !migrate_llc_task_wrong_dst(p, env);
 
 	degrades = migrate_degrades_locality(p, env);
 	if (!degrades) {
@@ -13156,6 +13176,20 @@ static int need_active_balance(struct lb_env *env)
 }
 
 static int active_load_balance_cpu_stop(void *data);
+static int active_load_balance_llc_cpu_stop(void *data);
+
+/*
+ * migration_type is checked elsewhere to decide migration policy, so
+ * it shouldn't be repurposed just to flag an LLC-directed active
+ * balance across the stopper. Pick the callback here instead.
+ */
+static inline cpu_stop_fn_t alb_stop_fn(struct lb_env *env)
+{
+	if (env->migration_type == migrate_llc_task)
+		return active_load_balance_llc_cpu_stop;
+
+	return active_load_balance_cpu_stop;
+}
 
 static int should_we_balance(struct lb_env *env)
 {
@@ -13492,7 +13526,7 @@ static int sched_balance_rq(int this_cpu, struct rq *this_rq,
 			raw_spin_rq_unlock_irqrestore(busiest, flags);
 			if (active_balance) {
 				stop_one_cpu_nowait(cpu_of(busiest),
-					active_load_balance_cpu_stop, busiest,
+					alb_stop_fn(&env), busiest,
 					&busiest->active_balance_work);
 			}
 			preempt_enable();
@@ -13603,7 +13637,7 @@ update_next_balance(struct sched_domain *sd, unsigned long *next_balance)
  * least 1 task to be running on each physical CPU where possible, and
  * avoids physical / logical imbalances.
  */
-static int active_load_balance_cpu_stop(void *data)
+static int __active_load_balance_cpu_stop(void *data, unsigned int lb_flags)
 {
 	struct rq *busiest_rq = data;
 	int busiest_cpu = cpu_of(busiest_rq);
@@ -13653,7 +13687,7 @@ static int active_load_balance_cpu_stop(void *data)
 			.src_cpu	= busiest_rq->cpu,
 			.src_rq		= busiest_rq,
 			.idle		= CPU_IDLE,
-			.flags		= LBF_ACTIVE_LB,
+			.flags		= LBF_ACTIVE_LB | lb_flags,
 		};
 
 		schedstat_inc(sd->alb_count);
@@ -13681,6 +13715,16 @@ static int active_load_balance_cpu_stop(void *data)
 	return 0;
 }
 
+static int active_load_balance_cpu_stop(void *data)
+{
+	return __active_load_balance_cpu_stop(data, 0);
+}
+
+static int active_load_balance_llc_cpu_stop(void *data)
+{
+	return __active_load_balance_cpu_stop(data, LBF_ACTIVE_LB_LLC);
+}
+
 /*
  * Scale the max sched_balance_rq interval with the number of CPUs in the system.
  * This trades load-balance latency on larger machines for less cross talk.
-- 
2.43.0


      parent reply	other threads:[~2026-08-09 10:54 UTC|newest]

Thread overview: 17+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-08-01 12:22 [PATCH] sched/cache: honor migrate_llc_task semantics in active load balance Lu Wang
2026-08-03  4:20 ` Chen, Yu C
2026-08-03 10:02   ` Lu Wang
2026-08-04  0:10     ` Tim Chen
2026-08-04  8:17       ` Chen, Yu C
2026-08-04 15:07         ` Lu Wang
2026-08-04 19:42           ` Tim Chen
2026-08-05  2:38             ` wanglu15
2026-08-05 16:04               ` Tim Chen
2026-08-05 16:43                 ` Chen, Yu C
2026-08-06 16:21                   ` Tim Chen
2026-08-04  8:30       ` Lu Wang
2026-08-06 15:35 ` Chen, Yu C
2026-08-06 17:22   ` Tim Chen
2026-08-07  6:48     ` Chen, Yu C
2026-08-07  9:58       ` Lu Wang
2026-08-09 10:53 ` Lu Wang [this message]

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260809105343.1189051-1-wanglu.priv@gmail.com \
    --to=wanglu.priv@gmail.com \
    --cc=bsegall@google.com \
    --cc=chen.yu@linux.dev \
    --cc=dietmar.eggemann@arm.com \
    --cc=juri.lelli@redhat.com \
    --cc=kprateek.nayak@amd.com \
    --cc=linux-kernel@vger.kernel.org \
    --cc=mgorman@suse.de \
    --cc=mingo@redhat.com \
    --cc=peterz@infradead.org \
    --cc=rostedt@goodmis.org \
    --cc=tim.c.chen@linux.intel.com \
    --cc=vincent.guittot@linaro.org \
    --cc=vschneid@redhat.com \
    --cc=yu.c.chen@intel.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.