Linux Power Management development
 help / color / mirror / Atom feed
From: Vincent Guittot <vincent.guittot@linaro.org>
To: mingo@redhat.com, peterz@infradead.org, juri.lelli@redhat.com,
	dietmar.eggemann@arm.com, rostedt@goodmis.org,
	bsegall@google.com, mgorman@suse.de, vschneid@redhat.com,
	kprateek.nayak@amd.com, linux-kernel@vger.kernel.org,
	lukasz.luba@arm.com, rafael@kernel.org, linux-pm@vger.kernel.org,
	tj@kernel.org, void@manifault.com, arighi@nvidia.com,
	changwoo@igalia.com, sched-ext@lists.linux.dev
Cc: qyousef@layalina.io, christian.loehle@arm.com,
	pierre.gondois@arm.com, sshegde@linux.ibm.com,
	Vincent Guittot <vincent.guittot@linaro.org>
Subject: [PATCH 01/18 v2] sched/eevdf: Decay positive lag of sleeping entities
Date: Fri,  2 Oct 2026 17:43:58 +0200	[thread overview]
Message-ID: <20261002154415.2270586-2-vincent.guittot@linaro.org> (raw)
In-Reply-To: <20261002154415.2270586-1-vincent.guittot@linaro.org>

Similarly to delayed dequeue that enables an entity to decay its negative
lag while sleeping, a task should not keep a positive lag forever.

The sleep duration and the weight of the entity is used to decay the
positive lag at wakeup.

Signed-off-by: Vincent Guittot <vincent.guittot@linaro.org>
---
 kernel/sched/fair.c | 51 +++++++++++++++++++++++++++++++++++++++++++--
 1 file changed, 49 insertions(+), 2 deletions(-)

diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
index 03206e15e6fe..8cda1d39b037 100644
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -897,6 +897,51 @@ bool update_entity_lag(struct cfs_rq *cfs_rq, struct sched_entity *se)
 	return avruntime - vlag != se->vruntime;
 }
 
+static inline unsigned long cfs_rq_load_avg(struct cfs_rq *cfs_rq);
+
+static __always_inline
+void decay_entity_lag(struct cfs_rq *cfs_rq, struct sched_entity *se, int flags)
+{
+	s64 delta_exec, vlag = se->vlag;
+	unsigned long cfs_load;
+	struct rq *rq;
+
+	WARN_ON_ONCE(se->on_rq);
+
+	/* Negative lag implies delayed dequeue */
+	if (vlag <= 0)
+		return;
+
+	rq = rq_of(cfs_rq);
+
+	if (flags & ENQUEUE_MIGRATED)
+		return;
+
+	/* Compute sleep time */
+	delta_exec = rq_clock_task(rq) - se->exec_start;
+	if (unlikely(delta_exec <= 0))
+		return;
+
+	/* For anything above ~4 seconds, save computation and clear the lag */
+	if (unlikely(delta_exec >> 32)) {
+		se->vlag = 0;
+		return;
+	}
+
+	cfs_load = cfs_rq_load_avg(cfs_rq);
+	if (cfs_load) {
+		unsigned long weight = scale_load_down(se->h_load.weight);
+
+		delta_exec *= weight;
+		delta_exec = div64_long(delta_exec, cfs_load + weight);
+	}
+
+	vlag -= calc_delta_fair(delta_exec, se);
+
+	/* vlag can't become neg while sleeping */
+	se->vlag = max(0, vlag);
+}
+
 /*
  * Entity is eligible once it received less service than it ought to have,
  * eg. lag >= 0.
@@ -7987,7 +8032,7 @@ static void
 enqueue_task_fair(struct rq *rq, struct task_struct *p, int flags)
 {
 	int rq_h_nr_queued = rq->cfs.h_nr_queued;
-	int task_new = !(flags & ENQUEUE_WAKEUP);
+	int task_wake = flags & ENQUEUE_WAKEUP;
 	struct sched_entity *se = &p->se;
 	struct cfs_rq *cfs_rq = &rq->cfs;
 	unsigned long weight;
@@ -8019,6 +8064,8 @@ enqueue_task_fair(struct rq *rq, struct task_struct *p, int flags)
 	if (p->in_iowait)
 		cpufreq_update_util(rq, SCHED_CPUFREQ_IOWAIT);
 
+	if (task_wake)
+		decay_entity_lag(cfs_rq, se, flags);
 
 	if (se->on_rq && se->sched_delayed)
 		requeue_delayed_entity(cfs_rq, se);
@@ -8048,7 +8095,7 @@ enqueue_task_fair(struct rq *rq, struct task_struct *p, int flags)
 	 * into account, but that is not straightforward to implement,
 	 * and the following generally works well enough in practice.
 	 */
-	if (!task_new)
+	if (task_wake)
 		check_update_overutilized_status(rq);
 
 	assert_list_leaf_cfs_rq(rq);
-- 
2.53.0


  reply	other threads:[~2026-10-02 15:45 UTC|newest]

Thread overview: 43+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-10-02 15:43 [PATCH 00/18 v2] Improving latency of short slice tasks Vincent Guittot
2026-10-02 15:43 ` Vincent Guittot [this message]
2026-10-02 15:43 ` [PATCH 02/18] sched/eevdf: Reset lag when waking up on idle cpu Vincent Guittot
2026-10-04 17:30   ` Kayra Cizmeci
2026-10-09 13:35     ` Vincent Guittot
2026-10-02 15:44 ` [PATCH 03/18 v2] sched/eevdf: Add per cpu cached min_slice Vincent Guittot
2026-10-02 15:44 ` [PATCH 04/18 v2] sched/eevdf: Compare min slice during wake_affine Vincent Guittot
2026-10-05 15:58   ` Kayra Cizmeci
2026-10-02 15:44 ` [PATCH 05/18 v2] sched/eevdf: Add min slice check when selecting CPU Vincent Guittot
2026-10-04 19:18   ` Kayra Cizmeci
2026-10-02 15:44 ` [PATCH 06/18 v2] sched/fair: Prepare select_task_rq_fair() to be called for new cases Vincent Guittot
2026-10-06 22:25   ` Tim Chen
2026-10-09 13:36     ` Vincent Guittot
2026-10-02 15:44 ` [PATCH 07/18] sched/fair: Add push task mechanism for fair Vincent Guittot
2026-10-07  2:45   ` Chen Yu
2026-10-09 13:38     ` Vincent Guittot
2026-10-09 14:31     ` Vincent Guittot
2026-10-02 15:44 ` [PATCH 08/18] sched/fair: Optimize " Vincent Guittot
2026-10-02 15:44 ` [PATCH 09/18 v2] sched/core: Add rq flag to tick parameters Vincent Guittot
2026-10-02 15:44 ` [PATCH 10/18 v2] sched/fair: Add force push task mechanism for fair Vincent Guittot
2026-10-06 19:24   ` Kayra Cizmeci
2026-10-09 14:01     ` Vincent Guittot
2026-10-09  5:53   ` Kayra Cizmeci
2026-10-09 14:04     ` Vincent Guittot
2026-10-09 14:47       ` Vincent Guittot
2026-10-02 15:44 ` [PATCH 11/18 v2] sched/fair: Support not wakeup case in select_idle_sibling Vincent Guittot
2026-10-07 17:55   ` Tim Chen
2026-10-09 14:06     ` Vincent Guittot
2026-10-02 15:44 ` [PATCH 12/18 v2] sched/eevdf: Try to push short slice task on a better CPU Vincent Guittot
2026-10-02 15:44 ` [PATCH 13/18 v2] sched/eevdf: Push short slice task that are not picked Vincent Guittot
2026-10-02 15:44 ` [PATCH 14/18 v2] sched/fair: Enable push task for preempt short Vincent Guittot
2026-10-09 11:16   ` Kayra Cizmeci
2026-10-09 14:07     ` Vincent Guittot
2026-10-02 15:44 ` [PATCH 15/18 v2] energy model: Add a get previous state function Vincent Guittot
2026-10-02 15:44 ` [PATCH 16/18 v2] sched/fair: Rework feec() to use cost instead of spare capacity Vincent Guittot
2026-10-02 15:44 ` [PATCH 17/18 v2] energy model: Remove unused em_cpu_energy() Vincent Guittot
2026-10-02 15:44 ` [PATCH 18/18 v2] sched/fair: Take into account slice in EAS Vincent Guittot
2026-10-07 14:36   ` Kayra Cizmeci
2026-10-08 22:53   ` Tim Chen
2026-10-09  6:31     ` Kayra Cizmeci
2026-10-09 14:17     ` Vincent Guittot
2026-10-08 19:12 ` [PATCH 00/18 v2] Improving latency of short slice tasks Kayra Cizmeci
2026-10-09 14:56   ` Vincent Guittot

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20261002154415.2270586-2-vincent.guittot@linaro.org \
    --to=vincent.guittot@linaro.org \
    --cc=arighi@nvidia.com \
    --cc=bsegall@google.com \
    --cc=changwoo@igalia.com \
    --cc=christian.loehle@arm.com \
    --cc=dietmar.eggemann@arm.com \
    --cc=juri.lelli@redhat.com \
    --cc=kprateek.nayak@amd.com \
    --cc=linux-kernel@vger.kernel.org \
    --cc=linux-pm@vger.kernel.org \
    --cc=lukasz.luba@arm.com \
    --cc=mgorman@suse.de \
    --cc=mingo@redhat.com \
    --cc=peterz@infradead.org \
    --cc=pierre.gondois@arm.com \
    --cc=qyousef@layalina.io \
    --cc=rafael@kernel.org \
    --cc=rostedt@goodmis.org \
    --cc=sched-ext@lists.linux.dev \
    --cc=sshegde@linux.ibm.com \
    --cc=tj@kernel.org \
    --cc=void@manifault.com \
    --cc=vschneid@redhat.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox