The Linux Kernel Mailing List
 help / color / mirror / Atom feed
* [PATCH v3 sched_ext/for-7.3] sched_ext: Bound per-task reenqueues and eject the owning scheduler
@ 2026-07-26 19:49 Tejun Heo
  2026-07-26 20:20 ` Andrea Righi
  2026-07-26 21:48 ` [PATCH v3 sched_ext/for-7.3] sched_ext: Bound per-task reenqueues and eject the owning scheduler Tejun Heo
  0 siblings, 2 replies; 4+ messages in thread
From: Tejun Heo @ 2026-07-26 19:49 UTC (permalink / raw)
  To: David Vernet, Andrea Righi, Changwoo Min
  Cc: sched-ext, Emil Tsalapatis, linux-kernel

Unlike local reenqueues, cap rejections have no repeat limit. A
malfunctioning scheduler can keep re-inserting a task to a cid it lacks caps
on, cycling the task through reject and reenqueue. This was assumed safe
because a task that never runs trips the stall watchdog. However, the
reenqueue irq_work re-arms itself and outranks the timer vector, blocking
everything else on the CPU including stall detection and recovery, until the
NMI hardlockup detector fires.

Local reenqueues already have a repeat cap, SCX_REENQ_LOCAL_MAX_REPEAT,
which needs generalizing to cover all reenqueues. It also has an attribution
problem. Counted per-cpu on root, it tears down the whole hierarchy even
when a sub-scheduler caused the repeated reenqueues.

Generalize by bounding every reenqueue with one per-task counter. reenq_cnt
is bumped in scx_do_enqueue_task() on each SCX_ENQ_REENQ, the single path
every reenqueue producer passes through, and cleared in clr_task_runnable()
when the task is picked to run and in scx_disable_task() when it leaves the
scheduler's control. Past SCX_REENQ_MAX_REPEAT the task's owning scheduler
is ejected with a new SCX_EXIT_ERROR_REENQ and the task is left stranded to
be picked up during sched exit.

The SCX_EV_REENQ_LOCAL_REPEAT event becomes SCX_EV_REENQ_REPEAT, counting
repeat reenqueues from all sources.

v2: Count SCX_EV_REENQ_REPEAT only when a reenqueue leads to another
    reenqueue, not on every reenqueue.

v3: - Also clear reenq_cnt in scx_disable_task() so that the count doesn't
      carry over to the next owner across sched class switches, scheduler
      replacement or sub-scheduler rehoming (Andrea Righi).

    - Update the stale SCX_EV_REENQ_LOCAL_REPEAT references in sched-ext.rst
      (Andrea Righi).

Signed-off-by: Tejun Heo <tj@kernel.org>
---
 Documentation/scheduler/sched-ext.rst |    8 ++--
 include/linux/sched/ext.h             |    1 
 kernel/sched/ext/ext.c                |   58 +++++++++++++++++++---------------
 kernel/sched/ext/internal.h           |   19 ++++-------
 kernel/sched/ext/sub.c                |    6 +--
 kernel/sched/ext/types.h              |    2 -
 kernel/sched/sched.h                  |    1 
 7 files changed, 51 insertions(+), 44 deletions(-)

--- a/Documentation/scheduler/sched-ext.rst
+++ b/Documentation/scheduler/sched-ext.rst
@@ -106,7 +106,7 @@ counters. Each counter occupies one ``na
     SCX_EV_ENQ_SKIP_EXITING 0
     SCX_EV_ENQ_SKIP_MIGRATION_DISABLED 0
     SCX_EV_REENQ_IMMED 0
-    SCX_EV_REENQ_LOCAL_REPEAT 0
+    SCX_EV_REENQ_REPEAT 0
     SCX_EV_REFILL_SLICE_DFL 456789
     SCX_EV_BYPASS_DURATION 0
     SCX_EV_BYPASS_DISPATCH 0
@@ -129,9 +129,9 @@ The counters are described in ``kernel/s
   ``SCX_OPS_ENQ_MIGRATION_DISABLED`` is not set).
 * ``SCX_EV_REENQ_IMMED``: a task dispatched with ``SCX_ENQ_IMMED`` was
   re-enqueued because the target CPU was not available for immediate execution.
-* ``SCX_EV_REENQ_LOCAL_REPEAT``: a reenqueue of the local DSQ triggered
-  another reenqueue; recurring counts indicate incorrect ``SCX_ENQ_REENQ``
-  handling in the BPF scheduler.
+* ``SCX_EV_REENQ_REPEAT``: a reenqueue led to another reenqueue without the
+  task running in between; recurring counts indicate that the BPF scheduler
+  keeps re-deciding placements it can't honor.
 * ``SCX_EV_REFILL_SLICE_DFL``: a task's time slice was refilled with the
   default value (``SCX_SLICE_DFL``).
 * ``SCX_EV_BYPASS_DURATION``: total nanoseconds spent in bypass mode.
--- a/include/linux/sched/ext.h
+++ b/include/linux/sched/ext.h
@@ -198,6 +198,7 @@ struct sched_ext_entity {
 	u32			dsq_flags;	/* protected by DSQ lock */
 	u32			flags;		/* protected by rq lock */
 	u32			weight;
+	u32			reenq_cnt;	/* reenqueues since last run */
 	s32			sticky_cpu;
 	s32			holding_cpu;
 	s32			selected_cpu;
--- a/kernel/sched/ext/ext.c
+++ b/kernel/sched/ext/ext.c
@@ -1904,6 +1904,24 @@ void scx_do_enqueue_task(struct rq *rq,
 	p->scx.flags &= ~SCX_TASK_IMMED;
 
 	/*
+	 * A task reenqueued too many times without running means the scheduler
+	 * keeps re-deciding a placement it can't honor, e.g. re-inserting to a
+	 * cid it lacks caps on. Eject the owning scheduler and strand the task
+	 * to be picked up during sched exit.
+	 */
+	if (enq_flags & SCX_ENQ_REENQ) {
+		if (++p->scx.reenq_cnt > 1)
+			__scx_add_event(sch, SCX_EV_REENQ_REPEAT, 1);
+
+		if (unlikely(p->scx.reenq_cnt > SCX_REENQ_MAX_REPEAT)) {
+			__scx_exit(sch, SCX_EXIT_ERROR_REENQ, 0, cpu_of(rq),
+				   "%s[%d] reenqueued %u times without running",
+				   p->comm, p->pid, p->scx.reenq_cnt);
+			return;
+		}
+	}
+
+	/*
 	 * If !scx_rq_online(), we already told the BPF scheduler that the CPU
 	 * is offline and are just running the hotplug path. Don't bother the
 	 * BPF scheduler.
@@ -2025,8 +2043,10 @@ static void clr_task_runnable(struct tas
 {
 	list_del_init(&p->scx.runnable_node);
 	WRITE_ONCE(p->scx.runnable_cpu, -1);
-	if (reset_runnable_at)
+	if (reset_runnable_at) {
 		p->scx.flags |= SCX_TASK_RESET_RUNNABLE_AT;
+		p->scx.reenq_cnt = 0;
+	}
 }
 
 static void enqueue_task_scx(struct rq *rq, struct task_struct *p, int core_enq_flags)
@@ -3669,6 +3689,7 @@ static void scx_disable_task(struct scx_
 	 */
 	p->scx.dsq_vtime = 0;
 	set_task_slice(p, 0);
+	p->scx.reenq_cnt = 0;
 
 	/*
 	 * Verify the task is not in BPF scheduler's custody. If flag
@@ -4066,8 +4087,8 @@ static void process_ddsp_deferred_locals
  * Reenqueued tasks go through ops.enqueue() with %SCX_ENQ_REENQ |
  * %SCX_TASK_REENQ_IMMED. If the BPF scheduler dispatches back to the same local
  * DSQ with %SCX_ENQ_IMMED while the CPU is still unavailable, this triggers
- * another reenq cycle. Repetitions are bounded by %SCX_REENQ_LOCAL_MAX_REPEAT
- * in process_deferred_reenq_locals().
+ * another reenq cycle. Repetitions are bounded by %SCX_REENQ_MAX_REPEAT in
+ * scx_do_enqueue_task(), which ejects the task's owning scheduler.
  */
 static bool local_task_should_reenq(struct rq *rq, struct task_struct *p,
 				    u64 *reenq_flags, u32 *reason)
@@ -4175,14 +4196,16 @@ static u32 reenq_local(struct scx_sched
 
 static void process_deferred_reenq_locals(struct rq *rq)
 {
-	u64 seq = ++rq->scx.deferred_reenq_locals_seq;
-
 	lockdep_assert_rq_held(rq);
 
+	/*
+	 * A task can be re-queued within this loop when a reenqueued task
+	 * bounces straight back to the local DSQ. That recursion is bounded by
+	 * the per-task reenqueue cap in scx_do_enqueue_task().
+	 */
 	while (true) {
 		struct scx_sched *sch;
 		u64 reenq_flags;
-		bool skip = false;
 
 		scoped_guard (raw_spinlock, &rq->scx.deferred_reenq_lock) {
 			struct scx_deferred_reenq_local *drl =
@@ -4201,27 +4224,12 @@ static void process_deferred_reenq_local
 			reenq_flags = drl->flags;
 			WRITE_ONCE(drl->flags, 0);
 			list_del_init(&drl->node);
-
-			if (likely(drl->seq != seq)) {
-				drl->seq = seq;
-				drl->cnt = 0;
-			} else {
-				if (unlikely(++drl->cnt > SCX_REENQ_LOCAL_MAX_REPEAT)) {
-					scx_error(sch, "SCX_ENQ_REENQ on SCX_DSQ_LOCAL repeated %u times",
-						  drl->cnt);
-					skip = true;
-				}
-
-				__scx_add_event(sch, SCX_EV_REENQ_LOCAL_REPEAT, 1);
-			}
 		}
 
-		if (!skip) {
-			/* see schedule_dsq_reenq() */
-			smp_mb();
+		/* see schedule_dsq_reenq() */
+		smp_mb();
 
-			reenq_local(sch, rq, reenq_flags);
-		}
+		reenq_local(sch, rq, reenq_flags);
 	}
 }
 
@@ -5941,6 +5949,8 @@ static const char *scx_exit_reason(enum
 		return "scx_bpf_error";
 	case SCX_EXIT_ERROR_STALL:
 		return "runnable task stall";
+	case SCX_EXIT_ERROR_REENQ:
+		return "reenqueue limit";
 	default:
 		return "<UNKNOWN>";
 	}
--- a/kernel/sched/ext/internal.h
+++ b/kernel/sched/ext/internal.h
@@ -56,6 +56,7 @@ enum scx_exit_kind {
 	SCX_EXIT_ERROR = 1024,	/* runtime error, error msg contains details */
 	SCX_EXIT_ERROR_BPF,	/* ERROR but triggered through scx_bpf_error() */
 	SCX_EXIT_ERROR_STALL,	/* watchdog detected stalled runnable tasks */
+	SCX_EXIT_ERROR_REENQ,	/* task hit reenqueue limit without running */
 };
 
 /*
@@ -1119,15 +1120,13 @@ struct scx_event_stats {
 	s64		SCX_EV_REENQ_IMMED;
 
 	/*
-	 * The number of times a reenq of local DSQ caused another reenq of
-	 * local DSQ. This can happen when %SCX_ENQ_IMMED races against a higher
-	 * priority class task even if the BPF scheduler always satisfies the
-	 * prerequisites for %SCX_ENQ_IMMED at the time of enqueue. However,
-	 * that scenario is very unlikely and this count going up regularly
-	 * indicates that the BPF scheduler is handling %SCX_ENQ_REENQ
-	 * incorrectly causing recursive reenqueues.
+	 * The number of times a reenqueue (%SCX_ENQ_REENQ) led to another
+	 * reenqueue without the task running in between. This count climbing
+	 * rapidly indicates that the BPF scheduler keeps re-deciding placements
+	 * it can't honor. A single task reenqueued more than
+	 * %SCX_REENQ_MAX_REPEAT times gets its owning scheduler ejected.
 	 */
-	s64		SCX_EV_REENQ_LOCAL_REPEAT;
+	s64		SCX_EV_REENQ_REPEAT;
 
 	/*
 	 * Total number of times a task's time slice was refilled with the
@@ -1221,7 +1220,7 @@ struct scx_event_stats {
 	SCX_EVENT(SCX_EV_ENQ_SKIP_EXITING);				\
 	SCX_EVENT(SCX_EV_ENQ_SKIP_MIGRATION_DISABLED);			\
 	SCX_EVENT(SCX_EV_REENQ_IMMED);					\
-	SCX_EVENT(SCX_EV_REENQ_LOCAL_REPEAT);				\
+	SCX_EVENT(SCX_EV_REENQ_REPEAT);					\
 	SCX_EVENT(SCX_EV_REFILL_SLICE_DFL);				\
 	SCX_EVENT(SCX_EV_SLICE_CLAMPED);				\
 	SCX_EVENT(SCX_EV_SLICE_DENIED);					\
@@ -1260,8 +1259,6 @@ struct scx_dsp_ctx {
 struct scx_deferred_reenq_local {
 	struct list_head	node;
 	u64			flags;
-	u64			seq;
-	u32			cnt;
 };
 
 struct scx_sched_pcpu {
--- a/kernel/sched/ext/sub.c
+++ b/kernel/sched/ext/sub.c
@@ -319,9 +319,9 @@ bool scx_task_reenq_on_cap_revoke(struct
  * Drain @rq->scx.reject_dsq, reenqueueing each task so the BPF re-decides
  * from p->scx.reenq_reason_*.
  *
- * A task can be re-rejected repeatedly, and there's no repeat limit here.
- * Rejection can't happen for root, and sub-scheds can be safely ejected after
- * triggering the stall watchdog.
+ * A task can be re-rejected repeatedly. The reenqueue is bounded per task in
+ * scx_do_enqueue_task(), which ejects the owning sub past SCX_REENQ_MAX_REPEAT.
+ * Rejection can't happen for root.
  */
 void scx_reenq_reject(struct rq *rq)
 {
--- a/kernel/sched/ext/types.h
+++ b/kernel/sched/ext/types.h
@@ -41,7 +41,7 @@ enum scx_consts {
 	SCX_BYPASS_LB_MIN_DELTA_DIV	= 4,
 	SCX_BYPASS_LB_BATCH		= 256,
 
-	SCX_REENQ_LOCAL_MAX_REPEAT	= 256,
+	SCX_REENQ_MAX_REPEAT		= 256,
 
 	SCX_SUB_MAX_DEPTH		= 4,
 };
--- a/kernel/sched/sched.h
+++ b/kernel/sched/sched.h
@@ -823,7 +823,6 @@ struct scx_rq {
 	struct list_head	sched_pcpus_to_kick;	/* see kick_cpus_irq_workfn() */
 
 	raw_spinlock_t		deferred_reenq_lock;
-	u64			deferred_reenq_locals_seq;
 	struct list_head	deferred_reenq_locals;	/* scheds requesting reenq of local DSQ */
 	struct list_head	deferred_reenq_users;	/* user DSQs requesting reenq */
 	struct balance_callback	deferred_bal_cb;

^ permalink raw reply	[flat|nested] 4+ messages in thread

* Re: [PATCH v3 sched_ext/for-7.3] sched_ext: Bound per-task reenqueues and eject the owning scheduler
  2026-07-26 19:49 [PATCH v3 sched_ext/for-7.3] sched_ext: Bound per-task reenqueues and eject the owning scheduler Tejun Heo
@ 2026-07-26 20:20 ` Andrea Righi
  2026-07-26 21:48   ` [PATCH sched_ext/for-7.3] tools/sched_ext/include: Regenerate enum_defs.autogen.h Tejun Heo
  2026-07-26 21:48 ` [PATCH v3 sched_ext/for-7.3] sched_ext: Bound per-task reenqueues and eject the owning scheduler Tejun Heo
  1 sibling, 1 reply; 4+ messages in thread
From: Andrea Righi @ 2026-07-26 20:20 UTC (permalink / raw)
  To: Tejun Heo
  Cc: David Vernet, Changwoo Min, sched-ext, Emil Tsalapatis,
	linux-kernel

Hi Tejun,

On Sun, Jul 26, 2026 at 09:49:54AM -1000, Tejun Heo wrote:
> Unlike local reenqueues, cap rejections have no repeat limit. A
> malfunctioning scheduler can keep re-inserting a task to a cid it lacks caps
> on, cycling the task through reject and reenqueue. This was assumed safe
> because a task that never runs trips the stall watchdog. However, the
> reenqueue irq_work re-arms itself and outranks the timer vector, blocking
> everything else on the CPU including stall detection and recovery, until the
> NMI hardlockup detector fires.
> 
> Local reenqueues already have a repeat cap, SCX_REENQ_LOCAL_MAX_REPEAT,
> which needs generalizing to cover all reenqueues. It also has an attribution
> problem. Counted per-cpu on root, it tears down the whole hierarchy even
> when a sub-scheduler caused the repeated reenqueues.
> 
> Generalize by bounding every reenqueue with one per-task counter. reenq_cnt
> is bumped in scx_do_enqueue_task() on each SCX_ENQ_REENQ, the single path
> every reenqueue producer passes through, and cleared in clr_task_runnable()
> when the task is picked to run and in scx_disable_task() when it leaves the
> scheduler's control. Past SCX_REENQ_MAX_REPEAT the task's owning scheduler
> is ejected with a new SCX_EXIT_ERROR_REENQ and the task is left stranded to
> be picked up during sched exit.
> 
> The SCX_EV_REENQ_LOCAL_REPEAT event becomes SCX_EV_REENQ_REPEAT, counting
> repeat reenqueues from all sources.
> 
> v2: Count SCX_EV_REENQ_REPEAT only when a reenqueue leads to another
>     reenqueue, not on every reenqueue.
> 
> v3: - Also clear reenq_cnt in scx_disable_task() so that the count doesn't
>       carry over to the next owner across sched class switches, scheduler
>       replacement or sub-scheduler rehoming (Andrea Righi).
> 
>     - Update the stale SCX_EV_REENQ_LOCAL_REPEAT references in sched-ext.rst
>       (Andrea Righi).
> 
> Signed-off-by: Tejun Heo <tj@kernel.org>

Ack about counting all reenqueues.

One minor nit: tools/sched_ext/include/scx/enum_defs.autogen.h still defines
HAVE_SCX_REENQ_LOCAL_MAX_REPEAT, should be renamed to HAVE_SCX_REENQ_MAX_REPEAT.

Other than that, looks good to me.

Reviewed-by: Andrea Righi <arighi@nvidia.com>

Thanks,
-Andrea

> ---
>  Documentation/scheduler/sched-ext.rst |    8 ++--
>  include/linux/sched/ext.h             |    1 
>  kernel/sched/ext/ext.c                |   58 +++++++++++++++++++---------------
>  kernel/sched/ext/internal.h           |   19 ++++-------
>  kernel/sched/ext/sub.c                |    6 +--
>  kernel/sched/ext/types.h              |    2 -
>  kernel/sched/sched.h                  |    1 
>  7 files changed, 51 insertions(+), 44 deletions(-)
> 
> --- a/Documentation/scheduler/sched-ext.rst
> +++ b/Documentation/scheduler/sched-ext.rst
> @@ -106,7 +106,7 @@ counters. Each counter occupies one ``na
>      SCX_EV_ENQ_SKIP_EXITING 0
>      SCX_EV_ENQ_SKIP_MIGRATION_DISABLED 0
>      SCX_EV_REENQ_IMMED 0
> -    SCX_EV_REENQ_LOCAL_REPEAT 0
> +    SCX_EV_REENQ_REPEAT 0
>      SCX_EV_REFILL_SLICE_DFL 456789
>      SCX_EV_BYPASS_DURATION 0
>      SCX_EV_BYPASS_DISPATCH 0
> @@ -129,9 +129,9 @@ The counters are described in ``kernel/s
>    ``SCX_OPS_ENQ_MIGRATION_DISABLED`` is not set).
>  * ``SCX_EV_REENQ_IMMED``: a task dispatched with ``SCX_ENQ_IMMED`` was
>    re-enqueued because the target CPU was not available for immediate execution.
> -* ``SCX_EV_REENQ_LOCAL_REPEAT``: a reenqueue of the local DSQ triggered
> -  another reenqueue; recurring counts indicate incorrect ``SCX_ENQ_REENQ``
> -  handling in the BPF scheduler.
> +* ``SCX_EV_REENQ_REPEAT``: a reenqueue led to another reenqueue without the
> +  task running in between; recurring counts indicate that the BPF scheduler
> +  keeps re-deciding placements it can't honor.
>  * ``SCX_EV_REFILL_SLICE_DFL``: a task's time slice was refilled with the
>    default value (``SCX_SLICE_DFL``).
>  * ``SCX_EV_BYPASS_DURATION``: total nanoseconds spent in bypass mode.
> --- a/include/linux/sched/ext.h
> +++ b/include/linux/sched/ext.h
> @@ -198,6 +198,7 @@ struct sched_ext_entity {
>  	u32			dsq_flags;	/* protected by DSQ lock */
>  	u32			flags;		/* protected by rq lock */
>  	u32			weight;
> +	u32			reenq_cnt;	/* reenqueues since last run */
>  	s32			sticky_cpu;
>  	s32			holding_cpu;
>  	s32			selected_cpu;
> --- a/kernel/sched/ext/ext.c
> +++ b/kernel/sched/ext/ext.c
> @@ -1904,6 +1904,24 @@ void scx_do_enqueue_task(struct rq *rq,
>  	p->scx.flags &= ~SCX_TASK_IMMED;
>  
>  	/*
> +	 * A task reenqueued too many times without running means the scheduler
> +	 * keeps re-deciding a placement it can't honor, e.g. re-inserting to a
> +	 * cid it lacks caps on. Eject the owning scheduler and strand the task
> +	 * to be picked up during sched exit.
> +	 */
> +	if (enq_flags & SCX_ENQ_REENQ) {
> +		if (++p->scx.reenq_cnt > 1)
> +			__scx_add_event(sch, SCX_EV_REENQ_REPEAT, 1);
> +
> +		if (unlikely(p->scx.reenq_cnt > SCX_REENQ_MAX_REPEAT)) {
> +			__scx_exit(sch, SCX_EXIT_ERROR_REENQ, 0, cpu_of(rq),
> +				   "%s[%d] reenqueued %u times without running",
> +				   p->comm, p->pid, p->scx.reenq_cnt);
> +			return;
> +		}
> +	}
> +
> +	/*
>  	 * If !scx_rq_online(), we already told the BPF scheduler that the CPU
>  	 * is offline and are just running the hotplug path. Don't bother the
>  	 * BPF scheduler.
> @@ -2025,8 +2043,10 @@ static void clr_task_runnable(struct tas
>  {
>  	list_del_init(&p->scx.runnable_node);
>  	WRITE_ONCE(p->scx.runnable_cpu, -1);
> -	if (reset_runnable_at)
> +	if (reset_runnable_at) {
>  		p->scx.flags |= SCX_TASK_RESET_RUNNABLE_AT;
> +		p->scx.reenq_cnt = 0;
> +	}
>  }
>  
>  static void enqueue_task_scx(struct rq *rq, struct task_struct *p, int core_enq_flags)
> @@ -3669,6 +3689,7 @@ static void scx_disable_task(struct scx_
>  	 */
>  	p->scx.dsq_vtime = 0;
>  	set_task_slice(p, 0);
> +	p->scx.reenq_cnt = 0;
>  
>  	/*
>  	 * Verify the task is not in BPF scheduler's custody. If flag
> @@ -4066,8 +4087,8 @@ static void process_ddsp_deferred_locals
>   * Reenqueued tasks go through ops.enqueue() with %SCX_ENQ_REENQ |
>   * %SCX_TASK_REENQ_IMMED. If the BPF scheduler dispatches back to the same local
>   * DSQ with %SCX_ENQ_IMMED while the CPU is still unavailable, this triggers
> - * another reenq cycle. Repetitions are bounded by %SCX_REENQ_LOCAL_MAX_REPEAT
> - * in process_deferred_reenq_locals().
> + * another reenq cycle. Repetitions are bounded by %SCX_REENQ_MAX_REPEAT in
> + * scx_do_enqueue_task(), which ejects the task's owning scheduler.
>   */
>  static bool local_task_should_reenq(struct rq *rq, struct task_struct *p,
>  				    u64 *reenq_flags, u32 *reason)
> @@ -4175,14 +4196,16 @@ static u32 reenq_local(struct scx_sched
>  
>  static void process_deferred_reenq_locals(struct rq *rq)
>  {
> -	u64 seq = ++rq->scx.deferred_reenq_locals_seq;
> -
>  	lockdep_assert_rq_held(rq);
>  
> +	/*
> +	 * A task can be re-queued within this loop when a reenqueued task
> +	 * bounces straight back to the local DSQ. That recursion is bounded by
> +	 * the per-task reenqueue cap in scx_do_enqueue_task().
> +	 */
>  	while (true) {
>  		struct scx_sched *sch;
>  		u64 reenq_flags;
> -		bool skip = false;
>  
>  		scoped_guard (raw_spinlock, &rq->scx.deferred_reenq_lock) {
>  			struct scx_deferred_reenq_local *drl =
> @@ -4201,27 +4224,12 @@ static void process_deferred_reenq_local
>  			reenq_flags = drl->flags;
>  			WRITE_ONCE(drl->flags, 0);
>  			list_del_init(&drl->node);
> -
> -			if (likely(drl->seq != seq)) {
> -				drl->seq = seq;
> -				drl->cnt = 0;
> -			} else {
> -				if (unlikely(++drl->cnt > SCX_REENQ_LOCAL_MAX_REPEAT)) {
> -					scx_error(sch, "SCX_ENQ_REENQ on SCX_DSQ_LOCAL repeated %u times",
> -						  drl->cnt);
> -					skip = true;
> -				}
> -
> -				__scx_add_event(sch, SCX_EV_REENQ_LOCAL_REPEAT, 1);
> -			}
>  		}
>  
> -		if (!skip) {
> -			/* see schedule_dsq_reenq() */
> -			smp_mb();
> +		/* see schedule_dsq_reenq() */
> +		smp_mb();
>  
> -			reenq_local(sch, rq, reenq_flags);
> -		}
> +		reenq_local(sch, rq, reenq_flags);
>  	}
>  }
>  
> @@ -5941,6 +5949,8 @@ static const char *scx_exit_reason(enum
>  		return "scx_bpf_error";
>  	case SCX_EXIT_ERROR_STALL:
>  		return "runnable task stall";
> +	case SCX_EXIT_ERROR_REENQ:
> +		return "reenqueue limit";
>  	default:
>  		return "<UNKNOWN>";
>  	}
> --- a/kernel/sched/ext/internal.h
> +++ b/kernel/sched/ext/internal.h
> @@ -56,6 +56,7 @@ enum scx_exit_kind {
>  	SCX_EXIT_ERROR = 1024,	/* runtime error, error msg contains details */
>  	SCX_EXIT_ERROR_BPF,	/* ERROR but triggered through scx_bpf_error() */
>  	SCX_EXIT_ERROR_STALL,	/* watchdog detected stalled runnable tasks */
> +	SCX_EXIT_ERROR_REENQ,	/* task hit reenqueue limit without running */
>  };
>  
>  /*
> @@ -1119,15 +1120,13 @@ struct scx_event_stats {
>  	s64		SCX_EV_REENQ_IMMED;
>  
>  	/*
> -	 * The number of times a reenq of local DSQ caused another reenq of
> -	 * local DSQ. This can happen when %SCX_ENQ_IMMED races against a higher
> -	 * priority class task even if the BPF scheduler always satisfies the
> -	 * prerequisites for %SCX_ENQ_IMMED at the time of enqueue. However,
> -	 * that scenario is very unlikely and this count going up regularly
> -	 * indicates that the BPF scheduler is handling %SCX_ENQ_REENQ
> -	 * incorrectly causing recursive reenqueues.
> +	 * The number of times a reenqueue (%SCX_ENQ_REENQ) led to another
> +	 * reenqueue without the task running in between. This count climbing
> +	 * rapidly indicates that the BPF scheduler keeps re-deciding placements
> +	 * it can't honor. A single task reenqueued more than
> +	 * %SCX_REENQ_MAX_REPEAT times gets its owning scheduler ejected.
>  	 */
> -	s64		SCX_EV_REENQ_LOCAL_REPEAT;
> +	s64		SCX_EV_REENQ_REPEAT;
>  
>  	/*
>  	 * Total number of times a task's time slice was refilled with the
> @@ -1221,7 +1220,7 @@ struct scx_event_stats {
>  	SCX_EVENT(SCX_EV_ENQ_SKIP_EXITING);				\
>  	SCX_EVENT(SCX_EV_ENQ_SKIP_MIGRATION_DISABLED);			\
>  	SCX_EVENT(SCX_EV_REENQ_IMMED);					\
> -	SCX_EVENT(SCX_EV_REENQ_LOCAL_REPEAT);				\
> +	SCX_EVENT(SCX_EV_REENQ_REPEAT);					\
>  	SCX_EVENT(SCX_EV_REFILL_SLICE_DFL);				\
>  	SCX_EVENT(SCX_EV_SLICE_CLAMPED);				\
>  	SCX_EVENT(SCX_EV_SLICE_DENIED);					\
> @@ -1260,8 +1259,6 @@ struct scx_dsp_ctx {
>  struct scx_deferred_reenq_local {
>  	struct list_head	node;
>  	u64			flags;
> -	u64			seq;
> -	u32			cnt;
>  };
>  
>  struct scx_sched_pcpu {
> --- a/kernel/sched/ext/sub.c
> +++ b/kernel/sched/ext/sub.c
> @@ -319,9 +319,9 @@ bool scx_task_reenq_on_cap_revoke(struct
>   * Drain @rq->scx.reject_dsq, reenqueueing each task so the BPF re-decides
>   * from p->scx.reenq_reason_*.
>   *
> - * A task can be re-rejected repeatedly, and there's no repeat limit here.
> - * Rejection can't happen for root, and sub-scheds can be safely ejected after
> - * triggering the stall watchdog.
> + * A task can be re-rejected repeatedly. The reenqueue is bounded per task in
> + * scx_do_enqueue_task(), which ejects the owning sub past SCX_REENQ_MAX_REPEAT.
> + * Rejection can't happen for root.
>   */
>  void scx_reenq_reject(struct rq *rq)
>  {
> --- a/kernel/sched/ext/types.h
> +++ b/kernel/sched/ext/types.h
> @@ -41,7 +41,7 @@ enum scx_consts {
>  	SCX_BYPASS_LB_MIN_DELTA_DIV	= 4,
>  	SCX_BYPASS_LB_BATCH		= 256,
>  
> -	SCX_REENQ_LOCAL_MAX_REPEAT	= 256,
> +	SCX_REENQ_MAX_REPEAT		= 256,
>  
>  	SCX_SUB_MAX_DEPTH		= 4,
>  };
> --- a/kernel/sched/sched.h
> +++ b/kernel/sched/sched.h
> @@ -823,7 +823,6 @@ struct scx_rq {
>  	struct list_head	sched_pcpus_to_kick;	/* see kick_cpus_irq_workfn() */
>  
>  	raw_spinlock_t		deferred_reenq_lock;
> -	u64			deferred_reenq_locals_seq;
>  	struct list_head	deferred_reenq_locals;	/* scheds requesting reenq of local DSQ */
>  	struct list_head	deferred_reenq_users;	/* user DSQs requesting reenq */
>  	struct balance_callback	deferred_bal_cb;

^ permalink raw reply	[flat|nested] 4+ messages in thread

* Re: [PATCH v3 sched_ext/for-7.3] sched_ext: Bound per-task reenqueues and eject the owning scheduler
  2026-07-26 19:49 [PATCH v3 sched_ext/for-7.3] sched_ext: Bound per-task reenqueues and eject the owning scheduler Tejun Heo
  2026-07-26 20:20 ` Andrea Righi
@ 2026-07-26 21:48 ` Tejun Heo
  1 sibling, 0 replies; 4+ messages in thread
From: Tejun Heo @ 2026-07-26 21:48 UTC (permalink / raw)
  To: David Vernet, Andrea Righi, Changwoo Min
  Cc: sched-ext, Emil Tsalapatis, linux-kernel

Applied to sched_ext/for-7.3 with Andrea's Reviewed-by. The stale
enum_defs.autogen.h is refreshed by the following patch.

Thanks.

--
tejun

^ permalink raw reply	[flat|nested] 4+ messages in thread

* [PATCH sched_ext/for-7.3] tools/sched_ext/include: Regenerate enum_defs.autogen.h
  2026-07-26 20:20 ` Andrea Righi
@ 2026-07-26 21:48   ` Tejun Heo
  0 siblings, 0 replies; 4+ messages in thread
From: Tejun Heo @ 2026-07-26 21:48 UTC (permalink / raw)
  To: David Vernet, Andrea Righi, Changwoo Min
  Cc: sched-ext, Emil Tsalapatis, linux-kernel

Regenerate enum_defs.autogen.h from the current vmlinux.h to pick up the SCX
enum changes accumulated since the last regeneration, including the
SCX_REENQ_LOCAL_MAX_REPEAT to SCX_REENQ_MAX_REPEAT rename.

Reported-by: Andrea Righi <arighi@nvidia.com>
Link: https://lore.kernel.org/all/amZsEbZJdDgjstPF@gpd4/
Signed-off-by: Tejun Heo <tj@kernel.org>
---
Applied to sched_ext/for-7.3.

 .../sched_ext/include/scx/enum_defs.autogen.h | 51 +++++++++++++++----
 1 file changed, 42 insertions(+), 9 deletions(-)

diff --git a/tools/sched_ext/include/scx/enum_defs.autogen.h b/tools/sched_ext/include/scx/enum_defs.autogen.h
index da4b459820fd..0379eff117c9 100644
--- a/tools/sched_ext/include/scx/enum_defs.autogen.h
+++ b/tools/sched_ext/include/scx/enum_defs.autogen.h
@@ -7,9 +7,26 @@
 #ifndef __ENUM_DEFS_AUTOGEN_H__
 #define __ENUM_DEFS_AUTOGEN_H__
 
+#define HAVE_SCX_ARENA_MIN_ORDER
+#define HAVE_SCX_ARENA_GROW_PAGES
+#define HAVE___SCX_CAP_ENQ_IMMED
+#define HAVE___SCX_CAP_ENQ
+#define HAVE___SCX_CAP_PREEMPT
+#define HAVE___SCX_CAP_PERF
+#define HAVE___SCX_NR_CAPS
+#define HAVE___SCX_CAP_ALL
+#define HAVE_SCX_CAP_ENQ_IMMED
+#define HAVE_SCX_CAP_ENQ
+#define HAVE_SCX_CAP_PREEMPT
+#define HAVE_SCX_CAP_PERF
+#define HAVE_SCX_CAP_BASE
+#define HAVE_SCX_CAPS_REENQ_ON_LOSS
+#define HAVE_SCX_CID_SHARD_SIZE_DFL
+#define HAVE_SCX_CID_SHARD_MAX_CPUS
 #define HAVE_SCX_DSP_DFL_MAX_BATCH
 #define HAVE_SCX_DSP_MAX_LOOPS
 #define HAVE_SCX_WATCHDOG_MAX_TIMEOUT
+#define HAVE_SCX_TID_CHUNK
 #define HAVE_SCX_EXIT_BT_LEN
 #define HAVE_SCX_EXIT_MSG_LEN
 #define HAVE_SCX_EXIT_DUMP_DFL_LEN
@@ -20,7 +37,7 @@
 #define HAVE_SCX_BYPASS_LB_DONOR_PCT
 #define HAVE_SCX_BYPASS_LB_MIN_DELTA_DIV
 #define HAVE_SCX_BYPASS_LB_BATCH
-#define HAVE_SCX_REENQ_LOCAL_MAX_REPEAT
+#define HAVE_SCX_REENQ_MAX_REPEAT
 #define HAVE_SCX_SUB_MAX_DEPTH
 #define HAVE_SCX_CPU_PREEMPT_RT
 #define HAVE_SCX_CPU_PREEMPT_DL
@@ -35,6 +52,7 @@
 #define HAVE_SCX_DSQ_GLOBAL
 #define HAVE_SCX_DSQ_LOCAL
 #define HAVE_SCX_DSQ_BYPASS
+#define HAVE_SCX_DSQ_REJECT
 #define HAVE_SCX_DSQ_LOCAL_ON
 #define HAVE_SCX_DSQ_LOCAL_CPU_MASK
 #define HAVE_SCX_DSQ_ITER_REV
@@ -60,6 +78,7 @@
 #define HAVE_SCX_ENQ_DSQ_PRIQ
 #define HAVE_SCX_ENQ_NESTED
 #define HAVE_SCX_ENQ_GDSQ_FALLBACK
+#define HAVE_SCX_ENQ_IGNORE_CAPS
 #define HAVE_SCX_TASK_DSQ_ON_PRIQ
 #define HAVE_SCX_TASK_QUEUED
 #define HAVE_SCX_TASK_IN_CUSTODY
@@ -71,9 +90,11 @@
 #define HAVE_SCX_TASK_STATE_BITS
 #define HAVE_SCX_TASK_STATE_MASK
 #define HAVE_SCX_TASK_NONE
+#define HAVE_SCX_TASK_INIT_BEGIN
 #define HAVE_SCX_TASK_INIT
 #define HAVE_SCX_TASK_READY
 #define HAVE_SCX_TASK_ENABLED
+#define HAVE_SCX_TASK_DEAD
 #define HAVE_SCX_TASK_REENQ_REASON_SHIFT
 #define HAVE_SCX_TASK_REENQ_REASON_BITS
 #define HAVE_SCX_TASK_REENQ_REASON_MASK
@@ -81,6 +102,7 @@
 #define HAVE_SCX_TASK_REENQ_KFUNC
 #define HAVE_SCX_TASK_REENQ_IMMED
 #define HAVE_SCX_TASK_REENQ_PREEMPTED
+#define HAVE_SCX_TASK_REENQ_CAP
 #define HAVE_SCX_TASK_CURSOR
 #define HAVE_SCX_ECODE_RSN_HOTPLUG
 #define HAVE_SCX_ECODE_RSN_CGROUP_OFFLINE
@@ -93,17 +115,17 @@
 #define HAVE_SCX_EXIT_UNREG_KERN
 #define HAVE_SCX_EXIT_SYSRQ
 #define HAVE_SCX_EXIT_PARENT
+#define HAVE_SCX_EXIT_PARENT_KILL
 #define HAVE_SCX_EXIT_ERROR
 #define HAVE_SCX_EXIT_ERROR_BPF
 #define HAVE_SCX_EXIT_ERROR_STALL
-#define HAVE_SCX_KF_UNLOCKED
-#define HAVE_SCX_KF_CPU_RELEASE
-#define HAVE_SCX_KF_DISPATCH
-#define HAVE_SCX_KF_ENQUEUE
-#define HAVE_SCX_KF_SELECT_CPU
-#define HAVE_SCX_KF_REST
-#define HAVE___SCX_KF_RQ_LOCKED
-#define HAVE___SCX_KF_TERMINAL
+#define HAVE_SCX_EXIT_ERROR_REENQ
+#define HAVE_SCX_KF_ALLOW_UNLOCKED
+#define HAVE_SCX_KF_ALLOW_INIT_CIDS
+#define HAVE_SCX_KF_ALLOW_CPU_RELEASE
+#define HAVE_SCX_KF_ALLOW_DISPATCH
+#define HAVE_SCX_KF_ALLOW_ENQUEUE
+#define HAVE_SCX_KF_ALLOW_SELECT_CPU
 #define HAVE_SCX_KICK_IDLE
 #define HAVE_SCX_KICK_PREEMPT
 #define HAVE_SCX_KICK_WAIT
@@ -121,6 +143,7 @@
 #define HAVE_SCX_OPS_ALLOW_QUEUED_WAKEUP
 #define HAVE_SCX_OPS_BUILTIN_IDLE_PER_NODE
 #define HAVE_SCX_OPS_ALWAYS_ENQ_IMMED
+#define HAVE_SCX_OPS_TID_TO_TASK
 #define HAVE_SCX_OPS_ALL_FLAGS
 #define HAVE___SCX_OPS_INTERNAL_MASK
 #define HAVE_SCX_OPS_HAS_CPU_PREEMPT
@@ -136,6 +159,7 @@
 #define HAVE_SCX_SLICE_BYPASS
 #define HAVE_SCX_SLICE_INF
 #define HAVE_SCX_REENQ_ANY
+#define HAVE_SCX_REENQ_CAP_REVOKE
 #define HAVE___SCX_REENQ_FILTER_MASK
 #define HAVE___SCX_REENQ_USER_MASK
 #define HAVE_SCX_REENQ_TSR_RQ_OPEN
@@ -146,11 +170,20 @@
 #define HAVE_SCX_RQ_BAL_KEEP
 #define HAVE_SCX_RQ_CLK_VALID
 #define HAVE_SCX_RQ_BAL_CB_PENDING
+#define HAVE_SCX_RQ_SUB_IDLE_RENOTIFY
+#define HAVE_SCX_RQ_ROOT_IDLE_RENOTIFY
 #define HAVE_SCX_RQ_IN_WAKEUP
 #define HAVE_SCX_RQ_IN_BALANCE
 #define HAVE_SCX_SCHED_PCPU_BYPASSING
+#define HAVE_SCX_SLICE_OOB_DUR_BITS
+#define HAVE_SCX_SLICE_OOB_ID_BITS
+#define HAVE_SCX_SLICE_OOB_DUR_MASK
+#define HAVE_SCX_SLICE_OOB_ID_SHIFT
+#define HAVE_SCX_SLICE_OOB_ID_MASK
+#define HAVE_SCX_SLICE_OOB_PENDING
 #define HAVE_SCX_TG_ONLINE
 #define HAVE_SCX_TG_INITED
+#define HAVE_SCX_TG_SUB_INIT
 #define HAVE_SCX_WAKE_FORK
 #define HAVE_SCX_WAKE_TTWU
 #define HAVE_SCX_WAKE_SYNC
-- 
2.55.0


^ permalink raw reply related	[flat|nested] 4+ messages in thread

end of thread, other threads:[~2026-07-26 21:48 UTC | newest]

Thread overview: 4+ messages (download: mbox.gz follow: Atom feed
-- links below jump to the message on this page --
2026-07-26 19:49 [PATCH v3 sched_ext/for-7.3] sched_ext: Bound per-task reenqueues and eject the owning scheduler Tejun Heo
2026-07-26 20:20 ` Andrea Righi
2026-07-26 21:48   ` [PATCH sched_ext/for-7.3] tools/sched_ext/include: Regenerate enum_defs.autogen.h Tejun Heo
2026-07-26 21:48 ` [PATCH v3 sched_ext/for-7.3] sched_ext: Bound per-task reenqueues and eject the owning scheduler Tejun Heo

This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox