* [PATCH 1/2] sched_ext: Add lazy preemption support
2026-09-11 19:56 [PATCHSET sched_ext/for-7.4] sched_ext: Add lazy preemption support Andrea Righi
@ 2026-09-11 19:56 ` Andrea Righi
2026-09-13 16:50 ` Tejun Heo
2026-09-11 19:56 ` [PATCH 2/2] selftests/sched_ext: Add kick selftest Andrea Righi
1 sibling, 1 reply; 4+ messages in thread
From: Andrea Righi @ 2026-09-11 19:56 UTC (permalink / raw)
To: Tejun Heo, David Vernet, Changwoo Min
Cc: Emil Tsalapatis, sched-ext, linux-kernel
The fair scheduling class supports lazy rescheduling: when lazy
preemption is enabled, a lazy reschedule request does not preempt the
current task immediately at the next kernel preemption point. Instead,
it is serviced when returning to user space or promoted to an immediate
reschedule by the next scheduler tick. This avoids unnecessary in-kernel
preemption while still bounding the scheduling delay.
sched_ext only supports immediate preemption, so BPF schedulers cannot
make the same trade-off.
Add lazy preemption support to sched_ext through SCX_ENQ_PREEMPT_LAZY
and SCX_KICK_PREEMPT_LAZY. Both operations still expire the current
sched_ext task's slice, but use resched_curr_lazy() so the scheduling
boundary can be deferred when lazy preemption is available.
On a non-local DSQ %SCX_ENQ_PREEMPT_LAZY means a head insertion, as
%SCX_ENQ_PREEMPT does there.
fair.c also expires a slice lazily: update_curr() calls
resched_curr_lazy() when the slice has run out at the tick. Let a BPF
scheduler opt into the same with SCX_OPS_LAZY_SLICE_EXPIRY, which makes
task_tick_scx() request lazy rescheduling for a depleted slice. A task
in user space still reschedules on the way back from the tick, a task in
the kernel runs on to its next return to user space or to the next tick.
The default stays immediate, as does rescheduling while the scheduler is
being disabled.
Reject requests which combine immediate and lazy preemption. Also reject
lazy kicks combined with SCX_KICK_WAIT, as waiting for a lazy scheduling
boundary would have a potentially unbounded contract. Unknown kick flags
are now rejected as well, where they used to be ignored.
Track lazy kicks in a separate mask, outside cpus_to_kick, so that a
lazy-only request can be told from one that coincides with a plain or
preempting kick. An immediate preemption supersedes a queued lazy one. A
plain kick reschedules at once, but leaves the slice alone, so a lazy
request next to it keeps clearing the slice while the reschedule stays
immediate: both are served.
Signed-off-by: Andrea Righi <arighi@nvidia.com>
---
kernel/sched/ext/ext.c | 115 ++++++++++++++----
kernel/sched/ext/internal.h | 36 +++++-
kernel/sched/ext/sub.c | 15 +--
.../sched_ext/include/scx/enum_defs.autogen.h | 3 +
.../sched_ext/include/scx/enums.autogen.bpf.h | 6 +
tools/sched_ext/include/scx/enums.autogen.h | 2 +
.../sched_ext/include/scx/enums_abi.autogen.h | 5 +-
7 files changed, 149 insertions(+), 33 deletions(-)
diff --git a/kernel/sched/ext/ext.c b/kernel/sched/ext/ext.c
index 4080902a528cd..a918e3ade1148 100644
--- a/kernel/sched/ext/ext.c
+++ b/kernel/sched/ext/ext.c
@@ -383,11 +383,11 @@ static bool rq_is_open(struct rq *rq, u64 enq_flags)
return true;
/*
- * %SCX_ENQ_PREEMPT clears $curr's slice if on SCX and kicks dispatch,
- * so allow it to avoid spuriously triggering reenq on a combined
+ * The preemption flags clear $curr's slice if on SCX and kick dispatch,
+ * so allow them to avoid spuriously triggering reenq on a combined
* PREEMPT|IMMED insertion.
*/
- if (enq_flags & SCX_ENQ_PREEMPT) {
+ if (enq_flags & (SCX_ENQ_PREEMPT | SCX_ENQ_PREEMPT_LAZY)) {
struct task_struct *curr = rq->curr;
/*
@@ -1577,12 +1577,16 @@ static void rq_owned_post_enq(struct scx_sched *sch, struct rq *rq,
if (rq->scx.flags & SCX_RQ_IN_DISPATCH)
return;
- if ((enq_flags & SCX_ENQ_PREEMPT) && p != rq->curr &&
+ if ((enq_flags & (SCX_ENQ_PREEMPT | SCX_ENQ_PREEMPT_LAZY)) && p != rq->curr &&
rq->curr->sched_class == &ext_sched_class) {
- if (likely(scx_set_task_slice(rq->curr, 0)))
- resched_curr(rq);
- else
+ if (likely(scx_set_task_slice(rq->curr, 0))) {
+ if (enq_flags & SCX_ENQ_PREEMPT)
+ resched_curr(rq);
+ else
+ resched_curr_lazy(rq);
+ } else {
__scx_add_event(sch, SCX_EV_SLICE_DENIED, 1);
+ }
}
}
@@ -1672,7 +1676,7 @@ static void scx_dispatch_enqueue(struct scx_sched *sch, struct rq *rq,
scx_error(sch, "DSQ ID 0x%016llx already had PRIQ-enqueued tasks",
dsq->id);
- if (enq_flags & (SCX_ENQ_HEAD | SCX_ENQ_PREEMPT)) {
+ if (enq_flags & (SCX_ENQ_HEAD | SCX_ENQ_PREEMPT | SCX_ENQ_PREEMPT_LAZY)) {
/* new task inserted at head - use fastpath */
if (dsq_insert_head(dsq, p) && !(dsq->id & SCX_DSQ_FLAG_BUILTIN))
rcu_assign_pointer(dsq->first_task, p);
@@ -2386,7 +2390,7 @@ void scx_move_local_task_to_local_dsq(struct scx_sched *sch, struct task_struct
WARN_ON_ONCE(p->scx.holding_cpu >= 0);
- if (enq_flags & (SCX_ENQ_HEAD | SCX_ENQ_PREEMPT))
+ if (enq_flags & (SCX_ENQ_HEAD | SCX_ENQ_PREEMPT | SCX_ENQ_PREEMPT_LAZY))
dsq_insert_head(dst_dsq, p);
else
list_add_tail(&p->scx.dsq_list.node, &dst_dsq->list);
@@ -3804,8 +3808,14 @@ static void task_tick_scx(struct rq *rq, struct task_struct *curr, int queued)
else if (SCX_HAS_OP(sch, tick))
SCX_CALL_OP_TASK(sch, tick, rq, curr);
- if (!curr->scx.slice)
- resched_curr(rq);
+ if (!curr->scx.slice) {
+ /* see SCX_OPS_LAZY_SLICE_EXPIRY; the slice can't be trusted while bypassing */
+ if ((sch->ops.flags & SCX_OPS_LAZY_SLICE_EXPIRY) &&
+ !scx_bypassing(sch, cpu_of(rq)))
+ resched_curr_lazy(rq);
+ else
+ resched_curr(rq);
+ }
}
#ifdef CONFIG_EXT_GROUP_SCHED
@@ -5371,6 +5381,7 @@ static void scx_sched_free_rcu_work(struct work_struct *work)
free_cpumask_var(pcpu->cpus_to_kick);
free_cpumask_var(pcpu->cpus_to_kick_if_idle);
free_cpumask_var(pcpu->cpus_to_preempt);
+ free_cpumask_var(pcpu->cpus_to_preempt_lazy);
free_cpumask_var(pcpu->cpus_to_wait);
exit_dsq(scx_bypass_dsq(sch, cpu));
@@ -6883,6 +6894,9 @@ static void scx_dump_cpu(struct scx_sched *sch, struct seq_buf *s,
if (!cpumask_empty(pcpu->cpus_to_preempt))
scx_dump_line(&ns, " cpus_to_preempt: %*pb",
cpumask_pr_args(pcpu->cpus_to_preempt));
+ if (!cpumask_empty(pcpu->cpus_to_preempt_lazy))
+ scx_dump_line(&ns, " preempt_lazy : %*pb",
+ cpumask_pr_args(pcpu->cpus_to_preempt_lazy));
if (!cpumask_empty(pcpu->cpus_to_wait))
scx_dump_line(&ns, " cpus_to_wait : %*pb",
cpumask_pr_args(pcpu->cpus_to_wait));
@@ -7200,6 +7214,7 @@ struct scx_sched *scx_alloc_and_add_sched(struct scx_enable_cmd *cmd,
if (!zalloc_cpumask_var_node(&pcpu->cpus_to_kick, GFP_KERNEL, node) ||
!zalloc_cpumask_var_node(&pcpu->cpus_to_kick_if_idle, GFP_KERNEL, node) ||
!zalloc_cpumask_var_node(&pcpu->cpus_to_preempt, GFP_KERNEL, node) ||
+ !zalloc_cpumask_var_node(&pcpu->cpus_to_preempt_lazy, GFP_KERNEL, node) ||
!zalloc_cpumask_var_node(&pcpu->cpus_to_wait, GFP_KERNEL, node)) {
ret = -ENOMEM;
goto err_free_pcpu;
@@ -7335,6 +7350,7 @@ struct scx_sched *scx_alloc_and_add_sched(struct scx_enable_cmd *cmd,
free_cpumask_var(pcpu->cpus_to_kick);
free_cpumask_var(pcpu->cpus_to_kick_if_idle);
free_cpumask_var(pcpu->cpus_to_preempt);
+ free_cpumask_var(pcpu->cpus_to_preempt_lazy);
free_cpumask_var(pcpu->cpus_to_wait);
}
for_each_possible_cpu(cpu) {
@@ -8462,10 +8478,20 @@ static bool kick_one_cpu(s32 cpu, struct scx_sched_pcpu *pcpu, struct rq *this_r
const struct sched_class *cur_class;
bool should_wait = false;
bool kickable;
+ bool preempt, preempt_lazy, immediate;
unsigned long flags;
raw_spin_rq_lock_irqsave(rq, flags);
cur_class = rq->curr->sched_class;
+ preempt = cpumask_test_cpu(cpu, pcpu->cpus_to_preempt);
+ preempt_lazy = cpumask_test_cpu(cpu, pcpu->cpus_to_preempt_lazy);
+ /*
+ * A lazy-only request sits in cpus_to_preempt_lazy alone. Any other
+ * kick for the same cpu, plain or preempting, is in cpus_to_kick and
+ * asks for an immediate reschedule; the lazy request still clears the
+ * slice, the immediate one still reschedules now, so both are served.
+ */
+ immediate = !preempt_lazy || cpumask_test_cpu(cpu, pcpu->cpus_to_kick);
/*
* During CPU hotplug, a CPU may depend on kicking itself to make
@@ -8479,7 +8505,7 @@ static bool kick_one_cpu(s32 cpu, struct scx_sched_pcpu *pcpu, struct rq *this_r
!sched_class_above(cur_class, &ext_sched_class);
if (kickable && !scx_missing_caps(pcpu->sch, cpu, SCX_CAP_BASE)) {
- if (cpumask_test_cpu(cpu, pcpu->cpus_to_preempt)) {
+ if (preempt || preempt_lazy) {
if (cur_class == &ext_sched_class) {
u64 caps = scx_caps_for_preempt(pcpu->sch, rq, 0);
@@ -8488,7 +8514,6 @@ static bool kick_one_cpu(s32 cpu, struct scx_sched_pcpu *pcpu, struct rq *this_r
else if (unlikely(!scx_set_task_slice(rq->curr, 0)))
__scx_add_event(pcpu->sch, SCX_EV_SLICE_DENIED, 1);
}
- cpumask_clear_cpu(cpu, pcpu->cpus_to_preempt);
}
if (cpumask_test_cpu(cpu, pcpu->cpus_to_wait)) {
@@ -8500,14 +8525,20 @@ static bool kick_one_cpu(s32 cpu, struct scx_sched_pcpu *pcpu, struct rq *this_r
cpumask_clear_cpu(cpu, pcpu->cpus_to_wait);
}
- resched_curr(rq);
+ if (immediate)
+ resched_curr(rq);
+ else
+ resched_curr_lazy(rq);
} else {
/* a kickable cpu was skipped solely for the missing caps */
if (kickable)
__scx_add_event(pcpu->sch, SCX_EV_SUB_KICK_DENIED, 1);
- cpumask_clear_cpu(cpu, pcpu->cpus_to_preempt);
cpumask_clear_cpu(cpu, pcpu->cpus_to_wait);
}
+ if (preempt)
+ cpumask_clear_cpu(cpu, pcpu->cpus_to_preempt);
+ if (preempt_lazy)
+ cpumask_clear_cpu(cpu, pcpu->cpus_to_preempt_lazy);
scx_rq_lock_drop(rq);
raw_spin_rq_unlock_irqrestore(rq, flags);
@@ -8566,6 +8597,13 @@ static void kick_cpus_irq_workfn(struct irq_work *irq_work)
cpumask_clear_cpu(cpu, pcpu->cpus_to_kick_if_idle);
}
+ /*
+ * kick_one_cpu() clears the lazy bit of every cpu it visited
+ * above; what remains are lazy-only requests, see scx_kick_cpu().
+ */
+ for_each_cpu(cpu, pcpu->cpus_to_preempt_lazy)
+ kick_one_cpu(cpu, pcpu, this_rq, ksyncs);
+
for_each_cpu(cpu, pcpu->cpus_to_kick_if_idle) {
kick_one_cpu_if_idle(cpu, pcpu, this_rq);
cpumask_clear_cpu(cpu, pcpu->cpus_to_kick_if_idle);
@@ -8734,6 +8772,11 @@ static bool scx_vet_enq_flags(struct scx_sched *sch, u64 dsq_id, u64 *enq_flags)
return false;
}
+ if (unlikely((*enq_flags & SCX_ENQ_PREEMPT) && (*enq_flags & SCX_ENQ_PREEMPT_LAZY))) {
+ scx_error(sch, "SCX_ENQ_PREEMPT and SCX_ENQ_PREEMPT_LAZY cannot be combined");
+ return false;
+ }
+
if (*enq_flags & SCX_ENQ_IMMED) {
if (unlikely(!is_local)) {
scx_error(sch, "SCX_ENQ_IMMED on a non-local DSQ 0x%llx", dsq_id);
@@ -9539,6 +9582,24 @@ void scx_kick_cpu(struct scx_sched *sch, s32 cpu, u64 flags)
if (!scx_kf_allowed_ctx(sch))
return;
+ if (unlikely(flags & ~(SCX_KICK_IDLE | SCX_KICK_PREEMPT | SCX_KICK_WAIT |
+ SCX_KICK_PREEMPT_LAZY))) {
+ scx_error(sch, "invalid kick flags 0x%llx", flags);
+ return;
+ }
+ if (unlikely((flags & SCX_KICK_PREEMPT) && (flags & SCX_KICK_PREEMPT_LAZY))) {
+ scx_error(sch, "SCX_KICK_PREEMPT and SCX_KICK_PREEMPT_LAZY cannot be combined");
+ return;
+ }
+ if (unlikely((flags & SCX_KICK_PREEMPT_LAZY) && (flags & SCX_KICK_WAIT))) {
+ scx_error(sch, "SCX_KICK_PREEMPT_LAZY cannot be used with SCX_KICK_WAIT");
+ return;
+ }
+ if (unlikely((flags & SCX_KICK_IDLE) &&
+ (flags & (SCX_KICK_PREEMPT | SCX_KICK_PREEMPT_LAZY | SCX_KICK_WAIT)))) {
+ scx_error(sch, "PREEMPT/WAIT cannot be used with SCX_KICK_IDLE");
+ return;
+ }
local_irq_save(irq_flags);
@@ -9564,9 +9625,6 @@ void scx_kick_cpu(struct scx_sched *sch, s32 cpu, u64 flags)
if (flags & SCX_KICK_IDLE) {
struct rq *target_rq = cpu_rq(cpu);
- if (unlikely(flags & (SCX_KICK_PREEMPT | SCX_KICK_WAIT)))
- scx_error(sch, "PREEMPT/WAIT cannot be used with SCX_KICK_IDLE");
-
if (raw_spin_rq_trylock(target_rq)) {
if (can_skip_idle_kick(target_rq)) {
scx_rq_lock_drop(target_rq);
@@ -9577,11 +9635,25 @@ void scx_kick_cpu(struct scx_sched *sch, s32 cpu, u64 flags)
raw_spin_rq_unlock(target_rq);
}
cpumask_set_cpu(cpu, pcpu->cpus_to_kick_if_idle);
+ } else if (flags & SCX_KICK_PREEMPT_LAZY) {
+ /*
+ * Recorded in its own mask and not in cpus_to_kick, so that
+ * kick_one_cpu() can tell a lazy-only request from one that
+ * coincides with a plain or preempting kick. A queued immediate
+ * preemption already clears the slice and reschedules now,
+ * which is more than asked for; a plain kick doesn't touch the
+ * slice, so the lazy request must stay recorded next to it.
+ */
+ if (!cpumask_test_cpu(cpu, pcpu->cpus_to_preempt))
+ cpumask_set_cpu(cpu, pcpu->cpus_to_preempt_lazy);
} else {
cpumask_set_cpu(cpu, pcpu->cpus_to_kick);
- if (flags & SCX_KICK_PREEMPT)
+ if (flags & SCX_KICK_PREEMPT) {
+ /* an immediate preemption subsumes a queued lazy one */
cpumask_set_cpu(cpu, pcpu->cpus_to_preempt);
+ cpumask_clear_cpu(cpu, pcpu->cpus_to_preempt_lazy);
+ }
if (flags & SCX_KICK_WAIT)
cpumask_set_cpu(cpu, pcpu->cpus_to_wait);
}
@@ -9623,8 +9695,9 @@ __bpf_kfunc void scx_bpf_kick_cpu(s32 cpu, u64 flags, const struct bpf_prog_aux
* cid-addressed equivalent of scx_bpf_kick_cpu(). An invalid @cid aborts the
* scheduler via scx_cid_to_cpu(). Caps are enforced on the delivery path: a
* kick is dropped if the caller lacks baseline access on @cid, and a
- * %SCX_KICK_PREEMPT degrades to a plain reschedule if the caller lacks
- * %SCX_CAP_PREEMPT for a task outside its subtree.
+ * %SCX_KICK_PREEMPT or %SCX_KICK_PREEMPT_LAZY request degrades to a plain
+ * reschedule if the caller lacks %SCX_CAP_PREEMPT for a task outside its
+ * subtree.
*/
__bpf_kfunc void scx_bpf_kick_cid(s32 cid, u64 flags, const struct bpf_prog_aux *aux)
{
diff --git a/kernel/sched/ext/internal.h b/kernel/sched/ext/internal.h
index e1eb3a0d456cb..0b38f84353eb8 100644
--- a/kernel/sched/ext/internal.h
+++ b/kernel/sched/ext/internal.h
@@ -215,6 +215,17 @@ enum scx_ops_flags {
*/
SCX_OPS_TID_TO_TASK = 1LLU << 8,
+ /*
+ * If set, a slice that runs out at the tick requests lazy rescheduling
+ * instead of an immediate one, the way fair.c expires a slice from
+ * update_curr(): a task in user space still reschedules on the way
+ * back from the tick, a task in the kernel runs on to its next return
+ * to user space or to the next tick, which promotes the request. No
+ * effect on kernels without lazy preemption. Rescheduling while the
+ * scheduler is being disabled stays immediate.
+ */
+ SCX_OPS_LAZY_SLICE_EXPIRY = 1LLU << 9,
+
SCX_OPS_ALL_FLAGS = SCX_OPS_KEEP_BUILTIN_IDLE |
SCX_OPS_ENQ_LAST |
SCX_OPS_ENQ_EXITING |
@@ -223,7 +234,8 @@ enum scx_ops_flags {
SCX_OPS_SWITCH_PARTIAL |
SCX_OPS_BUILTIN_IDLE_PER_NODE |
SCX_OPS_ALWAYS_ENQ_IMMED |
- SCX_OPS_TID_TO_TASK,
+ SCX_OPS_TID_TO_TASK |
+ SCX_OPS_LAZY_SLICE_EXPIRY,
/* high 8 bits are internal, don't include in SCX_OPS_ALL_FLAGS */
__SCX_OPS_INTERNAL_MASK = 0xffLLU << 56,
@@ -1327,6 +1339,7 @@ struct scx_sched_pcpu {
cpumask_var_t cpus_to_kick;
cpumask_var_t cpus_to_kick_if_idle;
cpumask_var_t cpus_to_preempt;
+ cpumask_var_t cpus_to_preempt_lazy;
cpumask_var_t cpus_to_wait;
struct list_head to_kick_node;
@@ -1407,15 +1420,16 @@ struct scx_sched_pnode {
* the allocation pattern.
*
* ENQ_IMMED insert an IMMED task onto the cid's local DSQ
- * - kick the cid's cpu (except SCX_KICK_PREEMPT)
+ * - kick the cid's cpu (except SCX_KICK_PREEMPT and
+ * SCX_KICK_PREEMPT_LAZY)
*
* ENQ insert any task onto the cid's local DSQ (implies ENQ_IMMED)
*
* PREEMPT preempt any task running on the cid regardless of the owning
* sched (implies ENQ). Preempting a task in the sched's own subtree
* doesn't require any cap.
- * - SCX_ENQ_PREEMPT inserts
- * - SCX_KICK_PREEMPT kicks
+ * - SCX_ENQ_PREEMPT and SCX_ENQ_PREEMPT_LAZY inserts
+ * - SCX_KICK_PREEMPT and SCX_KICK_PREEMPT_LAZY kicks
*
* PERF control the cid's cpu power/perf management state, currently the
* cpufreq target set through scx_bpf_cidperf_set(). Hardware
@@ -1685,6 +1699,14 @@ enum scx_enq_flags {
*/
SCX_ENQ_PREEMPT = 1LLU << 32,
+ /*
+ * Like %SCX_ENQ_PREEMPT, but request lazy rescheduling. The current
+ * task's slice is still cleared immediately so that the next scheduling
+ * boundary observes the new ordering. Implies %SCX_ENQ_HEAD, which is
+ * all it means on a non-local DSQ, as with %SCX_ENQ_PREEMPT.
+ */
+ SCX_ENQ_PREEMPT_LAZY = 1LLU << 35,
+
/*
* Only allowed on local DSQs. Guarantees that the task either gets
* on the CPU immediately and stays on it, or gets reenqueued back
@@ -1811,6 +1833,12 @@ enum scx_kick_flags {
* is not on SCX.
*/
SCX_KICK_WAIT = 1LLU << 2,
+
+ /*
+ * Like %SCX_KICK_PREEMPT, but request lazy rescheduling. This cannot be
+ * combined with %SCX_KICK_WAIT.
+ */
+ SCX_KICK_PREEMPT_LAZY = 1LLU << 3,
};
enum scx_tg_flags {
diff --git a/kernel/sched/ext/sub.c b/kernel/sched/ext/sub.c
index 380a5653dc529..a88b87614b55e 100644
--- a/kernel/sched/ext/sub.c
+++ b/kernel/sched/ext/sub.c
@@ -686,9 +686,10 @@ void scx_rescue_init(struct rq *rq)
* rescue is enabled, or @rq's reject DSQ after recording the reenq reason on
* @p.
*
- * %SCX_ENQ_IMMED, %SCX_ENQ_PREEMPT and %SCX_ENQ_HEAD are cleared when diverting
- * to rescue or reject. %SCX_ENQ_PREEMPT is also cleared on a fallback
- * migration-disabled admission.
+ * %SCX_ENQ_IMMED, %SCX_ENQ_PREEMPT, %SCX_ENQ_PREEMPT_LAZY and %SCX_ENQ_HEAD
+ * are cleared when diverting to rescue or reject. %SCX_ENQ_PREEMPT and
+ * %SCX_ENQ_PREEMPT_LAZY are also cleared on a fallback migration-disabled
+ * admission.
*
* Bypass doesn't need special-casing as a bypassing sched's tasks are enqueued
* to and run by its nearest non-bypassing ancestor. If root is bypassing, it
@@ -709,7 +710,7 @@ struct scx_dispatch_q *scx_resolve_local_dsq(struct scx_sched *sch, struct rq *r
* On a remote activation the scheduling sched (@asch) differs from
* @p's owner (@sch). Check caps against the scheduling sched.
*/
- if (*enq_flags & SCX_ENQ_PREEMPT)
+ if (*enq_flags & (SCX_ENQ_PREEMPT | SCX_ENQ_PREEMPT_LAZY))
needed |= scx_caps_for_preempt(asch, rq, *enq_flags);
missing = scx_missing_caps(asch, cpu_of(rq), needed);
@@ -726,7 +727,7 @@ struct scx_dispatch_q *scx_resolve_local_dsq(struct scx_sched *sch, struct rq *r
if (unlikely(!scx_rq_online(rq) || is_migration_disabled(p) ||
p->migration_pending)) {
__scx_add_event(sch, SCX_EV_SUB_FORCED_ADMIT, 1);
- *enq_flags &= ~SCX_ENQ_PREEMPT;
+ *enq_flags &= ~(SCX_ENQ_PREEMPT | SCX_ENQ_PREEMPT_LAZY);
return &rq->scx.local_dsq;
}
@@ -735,8 +736,8 @@ struct scx_dispatch_q *scx_resolve_local_dsq(struct scx_sched *sch, struct rq *r
* or HEAD - a diversion has no priority and IMMED is not allowed on
* non-local DSQs. Strip the enq and task flags along with the slice.
*/
- *enq_flags &= ~(SCX_ENQ_IMMED | SCX_ENQ_PREEMPT | SCX_ENQ_HEAD |
- SCX_ENQ_APPLY_SLICE | SCX_ENQ_SLICE_DFL);
+ *enq_flags &= ~(SCX_ENQ_IMMED | SCX_ENQ_PREEMPT | SCX_ENQ_PREEMPT_LAZY |
+ SCX_ENQ_HEAD | SCX_ENQ_APPLY_SLICE | SCX_ENQ_SLICE_DFL);
p->scx.flags &= ~SCX_TASK_IMMED;
/* the enqueuer opted for rescue instead of rejection and reenqueue */
diff --git a/tools/sched_ext/include/scx/enum_defs.autogen.h b/tools/sched_ext/include/scx/enum_defs.autogen.h
index 63b6b14b19bd4..f1554c22071ff 100644
--- a/tools/sched_ext/include/scx/enum_defs.autogen.h
+++ b/tools/sched_ext/include/scx/enum_defs.autogen.h
@@ -85,6 +85,7 @@
#define HAVE_SCX_ENQ_HEAD
#define HAVE_SCX_ENQ_CPU_SELECTED
#define HAVE_SCX_ENQ_PREEMPT
+#define HAVE_SCX_ENQ_PREEMPT_LAZY
#define HAVE_SCX_ENQ_IMMED
#define HAVE_SCX_ENQ_RESCUE
#define HAVE_SCX_ENQ_REENQ
@@ -148,6 +149,7 @@
#define HAVE_SCX_KF_ALLOW_SELECT_CPU
#define HAVE_SCX_KICK_IDLE
#define HAVE_SCX_KICK_PREEMPT
+#define HAVE_SCX_KICK_PREEMPT_LAZY
#define HAVE_SCX_KICK_WAIT
#define HAVE_SCX_OPI_BEGIN
#define HAVE_SCX_OPI_NORMAL_BEGIN
@@ -164,6 +166,7 @@
#define HAVE_SCX_OPS_BUILTIN_IDLE_PER_NODE
#define HAVE_SCX_OPS_ALWAYS_ENQ_IMMED
#define HAVE_SCX_OPS_TID_TO_TASK
+#define HAVE_SCX_OPS_LAZY_SLICE_EXPIRY
#define HAVE_SCX_OPS_ALL_FLAGS
#define HAVE___SCX_OPS_INTERNAL_MASK
#define HAVE_SCX_OPS_HAS_CPU_PREEMPT
diff --git a/tools/sched_ext/include/scx/enums.autogen.bpf.h b/tools/sched_ext/include/scx/enums.autogen.bpf.h
index 7268131010de3..2cec24beb2d80 100644
--- a/tools/sched_ext/include/scx/enums.autogen.bpf.h
+++ b/tools/sched_ext/include/scx/enums.autogen.bpf.h
@@ -109,6 +109,9 @@ const volatile u64 __SCX_KICK_IDLE __weak;
const volatile u64 __SCX_KICK_PREEMPT __weak;
#define SCX_KICK_PREEMPT __SCX_KICK_PREEMPT
+const volatile u64 __SCX_KICK_PREEMPT_LAZY __weak;
+#define SCX_KICK_PREEMPT_LAZY __SCX_KICK_PREEMPT_LAZY
+
const volatile u64 __SCX_KICK_WAIT __weak;
#define SCX_KICK_WAIT __SCX_KICK_WAIT
@@ -121,6 +124,9 @@ const volatile u64 __SCX_ENQ_HEAD __weak;
const volatile u64 __SCX_ENQ_PREEMPT __weak;
#define SCX_ENQ_PREEMPT __SCX_ENQ_PREEMPT
+const volatile u64 __SCX_ENQ_PREEMPT_LAZY __weak;
+#define SCX_ENQ_PREEMPT_LAZY __SCX_ENQ_PREEMPT_LAZY
+
const volatile u64 __SCX_ENQ_IMMED __weak;
#define SCX_ENQ_IMMED __SCX_ENQ_IMMED
diff --git a/tools/sched_ext/include/scx/enums.autogen.h b/tools/sched_ext/include/scx/enums.autogen.h
index e616326545172..dd762ee0ddb36 100644
--- a/tools/sched_ext/include/scx/enums.autogen.h
+++ b/tools/sched_ext/include/scx/enums.autogen.h
@@ -40,10 +40,12 @@
SCX_ENUM_SET(skel, scx_ent_dsq_flags, SCX_TASK_DSQ_ON_PRIQ); \
SCX_ENUM_SET(skel, scx_kick_flags, SCX_KICK_IDLE); \
SCX_ENUM_SET(skel, scx_kick_flags, SCX_KICK_PREEMPT); \
+ SCX_ENUM_SET(skel, scx_kick_flags, SCX_KICK_PREEMPT_LAZY); \
SCX_ENUM_SET(skel, scx_kick_flags, SCX_KICK_WAIT); \
SCX_ENUM_SET(skel, scx_enq_flags, SCX_ENQ_WAKEUP); \
SCX_ENUM_SET(skel, scx_enq_flags, SCX_ENQ_HEAD); \
SCX_ENUM_SET(skel, scx_enq_flags, SCX_ENQ_PREEMPT); \
+ SCX_ENUM_SET(skel, scx_enq_flags, SCX_ENQ_PREEMPT_LAZY); \
SCX_ENUM_SET(skel, scx_enq_flags, SCX_ENQ_IMMED); \
SCX_ENUM_SET(skel, scx_enq_flags, SCX_ENQ_RESCUE); \
SCX_ENUM_SET(skel, scx_enq_flags, SCX_ENQ_REENQ); \
diff --git a/tools/sched_ext/include/scx/enums_abi.autogen.h b/tools/sched_ext/include/scx/enums_abi.autogen.h
index d53899764f5ac..672c02a497d14 100644
--- a/tools/sched_ext/include/scx/enums_abi.autogen.h
+++ b/tools/sched_ext/include/scx/enums_abi.autogen.h
@@ -97,6 +97,7 @@ static const struct __scx_enum_abi_val __scx_enum_abi_vals[]
{ "scx_enq_flags", "SCX_ENQ_HEAD", 0x10000LLU },
{ "scx_enq_flags", "SCX_ENQ_CPU_SELECTED", 0x100000LLU },
{ "scx_enq_flags", "SCX_ENQ_PREEMPT", 0x100000000LLU },
+ { "scx_enq_flags", "SCX_ENQ_PREEMPT_LAZY", 0x800000000LLU },
{ "scx_enq_flags", "SCX_ENQ_IMMED", 0x200000000LLU },
{ "scx_enq_flags", "SCX_ENQ_RESCUE", 0x400000000LLU },
{ "scx_enq_flags", "SCX_ENQ_REENQ", 0x10000000000LLU },
@@ -160,6 +161,7 @@ static const struct __scx_enum_abi_val __scx_enum_abi_vals[]
{ "scx_kf_allow_flags", "SCX_KF_ALLOW_SELECT_CPU", 0x20LLU },
{ "scx_kick_flags", "SCX_KICK_IDLE", 0x1LLU },
{ "scx_kick_flags", "SCX_KICK_PREEMPT", 0x2LLU },
+ { "scx_kick_flags", "SCX_KICK_PREEMPT_LAZY", 0x8LLU },
{ "scx_kick_flags", "SCX_KICK_WAIT", 0x4LLU },
{ "scx_opi", "SCX_OPI_BEGIN", 0x0LLU },
{ "scx_opi", "SCX_OPI_NORMAL_BEGIN", 0x0LLU },
@@ -176,7 +178,8 @@ static const struct __scx_enum_abi_val __scx_enum_abi_vals[]
{ "scx_ops_flags", "SCX_OPS_BUILTIN_IDLE_PER_NODE", 0x40LLU },
{ "scx_ops_flags", "SCX_OPS_ALWAYS_ENQ_IMMED", 0x80LLU },
{ "scx_ops_flags", "SCX_OPS_TID_TO_TASK", 0x100LLU },
- { "scx_ops_flags", "SCX_OPS_ALL_FLAGS", 0x1ffLLU },
+ { "scx_ops_flags", "SCX_OPS_LAZY_SLICE_EXPIRY", 0x200LLU },
+ { "scx_ops_flags", "SCX_OPS_ALL_FLAGS", 0x3ffLLU },
{ "scx_ops_flags", "__SCX_OPS_INTERNAL_MASK", 0xff00000000000000LLU },
{ "scx_ops_flags", "SCX_OPS_HAS_CPU_PREEMPT", 0x100000000000000LLU },
{ "scx_ops_state", "SCX_OPSS_NONE", 0x0LLU },
--
2.55.0
^ permalink raw reply related [flat|nested] 4+ messages in thread* [PATCH 2/2] selftests/sched_ext: Add kick selftest
2026-09-11 19:56 [PATCHSET sched_ext/for-7.4] sched_ext: Add lazy preemption support Andrea Righi
2026-09-11 19:56 ` [PATCH 1/2] " Andrea Righi
@ 2026-09-11 19:56 ` Andrea Righi
1 sibling, 0 replies; 4+ messages in thread
From: Andrea Righi @ 2026-09-11 19:56 UTC (permalink / raw)
To: Tejun Heo, David Vernet, Changwoo Min
Cc: Emil Tsalapatis, sched-ext, linux-kernel
Add trace-based coverage for immediate and lazy preemption kicks. Run a
CPU-bound sched_ext victim and issue each kick after the victim starts
running.
Use an fexit probe on __resched_curr() and the stopping callback to
verify that the victim starts with a positive slice, has a zero slice
after the rescheduling request is processed and subsequently reaches a
scheduling boundary. The probe also records the requested TIF for
comparison against an immediate-kick baseline and the active runtime
preemption mode.
Issue lazy and immediate kicks in both orders and verify that the
recorded TIF matches the immediate baseline. Keep the invalid flag checks
from the original test.
Register immediate, lazy, coalescing, tick-expiry and invalid-request
coverage as separate tests. Probe the running kernel BTF before using
each kick mode to skip unsupported modes instead of failing or
accidentally issuing a zero-flag kick.
Signed-off-by: Andrea Righi <arighi@nvidia.com>
---
tools/testing/selftests/sched_ext/Makefile | 1 +
tools/testing/selftests/sched_ext/kick.bpf.c | 144 ++++++
tools/testing/selftests/sched_ext/kick.c | 479 +++++++++++++++++++
3 files changed, 624 insertions(+)
create mode 100644 tools/testing/selftests/sched_ext/kick.bpf.c
create mode 100644 tools/testing/selftests/sched_ext/kick.c
diff --git a/tools/testing/selftests/sched_ext/Makefile b/tools/testing/selftests/sched_ext/Makefile
index 5f5dd9ab903ae..c4ec9b21a4c2a 100644
--- a/tools/testing/selftests/sched_ext/Makefile
+++ b/tools/testing/selftests/sched_ext/Makefile
@@ -172,6 +172,7 @@ auto-test-targets := \
exit \
hotplug \
init_enable_count \
+ kick \
maximal \
maybe_null \
minimal \
diff --git a/tools/testing/selftests/sched_ext/kick.bpf.c b/tools/testing/selftests/sched_ext/kick.bpf.c
new file mode 100644
index 0000000000000..031856f990cef
--- /dev/null
+++ b/tools/testing/selftests/sched_ext/kick.bpf.c
@@ -0,0 +1,144 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES
+ */
+#include <scx/common.bpf.h>
+
+char _license[] SEC("license") = "GPL";
+
+enum kick_scenario {
+ KICK_IMMEDIATE,
+ KICK_LAZY,
+ KICK_LAZY_THEN_IMMEDIATE,
+ KICK_IMMEDIATE_THEN_LAZY,
+ KICK_PLAIN_THEN_LAZY,
+ KICK_LAZY_THEN_PLAIN,
+ TICK_EXPIRY,
+ INVALID_ENQ_BOTH,
+ INVALID_KICK_BOTH,
+ INVALID_KICK_WAIT,
+ INVALID_KICK_IDLE,
+ INVALID_KICK_UNKNOWN,
+};
+
+enum kick_state {
+ KICK_STATE_IDLE,
+ KICK_STATE_ARMED,
+ KICK_STATE_QUEUED,
+ KICK_STATE_RESCHED,
+ KICK_STATE_DONE,
+};
+
+const volatile u32 scenario;
+const volatile s32 victim_pid;
+const volatile s32 target_cpu;
+
+u32 state;
+u64 slice_before;
+u64 slice_at_resched;
+s32 resched_tif;
+
+UEI_DEFINE(uei);
+
+static bool is_trace_scenario(void)
+{
+ return scenario <= TICK_EXPIRY;
+}
+
+void BPF_STRUCT_OPS(kick_enqueue, struct task_struct *p, u64 enq_flags)
+{
+ switch (scenario) {
+ case INVALID_ENQ_BOTH:
+ scx_bpf_dsq_insert(p, SCX_DSQ_LOCAL, SCX_SLICE_DFL,
+ enq_flags | SCX_ENQ_PREEMPT | SCX_ENQ_PREEMPT_LAZY);
+ return;
+ case INVALID_KICK_BOTH:
+ scx_bpf_kick_cpu(scx_bpf_task_cpu(p), SCX_KICK_PREEMPT | SCX_KICK_PREEMPT_LAZY);
+ break;
+ case INVALID_KICK_WAIT:
+ scx_bpf_kick_cpu(scx_bpf_task_cpu(p), SCX_KICK_PREEMPT_LAZY | SCX_KICK_WAIT);
+ break;
+ case INVALID_KICK_IDLE:
+ scx_bpf_kick_cpu(scx_bpf_task_cpu(p), SCX_KICK_PREEMPT_LAZY | SCX_KICK_IDLE);
+ break;
+ case INVALID_KICK_UNKNOWN:
+ scx_bpf_kick_cpu(scx_bpf_task_cpu(p), 1LLU << 63);
+ break;
+ }
+
+ scx_bpf_dsq_insert(p, SCX_DSQ_GLOBAL, SCX_SLICE_DFL, enq_flags);
+}
+
+void BPF_STRUCT_OPS(kick_running, struct task_struct *p)
+{
+ if (!is_trace_scenario() || p->pid != victim_pid)
+ return;
+ if (__sync_val_compare_and_swap(&state, KICK_STATE_ARMED, KICK_STATE_QUEUED) !=
+ KICK_STATE_ARMED)
+ return;
+
+ slice_before = p->scx.slice;
+
+ switch (scenario) {
+ case KICK_IMMEDIATE:
+ scx_bpf_kick_cpu(target_cpu, SCX_KICK_PREEMPT);
+ break;
+ case KICK_LAZY:
+ scx_bpf_kick_cpu(target_cpu, SCX_KICK_PREEMPT_LAZY);
+ break;
+ case KICK_LAZY_THEN_IMMEDIATE:
+ scx_bpf_kick_cpu(target_cpu, SCX_KICK_PREEMPT_LAZY);
+ scx_bpf_kick_cpu(target_cpu, SCX_KICK_PREEMPT);
+ break;
+ case KICK_IMMEDIATE_THEN_LAZY:
+ scx_bpf_kick_cpu(target_cpu, SCX_KICK_PREEMPT);
+ scx_bpf_kick_cpu(target_cpu, SCX_KICK_PREEMPT_LAZY);
+ break;
+ case KICK_PLAIN_THEN_LAZY:
+ scx_bpf_kick_cpu(target_cpu, 0);
+ scx_bpf_kick_cpu(target_cpu, SCX_KICK_PREEMPT_LAZY);
+ break;
+ case KICK_LAZY_THEN_PLAIN:
+ scx_bpf_kick_cpu(target_cpu, SCX_KICK_PREEMPT_LAZY);
+ scx_bpf_kick_cpu(target_cpu, 0);
+ break;
+ case TICK_EXPIRY:
+ /* no kick: the slice runs out at the tick */
+ break;
+ }
+}
+
+SEC("fexit/__resched_curr")
+int BPF_PROG(kick_need_resched, struct rq *rq, int tif)
+{
+ struct task_struct *task = BPF_CORE_READ(rq, curr);
+
+ if (!task || BPF_CORE_READ(task, pid) != victim_pid ||
+ BPF_CORE_READ(rq, cpu) != target_cpu || state != KICK_STATE_QUEUED)
+ return 0;
+
+ slice_at_resched = BPF_CORE_READ(task, scx.slice);
+ resched_tif = tif;
+ state = KICK_STATE_RESCHED;
+ return 0;
+}
+
+void BPF_STRUCT_OPS(kick_stopping, struct task_struct *p, bool runnable)
+{
+ if (p->pid == victim_pid && state == KICK_STATE_RESCHED)
+ state = KICK_STATE_DONE;
+}
+
+void BPF_STRUCT_OPS(kick_exit, struct scx_exit_info *ei)
+{
+ UEI_RECORD(uei, ei);
+}
+
+SEC(".struct_ops.link")
+struct sched_ext_ops kick_ops = {
+ .enqueue = (void *)kick_enqueue,
+ .running = (void *)kick_running,
+ .stopping = (void *)kick_stopping,
+ .exit = (void *)kick_exit,
+ .name = "kick",
+};
diff --git a/tools/testing/selftests/sched_ext/kick.c b/tools/testing/selftests/sched_ext/kick.c
new file mode 100644
index 0000000000000..652f4c3add54b
--- /dev/null
+++ b/tools/testing/selftests/sched_ext/kick.c
@@ -0,0 +1,479 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES
+ */
+#define _GNU_SOURCE
+#include <bpf/bpf.h>
+#include <linux/sched.h>
+#include <sched.h>
+#include <signal.h>
+#include <stdio.h>
+#include <string.h>
+#include <sys/wait.h>
+#include <unistd.h>
+#include <scx/common.h>
+
+#include "kick.bpf.skel.h"
+#include "scx_test.h"
+
+#define WAIT_LOOPS 3000
+
+enum kick_scenario {
+ KICK_IMMEDIATE,
+ KICK_LAZY,
+ KICK_LAZY_THEN_IMMEDIATE,
+ KICK_IMMEDIATE_THEN_LAZY,
+ KICK_PLAIN_THEN_LAZY,
+ KICK_LAZY_THEN_PLAIN,
+ TICK_EXPIRY,
+ INVALID_ENQ_BOTH,
+ INVALID_KICK_BOTH,
+ INVALID_KICK_WAIT,
+ INVALID_KICK_IDLE,
+ INVALID_KICK_UNKNOWN,
+};
+
+enum kick_state {
+ KICK_STATE_IDLE,
+ KICK_STATE_ARMED,
+ KICK_STATE_QUEUED,
+ KICK_STATE_RESCHED,
+ KICK_STATE_DONE,
+};
+
+struct victim {
+ pid_t pid;
+ int start_fd;
+};
+
+struct observation {
+ u64 slice_before;
+ u64 slice_at_resched;
+ s32 resched_tif;
+};
+
+static bool enum_supported(const char *type, const char *name)
+{
+ u64 value;
+
+ return __COMPAT_read_enum(type, name, &value);
+}
+
+static int pick_target_cpu(void)
+{
+ cpu_set_t mask;
+ int cpu;
+
+ if (sched_getaffinity(0, sizeof(mask), &mask))
+ return -1;
+
+ for (cpu = 0; cpu < CPU_SETSIZE; cpu++)
+ if (CPU_ISSET(cpu, &mask))
+ return cpu;
+
+ return -1;
+}
+
+static struct victim spawn_victim(int cpu)
+{
+ struct victim victim = { .pid = -1, .start_fd = -1 };
+ int ready[2], start[2];
+ char byte = 1;
+
+ if (pipe(ready))
+ return victim;
+ if (pipe(start)) {
+ close(ready[0]);
+ close(ready[1]);
+ return victim;
+ }
+
+ victim.pid = fork();
+ if (!victim.pid) {
+ cpu_set_t mask;
+
+ close(ready[0]);
+ close(start[1]);
+ CPU_ZERO(&mask);
+ CPU_SET(cpu, &mask);
+ if (sched_setaffinity(0, sizeof(mask), &mask))
+ _exit(1);
+ if (write(ready[1], &byte, 1) != 1)
+ _exit(1);
+ close(ready[1]);
+ if (read(start[0], &byte, 1) != 1)
+ _exit(1);
+ close(start[0]);
+ for (;;)
+ asm volatile("" ::: "memory");
+ }
+ if (victim.pid < 0) {
+ close(ready[0]);
+ close(ready[1]);
+ close(start[0]);
+ close(start[1]);
+ return victim;
+ }
+
+ close(ready[1]);
+ close(start[0]);
+ if (read(ready[0], &byte, 1) != 1) {
+ close(ready[0]);
+ close(start[1]);
+ kill(victim.pid, SIGKILL);
+ waitpid(victim.pid, NULL, 0);
+ victim.pid = -1;
+ return victim;
+ }
+ close(ready[0]);
+ victim.start_fd = start[1];
+ return victim;
+}
+
+static void stop_victim(struct victim *victim)
+{
+ if (victim->start_fd >= 0)
+ close(victim->start_fd);
+ if (victim->pid > 0) {
+ kill(victim->pid, SIGKILL);
+ waitpid(victim->pid, NULL, 0);
+ }
+}
+
+static bool start_victim(struct victim *victim)
+{
+ char byte = 1;
+
+ if (write(victim->start_fd, &byte, 1) != 1)
+ return false;
+ close(victim->start_fd);
+ victim->start_fd = -1;
+ return true;
+}
+
+static bool wait_for_state(struct kick *skel, u32 wanted)
+{
+ int i;
+
+ for (i = 0; i < WAIT_LOOPS; i++) {
+ if (__atomic_load_n(&skel->bss->state, __ATOMIC_ACQUIRE) == wanted)
+ return true;
+ if (skel->data->uei.kind != EXIT_KIND(SCX_EXIT_NONE))
+ return false;
+ usleep(1000);
+ }
+ return false;
+}
+
+static enum scx_test_status trace_one(u32 scenario, u64 ops_flags, struct observation *obs)
+{
+ struct bpf_link *ops_link = NULL;
+ struct kick *skel = NULL;
+ struct victim victim;
+ enum scx_test_status ret = SCX_TEST_FAIL;
+ int cpu = pick_target_cpu();
+
+ if (cpu < 0) {
+ SCX_ERR("No available CPU");
+ return SCX_TEST_FAIL;
+ }
+ victim = spawn_victim(cpu);
+ if (victim.pid < 0) {
+ SCX_ERR("Failed to spawn victim");
+ return SCX_TEST_FAIL;
+ }
+
+ skel = kick__open();
+ if (!skel) {
+ SCX_ERR("Failed to open scenario %u", scenario);
+ goto out;
+ }
+ SCX_ENUM_INIT(skel);
+ skel->rodata->scenario = scenario;
+ skel->rodata->victim_pid = victim.pid;
+ skel->rodata->target_cpu = cpu;
+ skel->struct_ops.kick_ops->flags |= ops_flags;
+ if (kick__load(skel)) {
+ SCX_ERR("Failed to load scenario %u", scenario);
+ goto out;
+ }
+
+ bpf_map__set_autoattach(skel->maps.kick_ops, false);
+ if (kick__attach(skel)) {
+ SCX_ERR("Failed to attach __resched_curr tracer");
+ goto out;
+ }
+ skel->bss->state = KICK_STATE_ARMED;
+ ops_link = bpf_map__attach_struct_ops(skel->maps.kick_ops);
+ if (!ops_link) {
+ SCX_ERR("Failed to attach scenario %u", scenario);
+ goto out;
+ }
+ if (!start_victim(&victim)) {
+ SCX_ERR("Failed to start victim");
+ goto out;
+ }
+ if (!wait_for_state(skel, KICK_STATE_DONE)) {
+ SCX_ERR("Scenario %u stopped in state %u, exit kind %d", scenario, skel->bss->state,
+ skel->data->uei.kind);
+ goto out;
+ }
+
+ obs->slice_before = skel->bss->slice_before;
+ obs->slice_at_resched = skel->bss->slice_at_resched;
+ obs->resched_tif = skel->bss->resched_tif;
+ ret = SCX_TEST_PASS;
+out:
+ stop_victim(&victim);
+ if (ops_link)
+ bpf_link__destroy(ops_link);
+ if (skel)
+ kick__destroy(skel);
+ return ret;
+}
+
+static bool observation_valid(const struct observation *obs)
+{
+ return obs->slice_before > 0 && !obs->slice_at_resched;
+}
+
+static int active_lazy_mode(void)
+{
+ char buf[128];
+ FILE *file;
+
+ file = fopen("/sys/kernel/debug/sched/preempt", "r");
+ if (!file)
+ return -1;
+ if (!fgets(buf, sizeof(buf), file)) {
+ fclose(file);
+ return -1;
+ }
+ fclose(file);
+
+ if (strstr(buf, "(lazy)"))
+ return 1;
+ if (strstr(buf, "(full)"))
+ return 0;
+ return -1;
+}
+
+static enum scx_test_status setup_immediate(void **ctx)
+{
+ if (!enum_supported("scx_kick_flags", "SCX_KICK_PREEMPT")) {
+ printf("SKIP: SCX_KICK_PREEMPT is not supported\n");
+ return SCX_TEST_SKIP;
+ }
+ return SCX_TEST_PASS;
+}
+
+static enum scx_test_status setup_lazy(void **ctx)
+{
+ if (!enum_supported("scx_kick_flags", "SCX_KICK_PREEMPT_LAZY")) {
+ printf("SKIP: SCX_KICK_PREEMPT_LAZY is not supported\n");
+ return SCX_TEST_SKIP;
+ }
+ return setup_immediate(ctx);
+}
+
+static enum scx_test_status setup_tick(void **ctx)
+{
+ if (!enum_supported("scx_ops_flags", "SCX_OPS_LAZY_SLICE_EXPIRY")) {
+ printf("SKIP: SCX_OPS_LAZY_SLICE_EXPIRY is not supported\n");
+ return SCX_TEST_SKIP;
+ }
+ return setup_lazy(ctx);
+}
+
+static enum scx_test_status setup_invalid(void **ctx)
+{
+ if (!enum_supported("scx_enq_flags", "SCX_ENQ_PREEMPT_LAZY")) {
+ printf("SKIP: SCX_ENQ_PREEMPT_LAZY is not supported\n");
+ return SCX_TEST_SKIP;
+ }
+ return setup_lazy(ctx);
+}
+
+static enum scx_test_status run_immediate(void *ctx)
+{
+ struct observation obs;
+
+ SCX_EQ(trace_one(KICK_IMMEDIATE, 0, &obs), SCX_TEST_PASS);
+ SCX_ASSERT(observation_valid(&obs));
+ return SCX_TEST_PASS;
+}
+
+static enum scx_test_status run_lazy(void *ctx)
+{
+ struct observation immediate, lazy;
+ int lazy_mode;
+
+ SCX_EQ(trace_one(KICK_IMMEDIATE, 0, &immediate), SCX_TEST_PASS);
+ SCX_EQ(trace_one(KICK_LAZY, 0, &lazy), SCX_TEST_PASS);
+ SCX_ASSERT(observation_valid(&immediate));
+ SCX_ASSERT(observation_valid(&lazy));
+
+ lazy_mode = active_lazy_mode();
+ if (lazy_mode > 0)
+ SCX_FAIL_IF(lazy.resched_tif == immediate.resched_tif,
+ "Lazy mode used immediate TIF %d", lazy.resched_tif);
+ else if (!lazy_mode)
+ SCX_EQ(lazy.resched_tif, immediate.resched_tif);
+ else
+ printf("INFO: preemption mode unavailable; lazy TIF was %d, immediate TIF was %d\n",
+ lazy.resched_tif, immediate.resched_tif);
+
+ return SCX_TEST_PASS;
+}
+
+static enum scx_test_status run_coalesce(void *ctx)
+{
+ struct observation immediate, lazy_first, immediate_first;
+ struct observation plain_first, lazy_then_plain;
+
+ SCX_EQ(trace_one(KICK_IMMEDIATE, 0, &immediate), SCX_TEST_PASS);
+ SCX_EQ(trace_one(KICK_LAZY_THEN_IMMEDIATE, 0, &lazy_first), SCX_TEST_PASS);
+ SCX_EQ(trace_one(KICK_IMMEDIATE_THEN_LAZY, 0, &immediate_first), SCX_TEST_PASS);
+ SCX_ASSERT(observation_valid(&lazy_first));
+ SCX_ASSERT(observation_valid(&immediate_first));
+ SCX_EQ(lazy_first.resched_tif, immediate.resched_tif);
+ SCX_EQ(immediate_first.resched_tif, immediate.resched_tif);
+
+ /*
+ * A plain kick in the same batch doesn't clear the slice by itself.
+ * The lazy preemption must still expire it, and the plain kick must
+ * still reschedule immediately, whichever came first.
+ */
+ SCX_EQ(trace_one(KICK_PLAIN_THEN_LAZY, 0, &plain_first), SCX_TEST_PASS);
+ SCX_EQ(trace_one(KICK_LAZY_THEN_PLAIN, 0, &lazy_then_plain), SCX_TEST_PASS);
+ SCX_ASSERT(observation_valid(&plain_first));
+ SCX_ASSERT(observation_valid(&lazy_then_plain));
+ SCX_EQ(plain_first.resched_tif, immediate.resched_tif);
+ SCX_EQ(lazy_then_plain.resched_tif, immediate.resched_tif);
+ return SCX_TEST_PASS;
+}
+
+/*
+ * A slice running out at the tick reschedules immediately by default and
+ * lazily with SCX_OPS_LAZY_SLICE_EXPIRY, the way fair.c expires a slice.
+ */
+static enum scx_test_status run_tick(void *ctx)
+{
+ struct observation immediate, lazy;
+ u64 lazy_flag;
+ int lazy_mode;
+
+ SCX_ASSERT(__COMPAT_read_enum("scx_ops_flags", "SCX_OPS_LAZY_SLICE_EXPIRY", &lazy_flag));
+ SCX_EQ(trace_one(TICK_EXPIRY, 0, &immediate), SCX_TEST_PASS);
+ SCX_EQ(trace_one(TICK_EXPIRY, lazy_flag, &lazy), SCX_TEST_PASS);
+ SCX_ASSERT(observation_valid(&immediate));
+ SCX_ASSERT(observation_valid(&lazy));
+
+ lazy_mode = active_lazy_mode();
+ if (lazy_mode > 0)
+ SCX_FAIL_IF(lazy.resched_tif == immediate.resched_tif,
+ "Lazy slice expiry used immediate TIF %d", lazy.resched_tif);
+ else if (!lazy_mode)
+ SCX_EQ(lazy.resched_tif, immediate.resched_tif);
+ else
+ printf("INFO: preemption mode unavailable; lazy TIF was %d, immediate TIF was %d\n",
+ lazy.resched_tif, immediate.resched_tif);
+
+ return SCX_TEST_PASS;
+}
+
+static enum scx_test_status invalid_one(u32 scenario)
+{
+ struct bpf_link *ops_link = NULL;
+ struct kick *skel = NULL;
+ struct victim victim;
+ enum scx_test_status ret = SCX_TEST_FAIL;
+ int cpu = pick_target_cpu();
+ int i;
+
+ if (cpu < 0)
+ return SCX_TEST_FAIL;
+ victim = spawn_victim(cpu);
+ if (victim.pid < 0)
+ return SCX_TEST_FAIL;
+
+ skel = kick__open();
+ if (!skel)
+ goto out;
+ SCX_ENUM_INIT(skel);
+ skel->rodata->scenario = scenario;
+ if (kick__load(skel))
+ goto out;
+ ops_link = bpf_map__attach_struct_ops(skel->maps.kick_ops);
+ if (!ops_link || !start_victim(&victim))
+ goto out;
+
+ for (i = 0; i < WAIT_LOOPS; i++) {
+ if (skel->data->uei.kind == EXIT_KIND(SCX_EXIT_ERROR)) {
+ ret = SCX_TEST_PASS;
+ break;
+ }
+ usleep(1000);
+ }
+out:
+ stop_victim(&victim);
+ if (ops_link)
+ bpf_link__destroy(ops_link);
+ if (skel)
+ kick__destroy(skel);
+ return ret;
+}
+
+static enum scx_test_status run_invalid(void *ctx)
+{
+ u32 scenario;
+
+ for (scenario = INVALID_ENQ_BOTH; scenario <= INVALID_KICK_UNKNOWN; scenario++)
+ SCX_EQ(invalid_one(scenario), SCX_TEST_PASS);
+ return SCX_TEST_PASS;
+}
+
+static struct scx_test kick_immediate = {
+ .name = "kick_immediate",
+ .description = "Trace immediate kick slice expiration and rescheduling",
+ .setup = setup_immediate,
+ .run = run_immediate,
+};
+
+static struct scx_test kick_lazy = {
+ .name = "kick_lazy",
+ .description = "Trace lazy kick slice expiration and rescheduling",
+ .setup = setup_lazy,
+ .run = run_lazy,
+};
+
+static struct scx_test kick_coalesce = {
+ .name = "kick_coalesce",
+ .description = "Verify lazy kicks coalesce with immediate and plain kicks",
+ .setup = setup_lazy,
+ .run = run_coalesce,
+};
+
+static struct scx_test kick_tick = {
+ .name = "kick_tick",
+ .description = "Trace slice expiry at the tick, immediate and lazy",
+ .setup = setup_tick,
+ .run = run_tick,
+};
+
+static struct scx_test kick_invalid = {
+ .name = "kick_invalid",
+ .description = "Verify invalid enqueue and kick flag combinations fail",
+ .setup = setup_invalid,
+ .run = run_invalid,
+};
+
+__attribute__((constructor))
+static void register_kick_tests(void)
+{
+ scx_test_register(&kick_immediate);
+ scx_test_register(&kick_lazy);
+ scx_test_register(&kick_coalesce);
+ scx_test_register(&kick_tick);
+ scx_test_register(&kick_invalid);
+}
--
2.55.0
^ permalink raw reply related [flat|nested] 4+ messages in thread