* Re: [PATCH -next v5 09/22] entry: Split generic entry into irq and syscall
@ 2024-12-09 4:04 kernel test robot
0 siblings, 0 replies; 3+ messages in thread
From: kernel test robot @ 2024-12-09 4:04 UTC (permalink / raw)
To: oe-kbuild; +Cc: lkp
::::::
:::::: Manual check reason: "low confidence bisect report"
::::::
BCC: lkp@intel.com
CC: oe-kbuild-all@lists.linux.dev
In-Reply-To: <20241206101744.4161990-10-ruanjinjie@huawei.com>
References: <20241206101744.4161990-10-ruanjinjie@huawei.com>
TO: Jinjie Ruan <ruanjinjie@huawei.com>
TO: catalin.marinas@arm.com
TO: will@kernel.org
TO: oleg@redhat.com
TO: sstabellini@kernel.org
TO: tglx@linutronix.de
TO: peterz@infradead.org
TO: luto@kernel.org
TO: mingo@redhat.com
TO: juri.lelli@redhat.com
TO: vincent.guittot@linaro.org
TO: dietmar.eggemann@arm.com
TO: rostedt@goodmis.org
TO: bsegall@google.com
TO: mgorman@suse.de
TO: vschneid@redhat.com
TO: kees@kernel.org
TO: wad@chromium.org
TO: akpm@linux-foundation.org
TO: samitolvanen@google.com
TO: masahiroy@kernel.org
TO: hca@linux.ibm.com
TO: aliceryhl@google.com
TO: rppt@kernel.org
TO: xur@google.com
TO: paulmck@kernel.org
TO: arnd@arndb.de
TO: mbenes@suse.cz
TO: puranjay@kernel.org
TO: mark.rutland@arm.com
TO: ruanjinjie@huawei.com
Hi Jinjie,
kernel test robot noticed the following build warnings:
[auto build test WARNING on next-20241205]
url: https://github.com/intel-lab-lkp/linux/commits/Jinjie-Ruan/arm64-ptrace-Replace-interrupts_enabled-with-regs_irqs_disabled/20241206-183134
base: next-20241205
patch link: https://lore.kernel.org/r/20241206101744.4161990-10-ruanjinjie%40huawei.com
patch subject: [PATCH -next v5 09/22] entry: Split generic entry into irq and syscall
:::::: branch date: 17 hours ago
:::::: commit date: 17 hours ago
compiler: clang version 19.1.3 (https://github.com/llvm/llvm-project ab51eccf88f5321e7c60591c5546b254b6afab99)
If you fix the issue in a separate patch/commit (i.e. not just a new version of
the same patch/commit), kindly add following tags
| Reported-by: kernel test robot <lkp@intel.com>
| Closes: https://lore.kernel.org/r/202412071107.6IQ3yz7r-lkp@intel.com/
includecheck warnings: (new ones prefixed by >>)
kernel/sched/core.c: linux/sched/rseq_api.h is included more than once.
kernel/sched/core.c: stats.h is included more than once.
>> kernel/sched/core.c: linux/irq-entry-common.h is included more than once.
vim +72 kernel/sched/core.c
69
70 #ifdef CONFIG_PREEMPT_DYNAMIC
71 # ifdef CONFIG_GENERIC_IRQ_ENTRY
> 72 # include <linux/irq-entry-common.h>
73 # endif
74 #endif
75
76 #include <uapi/linux/sched/types.h>
77
78 #include <asm/irq_regs.h>
79 #include <asm/switch_to.h>
80 #include <asm/tlb.h>
81
82 #define CREATE_TRACE_POINTS
83 #include <linux/sched/rseq_api.h>
84 #include <trace/events/sched.h>
85 #include <trace/events/ipi.h>
86 #undef CREATE_TRACE_POINTS
87
88 #include "sched.h"
89 #include "stats.h"
90
91 #include "autogroup.h"
92 #include "pelt.h"
93 #include "smp.h"
94 #include "stats.h"
95
96 #include "../workqueue_internal.h"
97 #include "../../io_uring/io-wq.h"
98 #include "../smpboot.h"
99
100 EXPORT_TRACEPOINT_SYMBOL_GPL(ipi_send_cpu);
101 EXPORT_TRACEPOINT_SYMBOL_GPL(ipi_send_cpumask);
102
103 /*
104 * Export tracepoints that act as a bare tracehook (ie: have no trace event
105 * associated with them) to allow external modules to probe them.
106 */
107 EXPORT_TRACEPOINT_SYMBOL_GPL(pelt_cfs_tp);
108 EXPORT_TRACEPOINT_SYMBOL_GPL(pelt_rt_tp);
109 EXPORT_TRACEPOINT_SYMBOL_GPL(pelt_dl_tp);
110 EXPORT_TRACEPOINT_SYMBOL_GPL(pelt_irq_tp);
111 EXPORT_TRACEPOINT_SYMBOL_GPL(pelt_se_tp);
112 EXPORT_TRACEPOINT_SYMBOL_GPL(pelt_hw_tp);
113 EXPORT_TRACEPOINT_SYMBOL_GPL(sched_cpu_capacity_tp);
114 EXPORT_TRACEPOINT_SYMBOL_GPL(sched_overutilized_tp);
115 EXPORT_TRACEPOINT_SYMBOL_GPL(sched_util_est_cfs_tp);
116 EXPORT_TRACEPOINT_SYMBOL_GPL(sched_util_est_se_tp);
117 EXPORT_TRACEPOINT_SYMBOL_GPL(sched_update_nr_running_tp);
118 EXPORT_TRACEPOINT_SYMBOL_GPL(sched_compute_energy_tp);
119
120 DEFINE_PER_CPU_SHARED_ALIGNED(struct rq, runqueues);
121
122 #ifdef CONFIG_SCHED_DEBUG
123 /*
124 * Debugging: various feature bits
125 *
126 * If SCHED_DEBUG is disabled, each compilation unit has its own copy of
127 * sysctl_sched_features, defined in sched.h, to allow constants propagation
128 * at compile time and compiler optimization based on features default.
129 */
130 #define SCHED_FEAT(name, enabled) \
131 (1UL << __SCHED_FEAT_##name) * enabled |
132 const_debug unsigned int sysctl_sched_features =
133 #include "features.h"
134 0;
135 #undef SCHED_FEAT
136
137 /*
138 * Print a warning if need_resched is set for the given duration (if
139 * LATENCY_WARN is enabled).
140 *
141 * If sysctl_resched_latency_warn_once is set, only one warning will be shown
142 * per boot.
143 */
144 __read_mostly int sysctl_resched_latency_warn_ms = 100;
145 __read_mostly int sysctl_resched_latency_warn_once = 1;
146 #endif /* CONFIG_SCHED_DEBUG */
147
148 /*
149 * Number of tasks to iterate in a single balance run.
150 * Limited because this is done with IRQs disabled.
151 */
152 const_debug unsigned int sysctl_sched_nr_migrate = SCHED_NR_MIGRATE_BREAK;
153
154 __read_mostly int scheduler_running;
155
156 #ifdef CONFIG_SCHED_CORE
157
158 DEFINE_STATIC_KEY_FALSE(__sched_core_enabled);
159
160 /* kernel prio, less is more */
161 static inline int __task_prio(const struct task_struct *p)
162 {
163 if (p->sched_class == &stop_sched_class) /* trumps deadline */
164 return -2;
165
166 if (p->dl_server)
167 return -1; /* deadline */
168
169 if (rt_or_dl_prio(p->prio))
170 return p->prio; /* [-1, 99] */
171
172 if (p->sched_class == &idle_sched_class)
173 return MAX_RT_PRIO + NICE_WIDTH; /* 140 */
174
175 if (task_on_scx(p))
176 return MAX_RT_PRIO + MAX_NICE + 1; /* 120, squash ext */
177
178 return MAX_RT_PRIO + MAX_NICE; /* 119, squash fair */
179 }
180
181 /*
182 * l(a,b)
183 * le(a,b) := !l(b,a)
184 * g(a,b) := l(b,a)
185 * ge(a,b) := !l(a,b)
186 */
187
188 /* real prio, less is less */
189 static inline bool prio_less(const struct task_struct *a,
190 const struct task_struct *b, bool in_fi)
191 {
192
193 int pa = __task_prio(a), pb = __task_prio(b);
194
195 if (-pa < -pb)
196 return true;
197
198 if (-pb < -pa)
199 return false;
200
201 if (pa == -1) { /* dl_prio() doesn't work because of stop_class above */
202 const struct sched_dl_entity *a_dl, *b_dl;
203
204 a_dl = &a->dl;
205 /*
206 * Since,'a' and 'b' can be CFS tasks served by DL server,
207 * __task_prio() can return -1 (for DL) even for those. In that
208 * case, get to the dl_server's DL entity.
209 */
210 if (a->dl_server)
211 a_dl = a->dl_server;
212
213 b_dl = &b->dl;
214 if (b->dl_server)
215 b_dl = b->dl_server;
216
217 return !dl_time_before(a_dl->deadline, b_dl->deadline);
218 }
219
220 if (pa == MAX_RT_PRIO + MAX_NICE) /* fair */
221 return cfs_prio_less(a, b, in_fi);
222
223 #ifdef CONFIG_SCHED_CLASS_EXT
224 if (pa == MAX_RT_PRIO + MAX_NICE + 1) /* ext */
225 return scx_prio_less(a, b, in_fi);
226 #endif
227
228 return false;
229 }
230
231 static inline bool __sched_core_less(const struct task_struct *a,
232 const struct task_struct *b)
233 {
234 if (a->core_cookie < b->core_cookie)
235 return true;
236
237 if (a->core_cookie > b->core_cookie)
238 return false;
239
240 /* flip prio, so high prio is leftmost */
241 if (prio_less(b, a, !!task_rq(a)->core->core_forceidle_count))
242 return true;
243
244 return false;
245 }
246
247 #define __node_2_sc(node) rb_entry((node), struct task_struct, core_node)
248
249 static inline bool rb_sched_core_less(struct rb_node *a, const struct rb_node *b)
250 {
251 return __sched_core_less(__node_2_sc(a), __node_2_sc(b));
252 }
253
254 static inline int rb_sched_core_cmp(const void *key, const struct rb_node *node)
255 {
256 const struct task_struct *p = __node_2_sc(node);
257 unsigned long cookie = (unsigned long)key;
258
259 if (cookie < p->core_cookie)
260 return -1;
261
262 if (cookie > p->core_cookie)
263 return 1;
264
265 return 0;
266 }
267
268 void sched_core_enqueue(struct rq *rq, struct task_struct *p)
269 {
270 if (p->se.sched_delayed)
271 return;
272
273 rq->core->core_task_seq++;
274
275 if (!p->core_cookie)
276 return;
277
278 rb_add(&p->core_node, &rq->core_tree, rb_sched_core_less);
279 }
280
281 void sched_core_dequeue(struct rq *rq, struct task_struct *p, int flags)
282 {
283 if (p->se.sched_delayed)
284 return;
285
286 rq->core->core_task_seq++;
287
288 if (sched_core_enqueued(p)) {
289 rb_erase(&p->core_node, &rq->core_tree);
290 RB_CLEAR_NODE(&p->core_node);
291 }
292
293 /*
294 * Migrating the last task off the cpu, with the cpu in forced idle
295 * state. Reschedule to create an accounting edge for forced idle,
296 * and re-examine whether the core is still in forced idle state.
297 */
298 if (!(flags & DEQUEUE_SAVE) && rq->nr_running == 1 &&
299 rq->core->core_forceidle_count && rq->curr == rq->idle)
300 resched_curr(rq);
301 }
302
303 static int sched_task_is_throttled(struct task_struct *p, int cpu)
304 {
305 if (p->sched_class->task_is_throttled)
306 return p->sched_class->task_is_throttled(p, cpu);
307
308 return 0;
309 }
310
311 static struct task_struct *sched_core_next(struct task_struct *p, unsigned long cookie)
312 {
313 struct rb_node *node = &p->core_node;
314 int cpu = task_cpu(p);
315
316 do {
317 node = rb_next(node);
318 if (!node)
319 return NULL;
320
321 p = __node_2_sc(node);
322 if (p->core_cookie != cookie)
323 return NULL;
324
325 } while (sched_task_is_throttled(p, cpu));
326
327 return p;
328 }
329
330 /*
331 * Find left-most (aka, highest priority) and unthrottled task matching @cookie.
332 * If no suitable task is found, NULL will be returned.
333 */
334 static struct task_struct *sched_core_find(struct rq *rq, unsigned long cookie)
335 {
336 struct task_struct *p;
337 struct rb_node *node;
338
339 node = rb_find_first((void *)cookie, &rq->core_tree, rb_sched_core_cmp);
340 if (!node)
341 return NULL;
342
343 p = __node_2_sc(node);
344 if (!sched_task_is_throttled(p, rq->cpu))
345 return p;
346
347 return sched_core_next(p, cookie);
348 }
349
350 /*
351 * Magic required such that:
352 *
353 * raw_spin_rq_lock(rq);
354 * ...
355 * raw_spin_rq_unlock(rq);
356 *
357 * ends up locking and unlocking the _same_ lock, and all CPUs
358 * always agree on what rq has what lock.
359 *
360 * XXX entirely possible to selectively enable cores, don't bother for now.
361 */
362
363 static DEFINE_MUTEX(sched_core_mutex);
364 static atomic_t sched_core_count;
365 static struct cpumask sched_core_mask;
366
367 static void sched_core_lock(int cpu, unsigned long *flags)
368 {
369 const struct cpumask *smt_mask = cpu_smt_mask(cpu);
370 int t, i = 0;
371
372 local_irq_save(*flags);
373 for_each_cpu(t, smt_mask)
374 raw_spin_lock_nested(&cpu_rq(t)->__lock, i++);
375 }
376
377 static void sched_core_unlock(int cpu, unsigned long *flags)
378 {
379 const struct cpumask *smt_mask = cpu_smt_mask(cpu);
380 int t;
381
382 for_each_cpu(t, smt_mask)
383 raw_spin_unlock(&cpu_rq(t)->__lock);
384 local_irq_restore(*flags);
385 }
386
387 static void __sched_core_flip(bool enabled)
388 {
389 unsigned long flags;
390 int cpu, t;
391
392 cpus_read_lock();
393
394 /*
395 * Toggle the online cores, one by one.
396 */
397 cpumask_copy(&sched_core_mask, cpu_online_mask);
398 for_each_cpu(cpu, &sched_core_mask) {
399 const struct cpumask *smt_mask = cpu_smt_mask(cpu);
400
401 sched_core_lock(cpu, &flags);
402
403 for_each_cpu(t, smt_mask)
404 cpu_rq(t)->core_enabled = enabled;
405
406 cpu_rq(cpu)->core->core_forceidle_start = 0;
407
408 sched_core_unlock(cpu, &flags);
409
410 cpumask_andnot(&sched_core_mask, &sched_core_mask, smt_mask);
411 }
412
413 /*
414 * Toggle the offline CPUs.
415 */
416 for_each_cpu_andnot(cpu, cpu_possible_mask, cpu_online_mask)
417 cpu_rq(cpu)->core_enabled = enabled;
418
419 cpus_read_unlock();
420 }
421
422 static void sched_core_assert_empty(void)
423 {
424 int cpu;
425
426 for_each_possible_cpu(cpu)
427 WARN_ON_ONCE(!RB_EMPTY_ROOT(&cpu_rq(cpu)->core_tree));
428 }
429
430 static void __sched_core_enable(void)
431 {
432 static_branch_enable(&__sched_core_enabled);
433 /*
434 * Ensure all previous instances of raw_spin_rq_*lock() have finished
435 * and future ones will observe !sched_core_disabled().
436 */
437 synchronize_rcu();
438 __sched_core_flip(true);
439 sched_core_assert_empty();
440 }
441
442 static void __sched_core_disable(void)
443 {
444 sched_core_assert_empty();
445 __sched_core_flip(false);
446 static_branch_disable(&__sched_core_enabled);
447 }
448
449 void sched_core_get(void)
450 {
451 if (atomic_inc_not_zero(&sched_core_count))
452 return;
453
454 mutex_lock(&sched_core_mutex);
455 if (!atomic_read(&sched_core_count))
456 __sched_core_enable();
457
458 smp_mb__before_atomic();
459 atomic_inc(&sched_core_count);
460 mutex_unlock(&sched_core_mutex);
461 }
462
463 static void __sched_core_put(struct work_struct *work)
464 {
465 if (atomic_dec_and_mutex_lock(&sched_core_count, &sched_core_mutex)) {
466 __sched_core_disable();
467 mutex_unlock(&sched_core_mutex);
468 }
469 }
470
471 void sched_core_put(void)
472 {
473 static DECLARE_WORK(_work, __sched_core_put);
474
475 /*
476 * "There can be only one"
477 *
478 * Either this is the last one, or we don't actually need to do any
479 * 'work'. If it is the last *again*, we rely on
480 * WORK_STRUCT_PENDING_BIT.
481 */
482 if (!atomic_add_unless(&sched_core_count, -1, 1))
483 schedule_work(&_work);
484 }
485
486 #else /* !CONFIG_SCHED_CORE */
487
488 static inline void sched_core_enqueue(struct rq *rq, struct task_struct *p) { }
489 static inline void
490 sched_core_dequeue(struct rq *rq, struct task_struct *p, int flags) { }
491
492 #endif /* CONFIG_SCHED_CORE */
493
494 /*
495 * Serialization rules:
496 *
497 * Lock order:
498 *
499 * p->pi_lock
500 * rq->lock
501 * hrtimer_cpu_base->lock (hrtimer_start() for bandwidth controls)
502 *
503 * rq1->lock
504 * rq2->lock where: rq1 < rq2
505 *
506 * Regular state:
507 *
508 * Normal scheduling state is serialized by rq->lock. __schedule() takes the
509 * local CPU's rq->lock, it optionally removes the task from the runqueue and
510 * always looks at the local rq data structures to find the most eligible task
511 * to run next.
512 *
513 * Task enqueue is also under rq->lock, possibly taken from another CPU.
514 * Wakeups from another LLC domain might use an IPI to transfer the enqueue to
515 * the local CPU to avoid bouncing the runqueue state around [ see
516 * ttwu_queue_wakelist() ]
517 *
518 * Task wakeup, specifically wakeups that involve migration, are horribly
519 * complicated to avoid having to take two rq->locks.
520 *
521 * Special state:
522 *
523 * System-calls and anything external will use task_rq_lock() which acquires
524 * both p->pi_lock and rq->lock. As a consequence the state they change is
525 * stable while holding either lock:
526 *
527 * - sched_setaffinity()/
528 * set_cpus_allowed_ptr(): p->cpus_ptr, p->nr_cpus_allowed
529 * - set_user_nice(): p->se.load, p->*prio
530 * - __sched_setscheduler(): p->sched_class, p->policy, p->*prio,
531 * p->se.load, p->rt_priority,
532 * p->dl.dl_{runtime, deadline, period, flags, bw, density}
533 * - sched_setnuma(): p->numa_preferred_nid
534 * - sched_move_task(): p->sched_task_group
535 * - uclamp_update_active() p->uclamp*
536 *
537 * p->state <- TASK_*:
538 *
539 * is changed locklessly using set_current_state(), __set_current_state() or
540 * set_special_state(), see their respective comments, or by
541 * try_to_wake_up(). This latter uses p->pi_lock to serialize against
542 * concurrent self.
543 *
544 * p->on_rq <- { 0, 1 = TASK_ON_RQ_QUEUED, 2 = TASK_ON_RQ_MIGRATING }:
545 *
546 * is set by activate_task() and cleared by deactivate_task(), under
547 * rq->lock. Non-zero indicates the task is runnable, the special
548 * ON_RQ_MIGRATING state is used for migration without holding both
549 * rq->locks. It indicates task_cpu() is not stable, see task_rq_lock().
550 *
551 * Additionally it is possible to be ->on_rq but still be considered not
552 * runnable when p->se.sched_delayed is true. These tasks are on the runqueue
553 * but will be dequeued as soon as they get picked again. See the
554 * task_is_runnable() helper.
555 *
556 * p->on_cpu <- { 0, 1 }:
557 *
558 * is set by prepare_task() and cleared by finish_task() such that it will be
559 * set before p is scheduled-in and cleared after p is scheduled-out, both
560 * under rq->lock. Non-zero indicates the task is running on its CPU.
561 *
562 * [ The astute reader will observe that it is possible for two tasks on one
563 * CPU to have ->on_cpu = 1 at the same time. ]
564 *
565 * task_cpu(p): is changed by set_task_cpu(), the rules are:
566 *
567 * - Don't call set_task_cpu() on a blocked task:
568 *
569 * We don't care what CPU we're not running on, this simplifies hotplug,
570 * the CPU assignment of blocked tasks isn't required to be valid.
571 *
572 * - for try_to_wake_up(), called under p->pi_lock:
573 *
574 * This allows try_to_wake_up() to only take one rq->lock, see its comment.
575 *
576 * - for migration called under rq->lock:
577 * [ see task_on_rq_migrating() in task_rq_lock() ]
578 *
579 * o move_queued_task()
580 * o detach_task()
581 *
582 * - for migration called under double_rq_lock():
583 *
584 * o __migrate_swap_task()
585 * o push_rt_task() / pull_rt_task()
586 * o push_dl_task() / pull_dl_task()
587 * o dl_task_offline_migration()
588 *
589 */
590
591 void raw_spin_rq_lock_nested(struct rq *rq, int subclass)
592 {
593 raw_spinlock_t *lock;
594
595 /* Matches synchronize_rcu() in __sched_core_enable() */
596 preempt_disable();
597 if (sched_core_disabled()) {
598 raw_spin_lock_nested(&rq->__lock, subclass);
599 /* preempt_count *MUST* be > 1 */
600 preempt_enable_no_resched();
601 return;
602 }
603
604 for (;;) {
605 lock = __rq_lockp(rq);
606 raw_spin_lock_nested(lock, subclass);
607 if (likely(lock == __rq_lockp(rq))) {
608 /* preempt_count *MUST* be > 1 */
609 preempt_enable_no_resched();
610 return;
611 }
612 raw_spin_unlock(lock);
613 }
614 }
615
616 bool raw_spin_rq_trylock(struct rq *rq)
617 {
618 raw_spinlock_t *lock;
619 bool ret;
620
621 /* Matches synchronize_rcu() in __sched_core_enable() */
622 preempt_disable();
623 if (sched_core_disabled()) {
624 ret = raw_spin_trylock(&rq->__lock);
625 preempt_enable();
626 return ret;
627 }
628
629 for (;;) {
630 lock = __rq_lockp(rq);
631 ret = raw_spin_trylock(lock);
632 if (!ret || (likely(lock == __rq_lockp(rq)))) {
633 preempt_enable();
634 return ret;
635 }
636 raw_spin_unlock(lock);
637 }
638 }
639
640 void raw_spin_rq_unlock(struct rq *rq)
641 {
642 raw_spin_unlock(rq_lockp(rq));
643 }
644
645 #ifdef CONFIG_SMP
646 /*
647 * double_rq_lock - safely lock two runqueues
648 */
649 void double_rq_lock(struct rq *rq1, struct rq *rq2)
650 {
651 lockdep_assert_irqs_disabled();
652
653 if (rq_order_less(rq2, rq1))
654 swap(rq1, rq2);
655
656 raw_spin_rq_lock(rq1);
657 if (__rq_lockp(rq1) != __rq_lockp(rq2))
658 raw_spin_rq_lock_nested(rq2, SINGLE_DEPTH_NESTING);
659
660 double_rq_clock_clear_update(rq1, rq2);
661 }
662 #endif
663
664 /*
665 * __task_rq_lock - lock the rq @p resides on.
666 */
667 struct rq *__task_rq_lock(struct task_struct *p, struct rq_flags *rf)
668 __acquires(rq->lock)
669 {
670 struct rq *rq;
671
672 lockdep_assert_held(&p->pi_lock);
673
674 for (;;) {
675 rq = task_rq(p);
676 raw_spin_rq_lock(rq);
677 if (likely(rq == task_rq(p) && !task_on_rq_migrating(p))) {
678 rq_pin_lock(rq, rf);
679 return rq;
680 }
681 raw_spin_rq_unlock(rq);
682
683 while (unlikely(task_on_rq_migrating(p)))
684 cpu_relax();
685 }
686 }
687
688 /*
689 * task_rq_lock - lock p->pi_lock and lock the rq @p resides on.
690 */
691 struct rq *task_rq_lock(struct task_struct *p, struct rq_flags *rf)
692 __acquires(p->pi_lock)
693 __acquires(rq->lock)
694 {
695 struct rq *rq;
696
697 for (;;) {
698 raw_spin_lock_irqsave(&p->pi_lock, rf->flags);
699 rq = task_rq(p);
700 raw_spin_rq_lock(rq);
701 /*
702 * move_queued_task() task_rq_lock()
703 *
704 * ACQUIRE (rq->lock)
705 * [S] ->on_rq = MIGRATING [L] rq = task_rq()
706 * WMB (__set_task_cpu()) ACQUIRE (rq->lock);
707 * [S] ->cpu = new_cpu [L] task_rq()
708 * [L] ->on_rq
709 * RELEASE (rq->lock)
710 *
711 * If we observe the old CPU in task_rq_lock(), the acquire of
712 * the old rq->lock will fully serialize against the stores.
713 *
714 * If we observe the new CPU in task_rq_lock(), the address
715 * dependency headed by '[L] rq = task_rq()' and the acquire
716 * will pair with the WMB to ensure we then also see migrating.
717 */
718 if (likely(rq == task_rq(p) && !task_on_rq_migrating(p))) {
719 rq_pin_lock(rq, rf);
720 return rq;
721 }
722 raw_spin_rq_unlock(rq);
723 raw_spin_unlock_irqrestore(&p->pi_lock, rf->flags);
724
725 while (unlikely(task_on_rq_migrating(p)))
726 cpu_relax();
727 }
728 }
729
730 /*
731 * RQ-clock updating methods:
732 */
733
734 static void update_rq_clock_task(struct rq *rq, s64 delta)
735 {
736 /*
737 * In theory, the compile should just see 0 here, and optimize out the call
738 * to sched_rt_avg_update. But I don't trust it...
739 */
740 s64 __maybe_unused steal = 0, irq_delta = 0;
741
742 #ifdef CONFIG_IRQ_TIME_ACCOUNTING
743 irq_delta = irq_time_read(cpu_of(rq)) - rq->prev_irq_time;
744
745 /*
746 * Since irq_time is only updated on {soft,}irq_exit, we might run into
747 * this case when a previous update_rq_clock() happened inside a
748 * {soft,}IRQ region.
749 *
750 * When this happens, we stop ->clock_task and only update the
751 * prev_irq_time stamp to account for the part that fit, so that a next
752 * update will consume the rest. This ensures ->clock_task is
753 * monotonic.
754 *
755 * It does however cause some slight miss-attribution of {soft,}IRQ
756 * time, a more accurate solution would be to update the irq_time using
757 * the current rq->clock timestamp, except that would require using
758 * atomic ops.
759 */
760 if (irq_delta > delta)
761 irq_delta = delta;
762
763 rq->prev_irq_time += irq_delta;
764 delta -= irq_delta;
765 delayacct_irq(rq->curr, irq_delta);
766 #endif
767 #ifdef CONFIG_PARAVIRT_TIME_ACCOUNTING
768 if (static_key_false((¶virt_steal_rq_enabled))) {
769 u64 prev_steal;
770
771 steal = prev_steal = paravirt_steal_clock(cpu_of(rq));
772 steal -= rq->prev_steal_time_rq;
773
774 if (unlikely(steal > delta))
775 steal = delta;
776
777 rq->prev_steal_time_rq = prev_steal;
778 delta -= steal;
779 }
780 #endif
781
782 rq->clock_task += delta;
783
784 #ifdef CONFIG_HAVE_SCHED_AVG_IRQ
785 if ((irq_delta + steal) && sched_feat(NONTASK_CAPACITY))
786 update_irq_load_avg(rq, irq_delta + steal);
787 #endif
788 update_rq_clock_pelt(rq, delta);
789 }
790
791 void update_rq_clock(struct rq *rq)
792 {
793 s64 delta;
794
795 lockdep_assert_rq_held(rq);
796
797 if (rq->clock_update_flags & RQCF_ACT_SKIP)
798 return;
799
800 #ifdef CONFIG_SCHED_DEBUG
801 if (sched_feat(WARN_DOUBLE_CLOCK))
802 SCHED_WARN_ON(rq->clock_update_flags & RQCF_UPDATED);
803 rq->clock_update_flags |= RQCF_UPDATED;
804 #endif
805
806 delta = sched_clock_cpu(cpu_of(rq)) - rq->clock;
807 if (delta < 0)
808 return;
809 rq->clock += delta;
810 update_rq_clock_task(rq, delta);
811 }
812
813 #ifdef CONFIG_SCHED_HRTICK
814 /*
815 * Use HR-timers to deliver accurate preemption points.
816 */
817
818 static void hrtick_clear(struct rq *rq)
819 {
820 if (hrtimer_active(&rq->hrtick_timer))
821 hrtimer_cancel(&rq->hrtick_timer);
822 }
823
824 /*
825 * High-resolution timer tick.
826 * Runs from hardirq context with interrupts disabled.
827 */
828 static enum hrtimer_restart hrtick(struct hrtimer *timer)
829 {
830 struct rq *rq = container_of(timer, struct rq, hrtick_timer);
831 struct rq_flags rf;
832
833 WARN_ON_ONCE(cpu_of(rq) != smp_processor_id());
834
835 rq_lock(rq, &rf);
836 update_rq_clock(rq);
837 rq->donor->sched_class->task_tick(rq, rq->curr, 1);
838 rq_unlock(rq, &rf);
839
840 return HRTIMER_NORESTART;
841 }
842
843 #ifdef CONFIG_SMP
844
845 static void __hrtick_restart(struct rq *rq)
846 {
847 struct hrtimer *timer = &rq->hrtick_timer;
848 ktime_t time = rq->hrtick_time;
849
850 hrtimer_start(timer, time, HRTIMER_MODE_ABS_PINNED_HARD);
851 }
852
853 /*
854 * called from hardirq (IPI) context
855 */
856 static void __hrtick_start(void *arg)
857 {
858 struct rq *rq = arg;
859 struct rq_flags rf;
860
861 rq_lock(rq, &rf);
862 __hrtick_restart(rq);
863 rq_unlock(rq, &rf);
864 }
865
866 /*
867 * Called to set the hrtick timer state.
868 *
869 * called with rq->lock held and IRQs disabled
870 */
871 void hrtick_start(struct rq *rq, u64 delay)
872 {
873 struct hrtimer *timer = &rq->hrtick_timer;
874 s64 delta;
875
876 /*
877 * Don't schedule slices shorter than 10000ns, that just
878 * doesn't make sense and can cause timer DoS.
879 */
880 delta = max_t(s64, delay, 10000LL);
881 rq->hrtick_time = ktime_add_ns(timer->base->get_time(), delta);
882
883 if (rq == this_rq())
884 __hrtick_restart(rq);
885 else
886 smp_call_function_single_async(cpu_of(rq), &rq->hrtick_csd);
887 }
888
889 #else
890 /*
891 * Called to set the hrtick timer state.
892 *
893 * called with rq->lock held and IRQs disabled
894 */
895 void hrtick_start(struct rq *rq, u64 delay)
896 {
897 /*
898 * Don't schedule slices shorter than 10000ns, that just
899 * doesn't make sense. Rely on vruntime for fairness.
900 */
901 delay = max_t(u64, delay, 10000LL);
902 hrtimer_start(&rq->hrtick_timer, ns_to_ktime(delay),
903 HRTIMER_MODE_REL_PINNED_HARD);
904 }
905
906 #endif /* CONFIG_SMP */
907
908 static void hrtick_rq_init(struct rq *rq)
909 {
910 #ifdef CONFIG_SMP
911 INIT_CSD(&rq->hrtick_csd, __hrtick_start, rq);
912 #endif
913 hrtimer_init(&rq->hrtick_timer, CLOCK_MONOTONIC, HRTIMER_MODE_REL_HARD);
914 rq->hrtick_timer.function = hrtick;
915 }
916 #else /* CONFIG_SCHED_HRTICK */
917 static inline void hrtick_clear(struct rq *rq)
918 {
919 }
920
921 static inline void hrtick_rq_init(struct rq *rq)
922 {
923 }
924 #endif /* CONFIG_SCHED_HRTICK */
925
926 /*
927 * try_cmpxchg based fetch_or() macro so it works for different integer types:
928 */
929 #define fetch_or(ptr, mask) \
930 ({ \
931 typeof(ptr) _ptr = (ptr); \
932 typeof(mask) _mask = (mask); \
933 typeof(*_ptr) _val = *_ptr; \
934 \
935 do { \
936 } while (!try_cmpxchg(_ptr, &_val, _val | _mask)); \
937 _val; \
938 })
939
940 #if defined(CONFIG_SMP) && defined(TIF_POLLING_NRFLAG)
941 /*
942 * Atomically set TIF_NEED_RESCHED and test for TIF_POLLING_NRFLAG,
943 * this avoids any races wrt polling state changes and thereby avoids
944 * spurious IPIs.
945 */
946 static inline bool set_nr_and_not_polling(struct thread_info *ti, int tif)
947 {
948 return !(fetch_or(&ti->flags, 1 << tif) & _TIF_POLLING_NRFLAG);
949 }
950
951 /*
952 * Atomically set TIF_NEED_RESCHED if TIF_POLLING_NRFLAG is set.
953 *
954 * If this returns true, then the idle task promises to call
955 * sched_ttwu_pending() and reschedule soon.
956 */
957 static bool set_nr_if_polling(struct task_struct *p)
958 {
959 struct thread_info *ti = task_thread_info(p);
960 typeof(ti->flags) val = READ_ONCE(ti->flags);
961
962 do {
963 if (!(val & _TIF_POLLING_NRFLAG))
964 return false;
965 if (val & _TIF_NEED_RESCHED)
966 return true;
967 } while (!try_cmpxchg(&ti->flags, &val, val | _TIF_NEED_RESCHED));
968
969 return true;
970 }
971
972 #else
973 static inline bool set_nr_and_not_polling(struct thread_info *ti, int tif)
974 {
975 set_ti_thread_flag(ti, tif);
976 return true;
977 }
978
979 #ifdef CONFIG_SMP
980 static inline bool set_nr_if_polling(struct task_struct *p)
981 {
982 return false;
983 }
984 #endif
985 #endif
986
987 static bool __wake_q_add(struct wake_q_head *head, struct task_struct *task)
988 {
989 struct wake_q_node *node = &task->wake_q;
990
991 /*
992 * Atomically grab the task, if ->wake_q is !nil already it means
993 * it's already queued (either by us or someone else) and will get the
994 * wakeup due to that.
995 *
996 * In order to ensure that a pending wakeup will observe our pending
997 * state, even in the failed case, an explicit smp_mb() must be used.
998 */
999 smp_mb__before_atomic();
1000 if (unlikely(cmpxchg_relaxed(&node->next, NULL, WAKE_Q_TAIL)))
1001 return false;
1002
1003 /*
1004 * The head is context local, there can be no concurrency.
1005 */
1006 *head->lastp = node;
1007 head->lastp = &node->next;
1008 return true;
1009 }
1010
1011 /**
1012 * wake_q_add() - queue a wakeup for 'later' waking.
1013 * @head: the wake_q_head to add @task to
1014 * @task: the task to queue for 'later' wakeup
1015 *
1016 * Queue a task for later wakeup, most likely by the wake_up_q() call in the
1017 * same context, _HOWEVER_ this is not guaranteed, the wakeup can come
1018 * instantly.
1019 *
1020 * This function must be used as-if it were wake_up_process(); IOW the task
1021 * must be ready to be woken at this location.
1022 */
1023 void wake_q_add(struct wake_q_head *head, struct task_struct *task)
1024 {
1025 if (__wake_q_add(head, task))
1026 get_task_struct(task);
1027 }
1028
1029 /**
1030 * wake_q_add_safe() - safely queue a wakeup for 'later' waking.
1031 * @head: the wake_q_head to add @task to
1032 * @task: the task to queue for 'later' wakeup
1033 *
1034 * Queue a task for later wakeup, most likely by the wake_up_q() call in the
1035 * same context, _HOWEVER_ this is not guaranteed, the wakeup can come
1036 * instantly.
1037 *
1038 * This function must be used as-if it were wake_up_process(); IOW the task
1039 * must be ready to be woken at this location.
1040 *
1041 * This function is essentially a task-safe equivalent to wake_q_add(). Callers
1042 * that already hold reference to @task can call the 'safe' version and trust
1043 * wake_q to do the right thing depending whether or not the @task is already
1044 * queued for wakeup.
1045 */
1046 void wake_q_add_safe(struct wake_q_head *head, struct task_struct *task)
1047 {
1048 if (!__wake_q_add(head, task))
1049 put_task_struct(task);
1050 }
1051
1052 void wake_up_q(struct wake_q_head *head)
1053 {
1054 struct wake_q_node *node = head->first;
1055
1056 while (node != WAKE_Q_TAIL) {
1057 struct task_struct *task;
1058
1059 task = container_of(node, struct task_struct, wake_q);
1060 /* Task can safely be re-inserted now: */
1061 node = node->next;
1062 task->wake_q.next = NULL;
1063
1064 /*
1065 * wake_up_process() executes a full barrier, which pairs with
1066 * the queueing in wake_q_add() so as not to miss wakeups.
1067 */
1068 wake_up_process(task);
1069 put_task_struct(task);
1070 }
1071 }
1072
1073 /*
1074 * resched_curr - mark rq's current task 'to be rescheduled now'.
1075 *
1076 * On UP this means the setting of the need_resched flag, on SMP it
1077 * might also involve a cross-CPU call to trigger the scheduler on
1078 * the target CPU.
1079 */
1080 static void __resched_curr(struct rq *rq, int tif)
1081 {
1082 struct task_struct *curr = rq->curr;
1083 struct thread_info *cti = task_thread_info(curr);
1084 int cpu;
1085
1086 lockdep_assert_rq_held(rq);
1087
1088 /*
1089 * Always immediately preempt the idle task; no point in delaying doing
1090 * actual work.
1091 */
1092 if (is_idle_task(curr) && tif == TIF_NEED_RESCHED_LAZY)
1093 tif = TIF_NEED_RESCHED;
1094
1095 if (cti->flags & ((1 << tif) | _TIF_NEED_RESCHED))
1096 return;
1097
1098 cpu = cpu_of(rq);
1099
1100 if (cpu == smp_processor_id()) {
1101 set_ti_thread_flag(cti, tif);
1102 if (tif == TIF_NEED_RESCHED)
1103 set_preempt_need_resched();
1104 return;
1105 }
1106
1107 if (set_nr_and_not_polling(cti, tif)) {
1108 if (tif == TIF_NEED_RESCHED)
1109 smp_send_reschedule(cpu);
1110 } else {
1111 trace_sched_wake_idle_without_ipi(cpu);
1112 }
1113 }
1114
1115 void resched_curr(struct rq *rq)
1116 {
1117 __resched_curr(rq, TIF_NEED_RESCHED);
1118 }
1119
1120 #ifdef CONFIG_PREEMPT_DYNAMIC
1121 static DEFINE_STATIC_KEY_FALSE(sk_dynamic_preempt_lazy);
1122 static __always_inline bool dynamic_preempt_lazy(void)
1123 {
1124 return static_branch_unlikely(&sk_dynamic_preempt_lazy);
1125 }
1126 #else
1127 static __always_inline bool dynamic_preempt_lazy(void)
1128 {
1129 return IS_ENABLED(CONFIG_PREEMPT_LAZY);
1130 }
1131 #endif
1132
1133 static __always_inline int get_lazy_tif_bit(void)
1134 {
1135 if (dynamic_preempt_lazy())
1136 return TIF_NEED_RESCHED_LAZY;
1137
1138 return TIF_NEED_RESCHED;
1139 }
1140
1141 void resched_curr_lazy(struct rq *rq)
1142 {
1143 __resched_curr(rq, get_lazy_tif_bit());
1144 }
1145
1146 void resched_cpu(int cpu)
1147 {
1148 struct rq *rq = cpu_rq(cpu);
1149 unsigned long flags;
1150
1151 raw_spin_rq_lock_irqsave(rq, flags);
1152 if (cpu_online(cpu) || cpu == smp_processor_id())
1153 resched_curr(rq);
1154 raw_spin_rq_unlock_irqrestore(rq, flags);
1155 }
1156
1157 #ifdef CONFIG_SMP
1158 #ifdef CONFIG_NO_HZ_COMMON
1159 /*
1160 * In the semi idle case, use the nearest busy CPU for migrating timers
1161 * from an idle CPU. This is good for power-savings.
1162 *
1163 * We don't do similar optimization for completely idle system, as
1164 * selecting an idle CPU will add more delays to the timers than intended
1165 * (as that CPU's timer base may not be up to date wrt jiffies etc).
1166 */
1167 int get_nohz_timer_target(void)
1168 {
1169 int i, cpu = smp_processor_id(), default_cpu = -1;
1170 struct sched_domain *sd;
1171 const struct cpumask *hk_mask;
1172
1173 if (housekeeping_cpu(cpu, HK_TYPE_KERNEL_NOISE)) {
1174 if (!idle_cpu(cpu))
1175 return cpu;
1176 default_cpu = cpu;
1177 }
1178
1179 hk_mask = housekeeping_cpumask(HK_TYPE_KERNEL_NOISE);
1180
1181 guard(rcu)();
1182
1183 for_each_domain(cpu, sd) {
1184 for_each_cpu_and(i, sched_domain_span(sd), hk_mask) {
1185 if (cpu == i)
1186 continue;
1187
1188 if (!idle_cpu(i))
1189 return i;
1190 }
1191 }
1192
1193 if (default_cpu == -1)
1194 default_cpu = housekeeping_any_cpu(HK_TYPE_KERNEL_NOISE);
1195
1196 return default_cpu;
1197 }
1198
1199 /*
1200 * When add_timer_on() enqueues a timer into the timer wheel of an
1201 * idle CPU then this timer might expire before the next timer event
1202 * which is scheduled to wake up that CPU. In case of a completely
1203 * idle system the next event might even be infinite time into the
1204 * future. wake_up_idle_cpu() ensures that the CPU is woken up and
1205 * leaves the inner idle loop so the newly added timer is taken into
1206 * account when the CPU goes back to idle and evaluates the timer
1207 * wheel for the next timer event.
1208 */
1209 static void wake_up_idle_cpu(int cpu)
1210 {
1211 struct rq *rq = cpu_rq(cpu);
1212
1213 if (cpu == smp_processor_id())
1214 return;
1215
1216 /*
1217 * Set TIF_NEED_RESCHED and send an IPI if in the non-polling
1218 * part of the idle loop. This forces an exit from the idle loop
1219 * and a round trip to schedule(). Now this could be optimized
1220 * because a simple new idle loop iteration is enough to
1221 * re-evaluate the next tick. Provided some re-ordering of tick
1222 * nohz functions that would need to follow TIF_NR_POLLING
1223 * clearing:
1224 *
1225 * - On most architectures, a simple fetch_or on ti::flags with a
1226 * "0" value would be enough to know if an IPI needs to be sent.
1227 *
1228 * - x86 needs to perform a last need_resched() check between
1229 * monitor and mwait which doesn't take timers into account.
1230 * There a dedicated TIF_TIMER flag would be required to
1231 * fetch_or here and be checked along with TIF_NEED_RESCHED
1232 * before mwait().
1233 *
1234 * However, remote timer enqueue is not such a frequent event
1235 * and testing of the above solutions didn't appear to report
1236 * much benefits.
1237 */
1238 if (set_nr_and_not_polling(task_thread_info(rq->idle), TIF_NEED_RESCHED))
1239 smp_send_reschedule(cpu);
1240 else
1241 trace_sched_wake_idle_without_ipi(cpu);
1242 }
1243
1244 static bool wake_up_full_nohz_cpu(int cpu)
1245 {
1246 /*
1247 * We just need the target to call irq_exit() and re-evaluate
1248 * the next tick. The nohz full kick at least implies that.
1249 * If needed we can still optimize that later with an
1250 * empty IRQ.
1251 */
1252 if (cpu_is_offline(cpu))
1253 return true; /* Don't try to wake offline CPUs. */
1254 if (tick_nohz_full_cpu(cpu)) {
1255 if (cpu != smp_processor_id() ||
1256 tick_nohz_tick_stopped())
1257 tick_nohz_full_kick_cpu(cpu);
1258 return true;
1259 }
1260
1261 return false;
1262 }
1263
1264 /*
1265 * Wake up the specified CPU. If the CPU is going offline, it is the
1266 * caller's responsibility to deal with the lost wakeup, for example,
1267 * by hooking into the CPU_DEAD notifier like timers and hrtimers do.
1268 */
1269 void wake_up_nohz_cpu(int cpu)
1270 {
1271 if (!wake_up_full_nohz_cpu(cpu))
1272 wake_up_idle_cpu(cpu);
1273 }
1274
1275 static void nohz_csd_func(void *info)
1276 {
1277 struct rq *rq = info;
1278 int cpu = cpu_of(rq);
1279 unsigned int flags;
1280
1281 /*
1282 * Release the rq::nohz_csd.
1283 */
1284 flags = atomic_fetch_andnot(NOHZ_KICK_MASK | NOHZ_NEWILB_KICK, nohz_flags(cpu));
1285 WARN_ON(!(flags & NOHZ_KICK_MASK));
1286
1287 rq->idle_balance = idle_cpu(cpu);
1288 if (rq->idle_balance) {
1289 rq->nohz_idle_balance = flags;
1290 __raise_softirq_irqoff(SCHED_SOFTIRQ);
1291 }
1292 }
1293
1294 #endif /* CONFIG_NO_HZ_COMMON */
1295
1296 #ifdef CONFIG_NO_HZ_FULL
1297 static inline bool __need_bw_check(struct rq *rq, struct task_struct *p)
1298 {
1299 if (rq->nr_running != 1)
1300 return false;
1301
1302 if (p->sched_class != &fair_sched_class)
1303 return false;
1304
1305 if (!task_on_rq_queued(p))
1306 return false;
1307
1308 return true;
1309 }
1310
1311 bool sched_can_stop_tick(struct rq *rq)
1312 {
1313 int fifo_nr_running;
1314
1315 /* Deadline tasks, even if single, need the tick */
1316 if (rq->dl.dl_nr_running)
1317 return false;
1318
1319 /*
1320 * If there are more than one RR tasks, we need the tick to affect the
1321 * actual RR behaviour.
1322 */
1323 if (rq->rt.rr_nr_running) {
1324 if (rq->rt.rr_nr_running == 1)
1325 return true;
1326 else
1327 return false;
1328 }
1329
1330 /*
1331 * If there's no RR tasks, but FIFO tasks, we can skip the tick, no
1332 * forced preemption between FIFO tasks.
1333 */
1334 fifo_nr_running = rq->rt.rt_nr_running - rq->rt.rr_nr_running;
1335 if (fifo_nr_running)
1336 return true;
1337
1338 /*
1339 * If there are no DL,RR/FIFO tasks, there must only be CFS or SCX tasks
1340 * left. For CFS, if there's more than one we need the tick for
1341 * involuntary preemption. For SCX, ask.
1342 */
1343 if (scx_enabled() && !scx_can_stop_tick(rq))
1344 return false;
1345
1346 if (rq->cfs.nr_running > 1)
1347 return false;
1348
1349 /*
1350 * If there is one task and it has CFS runtime bandwidth constraints
1351 * and it's on the cpu now we don't want to stop the tick.
1352 * This check prevents clearing the bit if a newly enqueued task here is
1353 * dequeued by migrating while the constrained task continues to run.
1354 * E.g. going from 2->1 without going through pick_next_task().
1355 */
1356 if (__need_bw_check(rq, rq->curr)) {
1357 if (cfs_task_bw_constrained(rq->curr))
1358 return false;
1359 }
1360
1361 return true;
1362 }
1363 #endif /* CONFIG_NO_HZ_FULL */
1364 #endif /* CONFIG_SMP */
1365
1366 #if defined(CONFIG_RT_GROUP_SCHED) || (defined(CONFIG_FAIR_GROUP_SCHED) && \
1367 (defined(CONFIG_SMP) || defined(CONFIG_CFS_BANDWIDTH)))
1368 /*
1369 * Iterate task_group tree rooted at *from, calling @down when first entering a
1370 * node and @up when leaving it for the final time.
1371 *
1372 * Caller must hold rcu_lock or sufficient equivalent.
1373 */
1374 int walk_tg_tree_from(struct task_group *from,
1375 tg_visitor down, tg_visitor up, void *data)
1376 {
1377 struct task_group *parent, *child;
1378 int ret;
1379
1380 parent = from;
1381
1382 down:
1383 ret = (*down)(parent, data);
1384 if (ret)
1385 goto out;
1386 list_for_each_entry_rcu(child, &parent->children, siblings) {
1387 parent = child;
1388 goto down;
1389
1390 up:
1391 continue;
1392 }
1393 ret = (*up)(parent, data);
1394 if (ret || parent == from)
1395 goto out;
1396
1397 child = parent;
1398 parent = parent->parent;
1399 if (parent)
1400 goto up;
1401 out:
1402 return ret;
1403 }
1404
1405 int tg_nop(struct task_group *tg, void *data)
1406 {
1407 return 0;
1408 }
1409 #endif
1410
1411 void set_load_weight(struct task_struct *p, bool update_load)
1412 {
1413 int prio = p->static_prio - MAX_RT_PRIO;
1414 struct load_weight lw;
1415
1416 if (task_has_idle_policy(p)) {
1417 lw.weight = scale_load(WEIGHT_IDLEPRIO);
1418 lw.inv_weight = WMULT_IDLEPRIO;
1419 } else {
1420 lw.weight = scale_load(sched_prio_to_weight[prio]);
1421 lw.inv_weight = sched_prio_to_wmult[prio];
1422 }
1423
1424 /*
1425 * SCHED_OTHER tasks have to update their load when changing their
1426 * weight
1427 */
1428 if (update_load && p->sched_class->reweight_task)
1429 p->sched_class->reweight_task(task_rq(p), p, &lw);
1430 else
1431 p->se.load = lw;
1432 }
1433
1434 #ifdef CONFIG_UCLAMP_TASK
1435 /*
1436 * Serializes updates of utilization clamp values
1437 *
1438 * The (slow-path) user-space triggers utilization clamp value updates which
1439 * can require updates on (fast-path) scheduler's data structures used to
1440 * support enqueue/dequeue operations.
1441 * While the per-CPU rq lock protects fast-path update operations, user-space
1442 * requests are serialized using a mutex to reduce the risk of conflicting
1443 * updates or API abuses.
1444 */
1445 static __maybe_unused DEFINE_MUTEX(uclamp_mutex);
1446
1447 /* Max allowed minimum utilization */
1448 static unsigned int __maybe_unused sysctl_sched_uclamp_util_min = SCHED_CAPACITY_SCALE;
1449
1450 /* Max allowed maximum utilization */
1451 static unsigned int __maybe_unused sysctl_sched_uclamp_util_max = SCHED_CAPACITY_SCALE;
1452
1453 /*
1454 * By default RT tasks run at the maximum performance point/capacity of the
1455 * system. Uclamp enforces this by always setting UCLAMP_MIN of RT tasks to
1456 * SCHED_CAPACITY_SCALE.
1457 *
1458 * This knob allows admins to change the default behavior when uclamp is being
1459 * used. In battery powered devices, particularly, running at the maximum
1460 * capacity and frequency will increase energy consumption and shorten the
1461 * battery life.
1462 *
1463 * This knob only affects RT tasks that their uclamp_se->user_defined == false.
1464 *
1465 * This knob will not override the system default sched_util_clamp_min defined
1466 * above.
1467 */
1468 unsigned int sysctl_sched_uclamp_util_min_rt_default = SCHED_CAPACITY_SCALE;
1469
1470 /* All clamps are required to be less or equal than these values */
1471 static struct uclamp_se uclamp_default[UCLAMP_CNT];
1472
1473 /*
1474 * This static key is used to reduce the uclamp overhead in the fast path. It
1475 * primarily disables the call to uclamp_rq_{inc, dec}() in
1476 * enqueue/dequeue_task().
1477 *
1478 * This allows users to continue to enable uclamp in their kernel config with
1479 * minimum uclamp overhead in the fast path.
1480 *
1481 * As soon as userspace modifies any of the uclamp knobs, the static key is
1482 * enabled, since we have an actual users that make use of uclamp
1483 * functionality.
1484 *
1485 * The knobs that would enable this static key are:
1486 *
1487 * * A task modifying its uclamp value with sched_setattr().
1488 * * An admin modifying the sysctl_sched_uclamp_{min, max} via procfs.
1489 * * An admin modifying the cgroup cpu.uclamp.{min, max}
1490 */
1491 DEFINE_STATIC_KEY_FALSE(sched_uclamp_used);
1492
1493 static inline unsigned int
1494 uclamp_idle_value(struct rq *rq, enum uclamp_id clamp_id,
1495 unsigned int clamp_value)
1496 {
1497 /*
1498 * Avoid blocked utilization pushing up the frequency when we go
1499 * idle (which drops the max-clamp) by retaining the last known
1500 * max-clamp.
1501 */
1502 if (clamp_id == UCLAMP_MAX) {
1503 rq->uclamp_flags |= UCLAMP_FLAG_IDLE;
1504 return clamp_value;
1505 }
1506
1507 return uclamp_none(UCLAMP_MIN);
1508 }
1509
1510 static inline void uclamp_idle_reset(struct rq *rq, enum uclamp_id clamp_id,
1511 unsigned int clamp_value)
1512 {
1513 /* Reset max-clamp retention only on idle exit */
1514 if (!(rq->uclamp_flags & UCLAMP_FLAG_IDLE))
1515 return;
1516
1517 uclamp_rq_set(rq, clamp_id, clamp_value);
1518 }
1519
1520 static inline
1521 unsigned int uclamp_rq_max_value(struct rq *rq, enum uclamp_id clamp_id,
1522 unsigned int clamp_value)
1523 {
1524 struct uclamp_bucket *bucket = rq->uclamp[clamp_id].bucket;
1525 int bucket_id = UCLAMP_BUCKETS - 1;
1526
1527 /*
1528 * Since both min and max clamps are max aggregated, find the
1529 * top most bucket with tasks in.
1530 */
1531 for ( ; bucket_id >= 0; bucket_id--) {
1532 if (!bucket[bucket_id].tasks)
1533 continue;
1534 return bucket[bucket_id].value;
1535 }
1536
1537 /* No tasks -- default clamp values */
1538 return uclamp_idle_value(rq, clamp_id, clamp_value);
1539 }
1540
1541 static void __uclamp_update_util_min_rt_default(struct task_struct *p)
1542 {
1543 unsigned int default_util_min;
1544 struct uclamp_se *uc_se;
1545
1546 lockdep_assert_held(&p->pi_lock);
1547
1548 uc_se = &p->uclamp_req[UCLAMP_MIN];
1549
1550 /* Only sync if user didn't override the default */
1551 if (uc_se->user_defined)
1552 return;
1553
1554 default_util_min = sysctl_sched_uclamp_util_min_rt_default;
1555 uclamp_se_set(uc_se, default_util_min, false);
1556 }
1557
1558 static void uclamp_update_util_min_rt_default(struct task_struct *p)
1559 {
1560 if (!rt_task(p))
1561 return;
1562
1563 /* Protect updates to p->uclamp_* */
1564 guard(task_rq_lock)(p);
1565 __uclamp_update_util_min_rt_default(p);
1566 }
1567
1568 static inline struct uclamp_se
1569 uclamp_tg_restrict(struct task_struct *p, enum uclamp_id clamp_id)
1570 {
1571 /* Copy by value as we could modify it */
1572 struct uclamp_se uc_req = p->uclamp_req[clamp_id];
1573 #ifdef CONFIG_UCLAMP_TASK_GROUP
1574 unsigned int tg_min, tg_max, value;
1575
1576 /*
1577 * Tasks in autogroups or root task group will be
1578 * restricted by system defaults.
1579 */
1580 if (task_group_is_autogroup(task_group(p)))
1581 return uc_req;
1582 if (task_group(p) == &root_task_group)
1583 return uc_req;
1584
1585 tg_min = task_group(p)->uclamp[UCLAMP_MIN].value;
1586 tg_max = task_group(p)->uclamp[UCLAMP_MAX].value;
1587 value = uc_req.value;
1588 value = clamp(value, tg_min, tg_max);
1589 uclamp_se_set(&uc_req, value, false);
1590 #endif
1591
1592 return uc_req;
1593 }
1594
1595 /*
1596 * The effective clamp bucket index of a task depends on, by increasing
1597 * priority:
1598 * - the task specific clamp value, when explicitly requested from userspace
1599 * - the task group effective clamp value, for tasks not either in the root
1600 * group or in an autogroup
1601 * - the system default clamp value, defined by the sysadmin
1602 */
1603 static inline struct uclamp_se
1604 uclamp_eff_get(struct task_struct *p, enum uclamp_id clamp_id)
1605 {
1606 struct uclamp_se uc_req = uclamp_tg_restrict(p, clamp_id);
1607 struct uclamp_se uc_max = uclamp_default[clamp_id];
1608
1609 /* System default restrictions always apply */
1610 if (unlikely(uc_req.value > uc_max.value))
1611 return uc_max;
1612
1613 return uc_req;
1614 }
1615
1616 unsigned long uclamp_eff_value(struct task_struct *p, enum uclamp_id clamp_id)
1617 {
1618 struct uclamp_se uc_eff;
1619
1620 /* Task currently refcounted: use back-annotated (effective) value */
1621 if (p->uclamp[clamp_id].active)
1622 return (unsigned long)p->uclamp[clamp_id].value;
1623
1624 uc_eff = uclamp_eff_get(p, clamp_id);
1625
1626 return (unsigned long)uc_eff.value;
1627 }
1628
1629 /*
1630 * When a task is enqueued on a rq, the clamp bucket currently defined by the
1631 * task's uclamp::bucket_id is refcounted on that rq. This also immediately
1632 * updates the rq's clamp value if required.
1633 *
1634 * Tasks can have a task-specific value requested from user-space, track
1635 * within each bucket the maximum value for tasks refcounted in it.
1636 * This "local max aggregation" allows to track the exact "requested" value
1637 * for each bucket when all its RUNNABLE tasks require the same clamp.
1638 */
1639 static inline void uclamp_rq_inc_id(struct rq *rq, struct task_struct *p,
1640 enum uclamp_id clamp_id)
1641 {
1642 struct uclamp_rq *uc_rq = &rq->uclamp[clamp_id];
1643 struct uclamp_se *uc_se = &p->uclamp[clamp_id];
1644 struct uclamp_bucket *bucket;
1645
1646 lockdep_assert_rq_held(rq);
1647
1648 /* Update task effective clamp */
1649 p->uclamp[clamp_id] = uclamp_eff_get(p, clamp_id);
1650
1651 bucket = &uc_rq->bucket[uc_se->bucket_id];
1652 bucket->tasks++;
1653 uc_se->active = true;
1654
1655 uclamp_idle_reset(rq, clamp_id, uc_se->value);
1656
1657 /*
1658 * Local max aggregation: rq buckets always track the max
1659 * "requested" clamp value of its RUNNABLE tasks.
1660 */
1661 if (bucket->tasks == 1 || uc_se->value > bucket->value)
1662 bucket->value = uc_se->value;
1663
1664 if (uc_se->value > uclamp_rq_get(rq, clamp_id))
1665 uclamp_rq_set(rq, clamp_id, uc_se->value);
1666 }
1667
1668 /*
1669 * When a task is dequeued from a rq, the clamp bucket refcounted by the task
1670 * is released. If this is the last task reference counting the rq's max
1671 * active clamp value, then the rq's clamp value is updated.
1672 *
1673 * Both refcounted tasks and rq's cached clamp values are expected to be
1674 * always valid. If it's detected they are not, as defensive programming,
1675 * enforce the expected state and warn.
1676 */
1677 static inline void uclamp_rq_dec_id(struct rq *rq, struct task_struct *p,
1678 enum uclamp_id clamp_id)
1679 {
1680 struct uclamp_rq *uc_rq = &rq->uclamp[clamp_id];
1681 struct uclamp_se *uc_se = &p->uclamp[clamp_id];
1682 struct uclamp_bucket *bucket;
1683 unsigned int bkt_clamp;
1684 unsigned int rq_clamp;
1685
1686 lockdep_assert_rq_held(rq);
1687
1688 /*
1689 * If sched_uclamp_used was enabled after task @p was enqueued,
1690 * we could end up with unbalanced call to uclamp_rq_dec_id().
1691 *
1692 * In this case the uc_se->active flag should be false since no uclamp
1693 * accounting was performed at enqueue time and we can just return
1694 * here.
1695 *
1696 * Need to be careful of the following enqueue/dequeue ordering
1697 * problem too
1698 *
1699 * enqueue(taskA)
1700 * // sched_uclamp_used gets enabled
1701 * enqueue(taskB)
1702 * dequeue(taskA)
1703 * // Must not decrement bucket->tasks here
1704 * dequeue(taskB)
1705 *
1706 * where we could end up with stale data in uc_se and
1707 * bucket[uc_se->bucket_id].
1708 *
1709 * The following check here eliminates the possibility of such race.
1710 */
1711 if (unlikely(!uc_se->active))
1712 return;
1713
1714 bucket = &uc_rq->bucket[uc_se->bucket_id];
1715
1716 SCHED_WARN_ON(!bucket->tasks);
1717 if (likely(bucket->tasks))
1718 bucket->tasks--;
1719
1720 uc_se->active = false;
1721
1722 /*
1723 * Keep "local max aggregation" simple and accept to (possibly)
1724 * overboost some RUNNABLE tasks in the same bucket.
1725 * The rq clamp bucket value is reset to its base value whenever
1726 * there are no more RUNNABLE tasks refcounting it.
1727 */
1728 if (likely(bucket->tasks))
1729 return;
1730
1731 rq_clamp = uclamp_rq_get(rq, clamp_id);
1732 /*
1733 * Defensive programming: this should never happen. If it happens,
1734 * e.g. due to future modification, warn and fix up the expected value.
1735 */
1736 SCHED_WARN_ON(bucket->value > rq_clamp);
1737 if (bucket->value >= rq_clamp) {
1738 bkt_clamp = uclamp_rq_max_value(rq, clamp_id, uc_se->value);
1739 uclamp_rq_set(rq, clamp_id, bkt_clamp);
1740 }
1741 }
1742
1743 static inline void uclamp_rq_inc(struct rq *rq, struct task_struct *p)
1744 {
1745 enum uclamp_id clamp_id;
1746
1747 /*
1748 * Avoid any overhead until uclamp is actually used by the userspace.
1749 *
1750 * The condition is constructed such that a NOP is generated when
1751 * sched_uclamp_used is disabled.
1752 */
1753 if (!static_branch_unlikely(&sched_uclamp_used))
1754 return;
1755
1756 if (unlikely(!p->sched_class->uclamp_enabled))
1757 return;
1758
1759 if (p->se.sched_delayed)
1760 return;
1761
1762 for_each_clamp_id(clamp_id)
1763 uclamp_rq_inc_id(rq, p, clamp_id);
1764
1765 /* Reset clamp idle holding when there is one RUNNABLE task */
1766 if (rq->uclamp_flags & UCLAMP_FLAG_IDLE)
1767 rq->uclamp_flags &= ~UCLAMP_FLAG_IDLE;
1768 }
1769
1770 static inline void uclamp_rq_dec(struct rq *rq, struct task_struct *p)
1771 {
1772 enum uclamp_id clamp_id;
1773
1774 /*
1775 * Avoid any overhead until uclamp is actually used by the userspace.
1776 *
1777 * The condition is constructed such that a NOP is generated when
1778 * sched_uclamp_used is disabled.
1779 */
1780 if (!static_branch_unlikely(&sched_uclamp_used))
1781 return;
1782
1783 if (unlikely(!p->sched_class->uclamp_enabled))
1784 return;
1785
1786 if (p->se.sched_delayed)
1787 return;
1788
1789 for_each_clamp_id(clamp_id)
1790 uclamp_rq_dec_id(rq, p, clamp_id);
1791 }
1792
1793 static inline void uclamp_rq_reinc_id(struct rq *rq, struct task_struct *p,
1794 enum uclamp_id clamp_id)
1795 {
1796 if (!p->uclamp[clamp_id].active)
1797 return;
1798
1799 uclamp_rq_dec_id(rq, p, clamp_id);
1800 uclamp_rq_inc_id(rq, p, clamp_id);
1801
1802 /*
1803 * Make sure to clear the idle flag if we've transiently reached 0
1804 * active tasks on rq.
1805 */
1806 if (clamp_id == UCLAMP_MAX && (rq->uclamp_flags & UCLAMP_FLAG_IDLE))
1807 rq->uclamp_flags &= ~UCLAMP_FLAG_IDLE;
1808 }
1809
1810 static inline void
1811 uclamp_update_active(struct task_struct *p)
1812 {
1813 enum uclamp_id clamp_id;
1814 struct rq_flags rf;
1815 struct rq *rq;
1816
1817 /*
1818 * Lock the task and the rq where the task is (or was) queued.
1819 *
1820 * We might lock the (previous) rq of a !RUNNABLE task, but that's the
1821 * price to pay to safely serialize util_{min,max} updates with
1822 * enqueues, dequeues and migration operations.
1823 * This is the same locking schema used by __set_cpus_allowed_ptr().
1824 */
1825 rq = task_rq_lock(p, &rf);
1826
1827 /*
1828 * Setting the clamp bucket is serialized by task_rq_lock().
1829 * If the task is not yet RUNNABLE and its task_struct is not
1830 * affecting a valid clamp bucket, the next time it's enqueued,
1831 * it will already see the updated clamp bucket value.
1832 */
1833 for_each_clamp_id(clamp_id)
1834 uclamp_rq_reinc_id(rq, p, clamp_id);
1835
1836 task_rq_unlock(rq, p, &rf);
1837 }
1838
1839 #ifdef CONFIG_UCLAMP_TASK_GROUP
1840 static inline void
1841 uclamp_update_active_tasks(struct cgroup_subsys_state *css)
1842 {
1843 struct css_task_iter it;
1844 struct task_struct *p;
1845
1846 css_task_iter_start(css, 0, &it);
1847 while ((p = css_task_iter_next(&it)))
1848 uclamp_update_active(p);
1849 css_task_iter_end(&it);
1850 }
1851
1852 static void cpu_util_update_eff(struct cgroup_subsys_state *css);
1853 #endif
1854
1855 #ifdef CONFIG_SYSCTL
1856 #ifdef CONFIG_UCLAMP_TASK_GROUP
1857 static void uclamp_update_root_tg(void)
1858 {
1859 struct task_group *tg = &root_task_group;
1860
1861 uclamp_se_set(&tg->uclamp_req[UCLAMP_MIN],
1862 sysctl_sched_uclamp_util_min, false);
1863 uclamp_se_set(&tg->uclamp_req[UCLAMP_MAX],
1864 sysctl_sched_uclamp_util_max, false);
1865
1866 guard(rcu)();
1867 cpu_util_update_eff(&root_task_group.css);
1868 }
1869 #else
1870 static void uclamp_update_root_tg(void) { }
1871 #endif
1872
1873 static void uclamp_sync_util_min_rt_default(void)
1874 {
1875 struct task_struct *g, *p;
1876
1877 /*
1878 * copy_process() sysctl_uclamp
1879 * uclamp_min_rt = X;
1880 * write_lock(&tasklist_lock) read_lock(&tasklist_lock)
1881 * // link thread smp_mb__after_spinlock()
1882 * write_unlock(&tasklist_lock) read_unlock(&tasklist_lock);
1883 * sched_post_fork() for_each_process_thread()
1884 * __uclamp_sync_rt() __uclamp_sync_rt()
1885 *
1886 * Ensures that either sched_post_fork() will observe the new
1887 * uclamp_min_rt or for_each_process_thread() will observe the new
1888 * task.
1889 */
1890 read_lock(&tasklist_lock);
1891 smp_mb__after_spinlock();
1892 read_unlock(&tasklist_lock);
1893
1894 guard(rcu)();
1895 for_each_process_thread(g, p)
1896 uclamp_update_util_min_rt_default(p);
1897 }
1898
1899 static int sysctl_sched_uclamp_handler(const struct ctl_table *table, int write,
1900 void *buffer, size_t *lenp, loff_t *ppos)
1901 {
1902 bool update_root_tg = false;
1903 int old_min, old_max, old_min_rt;
1904 int result;
1905
1906 guard(mutex)(&uclamp_mutex);
1907
1908 old_min = sysctl_sched_uclamp_util_min;
1909 old_max = sysctl_sched_uclamp_util_max;
1910 old_min_rt = sysctl_sched_uclamp_util_min_rt_default;
1911
1912 result = proc_dointvec(table, write, buffer, lenp, ppos);
1913 if (result)
1914 goto undo;
1915 if (!write)
1916 return 0;
1917
1918 if (sysctl_sched_uclamp_util_min > sysctl_sched_uclamp_util_max ||
1919 sysctl_sched_uclamp_util_max > SCHED_CAPACITY_SCALE ||
1920 sysctl_sched_uclamp_util_min_rt_default > SCHED_CAPACITY_SCALE) {
1921
1922 result = -EINVAL;
1923 goto undo;
1924 }
1925
1926 if (old_min != sysctl_sched_uclamp_util_min) {
1927 uclamp_se_set(&uclamp_default[UCLAMP_MIN],
1928 sysctl_sched_uclamp_util_min, false);
1929 update_root_tg = true;
1930 }
1931 if (old_max != sysctl_sched_uclamp_util_max) {
1932 uclamp_se_set(&uclamp_default[UCLAMP_MAX],
1933 sysctl_sched_uclamp_util_max, false);
1934 update_root_tg = true;
1935 }
1936
1937 if (update_root_tg) {
1938 static_branch_enable(&sched_uclamp_used);
1939 uclamp_update_root_tg();
1940 }
1941
1942 if (old_min_rt != sysctl_sched_uclamp_util_min_rt_default) {
1943 static_branch_enable(&sched_uclamp_used);
1944 uclamp_sync_util_min_rt_default();
1945 }
1946
1947 /*
1948 * We update all RUNNABLE tasks only when task groups are in use.
1949 * Otherwise, keep it simple and do just a lazy update at each next
1950 * task enqueue time.
1951 */
1952 return 0;
1953
1954 undo:
1955 sysctl_sched_uclamp_util_min = old_min;
1956 sysctl_sched_uclamp_util_max = old_max;
1957 sysctl_sched_uclamp_util_min_rt_default = old_min_rt;
1958 return result;
1959 }
1960 #endif
1961
1962 static void uclamp_fork(struct task_struct *p)
1963 {
1964 enum uclamp_id clamp_id;
1965
1966 /*
1967 * We don't need to hold task_rq_lock() when updating p->uclamp_* here
1968 * as the task is still at its early fork stages.
1969 */
1970 for_each_clamp_id(clamp_id)
1971 p->uclamp[clamp_id].active = false;
1972
1973 if (likely(!p->sched_reset_on_fork))
1974 return;
1975
1976 for_each_clamp_id(clamp_id) {
1977 uclamp_se_set(&p->uclamp_req[clamp_id],
1978 uclamp_none(clamp_id), false);
1979 }
1980 }
1981
1982 static void uclamp_post_fork(struct task_struct *p)
1983 {
1984 uclamp_update_util_min_rt_default(p);
1985 }
1986
1987 static void __init init_uclamp_rq(struct rq *rq)
1988 {
1989 enum uclamp_id clamp_id;
1990 struct uclamp_rq *uc_rq = rq->uclamp;
1991
1992 for_each_clamp_id(clamp_id) {
1993 uc_rq[clamp_id] = (struct uclamp_rq) {
1994 .value = uclamp_none(clamp_id)
1995 };
1996 }
1997
1998 rq->uclamp_flags = UCLAMP_FLAG_IDLE;
1999 }
2000
2001 static void __init init_uclamp(void)
2002 {
2003 struct uclamp_se uc_max = {};
2004 enum uclamp_id clamp_id;
2005 int cpu;
2006
2007 for_each_possible_cpu(cpu)
2008 init_uclamp_rq(cpu_rq(cpu));
2009
2010 for_each_clamp_id(clamp_id) {
2011 uclamp_se_set(&init_task.uclamp_req[clamp_id],
2012 uclamp_none(clamp_id), false);
2013 }
2014
2015 /* System defaults allow max clamp values for both indexes */
2016 uclamp_se_set(&uc_max, uclamp_none(UCLAMP_MAX), false);
2017 for_each_clamp_id(clamp_id) {
2018 uclamp_default[clamp_id] = uc_max;
2019 #ifdef CONFIG_UCLAMP_TASK_GROUP
2020 root_task_group.uclamp_req[clamp_id] = uc_max;
2021 root_task_group.uclamp[clamp_id] = uc_max;
2022 #endif
2023 }
2024 }
2025
2026 #else /* !CONFIG_UCLAMP_TASK */
2027 static inline void uclamp_rq_inc(struct rq *rq, struct task_struct *p) { }
2028 static inline void uclamp_rq_dec(struct rq *rq, struct task_struct *p) { }
2029 static inline void uclamp_fork(struct task_struct *p) { }
2030 static inline void uclamp_post_fork(struct task_struct *p) { }
2031 static inline void init_uclamp(void) { }
2032 #endif /* CONFIG_UCLAMP_TASK */
2033
2034 bool sched_task_on_rq(struct task_struct *p)
2035 {
2036 return task_on_rq_queued(p);
2037 }
2038
2039 unsigned long get_wchan(struct task_struct *p)
2040 {
2041 unsigned long ip = 0;
2042 unsigned int state;
2043
2044 if (!p || p == current)
2045 return 0;
2046
2047 /* Only get wchan if task is blocked and we can keep it that way. */
2048 raw_spin_lock_irq(&p->pi_lock);
2049 state = READ_ONCE(p->__state);
2050 smp_rmb(); /* see try_to_wake_up() */
2051 if (state != TASK_RUNNING && state != TASK_WAKING && !p->on_rq)
2052 ip = __get_wchan(p);
2053 raw_spin_unlock_irq(&p->pi_lock);
2054
2055 return ip;
2056 }
2057
2058 void enqueue_task(struct rq *rq, struct task_struct *p, int flags)
2059 {
2060 if (!(flags & ENQUEUE_NOCLOCK))
2061 update_rq_clock(rq);
2062
2063 p->sched_class->enqueue_task(rq, p, flags);
2064 /*
2065 * Must be after ->enqueue_task() because ENQUEUE_DELAYED can clear
2066 * ->sched_delayed.
2067 */
2068 uclamp_rq_inc(rq, p);
2069
2070 psi_enqueue(p, flags);
2071
2072 if (!(flags & ENQUEUE_RESTORE))
2073 sched_info_enqueue(rq, p);
2074
2075 if (sched_core_enabled(rq))
2076 sched_core_enqueue(rq, p);
2077 }
2078
2079 /*
2080 * Must only return false when DEQUEUE_SLEEP.
2081 */
2082 inline bool dequeue_task(struct rq *rq, struct task_struct *p, int flags)
2083 {
2084 if (sched_core_enabled(rq))
2085 sched_core_dequeue(rq, p, flags);
2086
2087 if (!(flags & DEQUEUE_NOCLOCK))
2088 update_rq_clock(rq);
2089
2090 if (!(flags & DEQUEUE_SAVE))
2091 sched_info_dequeue(rq, p);
2092
2093 psi_dequeue(p, flags);
2094
2095 /*
2096 * Must be before ->dequeue_task() because ->dequeue_task() can 'fail'
2097 * and mark the task ->sched_delayed.
2098 */
2099 uclamp_rq_dec(rq, p);
2100 return p->sched_class->dequeue_task(rq, p, flags);
2101 }
2102
2103 void activate_task(struct rq *rq, struct task_struct *p, int flags)
2104 {
2105 if (task_on_rq_migrating(p))
2106 flags |= ENQUEUE_MIGRATED;
2107 if (flags & ENQUEUE_MIGRATED)
2108 sched_mm_cid_migrate_to(rq, p);
2109
2110 enqueue_task(rq, p, flags);
2111
2112 WRITE_ONCE(p->on_rq, TASK_ON_RQ_QUEUED);
2113 ASSERT_EXCLUSIVE_WRITER(p->on_rq);
2114 }
2115
2116 void deactivate_task(struct rq *rq, struct task_struct *p, int flags)
2117 {
2118 SCHED_WARN_ON(flags & DEQUEUE_SLEEP);
2119
2120 WRITE_ONCE(p->on_rq, TASK_ON_RQ_MIGRATING);
2121 ASSERT_EXCLUSIVE_WRITER(p->on_rq);
2122
2123 /*
2124 * Code explicitly relies on TASK_ON_RQ_MIGRATING begin set *before*
2125 * dequeue_task() and cleared *after* enqueue_task().
2126 */
2127
2128 dequeue_task(rq, p, flags);
2129 }
2130
2131 static void block_task(struct rq *rq, struct task_struct *p, int flags)
2132 {
2133 if (dequeue_task(rq, p, DEQUEUE_SLEEP | flags))
2134 __block_task(rq, p);
2135 }
2136
2137 /**
2138 * task_curr - is this task currently executing on a CPU?
2139 * @p: the task in question.
2140 *
2141 * Return: 1 if the task is currently executing. 0 otherwise.
2142 */
2143 inline int task_curr(const struct task_struct *p)
2144 {
2145 return cpu_curr(task_cpu(p)) == p;
2146 }
2147
2148 /*
2149 * ->switching_to() is called with the pi_lock and rq_lock held and must not
2150 * mess with locking.
2151 */
2152 void check_class_changing(struct rq *rq, struct task_struct *p,
2153 const struct sched_class *prev_class)
2154 {
2155 if (prev_class != p->sched_class && p->sched_class->switching_to)
2156 p->sched_class->switching_to(rq, p);
2157 }
2158
2159 /*
2160 * switched_from, switched_to and prio_changed must _NOT_ drop rq->lock,
2161 * use the balance_callback list if you want balancing.
2162 *
2163 * this means any call to check_class_changed() must be followed by a call to
2164 * balance_callback().
2165 */
2166 void check_class_changed(struct rq *rq, struct task_struct *p,
2167 const struct sched_class *prev_class,
2168 int oldprio)
2169 {
2170 if (prev_class != p->sched_class) {
2171 if (prev_class->switched_from)
2172 prev_class->switched_from(rq, p);
2173
2174 p->sched_class->switched_to(rq, p);
2175 } else if (oldprio != p->prio || dl_task(p))
2176 p->sched_class->prio_changed(rq, p, oldprio);
2177 }
2178
2179 void wakeup_preempt(struct rq *rq, struct task_struct *p, int flags)
2180 {
2181 struct task_struct *donor = rq->donor;
2182
2183 if (p->sched_class == donor->sched_class)
2184 donor->sched_class->wakeup_preempt(rq, p, flags);
2185 else if (sched_class_above(p->sched_class, donor->sched_class))
2186 resched_curr(rq);
2187
2188 /*
2189 * A queue event has occurred, and we're going to schedule. In
2190 * this case, we can save a useless back to back clock update.
2191 */
2192 if (task_on_rq_queued(donor) && test_tsk_need_resched(rq->curr))
2193 rq_clock_skip_update(rq);
2194 }
2195
2196 static __always_inline
2197 int __task_state_match(struct task_struct *p, unsigned int state)
2198 {
2199 if (READ_ONCE(p->__state) & state)
2200 return 1;
2201
2202 if (READ_ONCE(p->saved_state) & state)
2203 return -1;
2204
2205 return 0;
2206 }
2207
2208 static __always_inline
2209 int task_state_match(struct task_struct *p, unsigned int state)
2210 {
2211 /*
2212 * Serialize against current_save_and_set_rtlock_wait_state(),
2213 * current_restore_rtlock_saved_state(), and __refrigerator().
2214 */
2215 guard(raw_spinlock_irq)(&p->pi_lock);
2216 return __task_state_match(p, state);
2217 }
2218
2219 /*
2220 * wait_task_inactive - wait for a thread to unschedule.
2221 *
2222 * Wait for the thread to block in any of the states set in @match_state.
2223 * If it changes, i.e. @p might have woken up, then return zero. When we
2224 * succeed in waiting for @p to be off its CPU, we return a positive number
2225 * (its total switch count). If a second call a short while later returns the
2226 * same number, the caller can be sure that @p has remained unscheduled the
2227 * whole time.
2228 *
2229 * The caller must ensure that the task *will* unschedule sometime soon,
2230 * else this function might spin for a *long* time. This function can't
2231 * be called with interrupts off, or it may introduce deadlock with
2232 * smp_call_function() if an IPI is sent by the same process we are
2233 * waiting to become inactive.
2234 */
2235 unsigned long wait_task_inactive(struct task_struct *p, unsigned int match_state)
2236 {
2237 int running, queued, match;
2238 struct rq_flags rf;
2239 unsigned long ncsw;
2240 struct rq *rq;
2241
2242 for (;;) {
2243 /*
2244 * We do the initial early heuristics without holding
2245 * any task-queue locks at all. We'll only try to get
2246 * the runqueue lock when things look like they will
2247 * work out!
2248 */
2249 rq = task_rq(p);
2250
2251 /*
2252 * If the task is actively running on another CPU
2253 * still, just relax and busy-wait without holding
2254 * any locks.
2255 *
2256 * NOTE! Since we don't hold any locks, it's not
2257 * even sure that "rq" stays as the right runqueue!
2258 * But we don't care, since "task_on_cpu()" will
2259 * return false if the runqueue has changed and p
2260 * is actually now running somewhere else!
2261 */
2262 while (task_on_cpu(rq, p)) {
2263 if (!task_state_match(p, match_state))
2264 return 0;
2265 cpu_relax();
2266 }
2267
2268 /*
2269 * Ok, time to look more closely! We need the rq
2270 * lock now, to be *sure*. If we're wrong, we'll
2271 * just go back and repeat.
2272 */
2273 rq = task_rq_lock(p, &rf);
2274 trace_sched_wait_task(p);
2275 running = task_on_cpu(rq, p);
2276 queued = task_on_rq_queued(p);
2277 ncsw = 0;
2278 if ((match = __task_state_match(p, match_state))) {
2279 /*
2280 * When matching on p->saved_state, consider this task
2281 * still queued so it will wait.
2282 */
2283 if (match < 0)
2284 queued = 1;
2285 ncsw = p->nvcsw | LONG_MIN; /* sets MSB */
2286 }
2287 task_rq_unlock(rq, p, &rf);
2288
2289 /*
2290 * If it changed from the expected state, bail out now.
2291 */
2292 if (unlikely(!ncsw))
2293 break;
2294
2295 /*
2296 * Was it really running after all now that we
2297 * checked with the proper locks actually held?
2298 *
2299 * Oops. Go back and try again..
2300 */
2301 if (unlikely(running)) {
2302 cpu_relax();
2303 continue;
2304 }
2305
2306 /*
2307 * It's not enough that it's not actively running,
2308 * it must be off the runqueue _entirely_, and not
2309 * preempted!
2310 *
2311 * So if it was still runnable (but just not actively
2312 * running right now), it's preempted, and we should
2313 * yield - it could be a while.
2314 */
2315 if (unlikely(queued)) {
2316 ktime_t to = NSEC_PER_SEC / HZ;
2317
2318 set_current_state(TASK_UNINTERRUPTIBLE);
2319 schedule_hrtimeout(&to, HRTIMER_MODE_REL_HARD);
2320 continue;
2321 }
2322
2323 /*
2324 * Ahh, all good. It wasn't running, and it wasn't
2325 * runnable, which means that it will never become
2326 * running in the future either. We're all done!
2327 */
2328 break;
2329 }
2330
2331 return ncsw;
2332 }
2333
2334 #ifdef CONFIG_SMP
2335
2336 static void
2337 __do_set_cpus_allowed(struct task_struct *p, struct affinity_context *ctx);
2338
2339 static void migrate_disable_switch(struct rq *rq, struct task_struct *p)
2340 {
2341 struct affinity_context ac = {
2342 .new_mask = cpumask_of(rq->cpu),
2343 .flags = SCA_MIGRATE_DISABLE,
2344 };
2345
2346 if (likely(!p->migration_disabled))
2347 return;
2348
2349 if (p->cpus_ptr != &p->cpus_mask)
2350 return;
2351
2352 /*
2353 * Violates locking rules! See comment in __do_set_cpus_allowed().
2354 */
2355 __do_set_cpus_allowed(p, &ac);
2356 }
2357
2358 void migrate_disable(void)
2359 {
2360 struct task_struct *p = current;
2361
2362 if (p->migration_disabled) {
2363 #ifdef CONFIG_DEBUG_PREEMPT
2364 /*
2365 *Warn about overflow half-way through the range.
2366 */
2367 WARN_ON_ONCE((s16)p->migration_disabled < 0);
2368 #endif
2369 p->migration_disabled++;
2370 return;
2371 }
2372
2373 guard(preempt)();
2374 this_rq()->nr_pinned++;
2375 p->migration_disabled = 1;
2376 }
2377 EXPORT_SYMBOL_GPL(migrate_disable);
2378
2379 void migrate_enable(void)
2380 {
2381 struct task_struct *p = current;
2382 struct affinity_context ac = {
2383 .new_mask = &p->cpus_mask,
2384 .flags = SCA_MIGRATE_ENABLE,
2385 };
2386
2387 #ifdef CONFIG_DEBUG_PREEMPT
2388 /*
2389 * Check both overflow from migrate_disable() and superfluous
2390 * migrate_enable().
2391 */
2392 if (WARN_ON_ONCE((s16)p->migration_disabled <= 0))
2393 return;
2394 #endif
2395
2396 if (p->migration_disabled > 1) {
2397 p->migration_disabled--;
2398 return;
2399 }
2400
2401 /*
2402 * Ensure stop_task runs either before or after this, and that
2403 * __set_cpus_allowed_ptr(SCA_MIGRATE_ENABLE) doesn't schedule().
2404 */
2405 guard(preempt)();
2406 if (p->cpus_ptr != &p->cpus_mask)
2407 __set_cpus_allowed_ptr(p, &ac);
2408 /*
2409 * Mustn't clear migration_disabled() until cpus_ptr points back at the
2410 * regular cpus_mask, otherwise things that race (eg.
2411 * select_fallback_rq) get confused.
2412 */
2413 barrier();
2414 p->migration_disabled = 0;
2415 this_rq()->nr_pinned--;
2416 }
2417 EXPORT_SYMBOL_GPL(migrate_enable);
2418
2419 static inline bool rq_has_pinned_tasks(struct rq *rq)
2420 {
2421 return rq->nr_pinned;
2422 }
2423
2424 /*
2425 * Per-CPU kthreads are allowed to run on !active && online CPUs, see
2426 * __set_cpus_allowed_ptr() and select_fallback_rq().
2427 */
2428 static inline bool is_cpu_allowed(struct task_struct *p, int cpu)
2429 {
2430 /* When not in the task's cpumask, no point in looking further. */
2431 if (!task_allowed_on_cpu(p, cpu))
2432 return false;
2433
2434 /* migrate_disabled() must be allowed to finish. */
2435 if (is_migration_disabled(p))
2436 return cpu_online(cpu);
2437
2438 /* Non kernel threads are not allowed during either online or offline. */
2439 if (!(p->flags & PF_KTHREAD))
2440 return cpu_active(cpu);
2441
2442 /* KTHREAD_IS_PER_CPU is always allowed. */
2443 if (kthread_is_per_cpu(p))
2444 return cpu_online(cpu);
2445
2446 /* Regular kernel threads don't get to stay during offline. */
2447 if (cpu_dying(cpu))
2448 return false;
2449
2450 /* But are allowed during online. */
2451 return cpu_online(cpu);
2452 }
2453
2454 /*
2455 * This is how migration works:
2456 *
2457 * 1) we invoke migration_cpu_stop() on the target CPU using
2458 * stop_one_cpu().
2459 * 2) stopper starts to run (implicitly forcing the migrated thread
2460 * off the CPU)
2461 * 3) it checks whether the migrated task is still in the wrong runqueue.
2462 * 4) if it's in the wrong runqueue then the migration thread removes
2463 * it and puts it into the right queue.
2464 * 5) stopper completes and stop_one_cpu() returns and the migration
2465 * is done.
2466 */
2467
2468 /*
2469 * move_queued_task - move a queued task to new rq.
2470 *
2471 * Returns (locked) new rq. Old rq's lock is released.
2472 */
2473 static struct rq *move_queued_task(struct rq *rq, struct rq_flags *rf,
2474 struct task_struct *p, int new_cpu)
2475 {
2476 lockdep_assert_rq_held(rq);
2477
2478 deactivate_task(rq, p, DEQUEUE_NOCLOCK);
2479 set_task_cpu(p, new_cpu);
2480 rq_unlock(rq, rf);
2481
2482 rq = cpu_rq(new_cpu);
2483
2484 rq_lock(rq, rf);
2485 WARN_ON_ONCE(task_cpu(p) != new_cpu);
2486 activate_task(rq, p, 0);
2487 wakeup_preempt(rq, p, 0);
2488
2489 return rq;
2490 }
2491
2492 struct migration_arg {
2493 struct task_struct *task;
2494 int dest_cpu;
2495 struct set_affinity_pending *pending;
2496 };
2497
2498 /*
2499 * @refs: number of wait_for_completion()
2500 * @stop_pending: is @stop_work in use
2501 */
2502 struct set_affinity_pending {
2503 refcount_t refs;
2504 unsigned int stop_pending;
2505 struct completion done;
2506 struct cpu_stop_work stop_work;
2507 struct migration_arg arg;
2508 };
2509
2510 /*
2511 * Move (not current) task off this CPU, onto the destination CPU. We're doing
2512 * this because either it can't run here any more (set_cpus_allowed()
2513 * away from this CPU, or CPU going down), or because we're
2514 * attempting to rebalance this task on exec (sched_exec).
2515 *
2516 * So we race with normal scheduler movements, but that's OK, as long
2517 * as the task is no longer on this CPU.
2518 */
2519 static struct rq *__migrate_task(struct rq *rq, struct rq_flags *rf,
2520 struct task_struct *p, int dest_cpu)
2521 {
2522 /* Affinity changed (again). */
2523 if (!is_cpu_allowed(p, dest_cpu))
2524 return rq;
2525
2526 rq = move_queued_task(rq, rf, p, dest_cpu);
2527
2528 return rq;
2529 }
2530
2531 /*
2532 * migration_cpu_stop - this will be executed by a high-prio stopper thread
2533 * and performs thread migration by bumping thread off CPU then
2534 * 'pushing' onto another runqueue.
2535 */
2536 static int migration_cpu_stop(void *data)
2537 {
2538 struct migration_arg *arg = data;
2539 struct set_affinity_pending *pending = arg->pending;
2540 struct task_struct *p = arg->task;
2541 struct rq *rq = this_rq();
2542 bool complete = false;
2543 struct rq_flags rf;
2544
2545 /*
2546 * The original target CPU might have gone down and we might
2547 * be on another CPU but it doesn't matter.
2548 */
2549 local_irq_save(rf.flags);
2550 /*
2551 * We need to explicitly wake pending tasks before running
2552 * __migrate_task() such that we will not miss enforcing cpus_ptr
2553 * during wakeups, see set_cpus_allowed_ptr()'s TASK_WAKING test.
2554 */
2555 flush_smp_call_function_queue();
2556
2557 raw_spin_lock(&p->pi_lock);
2558 rq_lock(rq, &rf);
2559
2560 /*
2561 * If we were passed a pending, then ->stop_pending was set, thus
2562 * p->migration_pending must have remained stable.
2563 */
2564 WARN_ON_ONCE(pending && pending != p->migration_pending);
2565
2566 /*
2567 * If task_rq(p) != rq, it cannot be migrated here, because we're
2568 * holding rq->lock, if p->on_rq == 0 it cannot get enqueued because
2569 * we're holding p->pi_lock.
2570 */
2571 if (task_rq(p) == rq) {
2572 if (is_migration_disabled(p))
2573 goto out;
2574
2575 if (pending) {
2576 p->migration_pending = NULL;
2577 complete = true;
2578
2579 if (cpumask_test_cpu(task_cpu(p), &p->cpus_mask))
2580 goto out;
2581 }
2582
2583 if (task_on_rq_queued(p)) {
2584 update_rq_clock(rq);
2585 rq = __migrate_task(rq, &rf, p, arg->dest_cpu);
2586 } else {
2587 p->wake_cpu = arg->dest_cpu;
2588 }
2589
2590 /*
2591 * XXX __migrate_task() can fail, at which point we might end
2592 * up running on a dodgy CPU, AFAICT this can only happen
2593 * during CPU hotplug, at which point we'll get pushed out
2594 * anyway, so it's probably not a big deal.
2595 */
2596
2597 } else if (pending) {
2598 /*
2599 * This happens when we get migrated between migrate_enable()'s
2600 * preempt_enable() and scheduling the stopper task. At that
2601 * point we're a regular task again and not current anymore.
2602 *
2603 * A !PREEMPT kernel has a giant hole here, which makes it far
2604 * more likely.
2605 */
2606
2607 /*
2608 * The task moved before the stopper got to run. We're holding
2609 * ->pi_lock, so the allowed mask is stable - if it got
2610 * somewhere allowed, we're done.
2611 */
2612 if (cpumask_test_cpu(task_cpu(p), p->cpus_ptr)) {
2613 p->migration_pending = NULL;
2614 complete = true;
2615 goto out;
2616 }
2617
2618 /*
2619 * When migrate_enable() hits a rq mis-match we can't reliably
2620 * determine is_migration_disabled() and so have to chase after
2621 * it.
2622 */
2623 WARN_ON_ONCE(!pending->stop_pending);
2624 preempt_disable();
2625 task_rq_unlock(rq, p, &rf);
2626 stop_one_cpu_nowait(task_cpu(p), migration_cpu_stop,
2627 &pending->arg, &pending->stop_work);
2628 preempt_enable();
2629 return 0;
2630 }
2631 out:
2632 if (pending)
2633 pending->stop_pending = false;
2634 task_rq_unlock(rq, p, &rf);
2635
2636 if (complete)
2637 complete_all(&pending->done);
2638
2639 return 0;
2640 }
2641
2642 int push_cpu_stop(void *arg)
2643 {
2644 struct rq *lowest_rq = NULL, *rq = this_rq();
2645 struct task_struct *p = arg;
2646
2647 raw_spin_lock_irq(&p->pi_lock);
2648 raw_spin_rq_lock(rq);
2649
2650 if (task_rq(p) != rq)
2651 goto out_unlock;
2652
2653 if (is_migration_disabled(p)) {
2654 p->migration_flags |= MDF_PUSH;
2655 goto out_unlock;
2656 }
2657
2658 p->migration_flags &= ~MDF_PUSH;
2659
2660 if (p->sched_class->find_lock_rq)
2661 lowest_rq = p->sched_class->find_lock_rq(p, rq);
2662
2663 if (!lowest_rq)
2664 goto out_unlock;
2665
2666 // XXX validate p is still the highest prio task
2667 if (task_rq(p) == rq) {
2668 move_queued_task_locked(rq, lowest_rq, p);
2669 resched_curr(lowest_rq);
2670 }
2671
2672 double_unlock_balance(rq, lowest_rq);
2673
2674 out_unlock:
2675 rq->push_busy = false;
2676 raw_spin_rq_unlock(rq);
2677 raw_spin_unlock_irq(&p->pi_lock);
2678
2679 put_task_struct(p);
2680 return 0;
2681 }
2682
2683 /*
2684 * sched_class::set_cpus_allowed must do the below, but is not required to
2685 * actually call this function.
2686 */
2687 void set_cpus_allowed_common(struct task_struct *p, struct affinity_context *ctx)
2688 {
2689 if (ctx->flags & (SCA_MIGRATE_ENABLE | SCA_MIGRATE_DISABLE)) {
2690 p->cpus_ptr = ctx->new_mask;
2691 return;
2692 }
2693
2694 cpumask_copy(&p->cpus_mask, ctx->new_mask);
2695 p->nr_cpus_allowed = cpumask_weight(ctx->new_mask);
2696
2697 /*
2698 * Swap in a new user_cpus_ptr if SCA_USER flag set
2699 */
2700 if (ctx->flags & SCA_USER)
2701 swap(p->user_cpus_ptr, ctx->user_mask);
2702 }
2703
2704 static void
2705 __do_set_cpus_allowed(struct task_struct *p, struct affinity_context *ctx)
2706 {
2707 struct rq *rq = task_rq(p);
2708 bool queued, running;
2709
2710 /*
2711 * This here violates the locking rules for affinity, since we're only
2712 * supposed to change these variables while holding both rq->lock and
2713 * p->pi_lock.
2714 *
2715 * HOWEVER, it magically works, because ttwu() is the only code that
2716 * accesses these variables under p->pi_lock and only does so after
2717 * smp_cond_load_acquire(&p->on_cpu, !VAL), and we're in __schedule()
2718 * before finish_task().
2719 *
2720 * XXX do further audits, this smells like something putrid.
2721 */
2722 if (ctx->flags & SCA_MIGRATE_DISABLE)
2723 SCHED_WARN_ON(!p->on_cpu);
2724 else
2725 lockdep_assert_held(&p->pi_lock);
2726
2727 queued = task_on_rq_queued(p);
2728 running = task_current_donor(rq, p);
2729
2730 if (queued) {
2731 /*
2732 * Because __kthread_bind() calls this on blocked tasks without
2733 * holding rq->lock.
2734 */
2735 lockdep_assert_rq_held(rq);
2736 dequeue_task(rq, p, DEQUEUE_SAVE | DEQUEUE_NOCLOCK);
2737 }
2738 if (running)
2739 put_prev_task(rq, p);
2740
2741 p->sched_class->set_cpus_allowed(p, ctx);
2742 mm_set_cpus_allowed(p->mm, ctx->new_mask);
2743
2744 if (queued)
2745 enqueue_task(rq, p, ENQUEUE_RESTORE | ENQUEUE_NOCLOCK);
2746 if (running)
2747 set_next_task(rq, p);
2748 }
2749
2750 /*
2751 * Used for kthread_bind() and select_fallback_rq(), in both cases the user
2752 * affinity (if any) should be destroyed too.
2753 */
2754 void do_set_cpus_allowed(struct task_struct *p, const struct cpumask *new_mask)
2755 {
2756 struct affinity_context ac = {
2757 .new_mask = new_mask,
2758 .user_mask = NULL,
2759 .flags = SCA_USER, /* clear the user requested mask */
2760 };
2761 union cpumask_rcuhead {
2762 cpumask_t cpumask;
2763 struct rcu_head rcu;
2764 };
2765
2766 __do_set_cpus_allowed(p, &ac);
2767
2768 /*
2769 * Because this is called with p->pi_lock held, it is not possible
2770 * to use kfree() here (when PREEMPT_RT=y), therefore punt to using
2771 * kfree_rcu().
2772 */
2773 kfree_rcu((union cpumask_rcuhead *)ac.user_mask, rcu);
2774 }
2775
2776 int dup_user_cpus_ptr(struct task_struct *dst, struct task_struct *src,
2777 int node)
2778 {
2779 cpumask_t *user_mask;
2780 unsigned long flags;
2781
2782 /*
2783 * Always clear dst->user_cpus_ptr first as their user_cpus_ptr's
2784 * may differ by now due to racing.
2785 */
2786 dst->user_cpus_ptr = NULL;
2787
2788 /*
2789 * This check is racy and losing the race is a valid situation.
2790 * It is not worth the extra overhead of taking the pi_lock on
2791 * every fork/clone.
2792 */
2793 if (data_race(!src->user_cpus_ptr))
2794 return 0;
2795
2796 user_mask = alloc_user_cpus_ptr(node);
2797 if (!user_mask)
2798 return -ENOMEM;
2799
2800 /*
2801 * Use pi_lock to protect content of user_cpus_ptr
2802 *
2803 * Though unlikely, user_cpus_ptr can be reset to NULL by a concurrent
2804 * do_set_cpus_allowed().
2805 */
2806 raw_spin_lock_irqsave(&src->pi_lock, flags);
2807 if (src->user_cpus_ptr) {
2808 swap(dst->user_cpus_ptr, user_mask);
2809 cpumask_copy(dst->user_cpus_ptr, src->user_cpus_ptr);
2810 }
2811 raw_spin_unlock_irqrestore(&src->pi_lock, flags);
2812
2813 if (unlikely(user_mask))
2814 kfree(user_mask);
2815
2816 return 0;
2817 }
2818
2819 static inline struct cpumask *clear_user_cpus_ptr(struct task_struct *p)
2820 {
2821 struct cpumask *user_mask = NULL;
2822
2823 swap(p->user_cpus_ptr, user_mask);
2824
2825 return user_mask;
2826 }
2827
2828 void release_user_cpus_ptr(struct task_struct *p)
2829 {
2830 kfree(clear_user_cpus_ptr(p));
2831 }
2832
2833 /*
2834 * This function is wildly self concurrent; here be dragons.
2835 *
2836 *
2837 * When given a valid mask, __set_cpus_allowed_ptr() must block until the
2838 * designated task is enqueued on an allowed CPU. If that task is currently
2839 * running, we have to kick it out using the CPU stopper.
2840 *
2841 * Migrate-Disable comes along and tramples all over our nice sandcastle.
2842 * Consider:
2843 *
2844 * Initial conditions: P0->cpus_mask = [0, 1]
2845 *
2846 * P0@CPU0 P1
2847 *
2848 * migrate_disable();
2849 * <preempted>
2850 * set_cpus_allowed_ptr(P0, [1]);
2851 *
2852 * P1 *cannot* return from this set_cpus_allowed_ptr() call until P0 executes
2853 * its outermost migrate_enable() (i.e. it exits its Migrate-Disable region).
2854 * This means we need the following scheme:
2855 *
2856 * P0@CPU0 P1
2857 *
2858 * migrate_disable();
2859 * <preempted>
2860 * set_cpus_allowed_ptr(P0, [1]);
2861 * <blocks>
2862 * <resumes>
2863 * migrate_enable();
2864 * __set_cpus_allowed_ptr();
2865 * <wakes local stopper>
2866 * `--> <woken on migration completion>
2867 *
2868 * Now the fun stuff: there may be several P1-like tasks, i.e. multiple
2869 * concurrent set_cpus_allowed_ptr(P0, [*]) calls. CPU affinity changes of any
2870 * task p are serialized by p->pi_lock, which we can leverage: the one that
2871 * should come into effect at the end of the Migrate-Disable region is the last
2872 * one. This means we only need to track a single cpumask (i.e. p->cpus_mask),
2873 * but we still need to properly signal those waiting tasks at the appropriate
2874 * moment.
2875 *
2876 * This is implemented using struct set_affinity_pending. The first
2877 * __set_cpus_allowed_ptr() caller within a given Migrate-Disable region will
2878 * setup an instance of that struct and install it on the targeted task_struct.
2879 * Any and all further callers will reuse that instance. Those then wait for
2880 * a completion signaled at the tail of the CPU stopper callback (1), triggered
2881 * on the end of the Migrate-Disable region (i.e. outermost migrate_enable()).
2882 *
2883 *
2884 * (1) In the cases covered above. There is one more where the completion is
2885 * signaled within affine_move_task() itself: when a subsequent affinity request
2886 * occurs after the stopper bailed out due to the targeted task still being
2887 * Migrate-Disable. Consider:
2888 *
2889 * Initial conditions: P0->cpus_mask = [0, 1]
2890 *
2891 * CPU0 P1 P2
2892 * <P0>
2893 * migrate_disable();
2894 * <preempted>
2895 * set_cpus_allowed_ptr(P0, [1]);
2896 * <blocks>
2897 * <migration/0>
2898 * migration_cpu_stop()
2899 * is_migration_disabled()
2900 * <bails>
2901 * set_cpus_allowed_ptr(P0, [0, 1]);
2902 * <signal completion>
2903 * <awakes>
2904 *
2905 * Note that the above is safe vs a concurrent migrate_enable(), as any
2906 * pending affinity completion is preceded by an uninstallation of
2907 * p->migration_pending done with p->pi_lock held.
2908 */
2909 static int affine_move_task(struct rq *rq, struct task_struct *p, struct rq_flags *rf,
2910 int dest_cpu, unsigned int flags)
2911 __releases(rq->lock)
2912 __releases(p->pi_lock)
2913 {
2914 struct set_affinity_pending my_pending = { }, *pending = NULL;
2915 bool stop_pending, complete = false;
2916
2917 /* Can the task run on the task's current CPU? If so, we're done */
2918 if (cpumask_test_cpu(task_cpu(p), &p->cpus_mask)) {
2919 struct task_struct *push_task = NULL;
2920
2921 if ((flags & SCA_MIGRATE_ENABLE) &&
2922 (p->migration_flags & MDF_PUSH) && !rq->push_busy) {
2923 rq->push_busy = true;
2924 push_task = get_task_struct(p);
2925 }
2926
2927 /*
2928 * If there are pending waiters, but no pending stop_work,
2929 * then complete now.
2930 */
2931 pending = p->migration_pending;
2932 if (pending && !pending->stop_pending) {
2933 p->migration_pending = NULL;
2934 complete = true;
2935 }
2936
2937 preempt_disable();
2938 task_rq_unlock(rq, p, rf);
2939 if (push_task) {
2940 stop_one_cpu_nowait(rq->cpu, push_cpu_stop,
2941 p, &rq->push_work);
2942 }
2943 preempt_enable();
2944
2945 if (complete)
2946 complete_all(&pending->done);
2947
2948 return 0;
2949 }
2950
2951 if (!(flags & SCA_MIGRATE_ENABLE)) {
2952 /* serialized by p->pi_lock */
2953 if (!p->migration_pending) {
2954 /* Install the request */
2955 refcount_set(&my_pending.refs, 1);
2956 init_completion(&my_pending.done);
2957 my_pending.arg = (struct migration_arg) {
2958 .task = p,
2959 .dest_cpu = dest_cpu,
2960 .pending = &my_pending,
2961 };
2962
2963 p->migration_pending = &my_pending;
2964 } else {
2965 pending = p->migration_pending;
2966 refcount_inc(&pending->refs);
2967 /*
2968 * Affinity has changed, but we've already installed a
2969 * pending. migration_cpu_stop() *must* see this, else
2970 * we risk a completion of the pending despite having a
2971 * task on a disallowed CPU.
2972 *
2973 * Serialized by p->pi_lock, so this is safe.
2974 */
2975 pending->arg.dest_cpu = dest_cpu;
2976 }
2977 }
2978 pending = p->migration_pending;
2979 /*
2980 * - !MIGRATE_ENABLE:
2981 * we'll have installed a pending if there wasn't one already.
2982 *
2983 * - MIGRATE_ENABLE:
2984 * we're here because the current CPU isn't matching anymore,
2985 * the only way that can happen is because of a concurrent
2986 * set_cpus_allowed_ptr() call, which should then still be
2987 * pending completion.
2988 *
2989 * Either way, we really should have a @pending here.
2990 */
2991 if (WARN_ON_ONCE(!pending)) {
2992 task_rq_unlock(rq, p, rf);
2993 return -EINVAL;
2994 }
2995
2996 if (task_on_cpu(rq, p) || READ_ONCE(p->__state) == TASK_WAKING) {
2997 /*
2998 * MIGRATE_ENABLE gets here because 'p == current', but for
2999 * anything else we cannot do is_migration_disabled(), punt
3000 * and have the stopper function handle it all race-free.
3001 */
3002 stop_pending = pending->stop_pending;
3003 if (!stop_pending)
3004 pending->stop_pending = true;
3005
3006 if (flags & SCA_MIGRATE_ENABLE)
3007 p->migration_flags &= ~MDF_PUSH;
3008
3009 preempt_disable();
3010 task_rq_unlock(rq, p, rf);
3011 if (!stop_pending) {
3012 stop_one_cpu_nowait(cpu_of(rq), migration_cpu_stop,
3013 &pending->arg, &pending->stop_work);
3014 }
3015 preempt_enable();
3016
3017 if (flags & SCA_MIGRATE_ENABLE)
3018 return 0;
3019 } else {
3020
3021 if (!is_migration_disabled(p)) {
3022 if (task_on_rq_queued(p))
3023 rq = move_queued_task(rq, rf, p, dest_cpu);
3024
3025 if (!pending->stop_pending) {
3026 p->migration_pending = NULL;
3027 complete = true;
3028 }
3029 }
3030 task_rq_unlock(rq, p, rf);
3031
3032 if (complete)
3033 complete_all(&pending->done);
3034 }
3035
3036 wait_for_completion(&pending->done);
3037
3038 if (refcount_dec_and_test(&pending->refs))
3039 wake_up_var(&pending->refs); /* No UaF, just an address */
3040
3041 /*
3042 * Block the original owner of &pending until all subsequent callers
3043 * have seen the completion and decremented the refcount
3044 */
3045 wait_var_event(&my_pending.refs, !refcount_read(&my_pending.refs));
3046
3047 /* ARGH */
3048 WARN_ON_ONCE(my_pending.stop_pending);
3049
3050 return 0;
3051 }
3052
3053 /*
3054 * Called with both p->pi_lock and rq->lock held; drops both before returning.
3055 */
3056 static int __set_cpus_allowed_ptr_locked(struct task_struct *p,
3057 struct affinity_context *ctx,
3058 struct rq *rq,
3059 struct rq_flags *rf)
3060 __releases(rq->lock)
3061 __releases(p->pi_lock)
3062 {
3063 const struct cpumask *cpu_allowed_mask = task_cpu_possible_mask(p);
3064 const struct cpumask *cpu_valid_mask = cpu_active_mask;
3065 bool kthread = p->flags & PF_KTHREAD;
3066 unsigned int dest_cpu;
3067 int ret = 0;
3068
3069 update_rq_clock(rq);
3070
3071 if (kthread || is_migration_disabled(p)) {
3072 /*
3073 * Kernel threads are allowed on online && !active CPUs,
3074 * however, during cpu-hot-unplug, even these might get pushed
3075 * away if not KTHREAD_IS_PER_CPU.
3076 *
3077 * Specifically, migration_disabled() tasks must not fail the
3078 * cpumask_any_and_distribute() pick below, esp. so on
3079 * SCA_MIGRATE_ENABLE, otherwise we'll not call
3080 * set_cpus_allowed_common() and actually reset p->cpus_ptr.
3081 */
3082 cpu_valid_mask = cpu_online_mask;
3083 }
3084
3085 if (!kthread && !cpumask_subset(ctx->new_mask, cpu_allowed_mask)) {
3086 ret = -EINVAL;
3087 goto out;
3088 }
3089
3090 /*
3091 * Must re-check here, to close a race against __kthread_bind(),
3092 * sched_setaffinity() is not guaranteed to observe the flag.
3093 */
3094 if ((ctx->flags & SCA_CHECK) && (p->flags & PF_NO_SETAFFINITY)) {
3095 ret = -EINVAL;
3096 goto out;
3097 }
3098
3099 if (!(ctx->flags & SCA_MIGRATE_ENABLE)) {
3100 if (cpumask_equal(&p->cpus_mask, ctx->new_mask)) {
3101 if (ctx->flags & SCA_USER)
3102 swap(p->user_cpus_ptr, ctx->user_mask);
3103 goto out;
3104 }
3105
3106 if (WARN_ON_ONCE(p == current &&
3107 is_migration_disabled(p) &&
3108 !cpumask_test_cpu(task_cpu(p), ctx->new_mask))) {
3109 ret = -EBUSY;
3110 goto out;
3111 }
3112 }
3113
3114 /*
3115 * Picking a ~random cpu helps in cases where we are changing affinity
3116 * for groups of tasks (ie. cpuset), so that load balancing is not
3117 * immediately required to distribute the tasks within their new mask.
3118 */
3119 dest_cpu = cpumask_any_and_distribute(cpu_valid_mask, ctx->new_mask);
3120 if (dest_cpu >= nr_cpu_ids) {
3121 ret = -EINVAL;
3122 goto out;
3123 }
3124
3125 __do_set_cpus_allowed(p, ctx);
3126
3127 return affine_move_task(rq, p, rf, dest_cpu, ctx->flags);
3128
3129 out:
3130 task_rq_unlock(rq, p, rf);
3131
3132 return ret;
3133 }
3134
3135 /*
3136 * Change a given task's CPU affinity. Migrate the thread to a
3137 * proper CPU and schedule it away if the CPU it's executing on
3138 * is removed from the allowed bitmask.
3139 *
3140 * NOTE: the caller must have a valid reference to the task, the
3141 * task must not exit() & deallocate itself prematurely. The
3142 * call is not atomic; no spinlocks may be held.
3143 */
3144 int __set_cpus_allowed_ptr(struct task_struct *p, struct affinity_context *ctx)
3145 {
3146 struct rq_flags rf;
3147 struct rq *rq;
3148
3149 rq = task_rq_lock(p, &rf);
3150 /*
3151 * Masking should be skipped if SCA_USER or any of the SCA_MIGRATE_*
3152 * flags are set.
3153 */
3154 if (p->user_cpus_ptr &&
3155 !(ctx->flags & (SCA_USER | SCA_MIGRATE_ENABLE | SCA_MIGRATE_DISABLE)) &&
3156 cpumask_and(rq->scratch_mask, ctx->new_mask, p->user_cpus_ptr))
3157 ctx->new_mask = rq->scratch_mask;
3158
3159 return __set_cpus_allowed_ptr_locked(p, ctx, rq, &rf);
3160 }
3161
3162 int set_cpus_allowed_ptr(struct task_struct *p, const struct cpumask *new_mask)
3163 {
3164 struct affinity_context ac = {
3165 .new_mask = new_mask,
3166 .flags = 0,
3167 };
3168
3169 return __set_cpus_allowed_ptr(p, &ac);
3170 }
3171 EXPORT_SYMBOL_GPL(set_cpus_allowed_ptr);
3172
3173 /*
3174 * Change a given task's CPU affinity to the intersection of its current
3175 * affinity mask and @subset_mask, writing the resulting mask to @new_mask.
3176 * If user_cpus_ptr is defined, use it as the basis for restricting CPU
3177 * affinity or use cpu_online_mask instead.
3178 *
3179 * If the resulting mask is empty, leave the affinity unchanged and return
3180 * -EINVAL.
3181 */
3182 static int restrict_cpus_allowed_ptr(struct task_struct *p,
3183 struct cpumask *new_mask,
3184 const struct cpumask *subset_mask)
3185 {
3186 struct affinity_context ac = {
3187 .new_mask = new_mask,
3188 .flags = 0,
3189 };
3190 struct rq_flags rf;
3191 struct rq *rq;
3192 int err;
3193
3194 rq = task_rq_lock(p, &rf);
3195
3196 /*
3197 * Forcefully restricting the affinity of a deadline task is
3198 * likely to cause problems, so fail and noisily override the
3199 * mask entirely.
3200 */
3201 if (task_has_dl_policy(p) && dl_bandwidth_enabled()) {
3202 err = -EPERM;
3203 goto err_unlock;
3204 }
3205
3206 if (!cpumask_and(new_mask, task_user_cpus(p), subset_mask)) {
3207 err = -EINVAL;
3208 goto err_unlock;
3209 }
3210
3211 return __set_cpus_allowed_ptr_locked(p, &ac, rq, &rf);
3212
3213 err_unlock:
3214 task_rq_unlock(rq, p, &rf);
3215 return err;
3216 }
3217
3218 /*
3219 * Restrict the CPU affinity of task @p so that it is a subset of
3220 * task_cpu_possible_mask() and point @p->user_cpus_ptr to a copy of the
3221 * old affinity mask. If the resulting mask is empty, we warn and walk
3222 * up the cpuset hierarchy until we find a suitable mask.
3223 */
3224 void force_compatible_cpus_allowed_ptr(struct task_struct *p)
3225 {
3226 cpumask_var_t new_mask;
3227 const struct cpumask *override_mask = task_cpu_possible_mask(p);
3228
3229 alloc_cpumask_var(&new_mask, GFP_KERNEL);
3230
3231 /*
3232 * __migrate_task() can fail silently in the face of concurrent
3233 * offlining of the chosen destination CPU, so take the hotplug
3234 * lock to ensure that the migration succeeds.
3235 */
3236 cpus_read_lock();
3237 if (!cpumask_available(new_mask))
3238 goto out_set_mask;
3239
3240 if (!restrict_cpus_allowed_ptr(p, new_mask, override_mask))
3241 goto out_free_mask;
3242
3243 /*
3244 * We failed to find a valid subset of the affinity mask for the
3245 * task, so override it based on its cpuset hierarchy.
3246 */
3247 cpuset_cpus_allowed(p, new_mask);
3248 override_mask = new_mask;
3249
3250 out_set_mask:
3251 if (printk_ratelimit()) {
3252 printk_deferred("Overriding affinity for process %d (%s) to CPUs %*pbl\n",
3253 task_pid_nr(p), p->comm,
3254 cpumask_pr_args(override_mask));
3255 }
3256
3257 WARN_ON(set_cpus_allowed_ptr(p, override_mask));
3258 out_free_mask:
3259 cpus_read_unlock();
3260 free_cpumask_var(new_mask);
3261 }
3262
3263 /*
3264 * Restore the affinity of a task @p which was previously restricted by a
3265 * call to force_compatible_cpus_allowed_ptr().
3266 *
3267 * It is the caller's responsibility to serialise this with any calls to
3268 * force_compatible_cpus_allowed_ptr(@p).
3269 */
3270 void relax_compatible_cpus_allowed_ptr(struct task_struct *p)
3271 {
3272 struct affinity_context ac = {
3273 .new_mask = task_user_cpus(p),
3274 .flags = 0,
3275 };
3276 int ret;
3277
3278 /*
3279 * Try to restore the old affinity mask with __sched_setaffinity().
3280 * Cpuset masking will be done there too.
3281 */
3282 ret = __sched_setaffinity(p, &ac);
3283 WARN_ON_ONCE(ret);
3284 }
3285
3286 void set_task_cpu(struct task_struct *p, unsigned int new_cpu)
3287 {
3288 #ifdef CONFIG_SCHED_DEBUG
3289 unsigned int state = READ_ONCE(p->__state);
3290
3291 /*
3292 * We should never call set_task_cpu() on a blocked task,
3293 * ttwu() will sort out the placement.
3294 */
3295 WARN_ON_ONCE(state != TASK_RUNNING && state != TASK_WAKING && !p->on_rq);
3296
3297 /*
3298 * Migrating fair class task must have p->on_rq = TASK_ON_RQ_MIGRATING,
3299 * because schedstat_wait_{start,end} rebase migrating task's wait_start
3300 * time relying on p->on_rq.
3301 */
3302 WARN_ON_ONCE(state == TASK_RUNNING &&
3303 p->sched_class == &fair_sched_class &&
3304 (p->on_rq && !task_on_rq_migrating(p)));
3305
3306 #ifdef CONFIG_LOCKDEP
3307 /*
3308 * The caller should hold either p->pi_lock or rq->lock, when changing
3309 * a task's CPU. ->pi_lock for waking tasks, rq->lock for runnable tasks.
3310 *
3311 * sched_move_task() holds both and thus holding either pins the cgroup,
3312 * see task_group().
3313 *
3314 * Furthermore, all task_rq users should acquire both locks, see
3315 * task_rq_lock().
3316 */
3317 WARN_ON_ONCE(debug_locks && !(lockdep_is_held(&p->pi_lock) ||
3318 lockdep_is_held(__rq_lockp(task_rq(p)))));
3319 #endif
3320 /*
3321 * Clearly, migrating tasks to offline CPUs is a fairly daft thing.
3322 */
3323 WARN_ON_ONCE(!cpu_online(new_cpu));
3324
3325 WARN_ON_ONCE(is_migration_disabled(p));
3326 #endif
3327
3328 trace_sched_migrate_task(p, new_cpu);
3329
3330 if (task_cpu(p) != new_cpu) {
3331 if (p->sched_class->migrate_task_rq)
3332 p->sched_class->migrate_task_rq(p, new_cpu);
3333 p->se.nr_migrations++;
3334 rseq_migrate(p);
3335 sched_mm_cid_migrate_from(p);
3336 perf_event_task_migrate(p);
3337 }
3338
3339 __set_task_cpu(p, new_cpu);
3340 }
3341
3342 #ifdef CONFIG_NUMA_BALANCING
3343 static void __migrate_swap_task(struct task_struct *p, int cpu)
3344 {
3345 if (task_on_rq_queued(p)) {
3346 struct rq *src_rq, *dst_rq;
3347 struct rq_flags srf, drf;
3348
3349 src_rq = task_rq(p);
3350 dst_rq = cpu_rq(cpu);
3351
3352 rq_pin_lock(src_rq, &srf);
3353 rq_pin_lock(dst_rq, &drf);
3354
3355 move_queued_task_locked(src_rq, dst_rq, p);
3356 wakeup_preempt(dst_rq, p, 0);
3357
3358 rq_unpin_lock(dst_rq, &drf);
3359 rq_unpin_lock(src_rq, &srf);
3360
3361 } else {
3362 /*
3363 * Task isn't running anymore; make it appear like we migrated
3364 * it before it went to sleep. This means on wakeup we make the
3365 * previous CPU our target instead of where it really is.
3366 */
3367 p->wake_cpu = cpu;
3368 }
3369 }
3370
3371 struct migration_swap_arg {
3372 struct task_struct *src_task, *dst_task;
3373 int src_cpu, dst_cpu;
3374 };
3375
3376 static int migrate_swap_stop(void *data)
3377 {
3378 struct migration_swap_arg *arg = data;
3379 struct rq *src_rq, *dst_rq;
3380
3381 if (!cpu_active(arg->src_cpu) || !cpu_active(arg->dst_cpu))
3382 return -EAGAIN;
3383
3384 src_rq = cpu_rq(arg->src_cpu);
3385 dst_rq = cpu_rq(arg->dst_cpu);
3386
3387 guard(double_raw_spinlock)(&arg->src_task->pi_lock, &arg->dst_task->pi_lock);
3388 guard(double_rq_lock)(src_rq, dst_rq);
3389
3390 if (task_cpu(arg->dst_task) != arg->dst_cpu)
3391 return -EAGAIN;
3392
3393 if (task_cpu(arg->src_task) != arg->src_cpu)
3394 return -EAGAIN;
3395
3396 if (!cpumask_test_cpu(arg->dst_cpu, arg->src_task->cpus_ptr))
3397 return -EAGAIN;
3398
3399 if (!cpumask_test_cpu(arg->src_cpu, arg->dst_task->cpus_ptr))
3400 return -EAGAIN;
3401
3402 __migrate_swap_task(arg->src_task, arg->dst_cpu);
3403 __migrate_swap_task(arg->dst_task, arg->src_cpu);
3404
3405 return 0;
3406 }
3407
3408 /*
3409 * Cross migrate two tasks
3410 */
3411 int migrate_swap(struct task_struct *cur, struct task_struct *p,
3412 int target_cpu, int curr_cpu)
3413 {
3414 struct migration_swap_arg arg;
3415 int ret = -EINVAL;
3416
3417 arg = (struct migration_swap_arg){
3418 .src_task = cur,
3419 .src_cpu = curr_cpu,
3420 .dst_task = p,
3421 .dst_cpu = target_cpu,
3422 };
3423
3424 if (arg.src_cpu == arg.dst_cpu)
3425 goto out;
3426
3427 /*
3428 * These three tests are all lockless; this is OK since all of them
3429 * will be re-checked with proper locks held further down the line.
3430 */
3431 if (!cpu_active(arg.src_cpu) || !cpu_active(arg.dst_cpu))
3432 goto out;
3433
3434 if (!cpumask_test_cpu(arg.dst_cpu, arg.src_task->cpus_ptr))
3435 goto out;
3436
3437 if (!cpumask_test_cpu(arg.src_cpu, arg.dst_task->cpus_ptr))
3438 goto out;
3439
3440 trace_sched_swap_numa(cur, arg.src_cpu, p, arg.dst_cpu);
3441 ret = stop_two_cpus(arg.dst_cpu, arg.src_cpu, migrate_swap_stop, &arg);
3442
3443 out:
3444 return ret;
3445 }
3446 #endif /* CONFIG_NUMA_BALANCING */
3447
3448 /***
3449 * kick_process - kick a running thread to enter/exit the kernel
3450 * @p: the to-be-kicked thread
3451 *
3452 * Cause a process which is running on another CPU to enter
3453 * kernel-mode, without any delay. (to get signals handled.)
3454 *
3455 * NOTE: this function doesn't have to take the runqueue lock,
3456 * because all it wants to ensure is that the remote task enters
3457 * the kernel. If the IPI races and the task has been migrated
3458 * to another CPU then no harm is done and the purpose has been
3459 * achieved as well.
3460 */
3461 void kick_process(struct task_struct *p)
3462 {
3463 guard(preempt)();
3464 int cpu = task_cpu(p);
3465
3466 if ((cpu != smp_processor_id()) && task_curr(p))
3467 smp_send_reschedule(cpu);
3468 }
3469 EXPORT_SYMBOL_GPL(kick_process);
3470
3471 /*
3472 * ->cpus_ptr is protected by both rq->lock and p->pi_lock
3473 *
3474 * A few notes on cpu_active vs cpu_online:
3475 *
3476 * - cpu_active must be a subset of cpu_online
3477 *
3478 * - on CPU-up we allow per-CPU kthreads on the online && !active CPU,
3479 * see __set_cpus_allowed_ptr(). At this point the newly online
3480 * CPU isn't yet part of the sched domains, and balancing will not
3481 * see it.
3482 *
3483 * - on CPU-down we clear cpu_active() to mask the sched domains and
3484 * avoid the load balancer to place new tasks on the to be removed
3485 * CPU. Existing tasks will remain running there and will be taken
3486 * off.
3487 *
3488 * This means that fallback selection must not select !active CPUs.
3489 * And can assume that any active CPU must be online. Conversely
3490 * select_task_rq() below may allow selection of !active CPUs in order
3491 * to satisfy the above rules.
3492 */
3493 static int select_fallback_rq(int cpu, struct task_struct *p)
3494 {
3495 int nid = cpu_to_node(cpu);
3496 const struct cpumask *nodemask = NULL;
3497 enum { cpuset, possible, fail } state = cpuset;
3498 int dest_cpu;
3499
3500 /*
3501 * If the node that the CPU is on has been offlined, cpu_to_node()
3502 * will return -1. There is no CPU on the node, and we should
3503 * select the CPU on the other node.
3504 */
3505 if (nid != -1) {
3506 nodemask = cpumask_of_node(nid);
3507
3508 /* Look for allowed, online CPU in same node. */
3509 for_each_cpu(dest_cpu, nodemask) {
3510 if (is_cpu_allowed(p, dest_cpu))
3511 return dest_cpu;
3512 }
3513 }
3514
3515 for (;;) {
3516 /* Any allowed, online CPU? */
3517 for_each_cpu(dest_cpu, p->cpus_ptr) {
3518 if (!is_cpu_allowed(p, dest_cpu))
3519 continue;
3520
3521 goto out;
3522 }
3523
3524 /* No more Mr. Nice Guy. */
3525 switch (state) {
3526 case cpuset:
3527 if (cpuset_cpus_allowed_fallback(p)) {
3528 state = possible;
3529 break;
3530 }
3531 fallthrough;
3532 case possible:
3533 /*
3534 * XXX When called from select_task_rq() we only
3535 * hold p->pi_lock and again violate locking order.
3536 *
3537 * More yuck to audit.
3538 */
3539 do_set_cpus_allowed(p, task_cpu_possible_mask(p));
3540 state = fail;
3541 break;
3542 case fail:
3543 BUG();
3544 break;
3545 }
3546 }
3547
3548 out:
3549 if (state != cpuset) {
3550 /*
3551 * Don't tell them about moving exiting tasks or
3552 * kernel threads (both mm NULL), since they never
3553 * leave kernel.
3554 */
3555 if (p->mm && printk_ratelimit()) {
3556 printk_deferred("process %d (%s) no longer affine to cpu%d\n",
3557 task_pid_nr(p), p->comm, cpu);
3558 }
3559 }
3560
3561 return dest_cpu;
3562 }
3563
3564 /*
3565 * The caller (fork, wakeup) owns p->pi_lock, ->cpus_ptr is stable.
3566 */
3567 static inline
3568 int select_task_rq(struct task_struct *p, int cpu, int *wake_flags)
3569 {
3570 lockdep_assert_held(&p->pi_lock);
3571
3572 if (p->nr_cpus_allowed > 1 && !is_migration_disabled(p)) {
3573 cpu = p->sched_class->select_task_rq(p, cpu, *wake_flags);
3574 *wake_flags |= WF_RQ_SELECTED;
3575 } else {
3576 cpu = cpumask_any(p->cpus_ptr);
3577 }
3578
3579 /*
3580 * In order not to call set_task_cpu() on a blocking task we need
3581 * to rely on ttwu() to place the task on a valid ->cpus_ptr
3582 * CPU.
3583 *
3584 * Since this is common to all placement strategies, this lives here.
3585 *
3586 * [ this allows ->select_task() to simply return task_cpu(p) and
3587 * not worry about this generic constraint ]
3588 */
3589 if (unlikely(!is_cpu_allowed(p, cpu)))
3590 cpu = select_fallback_rq(task_cpu(p), p);
3591
3592 return cpu;
3593 }
3594
3595 void sched_set_stop_task(int cpu, struct task_struct *stop)
3596 {
3597 static struct lock_class_key stop_pi_lock;
3598 struct sched_param param = { .sched_priority = MAX_RT_PRIO - 1 };
3599 struct task_struct *old_stop = cpu_rq(cpu)->stop;
3600
3601 if (stop) {
3602 /*
3603 * Make it appear like a SCHED_FIFO task, its something
3604 * userspace knows about and won't get confused about.
3605 *
3606 * Also, it will make PI more or less work without too
3607 * much confusion -- but then, stop work should not
3608 * rely on PI working anyway.
3609 */
3610 sched_setscheduler_nocheck(stop, SCHED_FIFO, ¶m);
3611
3612 stop->sched_class = &stop_sched_class;
3613
3614 /*
3615 * The PI code calls rt_mutex_setprio() with ->pi_lock held to
3616 * adjust the effective priority of a task. As a result,
3617 * rt_mutex_setprio() can trigger (RT) balancing operations,
3618 * which can then trigger wakeups of the stop thread to push
3619 * around the current task.
3620 *
3621 * The stop task itself will never be part of the PI-chain, it
3622 * never blocks, therefore that ->pi_lock recursion is safe.
3623 * Tell lockdep about this by placing the stop->pi_lock in its
3624 * own class.
3625 */
3626 lockdep_set_class(&stop->pi_lock, &stop_pi_lock);
3627 }
3628
3629 cpu_rq(cpu)->stop = stop;
3630
3631 if (old_stop) {
3632 /*
3633 * Reset it back to a normal scheduling class so that
3634 * it can die in pieces.
3635 */
3636 old_stop->sched_class = &rt_sched_class;
3637 }
3638 }
3639
3640 #else /* CONFIG_SMP */
3641
3642 static inline void migrate_disable_switch(struct rq *rq, struct task_struct *p) { }
3643
3644 static inline bool rq_has_pinned_tasks(struct rq *rq)
3645 {
3646 return false;
3647 }
3648
3649 #endif /* !CONFIG_SMP */
3650
3651 static void
3652 ttwu_stat(struct task_struct *p, int cpu, int wake_flags)
3653 {
3654 struct rq *rq;
3655
3656 if (!schedstat_enabled())
3657 return;
3658
3659 rq = this_rq();
3660
3661 #ifdef CONFIG_SMP
3662 if (cpu == rq->cpu) {
3663 __schedstat_inc(rq->ttwu_local);
3664 __schedstat_inc(p->stats.nr_wakeups_local);
3665 } else {
3666 struct sched_domain *sd;
3667
3668 __schedstat_inc(p->stats.nr_wakeups_remote);
3669
3670 guard(rcu)();
3671 for_each_domain(rq->cpu, sd) {
3672 if (cpumask_test_cpu(cpu, sched_domain_span(sd))) {
3673 __schedstat_inc(sd->ttwu_wake_remote);
3674 break;
3675 }
3676 }
3677 }
3678
3679 if (wake_flags & WF_MIGRATED)
3680 __schedstat_inc(p->stats.nr_wakeups_migrate);
3681 #endif /* CONFIG_SMP */
3682
3683 __schedstat_inc(rq->ttwu_count);
3684 __schedstat_inc(p->stats.nr_wakeups);
3685
3686 if (wake_flags & WF_SYNC)
3687 __schedstat_inc(p->stats.nr_wakeups_sync);
3688 }
3689
3690 /*
3691 * Mark the task runnable.
3692 */
3693 static inline void ttwu_do_wakeup(struct task_struct *p)
3694 {
3695 WRITE_ONCE(p->__state, TASK_RUNNING);
3696 trace_sched_wakeup(p);
3697 }
3698
3699 static void
3700 ttwu_do_activate(struct rq *rq, struct task_struct *p, int wake_flags,
3701 struct rq_flags *rf)
3702 {
3703 int en_flags = ENQUEUE_WAKEUP | ENQUEUE_NOCLOCK;
3704
3705 lockdep_assert_rq_held(rq);
3706
3707 if (p->sched_contributes_to_load)
3708 rq->nr_uninterruptible--;
3709
3710 #ifdef CONFIG_SMP
3711 if (wake_flags & WF_RQ_SELECTED)
3712 en_flags |= ENQUEUE_RQ_SELECTED;
3713 if (wake_flags & WF_MIGRATED)
3714 en_flags |= ENQUEUE_MIGRATED;
3715 else
3716 #endif
3717 if (p->in_iowait) {
3718 delayacct_blkio_end(p);
3719 atomic_dec(&task_rq(p)->nr_iowait);
3720 }
3721
3722 activate_task(rq, p, en_flags);
3723 wakeup_preempt(rq, p, wake_flags);
3724
3725 ttwu_do_wakeup(p);
3726
3727 #ifdef CONFIG_SMP
3728 if (p->sched_class->task_woken) {
3729 /*
3730 * Our task @p is fully woken up and running; so it's safe to
3731 * drop the rq->lock, hereafter rq is only used for statistics.
3732 */
3733 rq_unpin_lock(rq, rf);
3734 p->sched_class->task_woken(rq, p);
3735 rq_repin_lock(rq, rf);
3736 }
3737
3738 if (rq->idle_stamp) {
3739 u64 delta = rq_clock(rq) - rq->idle_stamp;
3740 u64 max = 2*rq->max_idle_balance_cost;
3741
3742 update_avg(&rq->avg_idle, delta);
3743
3744 if (rq->avg_idle > max)
3745 rq->avg_idle = max;
3746
3747 rq->idle_stamp = 0;
3748 }
3749 #endif
3750 }
3751
3752 /*
3753 * Consider @p being inside a wait loop:
3754 *
3755 * for (;;) {
3756 * set_current_state(TASK_UNINTERRUPTIBLE);
3757 *
3758 * if (CONDITION)
3759 * break;
3760 *
3761 * schedule();
3762 * }
3763 * __set_current_state(TASK_RUNNING);
3764 *
3765 * between set_current_state() and schedule(). In this case @p is still
3766 * runnable, so all that needs doing is change p->state back to TASK_RUNNING in
3767 * an atomic manner.
3768 *
3769 * By taking task_rq(p)->lock we serialize against schedule(), if @p->on_rq
3770 * then schedule() must still happen and p->state can be changed to
3771 * TASK_RUNNING. Otherwise we lost the race, schedule() has happened, and we
3772 * need to do a full wakeup with enqueue.
3773 *
3774 * Returns: %true when the wakeup is done,
3775 * %false otherwise.
3776 */
3777 static int ttwu_runnable(struct task_struct *p, int wake_flags)
3778 {
3779 struct rq_flags rf;
3780 struct rq *rq;
3781 int ret = 0;
3782
3783 rq = __task_rq_lock(p, &rf);
3784 if (task_on_rq_queued(p)) {
3785 update_rq_clock(rq);
3786 if (p->se.sched_delayed)
3787 enqueue_task(rq, p, ENQUEUE_NOCLOCK | ENQUEUE_DELAYED);
3788 if (!task_on_cpu(rq, p)) {
3789 /*
3790 * When on_rq && !on_cpu the task is preempted, see if
3791 * it should preempt the task that is current now.
3792 */
3793 wakeup_preempt(rq, p, wake_flags);
3794 }
3795 ttwu_do_wakeup(p);
3796 ret = 1;
3797 }
3798 __task_rq_unlock(rq, &rf);
3799
3800 return ret;
3801 }
3802
3803 #ifdef CONFIG_SMP
3804 void sched_ttwu_pending(void *arg)
3805 {
3806 struct llist_node *llist = arg;
3807 struct rq *rq = this_rq();
3808 struct task_struct *p, *t;
3809 struct rq_flags rf;
3810
3811 if (!llist)
3812 return;
3813
3814 rq_lock_irqsave(rq, &rf);
3815 update_rq_clock(rq);
3816
3817 llist_for_each_entry_safe(p, t, llist, wake_entry.llist) {
3818 if (WARN_ON_ONCE(p->on_cpu))
3819 smp_cond_load_acquire(&p->on_cpu, !VAL);
3820
3821 if (WARN_ON_ONCE(task_cpu(p) != cpu_of(rq)))
3822 set_task_cpu(p, cpu_of(rq));
3823
3824 ttwu_do_activate(rq, p, p->sched_remote_wakeup ? WF_MIGRATED : 0, &rf);
3825 }
3826
3827 /*
3828 * Must be after enqueueing at least once task such that
3829 * idle_cpu() does not observe a false-negative -- if it does,
3830 * it is possible for select_idle_siblings() to stack a number
3831 * of tasks on this CPU during that window.
3832 *
3833 * It is OK to clear ttwu_pending when another task pending.
3834 * We will receive IPI after local IRQ enabled and then enqueue it.
3835 * Since now nr_running > 0, idle_cpu() will always get correct result.
3836 */
3837 WRITE_ONCE(rq->ttwu_pending, 0);
3838 rq_unlock_irqrestore(rq, &rf);
3839 }
3840
3841 /*
3842 * Prepare the scene for sending an IPI for a remote smp_call
3843 *
3844 * Returns true if the caller can proceed with sending the IPI.
3845 * Returns false otherwise.
3846 */
3847 bool call_function_single_prep_ipi(int cpu)
3848 {
3849 if (set_nr_if_polling(cpu_rq(cpu)->idle)) {
3850 trace_sched_wake_idle_without_ipi(cpu);
3851 return false;
3852 }
3853
3854 return true;
3855 }
3856
3857 /*
3858 * Queue a task on the target CPUs wake_list and wake the CPU via IPI if
3859 * necessary. The wakee CPU on receipt of the IPI will queue the task
3860 * via sched_ttwu_wakeup() for activation so the wakee incurs the cost
3861 * of the wakeup instead of the waker.
3862 */
3863 static void __ttwu_queue_wakelist(struct task_struct *p, int cpu, int wake_flags)
3864 {
3865 struct rq *rq = cpu_rq(cpu);
3866
3867 p->sched_remote_wakeup = !!(wake_flags & WF_MIGRATED);
3868
3869 WRITE_ONCE(rq->ttwu_pending, 1);
3870 __smp_call_single_queue(cpu, &p->wake_entry.llist);
3871 }
3872
3873 void wake_up_if_idle(int cpu)
3874 {
3875 struct rq *rq = cpu_rq(cpu);
3876
3877 guard(rcu)();
3878 if (is_idle_task(rcu_dereference(rq->curr))) {
3879 guard(rq_lock_irqsave)(rq);
3880 if (is_idle_task(rq->curr))
3881 resched_curr(rq);
3882 }
3883 }
3884
3885 bool cpus_equal_capacity(int this_cpu, int that_cpu)
3886 {
3887 if (!sched_asym_cpucap_active())
3888 return true;
3889
3890 if (this_cpu == that_cpu)
3891 return true;
3892
3893 return arch_scale_cpu_capacity(this_cpu) == arch_scale_cpu_capacity(that_cpu);
3894 }
3895
3896 bool cpus_share_cache(int this_cpu, int that_cpu)
3897 {
3898 if (this_cpu == that_cpu)
3899 return true;
3900
3901 return per_cpu(sd_llc_id, this_cpu) == per_cpu(sd_llc_id, that_cpu);
3902 }
3903
3904 /*
3905 * Whether CPUs are share cache resources, which means LLC on non-cluster
3906 * machines and LLC tag or L2 on machines with clusters.
3907 */
3908 bool cpus_share_resources(int this_cpu, int that_cpu)
3909 {
3910 if (this_cpu == that_cpu)
3911 return true;
3912
3913 return per_cpu(sd_share_id, this_cpu) == per_cpu(sd_share_id, that_cpu);
3914 }
3915
3916 static inline bool ttwu_queue_cond(struct task_struct *p, int cpu)
3917 {
3918 /*
3919 * The BPF scheduler may depend on select_task_rq() being invoked during
3920 * wakeups. In addition, @p may end up executing on a different CPU
3921 * regardless of what happens in the wakeup path making the ttwu_queue
3922 * optimization less meaningful. Skip if on SCX.
3923 */
3924 if (task_on_scx(p))
3925 return false;
3926
3927 /*
3928 * Do not complicate things with the async wake_list while the CPU is
3929 * in hotplug state.
3930 */
3931 if (!cpu_active(cpu))
3932 return false;
3933
3934 /* Ensure the task will still be allowed to run on the CPU. */
3935 if (!cpumask_test_cpu(cpu, p->cpus_ptr))
3936 return false;
3937
3938 /*
3939 * If the CPU does not share cache, then queue the task on the
3940 * remote rqs wakelist to avoid accessing remote data.
3941 */
3942 if (!cpus_share_cache(smp_processor_id(), cpu))
3943 return true;
3944
3945 if (cpu == smp_processor_id())
3946 return false;
3947
3948 /*
3949 * If the wakee cpu is idle, or the task is descheduling and the
3950 * only running task on the CPU, then use the wakelist to offload
3951 * the task activation to the idle (or soon-to-be-idle) CPU as
3952 * the current CPU is likely busy. nr_running is checked to
3953 * avoid unnecessary task stacking.
3954 *
3955 * Note that we can only get here with (wakee) p->on_rq=0,
3956 * p->on_cpu can be whatever, we've done the dequeue, so
3957 * the wakee has been accounted out of ->nr_running.
3958 */
3959 if (!cpu_rq(cpu)->nr_running)
3960 return true;
3961
3962 return false;
3963 }
3964
3965 static bool ttwu_queue_wakelist(struct task_struct *p, int cpu, int wake_flags)
3966 {
3967 if (sched_feat(TTWU_QUEUE) && ttwu_queue_cond(p, cpu)) {
3968 sched_clock_cpu(cpu); /* Sync clocks across CPUs */
3969 __ttwu_queue_wakelist(p, cpu, wake_flags);
3970 return true;
3971 }
3972
3973 return false;
3974 }
3975
3976 #else /* !CONFIG_SMP */
3977
3978 static inline bool ttwu_queue_wakelist(struct task_struct *p, int cpu, int wake_flags)
3979 {
3980 return false;
3981 }
3982
3983 #endif /* CONFIG_SMP */
3984
3985 static void ttwu_queue(struct task_struct *p, int cpu, int wake_flags)
3986 {
3987 struct rq *rq = cpu_rq(cpu);
3988 struct rq_flags rf;
3989
3990 if (ttwu_queue_wakelist(p, cpu, wake_flags))
3991 return;
3992
3993 rq_lock(rq, &rf);
3994 update_rq_clock(rq);
3995 ttwu_do_activate(rq, p, wake_flags, &rf);
3996 rq_unlock(rq, &rf);
3997 }
3998
3999 /*
4000 * Invoked from try_to_wake_up() to check whether the task can be woken up.
4001 *
4002 * The caller holds p::pi_lock if p != current or has preemption
4003 * disabled when p == current.
4004 *
4005 * The rules of saved_state:
4006 *
4007 * The related locking code always holds p::pi_lock when updating
4008 * p::saved_state, which means the code is fully serialized in both cases.
4009 *
4010 * For PREEMPT_RT, the lock wait and lock wakeups happen via TASK_RTLOCK_WAIT.
4011 * No other bits set. This allows to distinguish all wakeup scenarios.
4012 *
4013 * For FREEZER, the wakeup happens via TASK_FROZEN. No other bits set. This
4014 * allows us to prevent early wakeup of tasks before they can be run on
4015 * asymmetric ISA architectures (eg ARMv9).
4016 */
4017 static __always_inline
4018 bool ttwu_state_match(struct task_struct *p, unsigned int state, int *success)
4019 {
4020 int match;
4021
4022 if (IS_ENABLED(CONFIG_DEBUG_PREEMPT)) {
4023 WARN_ON_ONCE((state & TASK_RTLOCK_WAIT) &&
4024 state != TASK_RTLOCK_WAIT);
4025 }
4026
4027 *success = !!(match = __task_state_match(p, state));
4028
4029 /*
4030 * Saved state preserves the task state across blocking on
4031 * an RT lock or TASK_FREEZABLE tasks. If the state matches,
4032 * set p::saved_state to TASK_RUNNING, but do not wake the task
4033 * because it waits for a lock wakeup or __thaw_task(). Also
4034 * indicate success because from the regular waker's point of
4035 * view this has succeeded.
4036 *
4037 * After acquiring the lock the task will restore p::__state
4038 * from p::saved_state which ensures that the regular
4039 * wakeup is not lost. The restore will also set
4040 * p::saved_state to TASK_RUNNING so any further tests will
4041 * not result in false positives vs. @success
4042 */
4043 if (match < 0)
4044 p->saved_state = TASK_RUNNING;
4045
4046 return match > 0;
4047 }
4048
4049 /*
4050 * Notes on Program-Order guarantees on SMP systems.
4051 *
4052 * MIGRATION
4053 *
4054 * The basic program-order guarantee on SMP systems is that when a task [t]
4055 * migrates, all its activity on its old CPU [c0] happens-before any subsequent
4056 * execution on its new CPU [c1].
4057 *
4058 * For migration (of runnable tasks) this is provided by the following means:
4059 *
4060 * A) UNLOCK of the rq(c0)->lock scheduling out task t
4061 * B) migration for t is required to synchronize *both* rq(c0)->lock and
4062 * rq(c1)->lock (if not at the same time, then in that order).
4063 * C) LOCK of the rq(c1)->lock scheduling in task
4064 *
4065 * Release/acquire chaining guarantees that B happens after A and C after B.
4066 * Note: the CPU doing B need not be c0 or c1
4067 *
4068 * Example:
4069 *
4070 * CPU0 CPU1 CPU2
4071 *
4072 * LOCK rq(0)->lock
4073 * sched-out X
4074 * sched-in Y
4075 * UNLOCK rq(0)->lock
4076 *
4077 * LOCK rq(0)->lock // orders against CPU0
4078 * dequeue X
4079 * UNLOCK rq(0)->lock
4080 *
4081 * LOCK rq(1)->lock
4082 * enqueue X
4083 * UNLOCK rq(1)->lock
4084 *
4085 * LOCK rq(1)->lock // orders against CPU2
4086 * sched-out Z
4087 * sched-in X
4088 * UNLOCK rq(1)->lock
4089 *
4090 *
4091 * BLOCKING -- aka. SLEEP + WAKEUP
4092 *
4093 * For blocking we (obviously) need to provide the same guarantee as for
4094 * migration. However the means are completely different as there is no lock
4095 * chain to provide order. Instead we do:
4096 *
4097 * 1) smp_store_release(X->on_cpu, 0) -- finish_task()
4098 * 2) smp_cond_load_acquire(!X->on_cpu) -- try_to_wake_up()
4099 *
4100 * Example:
4101 *
4102 * CPU0 (schedule) CPU1 (try_to_wake_up) CPU2 (schedule)
4103 *
4104 * LOCK rq(0)->lock LOCK X->pi_lock
4105 * dequeue X
4106 * sched-out X
4107 * smp_store_release(X->on_cpu, 0);
4108 *
4109 * smp_cond_load_acquire(&X->on_cpu, !VAL);
4110 * X->state = WAKING
4111 * set_task_cpu(X,2)
4112 *
4113 * LOCK rq(2)->lock
4114 * enqueue X
4115 * X->state = RUNNING
4116 * UNLOCK rq(2)->lock
4117 *
4118 * LOCK rq(2)->lock // orders against CPU1
4119 * sched-out Z
4120 * sched-in X
4121 * UNLOCK rq(2)->lock
4122 *
4123 * UNLOCK X->pi_lock
4124 * UNLOCK rq(0)->lock
4125 *
4126 *
4127 * However, for wakeups there is a second guarantee we must provide, namely we
4128 * must ensure that CONDITION=1 done by the caller can not be reordered with
4129 * accesses to the task state; see try_to_wake_up() and set_current_state().
4130 */
4131
4132 /**
4133 * try_to_wake_up - wake up a thread
4134 * @p: the thread to be awakened
4135 * @state: the mask of task states that can be woken
4136 * @wake_flags: wake modifier flags (WF_*)
4137 *
4138 * Conceptually does:
4139 *
4140 * If (@state & @p->state) @p->state = TASK_RUNNING.
4141 *
4142 * If the task was not queued/runnable, also place it back on a runqueue.
4143 *
4144 * This function is atomic against schedule() which would dequeue the task.
4145 *
4146 * It issues a full memory barrier before accessing @p->state, see the comment
4147 * with set_current_state().
4148 *
4149 * Uses p->pi_lock to serialize against concurrent wake-ups.
4150 *
4151 * Relies on p->pi_lock stabilizing:
4152 * - p->sched_class
4153 * - p->cpus_ptr
4154 * - p->sched_task_group
4155 * in order to do migration, see its use of select_task_rq()/set_task_cpu().
4156 *
4157 * Tries really hard to only take one task_rq(p)->lock for performance.
4158 * Takes rq->lock in:
4159 * - ttwu_runnable() -- old rq, unavoidable, see comment there;
4160 * - ttwu_queue() -- new rq, for enqueue of the task;
4161 * - psi_ttwu_dequeue() -- much sadness :-( accounting will kill us.
4162 *
4163 * As a consequence we race really badly with just about everything. See the
4164 * many memory barriers and their comments for details.
4165 *
4166 * Return: %true if @p->state changes (an actual wakeup was done),
4167 * %false otherwise.
4168 */
4169 int try_to_wake_up(struct task_struct *p, unsigned int state, int wake_flags)
4170 {
4171 guard(preempt)();
4172 int cpu, success = 0;
4173
4174 wake_flags |= WF_TTWU;
4175
4176 if (p == current) {
4177 /*
4178 * We're waking current, this means 'p->on_rq' and 'task_cpu(p)
4179 * == smp_processor_id()'. Together this means we can special
4180 * case the whole 'p->on_rq && ttwu_runnable()' case below
4181 * without taking any locks.
4182 *
4183 * Specifically, given current runs ttwu() we must be before
4184 * schedule()'s block_task(), as such this must not observe
4185 * sched_delayed.
4186 *
4187 * In particular:
4188 * - we rely on Program-Order guarantees for all the ordering,
4189 * - we're serialized against set_special_state() by virtue of
4190 * it disabling IRQs (this allows not taking ->pi_lock).
4191 */
4192 SCHED_WARN_ON(p->se.sched_delayed);
4193 if (!ttwu_state_match(p, state, &success))
4194 goto out;
4195
4196 trace_sched_waking(p);
4197 ttwu_do_wakeup(p);
4198 goto out;
4199 }
4200
4201 /*
4202 * If we are going to wake up a thread waiting for CONDITION we
4203 * need to ensure that CONDITION=1 done by the caller can not be
4204 * reordered with p->state check below. This pairs with smp_store_mb()
4205 * in set_current_state() that the waiting thread does.
4206 */
4207 scoped_guard (raw_spinlock_irqsave, &p->pi_lock) {
4208 smp_mb__after_spinlock();
4209 if (!ttwu_state_match(p, state, &success))
4210 break;
4211
4212 trace_sched_waking(p);
4213
4214 /*
4215 * Ensure we load p->on_rq _after_ p->state, otherwise it would
4216 * be possible to, falsely, observe p->on_rq == 0 and get stuck
4217 * in smp_cond_load_acquire() below.
4218 *
4219 * sched_ttwu_pending() try_to_wake_up()
4220 * STORE p->on_rq = 1 LOAD p->state
4221 * UNLOCK rq->lock
4222 *
4223 * __schedule() (switch to task 'p')
4224 * LOCK rq->lock smp_rmb();
4225 * smp_mb__after_spinlock();
4226 * UNLOCK rq->lock
4227 *
4228 * [task p]
4229 * STORE p->state = UNINTERRUPTIBLE LOAD p->on_rq
4230 *
4231 * Pairs with the LOCK+smp_mb__after_spinlock() on rq->lock in
4232 * __schedule(). See the comment for smp_mb__after_spinlock().
4233 *
4234 * A similar smp_rmb() lives in __task_needs_rq_lock().
4235 */
4236 smp_rmb();
4237 if (READ_ONCE(p->on_rq) && ttwu_runnable(p, wake_flags))
4238 break;
4239
4240 #ifdef CONFIG_SMP
4241 /*
4242 * Ensure we load p->on_cpu _after_ p->on_rq, otherwise it would be
4243 * possible to, falsely, observe p->on_cpu == 0.
4244 *
4245 * One must be running (->on_cpu == 1) in order to remove oneself
4246 * from the runqueue.
4247 *
4248 * __schedule() (switch to task 'p') try_to_wake_up()
4249 * STORE p->on_cpu = 1 LOAD p->on_rq
4250 * UNLOCK rq->lock
4251 *
4252 * __schedule() (put 'p' to sleep)
4253 * LOCK rq->lock smp_rmb();
4254 * smp_mb__after_spinlock();
4255 * STORE p->on_rq = 0 LOAD p->on_cpu
4256 *
4257 * Pairs with the LOCK+smp_mb__after_spinlock() on rq->lock in
4258 * __schedule(). See the comment for smp_mb__after_spinlock().
4259 *
4260 * Form a control-dep-acquire with p->on_rq == 0 above, to ensure
4261 * schedule()'s deactivate_task() has 'happened' and p will no longer
4262 * care about it's own p->state. See the comment in __schedule().
4263 */
4264 smp_acquire__after_ctrl_dep();
4265
4266 /*
4267 * We're doing the wakeup (@success == 1), they did a dequeue (p->on_rq
4268 * == 0), which means we need to do an enqueue, change p->state to
4269 * TASK_WAKING such that we can unlock p->pi_lock before doing the
4270 * enqueue, such as ttwu_queue_wakelist().
4271 */
4272 WRITE_ONCE(p->__state, TASK_WAKING);
4273
4274 /*
4275 * If the owning (remote) CPU is still in the middle of schedule() with
4276 * this task as prev, considering queueing p on the remote CPUs wake_list
4277 * which potentially sends an IPI instead of spinning on p->on_cpu to
4278 * let the waker make forward progress. This is safe because IRQs are
4279 * disabled and the IPI will deliver after on_cpu is cleared.
4280 *
4281 * Ensure we load task_cpu(p) after p->on_cpu:
4282 *
4283 * set_task_cpu(p, cpu);
4284 * STORE p->cpu = @cpu
4285 * __schedule() (switch to task 'p')
4286 * LOCK rq->lock
4287 * smp_mb__after_spin_lock() smp_cond_load_acquire(&p->on_cpu)
4288 * STORE p->on_cpu = 1 LOAD p->cpu
4289 *
4290 * to ensure we observe the correct CPU on which the task is currently
4291 * scheduling.
4292 */
4293 if (smp_load_acquire(&p->on_cpu) &&
4294 ttwu_queue_wakelist(p, task_cpu(p), wake_flags))
4295 break;
4296
4297 /*
4298 * If the owning (remote) CPU is still in the middle of schedule() with
4299 * this task as prev, wait until it's done referencing the task.
4300 *
4301 * Pairs with the smp_store_release() in finish_task().
4302 *
4303 * This ensures that tasks getting woken will be fully ordered against
4304 * their previous state and preserve Program Order.
4305 */
4306 smp_cond_load_acquire(&p->on_cpu, !VAL);
4307
4308 cpu = select_task_rq(p, p->wake_cpu, &wake_flags);
4309 if (task_cpu(p) != cpu) {
4310 if (p->in_iowait) {
4311 delayacct_blkio_end(p);
4312 atomic_dec(&task_rq(p)->nr_iowait);
4313 }
4314
4315 wake_flags |= WF_MIGRATED;
4316 psi_ttwu_dequeue(p);
4317 set_task_cpu(p, cpu);
4318 }
4319 #else
4320 cpu = task_cpu(p);
4321 #endif /* CONFIG_SMP */
4322
4323 ttwu_queue(p, cpu, wake_flags);
4324 }
4325 out:
4326 if (success)
4327 ttwu_stat(p, task_cpu(p), wake_flags);
4328
4329 return success;
4330 }
4331
4332 static bool __task_needs_rq_lock(struct task_struct *p)
4333 {
4334 unsigned int state = READ_ONCE(p->__state);
4335
4336 /*
4337 * Since pi->lock blocks try_to_wake_up(), we don't need rq->lock when
4338 * the task is blocked. Make sure to check @state since ttwu() can drop
4339 * locks at the end, see ttwu_queue_wakelist().
4340 */
4341 if (state == TASK_RUNNING || state == TASK_WAKING)
4342 return true;
4343
4344 /*
4345 * Ensure we load p->on_rq after p->__state, otherwise it would be
4346 * possible to, falsely, observe p->on_rq == 0.
4347 *
4348 * See try_to_wake_up() for a longer comment.
4349 */
4350 smp_rmb();
4351 if (p->on_rq)
4352 return true;
4353
4354 #ifdef CONFIG_SMP
4355 /*
4356 * Ensure the task has finished __schedule() and will not be referenced
4357 * anymore. Again, see try_to_wake_up() for a longer comment.
4358 */
4359 smp_rmb();
4360 smp_cond_load_acquire(&p->on_cpu, !VAL);
4361 #endif
4362
4363 return false;
4364 }
4365
4366 /**
4367 * task_call_func - Invoke a function on task in fixed state
4368 * @p: Process for which the function is to be invoked, can be @current.
4369 * @func: Function to invoke.
4370 * @arg: Argument to function.
4371 *
4372 * Fix the task in it's current state by avoiding wakeups and or rq operations
4373 * and call @func(@arg) on it. This function can use task_is_runnable() and
4374 * task_curr() to work out what the state is, if required. Given that @func
4375 * can be invoked with a runqueue lock held, it had better be quite
4376 * lightweight.
4377 *
4378 * Returns:
4379 * Whatever @func returns
4380 */
4381 int task_call_func(struct task_struct *p, task_call_f func, void *arg)
4382 {
4383 struct rq *rq = NULL;
4384 struct rq_flags rf;
4385 int ret;
4386
4387 raw_spin_lock_irqsave(&p->pi_lock, rf.flags);
4388
4389 if (__task_needs_rq_lock(p))
4390 rq = __task_rq_lock(p, &rf);
4391
4392 /*
4393 * At this point the task is pinned; either:
4394 * - blocked and we're holding off wakeups (pi->lock)
4395 * - woken, and we're holding off enqueue (rq->lock)
4396 * - queued, and we're holding off schedule (rq->lock)
4397 * - running, and we're holding off de-schedule (rq->lock)
4398 *
4399 * The called function (@func) can use: task_curr(), p->on_rq and
4400 * p->__state to differentiate between these states.
4401 */
4402 ret = func(p, arg);
4403
4404 if (rq)
4405 rq_unlock(rq, &rf);
4406
4407 raw_spin_unlock_irqrestore(&p->pi_lock, rf.flags);
4408 return ret;
4409 }
4410
4411 /**
4412 * cpu_curr_snapshot - Return a snapshot of the currently running task
4413 * @cpu: The CPU on which to snapshot the task.
4414 *
4415 * Returns the task_struct pointer of the task "currently" running on
4416 * the specified CPU.
4417 *
4418 * If the specified CPU was offline, the return value is whatever it
4419 * is, perhaps a pointer to the task_struct structure of that CPU's idle
4420 * task, but there is no guarantee. Callers wishing a useful return
4421 * value must take some action to ensure that the specified CPU remains
4422 * online throughout.
4423 *
4424 * This function executes full memory barriers before and after fetching
4425 * the pointer, which permits the caller to confine this function's fetch
4426 * with respect to the caller's accesses to other shared variables.
4427 */
4428 struct task_struct *cpu_curr_snapshot(int cpu)
4429 {
4430 struct rq *rq = cpu_rq(cpu);
4431 struct task_struct *t;
4432 struct rq_flags rf;
4433
4434 rq_lock_irqsave(rq, &rf);
4435 smp_mb__after_spinlock(); /* Pairing determined by caller's synchronization design. */
4436 t = rcu_dereference(cpu_curr(cpu));
4437 rq_unlock_irqrestore(rq, &rf);
4438 smp_mb(); /* Pairing determined by caller's synchronization design. */
4439
4440 return t;
4441 }
4442
4443 /**
4444 * wake_up_process - Wake up a specific process
4445 * @p: The process to be woken up.
4446 *
4447 * Attempt to wake up the nominated process and move it to the set of runnable
4448 * processes.
4449 *
4450 * Return: 1 if the process was woken up, 0 if it was already running.
4451 *
4452 * This function executes a full memory barrier before accessing the task state.
4453 */
4454 int wake_up_process(struct task_struct *p)
4455 {
4456 return try_to_wake_up(p, TASK_NORMAL, 0);
4457 }
4458 EXPORT_SYMBOL(wake_up_process);
4459
4460 int wake_up_state(struct task_struct *p, unsigned int state)
4461 {
4462 return try_to_wake_up(p, state, 0);
4463 }
4464
4465 /*
4466 * Perform scheduler related setup for a newly forked process p.
4467 * p is forked by current.
4468 *
4469 * __sched_fork() is basic setup which is also used by sched_init() to
4470 * initialize the boot CPU's idle task.
4471 */
4472 static void __sched_fork(unsigned long clone_flags, struct task_struct *p)
4473 {
4474 p->on_rq = 0;
4475
4476 p->se.on_rq = 0;
4477 p->se.exec_start = 0;
4478 p->se.sum_exec_runtime = 0;
4479 p->se.prev_sum_exec_runtime = 0;
4480 p->se.nr_migrations = 0;
4481 p->se.vruntime = 0;
4482 p->se.vlag = 0;
4483 INIT_LIST_HEAD(&p->se.group_node);
4484
4485 /* A delayed task cannot be in clone(). */
4486 SCHED_WARN_ON(p->se.sched_delayed);
4487
4488 #ifdef CONFIG_FAIR_GROUP_SCHED
4489 p->se.cfs_rq = NULL;
4490 #endif
4491
4492 #ifdef CONFIG_SCHEDSTATS
4493 /* Even if schedstat is disabled, there should not be garbage */
4494 memset(&p->stats, 0, sizeof(p->stats));
4495 #endif
4496
4497 init_dl_entity(&p->dl);
4498
4499 INIT_LIST_HEAD(&p->rt.run_list);
4500 p->rt.timeout = 0;
4501 p->rt.time_slice = sched_rr_timeslice;
4502 p->rt.on_rq = 0;
4503 p->rt.on_list = 0;
4504
4505 #ifdef CONFIG_SCHED_CLASS_EXT
4506 init_scx_entity(&p->scx);
4507 #endif
4508
4509 #ifdef CONFIG_PREEMPT_NOTIFIERS
4510 INIT_HLIST_HEAD(&p->preempt_notifiers);
4511 #endif
4512
4513 #ifdef CONFIG_COMPACTION
4514 p->capture_control = NULL;
4515 #endif
4516 init_numa_balancing(clone_flags, p);
4517 #ifdef CONFIG_SMP
4518 p->wake_entry.u_flags = CSD_TYPE_TTWU;
4519 p->migration_pending = NULL;
4520 #endif
4521 init_sched_mm_cid(p);
4522 }
4523
4524 DEFINE_STATIC_KEY_FALSE(sched_numa_balancing);
4525
4526 #ifdef CONFIG_NUMA_BALANCING
4527
4528 int sysctl_numa_balancing_mode;
4529
4530 static void __set_numabalancing_state(bool enabled)
4531 {
4532 if (enabled)
4533 static_branch_enable(&sched_numa_balancing);
4534 else
4535 static_branch_disable(&sched_numa_balancing);
4536 }
4537
4538 void set_numabalancing_state(bool enabled)
4539 {
4540 if (enabled)
4541 sysctl_numa_balancing_mode = NUMA_BALANCING_NORMAL;
4542 else
4543 sysctl_numa_balancing_mode = NUMA_BALANCING_DISABLED;
4544 __set_numabalancing_state(enabled);
4545 }
4546
4547 #ifdef CONFIG_PROC_SYSCTL
4548 static void reset_memory_tiering(void)
4549 {
4550 struct pglist_data *pgdat;
4551
4552 for_each_online_pgdat(pgdat) {
4553 pgdat->nbp_threshold = 0;
4554 pgdat->nbp_th_nr_cand = node_page_state(pgdat, PGPROMOTE_CANDIDATE);
4555 pgdat->nbp_th_start = jiffies_to_msecs(jiffies);
4556 }
4557 }
4558
4559 static int sysctl_numa_balancing(const struct ctl_table *table, int write,
4560 void *buffer, size_t *lenp, loff_t *ppos)
4561 {
4562 struct ctl_table t;
4563 int err;
4564 int state = sysctl_numa_balancing_mode;
4565
4566 if (write && !capable(CAP_SYS_ADMIN))
4567 return -EPERM;
4568
4569 t = *table;
4570 t.data = &state;
4571 err = proc_dointvec_minmax(&t, write, buffer, lenp, ppos);
4572 if (err < 0)
4573 return err;
4574 if (write) {
4575 if (!(sysctl_numa_balancing_mode & NUMA_BALANCING_MEMORY_TIERING) &&
4576 (state & NUMA_BALANCING_MEMORY_TIERING))
4577 reset_memory_tiering();
4578 sysctl_numa_balancing_mode = state;
4579 __set_numabalancing_state(state);
4580 }
4581 return err;
4582 }
4583 #endif
4584 #endif
4585
4586 #ifdef CONFIG_SCHEDSTATS
4587
4588 DEFINE_STATIC_KEY_FALSE(sched_schedstats);
4589
4590 static void set_schedstats(bool enabled)
4591 {
4592 if (enabled)
4593 static_branch_enable(&sched_schedstats);
4594 else
4595 static_branch_disable(&sched_schedstats);
4596 }
4597
4598 void force_schedstat_enabled(void)
4599 {
4600 if (!schedstat_enabled()) {
4601 pr_info("kernel profiling enabled schedstats, disable via kernel.sched_schedstats.\n");
4602 static_branch_enable(&sched_schedstats);
4603 }
4604 }
4605
4606 static int __init setup_schedstats(char *str)
4607 {
4608 int ret = 0;
4609 if (!str)
4610 goto out;
4611
4612 if (!strcmp(str, "enable")) {
4613 set_schedstats(true);
4614 ret = 1;
4615 } else if (!strcmp(str, "disable")) {
4616 set_schedstats(false);
4617 ret = 1;
4618 }
4619 out:
4620 if (!ret)
4621 pr_warn("Unable to parse schedstats=\n");
4622
4623 return ret;
4624 }
4625 __setup("schedstats=", setup_schedstats);
4626
4627 #ifdef CONFIG_PROC_SYSCTL
4628 static int sysctl_schedstats(const struct ctl_table *table, int write, void *buffer,
4629 size_t *lenp, loff_t *ppos)
4630 {
4631 struct ctl_table t;
4632 int err;
4633 int state = static_branch_likely(&sched_schedstats);
4634
4635 if (write && !capable(CAP_SYS_ADMIN))
4636 return -EPERM;
4637
4638 t = *table;
4639 t.data = &state;
4640 err = proc_dointvec_minmax(&t, write, buffer, lenp, ppos);
4641 if (err < 0)
4642 return err;
4643 if (write)
4644 set_schedstats(state);
4645 return err;
4646 }
4647 #endif /* CONFIG_PROC_SYSCTL */
4648 #endif /* CONFIG_SCHEDSTATS */
4649
4650 #ifdef CONFIG_SYSCTL
4651 static struct ctl_table sched_core_sysctls[] = {
4652 #ifdef CONFIG_SCHEDSTATS
4653 {
4654 .procname = "sched_schedstats",
4655 .data = NULL,
4656 .maxlen = sizeof(unsigned int),
4657 .mode = 0644,
4658 .proc_handler = sysctl_schedstats,
4659 .extra1 = SYSCTL_ZERO,
4660 .extra2 = SYSCTL_ONE,
4661 },
4662 #endif /* CONFIG_SCHEDSTATS */
4663 #ifdef CONFIG_UCLAMP_TASK
4664 {
4665 .procname = "sched_util_clamp_min",
4666 .data = &sysctl_sched_uclamp_util_min,
4667 .maxlen = sizeof(unsigned int),
4668 .mode = 0644,
4669 .proc_handler = sysctl_sched_uclamp_handler,
4670 },
4671 {
4672 .procname = "sched_util_clamp_max",
4673 .data = &sysctl_sched_uclamp_util_max,
4674 .maxlen = sizeof(unsigned int),
4675 .mode = 0644,
4676 .proc_handler = sysctl_sched_uclamp_handler,
4677 },
4678 {
4679 .procname = "sched_util_clamp_min_rt_default",
4680 .data = &sysctl_sched_uclamp_util_min_rt_default,
4681 .maxlen = sizeof(unsigned int),
4682 .mode = 0644,
4683 .proc_handler = sysctl_sched_uclamp_handler,
4684 },
4685 #endif /* CONFIG_UCLAMP_TASK */
4686 #ifdef CONFIG_NUMA_BALANCING
4687 {
4688 .procname = "numa_balancing",
4689 .data = NULL, /* filled in by handler */
4690 .maxlen = sizeof(unsigned int),
4691 .mode = 0644,
4692 .proc_handler = sysctl_numa_balancing,
4693 .extra1 = SYSCTL_ZERO,
4694 .extra2 = SYSCTL_FOUR,
4695 },
4696 #endif /* CONFIG_NUMA_BALANCING */
4697 };
4698 static int __init sched_core_sysctl_init(void)
4699 {
4700 register_sysctl_init("kernel", sched_core_sysctls);
4701 return 0;
4702 }
4703 late_initcall(sched_core_sysctl_init);
4704 #endif /* CONFIG_SYSCTL */
4705
4706 /*
4707 * fork()/clone()-time setup:
4708 */
4709 int sched_fork(unsigned long clone_flags, struct task_struct *p)
4710 {
4711 __sched_fork(clone_flags, p);
4712 /*
4713 * We mark the process as NEW here. This guarantees that
4714 * nobody will actually run it, and a signal or other external
4715 * event cannot wake it up and insert it on the runqueue either.
4716 */
4717 p->__state = TASK_NEW;
4718
4719 /*
4720 * Make sure we do not leak PI boosting priority to the child.
4721 */
4722 p->prio = current->normal_prio;
4723
4724 uclamp_fork(p);
4725
4726 /*
4727 * Revert to default priority/policy on fork if requested.
4728 */
4729 if (unlikely(p->sched_reset_on_fork)) {
4730 if (task_has_dl_policy(p) || task_has_rt_policy(p)) {
4731 p->policy = SCHED_NORMAL;
4732 p->static_prio = NICE_TO_PRIO(0);
4733 p->rt_priority = 0;
4734 } else if (PRIO_TO_NICE(p->static_prio) < 0)
4735 p->static_prio = NICE_TO_PRIO(0);
4736
4737 p->prio = p->normal_prio = p->static_prio;
4738 set_load_weight(p, false);
4739 p->se.custom_slice = 0;
4740 p->se.slice = sysctl_sched_base_slice;
4741
4742 /*
4743 * We don't need the reset flag anymore after the fork. It has
4744 * fulfilled its duty:
4745 */
4746 p->sched_reset_on_fork = 0;
4747 }
4748
4749 if (dl_prio(p->prio))
4750 return -EAGAIN;
4751
4752 scx_pre_fork(p);
4753
4754 if (rt_prio(p->prio)) {
4755 p->sched_class = &rt_sched_class;
4756 #ifdef CONFIG_SCHED_CLASS_EXT
4757 } else if (task_should_scx(p->policy)) {
4758 p->sched_class = &ext_sched_class;
4759 #endif
4760 } else {
4761 p->sched_class = &fair_sched_class;
4762 }
4763
4764 init_entity_runnable_average(&p->se);
4765
4766
4767 #ifdef CONFIG_SCHED_INFO
4768 if (likely(sched_info_on()))
4769 memset(&p->sched_info, 0, sizeof(p->sched_info));
4770 #endif
4771 #if defined(CONFIG_SMP)
4772 p->on_cpu = 0;
4773 #endif
4774 init_task_preempt_count(p);
4775 #ifdef CONFIG_SMP
4776 plist_node_init(&p->pushable_tasks, MAX_PRIO);
4777 RB_CLEAR_NODE(&p->pushable_dl_tasks);
4778 #endif
4779 return 0;
4780 }
4781
4782 int sched_cgroup_fork(struct task_struct *p, struct kernel_clone_args *kargs)
4783 {
4784 unsigned long flags;
4785
4786 /*
4787 * Because we're not yet on the pid-hash, p->pi_lock isn't strictly
4788 * required yet, but lockdep gets upset if rules are violated.
4789 */
4790 raw_spin_lock_irqsave(&p->pi_lock, flags);
4791 #ifdef CONFIG_CGROUP_SCHED
4792 if (1) {
4793 struct task_group *tg;
4794 tg = container_of(kargs->cset->subsys[cpu_cgrp_id],
4795 struct task_group, css);
4796 tg = autogroup_task_group(p, tg);
4797 p->sched_task_group = tg;
4798 }
4799 #endif
4800 rseq_migrate(p);
4801 /*
4802 * We're setting the CPU for the first time, we don't migrate,
4803 * so use __set_task_cpu().
4804 */
4805 __set_task_cpu(p, smp_processor_id());
4806 if (p->sched_class->task_fork)
4807 p->sched_class->task_fork(p);
4808 raw_spin_unlock_irqrestore(&p->pi_lock, flags);
4809
4810 return scx_fork(p);
4811 }
4812
4813 void sched_cancel_fork(struct task_struct *p)
4814 {
4815 scx_cancel_fork(p);
4816 }
4817
4818 void sched_post_fork(struct task_struct *p)
4819 {
4820 uclamp_post_fork(p);
4821 scx_post_fork(p);
4822 }
4823
4824 unsigned long to_ratio(u64 period, u64 runtime)
4825 {
4826 if (runtime == RUNTIME_INF)
4827 return BW_UNIT;
4828
4829 /*
4830 * Doing this here saves a lot of checks in all
4831 * the calling paths, and returning zero seems
4832 * safe for them anyway.
4833 */
4834 if (period == 0)
4835 return 0;
4836
4837 return div64_u64(runtime << BW_SHIFT, period);
4838 }
4839
4840 /*
4841 * wake_up_new_task - wake up a newly created task for the first time.
4842 *
4843 * This function will do some initial scheduler statistics housekeeping
4844 * that must be done for every newly created context, then puts the task
4845 * on the runqueue and wakes it.
4846 */
4847 void wake_up_new_task(struct task_struct *p)
4848 {
4849 struct rq_flags rf;
4850 struct rq *rq;
4851 int wake_flags = WF_FORK;
4852
4853 raw_spin_lock_irqsave(&p->pi_lock, rf.flags);
4854 WRITE_ONCE(p->__state, TASK_RUNNING);
4855 #ifdef CONFIG_SMP
4856 /*
4857 * Fork balancing, do it here and not earlier because:
4858 * - cpus_ptr can change in the fork path
4859 * - any previously selected CPU might disappear through hotplug
4860 *
4861 * Use __set_task_cpu() to avoid calling sched_class::migrate_task_rq,
4862 * as we're not fully set-up yet.
4863 */
4864 p->recent_used_cpu = task_cpu(p);
4865 rseq_migrate(p);
4866 __set_task_cpu(p, select_task_rq(p, task_cpu(p), &wake_flags));
4867 #endif
4868 rq = __task_rq_lock(p, &rf);
4869 update_rq_clock(rq);
4870 post_init_entity_util_avg(p);
4871
4872 activate_task(rq, p, ENQUEUE_NOCLOCK | ENQUEUE_INITIAL);
4873 trace_sched_wakeup_new(p);
4874 wakeup_preempt(rq, p, wake_flags);
4875 #ifdef CONFIG_SMP
4876 if (p->sched_class->task_woken) {
4877 /*
4878 * Nothing relies on rq->lock after this, so it's fine to
4879 * drop it.
4880 */
4881 rq_unpin_lock(rq, &rf);
4882 p->sched_class->task_woken(rq, p);
4883 rq_repin_lock(rq, &rf);
4884 }
4885 #endif
4886 task_rq_unlock(rq, p, &rf);
4887 }
4888
4889 #ifdef CONFIG_PREEMPT_NOTIFIERS
4890
4891 static DEFINE_STATIC_KEY_FALSE(preempt_notifier_key);
4892
4893 void preempt_notifier_inc(void)
4894 {
4895 static_branch_inc(&preempt_notifier_key);
4896 }
4897 EXPORT_SYMBOL_GPL(preempt_notifier_inc);
4898
4899 void preempt_notifier_dec(void)
4900 {
4901 static_branch_dec(&preempt_notifier_key);
4902 }
4903 EXPORT_SYMBOL_GPL(preempt_notifier_dec);
4904
4905 /**
4906 * preempt_notifier_register - tell me when current is being preempted & rescheduled
4907 * @notifier: notifier struct to register
4908 */
4909 void preempt_notifier_register(struct preempt_notifier *notifier)
4910 {
4911 if (!static_branch_unlikely(&preempt_notifier_key))
4912 WARN(1, "registering preempt_notifier while notifiers disabled\n");
4913
4914 hlist_add_head(¬ifier->link, ¤t->preempt_notifiers);
4915 }
4916 EXPORT_SYMBOL_GPL(preempt_notifier_register);
4917
4918 /**
4919 * preempt_notifier_unregister - no longer interested in preemption notifications
4920 * @notifier: notifier struct to unregister
4921 *
4922 * This is *not* safe to call from within a preemption notifier.
4923 */
4924 void preempt_notifier_unregister(struct preempt_notifier *notifier)
4925 {
4926 hlist_del(¬ifier->link);
4927 }
4928 EXPORT_SYMBOL_GPL(preempt_notifier_unregister);
4929
4930 static void __fire_sched_in_preempt_notifiers(struct task_struct *curr)
4931 {
4932 struct preempt_notifier *notifier;
4933
4934 hlist_for_each_entry(notifier, &curr->preempt_notifiers, link)
4935 notifier->ops->sched_in(notifier, raw_smp_processor_id());
4936 }
4937
4938 static __always_inline void fire_sched_in_preempt_notifiers(struct task_struct *curr)
4939 {
4940 if (static_branch_unlikely(&preempt_notifier_key))
4941 __fire_sched_in_preempt_notifiers(curr);
4942 }
4943
4944 static void
4945 __fire_sched_out_preempt_notifiers(struct task_struct *curr,
4946 struct task_struct *next)
4947 {
4948 struct preempt_notifier *notifier;
4949
4950 hlist_for_each_entry(notifier, &curr->preempt_notifiers, link)
4951 notifier->ops->sched_out(notifier, next);
4952 }
4953
4954 static __always_inline void
4955 fire_sched_out_preempt_notifiers(struct task_struct *curr,
4956 struct task_struct *next)
4957 {
4958 if (static_branch_unlikely(&preempt_notifier_key))
4959 __fire_sched_out_preempt_notifiers(curr, next);
4960 }
4961
4962 #else /* !CONFIG_PREEMPT_NOTIFIERS */
4963
4964 static inline void fire_sched_in_preempt_notifiers(struct task_struct *curr)
4965 {
4966 }
4967
4968 static inline void
4969 fire_sched_out_preempt_notifiers(struct task_struct *curr,
4970 struct task_struct *next)
4971 {
4972 }
4973
4974 #endif /* CONFIG_PREEMPT_NOTIFIERS */
4975
4976 static inline void prepare_task(struct task_struct *next)
4977 {
4978 #ifdef CONFIG_SMP
4979 /*
4980 * Claim the task as running, we do this before switching to it
4981 * such that any running task will have this set.
4982 *
4983 * See the smp_load_acquire(&p->on_cpu) case in ttwu() and
4984 * its ordering comment.
4985 */
4986 WRITE_ONCE(next->on_cpu, 1);
4987 #endif
4988 }
4989
4990 static inline void finish_task(struct task_struct *prev)
4991 {
4992 #ifdef CONFIG_SMP
4993 /*
4994 * This must be the very last reference to @prev from this CPU. After
4995 * p->on_cpu is cleared, the task can be moved to a different CPU. We
4996 * must ensure this doesn't happen until the switch is completely
4997 * finished.
4998 *
4999 * In particular, the load of prev->state in finish_task_switch() must
5000 * happen before this.
5001 *
5002 * Pairs with the smp_cond_load_acquire() in try_to_wake_up().
5003 */
5004 smp_store_release(&prev->on_cpu, 0);
5005 #endif
5006 }
5007
5008 #ifdef CONFIG_SMP
5009
5010 static void do_balance_callbacks(struct rq *rq, struct balance_callback *head)
5011 {
5012 void (*func)(struct rq *rq);
5013 struct balance_callback *next;
5014
5015 lockdep_assert_rq_held(rq);
5016
5017 while (head) {
5018 func = (void (*)(struct rq *))head->func;
5019 next = head->next;
5020 head->next = NULL;
5021 head = next;
5022
5023 func(rq);
5024 }
5025 }
5026
5027 static void balance_push(struct rq *rq);
5028
5029 /*
5030 * balance_push_callback is a right abuse of the callback interface and plays
5031 * by significantly different rules.
5032 *
5033 * Where the normal balance_callback's purpose is to be ran in the same context
5034 * that queued it (only later, when it's safe to drop rq->lock again),
5035 * balance_push_callback is specifically targeted at __schedule().
5036 *
5037 * This abuse is tolerated because it places all the unlikely/odd cases behind
5038 * a single test, namely: rq->balance_callback == NULL.
5039 */
5040 struct balance_callback balance_push_callback = {
5041 .next = NULL,
5042 .func = balance_push,
5043 };
5044
5045 static inline struct balance_callback *
5046 __splice_balance_callbacks(struct rq *rq, bool split)
5047 {
5048 struct balance_callback *head = rq->balance_callback;
5049
5050 if (likely(!head))
5051 return NULL;
5052
5053 lockdep_assert_rq_held(rq);
5054 /*
5055 * Must not take balance_push_callback off the list when
5056 * splice_balance_callbacks() and balance_callbacks() are not
5057 * in the same rq->lock section.
5058 *
5059 * In that case it would be possible for __schedule() to interleave
5060 * and observe the list empty.
5061 */
5062 if (split && head == &balance_push_callback)
5063 head = NULL;
5064 else
5065 rq->balance_callback = NULL;
5066
5067 return head;
5068 }
5069
5070 struct balance_callback *splice_balance_callbacks(struct rq *rq)
5071 {
5072 return __splice_balance_callbacks(rq, true);
5073 }
5074
5075 static void __balance_callbacks(struct rq *rq)
5076 {
5077 do_balance_callbacks(rq, __splice_balance_callbacks(rq, false));
5078 }
5079
5080 void balance_callbacks(struct rq *rq, struct balance_callback *head)
5081 {
5082 unsigned long flags;
5083
5084 if (unlikely(head)) {
5085 raw_spin_rq_lock_irqsave(rq, flags);
5086 do_balance_callbacks(rq, head);
5087 raw_spin_rq_unlock_irqrestore(rq, flags);
5088 }
5089 }
5090
5091 #else
5092
5093 static inline void __balance_callbacks(struct rq *rq)
5094 {
5095 }
5096
5097 #endif
5098
5099 static inline void
5100 prepare_lock_switch(struct rq *rq, struct task_struct *next, struct rq_flags *rf)
5101 {
5102 /*
5103 * Since the runqueue lock will be released by the next
5104 * task (which is an invalid locking op but in the case
5105 * of the scheduler it's an obvious special-case), so we
5106 * do an early lockdep release here:
5107 */
5108 rq_unpin_lock(rq, rf);
5109 spin_release(&__rq_lockp(rq)->dep_map, _THIS_IP_);
5110 #ifdef CONFIG_DEBUG_SPINLOCK
5111 /* this is a valid case when another task releases the spinlock */
5112 rq_lockp(rq)->owner = next;
5113 #endif
5114 }
5115
5116 static inline void finish_lock_switch(struct rq *rq)
5117 {
5118 /*
5119 * If we are tracking spinlock dependencies then we have to
5120 * fix up the runqueue lock - which gets 'carried over' from
5121 * prev into current:
5122 */
5123 spin_acquire(&__rq_lockp(rq)->dep_map, 0, 0, _THIS_IP_);
5124 __balance_callbacks(rq);
5125 raw_spin_rq_unlock_irq(rq);
5126 }
5127
5128 /*
5129 * NOP if the arch has not defined these:
5130 */
5131
5132 #ifndef prepare_arch_switch
5133 # define prepare_arch_switch(next) do { } while (0)
5134 #endif
5135
5136 #ifndef finish_arch_post_lock_switch
5137 # define finish_arch_post_lock_switch() do { } while (0)
5138 #endif
5139
5140 static inline void kmap_local_sched_out(void)
5141 {
5142 #ifdef CONFIG_KMAP_LOCAL
5143 if (unlikely(current->kmap_ctrl.idx))
5144 __kmap_local_sched_out();
5145 #endif
5146 }
5147
5148 static inline void kmap_local_sched_in(void)
5149 {
5150 #ifdef CONFIG_KMAP_LOCAL
5151 if (unlikely(current->kmap_ctrl.idx))
5152 __kmap_local_sched_in();
5153 #endif
5154 }
5155
5156 /**
5157 * prepare_task_switch - prepare to switch tasks
5158 * @rq: the runqueue preparing to switch
5159 * @prev: the current task that is being switched out
5160 * @next: the task we are going to switch to.
5161 *
5162 * This is called with the rq lock held and interrupts off. It must
5163 * be paired with a subsequent finish_task_switch after the context
5164 * switch.
5165 *
5166 * prepare_task_switch sets up locking and calls architecture specific
5167 * hooks.
5168 */
5169 static inline void
5170 prepare_task_switch(struct rq *rq, struct task_struct *prev,
5171 struct task_struct *next)
5172 {
5173 kcov_prepare_switch(prev);
5174 sched_info_switch(rq, prev, next);
5175 perf_event_task_sched_out(prev, next);
5176 rseq_preempt(prev);
5177 fire_sched_out_preempt_notifiers(prev, next);
5178 kmap_local_sched_out();
5179 prepare_task(next);
5180 prepare_arch_switch(next);
5181 }
5182
5183 /**
5184 * finish_task_switch - clean up after a task-switch
5185 * @prev: the thread we just switched away from.
5186 *
5187 * finish_task_switch must be called after the context switch, paired
5188 * with a prepare_task_switch call before the context switch.
5189 * finish_task_switch will reconcile locking set up by prepare_task_switch,
5190 * and do any other architecture-specific cleanup actions.
5191 *
5192 * Note that we may have delayed dropping an mm in context_switch(). If
5193 * so, we finish that here outside of the runqueue lock. (Doing it
5194 * with the lock held can cause deadlocks; see schedule() for
5195 * details.)
5196 *
5197 * The context switch have flipped the stack from under us and restored the
5198 * local variables which were saved when this task called schedule() in the
5199 * past. 'prev == current' is still correct but we need to recalculate this_rq
5200 * because prev may have moved to another CPU.
5201 */
5202 static struct rq *finish_task_switch(struct task_struct *prev)
5203 __releases(rq->lock)
5204 {
5205 struct rq *rq = this_rq();
5206 struct mm_struct *mm = rq->prev_mm;
5207 unsigned int prev_state;
5208
5209 /*
5210 * The previous task will have left us with a preempt_count of 2
5211 * because it left us after:
5212 *
5213 * schedule()
5214 * preempt_disable(); // 1
5215 * __schedule()
5216 * raw_spin_lock_irq(&rq->lock) // 2
5217 *
5218 * Also, see FORK_PREEMPT_COUNT.
5219 */
5220 if (WARN_ONCE(preempt_count() != 2*PREEMPT_DISABLE_OFFSET,
5221 "corrupted preempt_count: %s/%d/0x%x\n",
5222 current->comm, current->pid, preempt_count()))
5223 preempt_count_set(FORK_PREEMPT_COUNT);
5224
5225 rq->prev_mm = NULL;
5226
5227 /*
5228 * A task struct has one reference for the use as "current".
5229 * If a task dies, then it sets TASK_DEAD in tsk->state and calls
5230 * schedule one last time. The schedule call will never return, and
5231 * the scheduled task must drop that reference.
5232 *
5233 * We must observe prev->state before clearing prev->on_cpu (in
5234 * finish_task), otherwise a concurrent wakeup can get prev
5235 * running on another CPU and we could rave with its RUNNING -> DEAD
5236 * transition, resulting in a double drop.
5237 */
5238 prev_state = READ_ONCE(prev->__state);
5239 vtime_task_switch(prev);
5240 perf_event_task_sched_in(prev, current);
5241 finish_task(prev);
5242 tick_nohz_task_switch();
5243 finish_lock_switch(rq);
5244 finish_arch_post_lock_switch();
5245 kcov_finish_switch(current);
5246 /*
5247 * kmap_local_sched_out() is invoked with rq::lock held and
5248 * interrupts disabled. There is no requirement for that, but the
5249 * sched out code does not have an interrupt enabled section.
5250 * Restoring the maps on sched in does not require interrupts being
5251 * disabled either.
5252 */
5253 kmap_local_sched_in();
5254
5255 fire_sched_in_preempt_notifiers(current);
5256 /*
5257 * When switching through a kernel thread, the loop in
5258 * membarrier_{private,global}_expedited() may have observed that
5259 * kernel thread and not issued an IPI. It is therefore possible to
5260 * schedule between user->kernel->user threads without passing though
5261 * switch_mm(). Membarrier requires a barrier after storing to
5262 * rq->curr, before returning to userspace, so provide them here:
5263 *
5264 * - a full memory barrier for {PRIVATE,GLOBAL}_EXPEDITED, implicitly
5265 * provided by mmdrop_lazy_tlb(),
5266 * - a sync_core for SYNC_CORE.
5267 */
5268 if (mm) {
5269 membarrier_mm_sync_core_before_usermode(mm);
5270 mmdrop_lazy_tlb_sched(mm);
5271 }
5272
5273 if (unlikely(prev_state == TASK_DEAD)) {
5274 if (prev->sched_class->task_dead)
5275 prev->sched_class->task_dead(prev);
5276
5277 /* Task is done with its stack. */
5278 put_task_stack(prev);
5279
5280 put_task_struct_rcu_user(prev);
5281 }
5282
5283 return rq;
5284 }
5285
5286 /**
5287 * schedule_tail - first thing a freshly forked thread must call.
5288 * @prev: the thread we just switched away from.
5289 */
5290 asmlinkage __visible void schedule_tail(struct task_struct *prev)
5291 __releases(rq->lock)
5292 {
5293 /*
5294 * New tasks start with FORK_PREEMPT_COUNT, see there and
5295 * finish_task_switch() for details.
5296 *
5297 * finish_task_switch() will drop rq->lock() and lower preempt_count
5298 * and the preempt_enable() will end up enabling preemption (on
5299 * PREEMPT_COUNT kernels).
5300 */
5301
5302 finish_task_switch(prev);
5303 preempt_enable();
5304
5305 if (current->set_child_tid)
5306 put_user(task_pid_vnr(current), current->set_child_tid);
5307
5308 calculate_sigpending();
5309 }
5310
5311 /*
5312 * context_switch - switch to the new MM and the new thread's register state.
5313 */
5314 static __always_inline struct rq *
5315 context_switch(struct rq *rq, struct task_struct *prev,
5316 struct task_struct *next, struct rq_flags *rf)
5317 {
5318 prepare_task_switch(rq, prev, next);
5319
5320 /*
5321 * For paravirt, this is coupled with an exit in switch_to to
5322 * combine the page table reload and the switch backend into
5323 * one hypercall.
5324 */
5325 arch_start_context_switch(prev);
5326
5327 /*
5328 * kernel -> kernel lazy + transfer active
5329 * user -> kernel lazy + mmgrab_lazy_tlb() active
5330 *
5331 * kernel -> user switch + mmdrop_lazy_tlb() active
5332 * user -> user switch
5333 *
5334 * switch_mm_cid() needs to be updated if the barriers provided
5335 * by context_switch() are modified.
5336 */
5337 if (!next->mm) { // to kernel
5338 enter_lazy_tlb(prev->active_mm, next);
5339
5340 next->active_mm = prev->active_mm;
5341 if (prev->mm) // from user
5342 mmgrab_lazy_tlb(prev->active_mm);
5343 else
5344 prev->active_mm = NULL;
5345 } else { // to user
5346 membarrier_switch_mm(rq, prev->active_mm, next->mm);
5347 /*
5348 * sys_membarrier() requires an smp_mb() between setting
5349 * rq->curr / membarrier_switch_mm() and returning to userspace.
5350 *
5351 * The below provides this either through switch_mm(), or in
5352 * case 'prev->active_mm == next->mm' through
5353 * finish_task_switch()'s mmdrop().
5354 */
5355 switch_mm_irqs_off(prev->active_mm, next->mm, next);
5356 lru_gen_use_mm(next->mm);
5357
5358 if (!prev->mm) { // from kernel
5359 /* will mmdrop_lazy_tlb() in finish_task_switch(). */
5360 rq->prev_mm = prev->active_mm;
5361 prev->active_mm = NULL;
5362 }
5363 }
5364
5365 /* switch_mm_cid() requires the memory barriers above. */
5366 switch_mm_cid(rq, prev, next);
5367
5368 prepare_lock_switch(rq, next, rf);
5369
5370 /* Here we just switch the register state and the stack. */
5371 switch_to(prev, next, prev);
5372 barrier();
5373
5374 return finish_task_switch(prev);
5375 }
5376
5377 /*
5378 * nr_running and nr_context_switches:
5379 *
5380 * externally visible scheduler statistics: current number of runnable
5381 * threads, total number of context switches performed since bootup.
5382 */
5383 unsigned int nr_running(void)
5384 {
5385 unsigned int i, sum = 0;
5386
5387 for_each_online_cpu(i)
5388 sum += cpu_rq(i)->nr_running;
5389
5390 return sum;
5391 }
5392
5393 /*
5394 * Check if only the current task is running on the CPU.
5395 *
5396 * Caution: this function does not check that the caller has disabled
5397 * preemption, thus the result might have a time-of-check-to-time-of-use
5398 * race. The caller is responsible to use it correctly, for example:
5399 *
5400 * - from a non-preemptible section (of course)
5401 *
5402 * - from a thread that is bound to a single CPU
5403 *
5404 * - in a loop with very short iterations (e.g. a polling loop)
5405 */
5406 bool single_task_running(void)
5407 {
5408 return raw_rq()->nr_running == 1;
5409 }
5410 EXPORT_SYMBOL(single_task_running);
5411
5412 unsigned long long nr_context_switches_cpu(int cpu)
5413 {
5414 return cpu_rq(cpu)->nr_switches;
5415 }
5416
5417 unsigned long long nr_context_switches(void)
5418 {
5419 int i;
5420 unsigned long long sum = 0;
5421
5422 for_each_possible_cpu(i)
5423 sum += cpu_rq(i)->nr_switches;
5424
5425 return sum;
5426 }
5427
5428 /*
5429 * Consumers of these two interfaces, like for example the cpuidle menu
5430 * governor, are using nonsensical data. Preferring shallow idle state selection
5431 * for a CPU that has IO-wait which might not even end up running the task when
5432 * it does become runnable.
5433 */
5434
5435 unsigned int nr_iowait_cpu(int cpu)
5436 {
5437 return atomic_read(&cpu_rq(cpu)->nr_iowait);
5438 }
5439
5440 /*
5441 * IO-wait accounting, and how it's mostly bollocks (on SMP).
5442 *
5443 * The idea behind IO-wait account is to account the idle time that we could
5444 * have spend running if it were not for IO. That is, if we were to improve the
5445 * storage performance, we'd have a proportional reduction in IO-wait time.
5446 *
5447 * This all works nicely on UP, where, when a task blocks on IO, we account
5448 * idle time as IO-wait, because if the storage were faster, it could've been
5449 * running and we'd not be idle.
5450 *
5451 * This has been extended to SMP, by doing the same for each CPU. This however
5452 * is broken.
5453 *
5454 * Imagine for instance the case where two tasks block on one CPU, only the one
5455 * CPU will have IO-wait accounted, while the other has regular idle. Even
5456 * though, if the storage were faster, both could've ran at the same time,
5457 * utilising both CPUs.
5458 *
5459 * This means, that when looking globally, the current IO-wait accounting on
5460 * SMP is a lower bound, by reason of under accounting.
5461 *
5462 * Worse, since the numbers are provided per CPU, they are sometimes
5463 * interpreted per CPU, and that is nonsensical. A blocked task isn't strictly
5464 * associated with any one particular CPU, it can wake to another CPU than it
5465 * blocked on. This means the per CPU IO-wait number is meaningless.
5466 *
5467 * Task CPU affinities can make all that even more 'interesting'.
5468 */
5469
5470 unsigned int nr_iowait(void)
5471 {
5472 unsigned int i, sum = 0;
5473
5474 for_each_possible_cpu(i)
5475 sum += nr_iowait_cpu(i);
5476
5477 return sum;
5478 }
5479
5480 #ifdef CONFIG_SMP
5481
5482 /*
5483 * sched_exec - execve() is a valuable balancing opportunity, because at
5484 * this point the task has the smallest effective memory and cache footprint.
5485 */
5486 void sched_exec(void)
5487 {
5488 struct task_struct *p = current;
5489 struct migration_arg arg;
5490 int dest_cpu;
5491
5492 scoped_guard (raw_spinlock_irqsave, &p->pi_lock) {
5493 dest_cpu = p->sched_class->select_task_rq(p, task_cpu(p), WF_EXEC);
5494 if (dest_cpu == smp_processor_id())
5495 return;
5496
5497 if (unlikely(!cpu_active(dest_cpu)))
5498 return;
5499
5500 arg = (struct migration_arg){ p, dest_cpu };
5501 }
5502 stop_one_cpu(task_cpu(p), migration_cpu_stop, &arg);
5503 }
5504
5505 #endif
5506
5507 DEFINE_PER_CPU(struct kernel_stat, kstat);
5508 DEFINE_PER_CPU(struct kernel_cpustat, kernel_cpustat);
5509
5510 EXPORT_PER_CPU_SYMBOL(kstat);
5511 EXPORT_PER_CPU_SYMBOL(kernel_cpustat);
5512
5513 /*
5514 * The function fair_sched_class.update_curr accesses the struct curr
5515 * and its field curr->exec_start; when called from task_sched_runtime(),
5516 * we observe a high rate of cache misses in practice.
5517 * Prefetching this data results in improved performance.
5518 */
5519 static inline void prefetch_curr_exec_start(struct task_struct *p)
5520 {
5521 #ifdef CONFIG_FAIR_GROUP_SCHED
5522 struct sched_entity *curr = p->se.cfs_rq->curr;
5523 #else
5524 struct sched_entity *curr = task_rq(p)->cfs.curr;
5525 #endif
5526 prefetch(curr);
5527 prefetch(&curr->exec_start);
5528 }
5529
5530 /*
5531 * Return accounted runtime for the task.
5532 * In case the task is currently running, return the runtime plus current's
5533 * pending runtime that have not been accounted yet.
5534 */
5535 unsigned long long task_sched_runtime(struct task_struct *p)
5536 {
5537 struct rq_flags rf;
5538 struct rq *rq;
5539 u64 ns;
5540
5541 #if defined(CONFIG_64BIT) && defined(CONFIG_SMP)
5542 /*
5543 * 64-bit doesn't need locks to atomically read a 64-bit value.
5544 * So we have a optimization chance when the task's delta_exec is 0.
5545 * Reading ->on_cpu is racy, but this is OK.
5546 *
5547 * If we race with it leaving CPU, we'll take a lock. So we're correct.
5548 * If we race with it entering CPU, unaccounted time is 0. This is
5549 * indistinguishable from the read occurring a few cycles earlier.
5550 * If we see ->on_cpu without ->on_rq, the task is leaving, and has
5551 * been accounted, so we're correct here as well.
5552 */
5553 if (!p->on_cpu || !task_on_rq_queued(p))
5554 return p->se.sum_exec_runtime;
5555 #endif
5556
5557 rq = task_rq_lock(p, &rf);
5558 /*
5559 * Must be ->curr _and_ ->on_rq. If dequeued, we would
5560 * project cycles that may never be accounted to this
5561 * thread, breaking clock_gettime().
5562 */
5563 if (task_current_donor(rq, p) && task_on_rq_queued(p)) {
5564 prefetch_curr_exec_start(p);
5565 update_rq_clock(rq);
5566 p->sched_class->update_curr(rq);
5567 }
5568 ns = p->se.sum_exec_runtime;
5569 task_rq_unlock(rq, p, &rf);
5570
5571 return ns;
5572 }
5573
5574 #ifdef CONFIG_SCHED_DEBUG
5575 static u64 cpu_resched_latency(struct rq *rq)
5576 {
5577 int latency_warn_ms = READ_ONCE(sysctl_resched_latency_warn_ms);
5578 u64 resched_latency, now = rq_clock(rq);
5579 static bool warned_once;
5580
5581 if (sysctl_resched_latency_warn_once && warned_once)
5582 return 0;
5583
5584 if (!need_resched() || !latency_warn_ms)
5585 return 0;
5586
5587 if (system_state == SYSTEM_BOOTING)
5588 return 0;
5589
5590 if (!rq->last_seen_need_resched_ns) {
5591 rq->last_seen_need_resched_ns = now;
5592 rq->ticks_without_resched = 0;
5593 return 0;
5594 }
5595
5596 rq->ticks_without_resched++;
5597 resched_latency = now - rq->last_seen_need_resched_ns;
5598 if (resched_latency <= latency_warn_ms * NSEC_PER_MSEC)
5599 return 0;
5600
5601 warned_once = true;
5602
5603 return resched_latency;
5604 }
5605
5606 static int __init setup_resched_latency_warn_ms(char *str)
5607 {
5608 long val;
5609
5610 if ((kstrtol(str, 0, &val))) {
5611 pr_warn("Unable to set resched_latency_warn_ms\n");
5612 return 1;
5613 }
5614
5615 sysctl_resched_latency_warn_ms = val;
5616 return 1;
5617 }
5618 __setup("resched_latency_warn_ms=", setup_resched_latency_warn_ms);
5619 #else
5620 static inline u64 cpu_resched_latency(struct rq *rq) { return 0; }
5621 #endif /* CONFIG_SCHED_DEBUG */
5622
5623 /*
5624 * This function gets called by the timer code, with HZ frequency.
5625 * We call it with interrupts disabled.
5626 */
5627 void sched_tick(void)
5628 {
5629 int cpu = smp_processor_id();
5630 struct rq *rq = cpu_rq(cpu);
5631 /* accounting goes to the donor task */
5632 struct task_struct *donor;
5633 struct rq_flags rf;
5634 unsigned long hw_pressure;
5635 u64 resched_latency;
5636
5637 if (housekeeping_cpu(cpu, HK_TYPE_KERNEL_NOISE))
5638 arch_scale_freq_tick();
5639
5640 sched_clock_tick();
5641
5642 rq_lock(rq, &rf);
5643 donor = rq->donor;
5644
5645 psi_account_irqtime(rq, donor, NULL);
5646
5647 update_rq_clock(rq);
5648 hw_pressure = arch_scale_hw_pressure(cpu_of(rq));
5649 update_hw_load_avg(rq_clock_task(rq), rq, hw_pressure);
5650
5651 if (dynamic_preempt_lazy() && tif_test_bit(TIF_NEED_RESCHED_LAZY))
5652 resched_curr(rq);
5653
5654 donor->sched_class->task_tick(rq, donor, 0);
5655 if (sched_feat(LATENCY_WARN))
5656 resched_latency = cpu_resched_latency(rq);
5657 calc_global_load_tick(rq);
5658 sched_core_tick(rq);
5659 task_tick_mm_cid(rq, donor);
5660 scx_tick(rq);
5661
5662 rq_unlock(rq, &rf);
5663
5664 if (sched_feat(LATENCY_WARN) && resched_latency)
5665 resched_latency_warn(cpu, resched_latency);
5666
5667 perf_event_task_tick();
5668
5669 if (donor->flags & PF_WQ_WORKER)
5670 wq_worker_tick(donor);
5671
5672 #ifdef CONFIG_SMP
5673 if (!scx_switched_all()) {
5674 rq->idle_balance = idle_cpu(cpu);
5675 sched_balance_trigger(rq);
5676 }
5677 #endif
5678 }
5679
5680 #ifdef CONFIG_NO_HZ_FULL
5681
5682 struct tick_work {
5683 int cpu;
5684 atomic_t state;
5685 struct delayed_work work;
5686 };
5687 /* Values for ->state, see diagram below. */
5688 #define TICK_SCHED_REMOTE_OFFLINE 0
5689 #define TICK_SCHED_REMOTE_OFFLINING 1
5690 #define TICK_SCHED_REMOTE_RUNNING 2
5691
5692 /*
5693 * State diagram for ->state:
5694 *
5695 *
5696 * TICK_SCHED_REMOTE_OFFLINE
5697 * | ^
5698 * | |
5699 * | | sched_tick_remote()
5700 * | |
5701 * | |
5702 * +--TICK_SCHED_REMOTE_OFFLINING
5703 * | ^
5704 * | |
5705 * sched_tick_start() | | sched_tick_stop()
5706 * | |
5707 * V |
5708 * TICK_SCHED_REMOTE_RUNNING
5709 *
5710 *
5711 * Other transitions get WARN_ON_ONCE(), except that sched_tick_remote()
5712 * and sched_tick_start() are happy to leave the state in RUNNING.
5713 */
5714
5715 static struct tick_work __percpu *tick_work_cpu;
5716
5717 static void sched_tick_remote(struct work_struct *work)
5718 {
5719 struct delayed_work *dwork = to_delayed_work(work);
5720 struct tick_work *twork = container_of(dwork, struct tick_work, work);
5721 int cpu = twork->cpu;
5722 struct rq *rq = cpu_rq(cpu);
5723 int os;
5724
5725 /*
5726 * Handle the tick only if it appears the remote CPU is running in full
5727 * dynticks mode. The check is racy by nature, but missing a tick or
5728 * having one too much is no big deal because the scheduler tick updates
5729 * statistics and checks timeslices in a time-independent way, regardless
5730 * of when exactly it is running.
5731 */
5732 if (tick_nohz_tick_stopped_cpu(cpu)) {
5733 guard(rq_lock_irq)(rq);
5734 struct task_struct *curr = rq->curr;
5735
5736 if (cpu_online(cpu)) {
5737 /*
5738 * Since this is a remote tick for full dynticks mode,
5739 * we are always sure that there is no proxy (only a
5740 * single task is running).
5741 */
5742 SCHED_WARN_ON(rq->curr != rq->donor);
5743 update_rq_clock(rq);
5744
5745 if (!is_idle_task(curr)) {
5746 /*
5747 * Make sure the next tick runs within a
5748 * reasonable amount of time.
5749 */
5750 u64 delta = rq_clock_task(rq) - curr->se.exec_start;
5751 WARN_ON_ONCE(delta > (u64)NSEC_PER_SEC * 3);
5752 }
5753 curr->sched_class->task_tick(rq, curr, 0);
5754
5755 calc_load_nohz_remote(rq);
5756 }
5757 }
5758
5759 /*
5760 * Run the remote tick once per second (1Hz). This arbitrary
5761 * frequency is large enough to avoid overload but short enough
5762 * to keep scheduler internal stats reasonably up to date. But
5763 * first update state to reflect hotplug activity if required.
5764 */
5765 os = atomic_fetch_add_unless(&twork->state, -1, TICK_SCHED_REMOTE_RUNNING);
5766 WARN_ON_ONCE(os == TICK_SCHED_REMOTE_OFFLINE);
5767 if (os == TICK_SCHED_REMOTE_RUNNING)
5768 queue_delayed_work(system_unbound_wq, dwork, HZ);
5769 }
5770
5771 static void sched_tick_start(int cpu)
5772 {
5773 int os;
5774 struct tick_work *twork;
5775
5776 if (housekeeping_cpu(cpu, HK_TYPE_KERNEL_NOISE))
5777 return;
5778
5779 WARN_ON_ONCE(!tick_work_cpu);
5780
5781 twork = per_cpu_ptr(tick_work_cpu, cpu);
5782 os = atomic_xchg(&twork->state, TICK_SCHED_REMOTE_RUNNING);
5783 WARN_ON_ONCE(os == TICK_SCHED_REMOTE_RUNNING);
5784 if (os == TICK_SCHED_REMOTE_OFFLINE) {
5785 twork->cpu = cpu;
5786 INIT_DELAYED_WORK(&twork->work, sched_tick_remote);
5787 queue_delayed_work(system_unbound_wq, &twork->work, HZ);
5788 }
5789 }
5790
5791 #ifdef CONFIG_HOTPLUG_CPU
5792 static void sched_tick_stop(int cpu)
5793 {
5794 struct tick_work *twork;
5795 int os;
5796
5797 if (housekeeping_cpu(cpu, HK_TYPE_KERNEL_NOISE))
5798 return;
5799
5800 WARN_ON_ONCE(!tick_work_cpu);
5801
5802 twork = per_cpu_ptr(tick_work_cpu, cpu);
5803 /* There cannot be competing actions, but don't rely on stop-machine. */
5804 os = atomic_xchg(&twork->state, TICK_SCHED_REMOTE_OFFLINING);
5805 WARN_ON_ONCE(os != TICK_SCHED_REMOTE_RUNNING);
5806 /* Don't cancel, as this would mess up the state machine. */
5807 }
5808 #endif /* CONFIG_HOTPLUG_CPU */
5809
5810 int __init sched_tick_offload_init(void)
5811 {
5812 tick_work_cpu = alloc_percpu(struct tick_work);
5813 BUG_ON(!tick_work_cpu);
5814 return 0;
5815 }
5816
5817 #else /* !CONFIG_NO_HZ_FULL */
5818 static inline void sched_tick_start(int cpu) { }
5819 static inline void sched_tick_stop(int cpu) { }
5820 #endif
5821
5822 #if defined(CONFIG_PREEMPTION) && (defined(CONFIG_DEBUG_PREEMPT) || \
5823 defined(CONFIG_TRACE_PREEMPT_TOGGLE))
5824 /*
5825 * If the value passed in is equal to the current preempt count
5826 * then we just disabled preemption. Start timing the latency.
5827 */
5828 static inline void preempt_latency_start(int val)
5829 {
5830 if (preempt_count() == val) {
5831 unsigned long ip = get_lock_parent_ip();
5832 #ifdef CONFIG_DEBUG_PREEMPT
5833 current->preempt_disable_ip = ip;
5834 #endif
5835 trace_preempt_off(CALLER_ADDR0, ip);
5836 }
5837 }
5838
5839 void preempt_count_add(int val)
5840 {
5841 #ifdef CONFIG_DEBUG_PREEMPT
5842 /*
5843 * Underflow?
5844 */
5845 if (DEBUG_LOCKS_WARN_ON((preempt_count() < 0)))
5846 return;
5847 #endif
5848 __preempt_count_add(val);
5849 #ifdef CONFIG_DEBUG_PREEMPT
5850 /*
5851 * Spinlock count overflowing soon?
5852 */
5853 DEBUG_LOCKS_WARN_ON((preempt_count() & PREEMPT_MASK) >=
5854 PREEMPT_MASK - 10);
5855 #endif
5856 preempt_latency_start(val);
5857 }
5858 EXPORT_SYMBOL(preempt_count_add);
5859 NOKPROBE_SYMBOL(preempt_count_add);
5860
5861 /*
5862 * If the value passed in equals to the current preempt count
5863 * then we just enabled preemption. Stop timing the latency.
5864 */
5865 static inline void preempt_latency_stop(int val)
5866 {
5867 if (preempt_count() == val)
5868 trace_preempt_on(CALLER_ADDR0, get_lock_parent_ip());
5869 }
5870
5871 void preempt_count_sub(int val)
5872 {
5873 #ifdef CONFIG_DEBUG_PREEMPT
5874 /*
5875 * Underflow?
5876 */
5877 if (DEBUG_LOCKS_WARN_ON(val > preempt_count()))
5878 return;
5879 /*
5880 * Is the spinlock portion underflowing?
5881 */
5882 if (DEBUG_LOCKS_WARN_ON((val < PREEMPT_MASK) &&
5883 !(preempt_count() & PREEMPT_MASK)))
5884 return;
5885 #endif
5886
5887 preempt_latency_stop(val);
5888 __preempt_count_sub(val);
5889 }
5890 EXPORT_SYMBOL(preempt_count_sub);
5891 NOKPROBE_SYMBOL(preempt_count_sub);
5892
5893 #else
5894 static inline void preempt_latency_start(int val) { }
5895 static inline void preempt_latency_stop(int val) { }
5896 #endif
5897
5898 static inline unsigned long get_preempt_disable_ip(struct task_struct *p)
5899 {
5900 #ifdef CONFIG_DEBUG_PREEMPT
5901 return p->preempt_disable_ip;
5902 #else
5903 return 0;
5904 #endif
5905 }
5906
5907 /*
5908 * Print scheduling while atomic bug:
5909 */
5910 static noinline void __schedule_bug(struct task_struct *prev)
5911 {
5912 /* Save this before calling printk(), since that will clobber it */
5913 unsigned long preempt_disable_ip = get_preempt_disable_ip(current);
5914
5915 if (oops_in_progress)
5916 return;
5917
5918 printk(KERN_ERR "BUG: scheduling while atomic: %s/%d/0x%08x\n",
5919 prev->comm, prev->pid, preempt_count());
5920
5921 debug_show_held_locks(prev);
5922 print_modules();
5923 if (irqs_disabled())
5924 print_irqtrace_events(prev);
5925 if (IS_ENABLED(CONFIG_DEBUG_PREEMPT)) {
5926 pr_err("Preemption disabled at:");
5927 print_ip_sym(KERN_ERR, preempt_disable_ip);
5928 }
5929 check_panic_on_warn("scheduling while atomic");
5930
5931 dump_stack();
5932 add_taint(TAINT_WARN, LOCKDEP_STILL_OK);
5933 }
5934
5935 /*
5936 * Various schedule()-time debugging checks and statistics:
5937 */
5938 static inline void schedule_debug(struct task_struct *prev, bool preempt)
5939 {
5940 #ifdef CONFIG_SCHED_STACK_END_CHECK
5941 if (task_stack_end_corrupted(prev))
5942 panic("corrupted stack end detected inside scheduler\n");
5943
5944 if (task_scs_end_corrupted(prev))
5945 panic("corrupted shadow stack detected inside scheduler\n");
5946 #endif
5947
5948 #ifdef CONFIG_DEBUG_ATOMIC_SLEEP
5949 if (!preempt && READ_ONCE(prev->__state) && prev->non_block_count) {
5950 printk(KERN_ERR "BUG: scheduling in a non-blocking section: %s/%d/%i\n",
5951 prev->comm, prev->pid, prev->non_block_count);
5952 dump_stack();
5953 add_taint(TAINT_WARN, LOCKDEP_STILL_OK);
5954 }
5955 #endif
5956
5957 if (unlikely(in_atomic_preempt_off())) {
5958 __schedule_bug(prev);
5959 preempt_count_set(PREEMPT_DISABLED);
5960 }
5961 rcu_sleep_check();
5962 SCHED_WARN_ON(ct_state() == CT_STATE_USER);
5963
5964 profile_hit(SCHED_PROFILING, __builtin_return_address(0));
5965
5966 schedstat_inc(this_rq()->sched_count);
5967 }
5968
5969 static void prev_balance(struct rq *rq, struct task_struct *prev,
5970 struct rq_flags *rf)
5971 {
5972 const struct sched_class *start_class = prev->sched_class;
5973 const struct sched_class *class;
5974
5975 #ifdef CONFIG_SCHED_CLASS_EXT
5976 /*
5977 * SCX requires a balance() call before every pick_task() including when
5978 * waking up from SCHED_IDLE. If @start_class is below SCX, start from
5979 * SCX instead. Also, set a flag to detect missing balance() call.
5980 */
5981 if (scx_enabled()) {
5982 rq->scx.flags |= SCX_RQ_BAL_PENDING;
5983 if (sched_class_above(&ext_sched_class, start_class))
5984 start_class = &ext_sched_class;
5985 }
5986 #endif
5987
5988 /*
5989 * We must do the balancing pass before put_prev_task(), such
5990 * that when we release the rq->lock the task is in the same
5991 * state as before we took rq->lock.
5992 *
5993 * We can terminate the balance pass as soon as we know there is
5994 * a runnable task of @class priority or higher.
5995 */
5996 for_active_class_range(class, start_class, &idle_sched_class) {
5997 if (class->balance && class->balance(rq, prev, rf))
5998 break;
5999 }
6000 }
6001
6002 /*
6003 * Pick up the highest-prio task:
6004 */
6005 static inline struct task_struct *
6006 __pick_next_task(struct rq *rq, struct task_struct *prev, struct rq_flags *rf)
6007 {
6008 const struct sched_class *class;
6009 struct task_struct *p;
6010
6011 rq->dl_server = NULL;
6012
6013 if (scx_enabled())
6014 goto restart;
6015
6016 /*
6017 * Optimization: we know that if all tasks are in the fair class we can
6018 * call that function directly, but only if the @prev task wasn't of a
6019 * higher scheduling class, because otherwise those lose the
6020 * opportunity to pull in more work from other CPUs.
6021 */
6022 if (likely(!sched_class_above(prev->sched_class, &fair_sched_class) &&
6023 rq->nr_running == rq->cfs.h_nr_running)) {
6024
6025 p = pick_next_task_fair(rq, prev, rf);
6026 if (unlikely(p == RETRY_TASK))
6027 goto restart;
6028
6029 /* Assume the next prioritized class is idle_sched_class */
6030 if (!p) {
6031 p = pick_task_idle(rq);
6032 put_prev_set_next_task(rq, prev, p);
6033 }
6034
6035 return p;
6036 }
6037
6038 restart:
6039 prev_balance(rq, prev, rf);
6040
6041 for_each_active_class(class) {
6042 if (class->pick_next_task) {
6043 p = class->pick_next_task(rq, prev);
6044 if (p)
6045 return p;
6046 } else {
6047 p = class->pick_task(rq);
6048 if (p) {
6049 put_prev_set_next_task(rq, prev, p);
6050 return p;
6051 }
6052 }
6053 }
6054
6055 BUG(); /* The idle class should always have a runnable task. */
6056 }
6057
6058 #ifdef CONFIG_SCHED_CORE
6059 static inline bool is_task_rq_idle(struct task_struct *t)
6060 {
6061 return (task_rq(t)->idle == t);
6062 }
6063
6064 static inline bool cookie_equals(struct task_struct *a, unsigned long cookie)
6065 {
6066 return is_task_rq_idle(a) || (a->core_cookie == cookie);
6067 }
6068
6069 static inline bool cookie_match(struct task_struct *a, struct task_struct *b)
6070 {
6071 if (is_task_rq_idle(a) || is_task_rq_idle(b))
6072 return true;
6073
6074 return a->core_cookie == b->core_cookie;
6075 }
6076
6077 static inline struct task_struct *pick_task(struct rq *rq)
6078 {
6079 const struct sched_class *class;
6080 struct task_struct *p;
6081
6082 rq->dl_server = NULL;
6083
6084 for_each_active_class(class) {
6085 p = class->pick_task(rq);
6086 if (p)
6087 return p;
6088 }
6089
6090 BUG(); /* The idle class should always have a runnable task. */
6091 }
6092
6093 extern void task_vruntime_update(struct rq *rq, struct task_struct *p, bool in_fi);
6094
6095 static void queue_core_balance(struct rq *rq);
6096
6097 static struct task_struct *
6098 pick_next_task(struct rq *rq, struct task_struct *prev, struct rq_flags *rf)
6099 {
6100 struct task_struct *next, *p, *max = NULL;
6101 const struct cpumask *smt_mask;
6102 bool fi_before = false;
6103 bool core_clock_updated = (rq == rq->core);
6104 unsigned long cookie;
6105 int i, cpu, occ = 0;
6106 struct rq *rq_i;
6107 bool need_sync;
6108
6109 if (!sched_core_enabled(rq))
6110 return __pick_next_task(rq, prev, rf);
6111
6112 cpu = cpu_of(rq);
6113
6114 /* Stopper task is switching into idle, no need core-wide selection. */
6115 if (cpu_is_offline(cpu)) {
6116 /*
6117 * Reset core_pick so that we don't enter the fastpath when
6118 * coming online. core_pick would already be migrated to
6119 * another cpu during offline.
6120 */
6121 rq->core_pick = NULL;
6122 rq->core_dl_server = NULL;
6123 return __pick_next_task(rq, prev, rf);
6124 }
6125
6126 /*
6127 * If there were no {en,de}queues since we picked (IOW, the task
6128 * pointers are all still valid), and we haven't scheduled the last
6129 * pick yet, do so now.
6130 *
6131 * rq->core_pick can be NULL if no selection was made for a CPU because
6132 * it was either offline or went offline during a sibling's core-wide
6133 * selection. In this case, do a core-wide selection.
6134 */
6135 if (rq->core->core_pick_seq == rq->core->core_task_seq &&
6136 rq->core->core_pick_seq != rq->core_sched_seq &&
6137 rq->core_pick) {
6138 WRITE_ONCE(rq->core_sched_seq, rq->core->core_pick_seq);
6139
6140 next = rq->core_pick;
6141 rq->dl_server = rq->core_dl_server;
6142 rq->core_pick = NULL;
6143 rq->core_dl_server = NULL;
6144 goto out_set_next;
6145 }
6146
6147 prev_balance(rq, prev, rf);
6148
6149 smt_mask = cpu_smt_mask(cpu);
6150 need_sync = !!rq->core->core_cookie;
6151
6152 /* reset state */
6153 rq->core->core_cookie = 0UL;
6154 if (rq->core->core_forceidle_count) {
6155 if (!core_clock_updated) {
6156 update_rq_clock(rq->core);
6157 core_clock_updated = true;
6158 }
6159 sched_core_account_forceidle(rq);
6160 /* reset after accounting force idle */
6161 rq->core->core_forceidle_start = 0;
6162 rq->core->core_forceidle_count = 0;
6163 rq->core->core_forceidle_occupation = 0;
6164 need_sync = true;
6165 fi_before = true;
6166 }
6167
6168 /*
6169 * core->core_task_seq, core->core_pick_seq, rq->core_sched_seq
6170 *
6171 * @task_seq guards the task state ({en,de}queues)
6172 * @pick_seq is the @task_seq we did a selection on
6173 * @sched_seq is the @pick_seq we scheduled
6174 *
6175 * However, preemptions can cause multiple picks on the same task set.
6176 * 'Fix' this by also increasing @task_seq for every pick.
6177 */
6178 rq->core->core_task_seq++;
6179
6180 /*
6181 * Optimize for common case where this CPU has no cookies
6182 * and there are no cookied tasks running on siblings.
6183 */
6184 if (!need_sync) {
6185 next = pick_task(rq);
6186 if (!next->core_cookie) {
6187 rq->core_pick = NULL;
6188 rq->core_dl_server = NULL;
6189 /*
6190 * For robustness, update the min_vruntime_fi for
6191 * unconstrained picks as well.
6192 */
6193 WARN_ON_ONCE(fi_before);
6194 task_vruntime_update(rq, next, false);
6195 goto out_set_next;
6196 }
6197 }
6198
6199 /*
6200 * For each thread: do the regular task pick and find the max prio task
6201 * amongst them.
6202 *
6203 * Tie-break prio towards the current CPU
6204 */
6205 for_each_cpu_wrap(i, smt_mask, cpu) {
6206 rq_i = cpu_rq(i);
6207
6208 /*
6209 * Current cpu always has its clock updated on entrance to
6210 * pick_next_task(). If the current cpu is not the core,
6211 * the core may also have been updated above.
6212 */
6213 if (i != cpu && (rq_i != rq->core || !core_clock_updated))
6214 update_rq_clock(rq_i);
6215
6216 rq_i->core_pick = p = pick_task(rq_i);
6217 rq_i->core_dl_server = rq_i->dl_server;
6218
6219 if (!max || prio_less(max, p, fi_before))
6220 max = p;
6221 }
6222
6223 cookie = rq->core->core_cookie = max->core_cookie;
6224
6225 /*
6226 * For each thread: try and find a runnable task that matches @max or
6227 * force idle.
6228 */
6229 for_each_cpu(i, smt_mask) {
6230 rq_i = cpu_rq(i);
6231 p = rq_i->core_pick;
6232
6233 if (!cookie_equals(p, cookie)) {
6234 p = NULL;
6235 if (cookie)
6236 p = sched_core_find(rq_i, cookie);
6237 if (!p)
6238 p = idle_sched_class.pick_task(rq_i);
6239 }
6240
6241 rq_i->core_pick = p;
6242 rq_i->core_dl_server = NULL;
6243
6244 if (p == rq_i->idle) {
6245 if (rq_i->nr_running) {
6246 rq->core->core_forceidle_count++;
6247 if (!fi_before)
6248 rq->core->core_forceidle_seq++;
6249 }
6250 } else {
6251 occ++;
6252 }
6253 }
6254
6255 if (schedstat_enabled() && rq->core->core_forceidle_count) {
6256 rq->core->core_forceidle_start = rq_clock(rq->core);
6257 rq->core->core_forceidle_occupation = occ;
6258 }
6259
6260 rq->core->core_pick_seq = rq->core->core_task_seq;
6261 next = rq->core_pick;
6262 rq->core_sched_seq = rq->core->core_pick_seq;
6263
6264 /* Something should have been selected for current CPU */
6265 WARN_ON_ONCE(!next);
6266
6267 /*
6268 * Reschedule siblings
6269 *
6270 * NOTE: L1TF -- at this point we're no longer running the old task and
6271 * sending an IPI (below) ensures the sibling will no longer be running
6272 * their task. This ensures there is no inter-sibling overlap between
6273 * non-matching user state.
6274 */
6275 for_each_cpu(i, smt_mask) {
6276 rq_i = cpu_rq(i);
6277
6278 /*
6279 * An online sibling might have gone offline before a task
6280 * could be picked for it, or it might be offline but later
6281 * happen to come online, but its too late and nothing was
6282 * picked for it. That's Ok - it will pick tasks for itself,
6283 * so ignore it.
6284 */
6285 if (!rq_i->core_pick)
6286 continue;
6287
6288 /*
6289 * Update for new !FI->FI transitions, or if continuing to be in !FI:
6290 * fi_before fi update?
6291 * 0 0 1
6292 * 0 1 1
6293 * 1 0 1
6294 * 1 1 0
6295 */
6296 if (!(fi_before && rq->core->core_forceidle_count))
6297 task_vruntime_update(rq_i, rq_i->core_pick, !!rq->core->core_forceidle_count);
6298
6299 rq_i->core_pick->core_occupation = occ;
6300
6301 if (i == cpu) {
6302 rq_i->core_pick = NULL;
6303 rq_i->core_dl_server = NULL;
6304 continue;
6305 }
6306
6307 /* Did we break L1TF mitigation requirements? */
6308 WARN_ON_ONCE(!cookie_match(next, rq_i->core_pick));
6309
6310 if (rq_i->curr == rq_i->core_pick) {
6311 rq_i->core_pick = NULL;
6312 rq_i->core_dl_server = NULL;
6313 continue;
6314 }
6315
6316 resched_curr(rq_i);
6317 }
6318
6319 out_set_next:
6320 put_prev_set_next_task(rq, prev, next);
6321 if (rq->core->core_forceidle_count && next == rq->idle)
6322 queue_core_balance(rq);
6323
6324 return next;
6325 }
6326
6327 static bool try_steal_cookie(int this, int that)
6328 {
6329 struct rq *dst = cpu_rq(this), *src = cpu_rq(that);
6330 struct task_struct *p;
6331 unsigned long cookie;
6332 bool success = false;
6333
6334 guard(irq)();
6335 guard(double_rq_lock)(dst, src);
6336
6337 cookie = dst->core->core_cookie;
6338 if (!cookie)
6339 return false;
6340
6341 if (dst->curr != dst->idle)
6342 return false;
6343
6344 p = sched_core_find(src, cookie);
6345 if (!p)
6346 return false;
6347
6348 do {
6349 if (p == src->core_pick || p == src->curr)
6350 goto next;
6351
6352 if (!is_cpu_allowed(p, this))
6353 goto next;
6354
6355 if (p->core_occupation > dst->idle->core_occupation)
6356 goto next;
6357 /*
6358 * sched_core_find() and sched_core_next() will ensure
6359 * that task @p is not throttled now, we also need to
6360 * check whether the runqueue of the destination CPU is
6361 * being throttled.
6362 */
6363 if (sched_task_is_throttled(p, this))
6364 goto next;
6365
6366 move_queued_task_locked(src, dst, p);
6367 resched_curr(dst);
6368
6369 success = true;
6370 break;
6371
6372 next:
6373 p = sched_core_next(p, cookie);
6374 } while (p);
6375
6376 return success;
6377 }
6378
6379 static bool steal_cookie_task(int cpu, struct sched_domain *sd)
6380 {
6381 int i;
6382
6383 for_each_cpu_wrap(i, sched_domain_span(sd), cpu + 1) {
6384 if (i == cpu)
6385 continue;
6386
6387 if (need_resched())
6388 break;
6389
6390 if (try_steal_cookie(cpu, i))
6391 return true;
6392 }
6393
6394 return false;
6395 }
6396
6397 static void sched_core_balance(struct rq *rq)
6398 {
6399 struct sched_domain *sd;
6400 int cpu = cpu_of(rq);
6401
6402 guard(preempt)();
6403 guard(rcu)();
6404
6405 raw_spin_rq_unlock_irq(rq);
6406 for_each_domain(cpu, sd) {
6407 if (need_resched())
6408 break;
6409
6410 if (steal_cookie_task(cpu, sd))
6411 break;
6412 }
6413 raw_spin_rq_lock_irq(rq);
6414 }
6415
6416 static DEFINE_PER_CPU(struct balance_callback, core_balance_head);
6417
6418 static void queue_core_balance(struct rq *rq)
6419 {
6420 if (!sched_core_enabled(rq))
6421 return;
6422
6423 if (!rq->core->core_cookie)
6424 return;
6425
6426 if (!rq->nr_running) /* not forced idle */
6427 return;
6428
6429 queue_balance_callback(rq, &per_cpu(core_balance_head, rq->cpu), sched_core_balance);
6430 }
6431
6432 DEFINE_LOCK_GUARD_1(core_lock, int,
6433 sched_core_lock(*_T->lock, &_T->flags),
6434 sched_core_unlock(*_T->lock, &_T->flags),
6435 unsigned long flags)
6436
6437 static void sched_core_cpu_starting(unsigned int cpu)
6438 {
6439 const struct cpumask *smt_mask = cpu_smt_mask(cpu);
6440 struct rq *rq = cpu_rq(cpu), *core_rq = NULL;
6441 int t;
6442
6443 guard(core_lock)(&cpu);
6444
6445 WARN_ON_ONCE(rq->core != rq);
6446
6447 /* if we're the first, we'll be our own leader */
6448 if (cpumask_weight(smt_mask) == 1)
6449 return;
6450
6451 /* find the leader */
6452 for_each_cpu(t, smt_mask) {
6453 if (t == cpu)
6454 continue;
6455 rq = cpu_rq(t);
6456 if (rq->core == rq) {
6457 core_rq = rq;
6458 break;
6459 }
6460 }
6461
6462 if (WARN_ON_ONCE(!core_rq)) /* whoopsie */
6463 return;
6464
6465 /* install and validate core_rq */
6466 for_each_cpu(t, smt_mask) {
6467 rq = cpu_rq(t);
6468
6469 if (t == cpu)
6470 rq->core = core_rq;
6471
6472 WARN_ON_ONCE(rq->core != core_rq);
6473 }
6474 }
6475
6476 static void sched_core_cpu_deactivate(unsigned int cpu)
6477 {
6478 const struct cpumask *smt_mask = cpu_smt_mask(cpu);
6479 struct rq *rq = cpu_rq(cpu), *core_rq = NULL;
6480 int t;
6481
6482 guard(core_lock)(&cpu);
6483
6484 /* if we're the last man standing, nothing to do */
6485 if (cpumask_weight(smt_mask) == 1) {
6486 WARN_ON_ONCE(rq->core != rq);
6487 return;
6488 }
6489
6490 /* if we're not the leader, nothing to do */
6491 if (rq->core != rq)
6492 return;
6493
6494 /* find a new leader */
6495 for_each_cpu(t, smt_mask) {
6496 if (t == cpu)
6497 continue;
6498 core_rq = cpu_rq(t);
6499 break;
6500 }
6501
6502 if (WARN_ON_ONCE(!core_rq)) /* impossible */
6503 return;
6504
6505 /* copy the shared state to the new leader */
6506 core_rq->core_task_seq = rq->core_task_seq;
6507 core_rq->core_pick_seq = rq->core_pick_seq;
6508 core_rq->core_cookie = rq->core_cookie;
6509 core_rq->core_forceidle_count = rq->core_forceidle_count;
6510 core_rq->core_forceidle_seq = rq->core_forceidle_seq;
6511 core_rq->core_forceidle_occupation = rq->core_forceidle_occupation;
6512
6513 /*
6514 * Accounting edge for forced idle is handled in pick_next_task().
6515 * Don't need another one here, since the hotplug thread shouldn't
6516 * have a cookie.
6517 */
6518 core_rq->core_forceidle_start = 0;
6519
6520 /* install new leader */
6521 for_each_cpu(t, smt_mask) {
6522 rq = cpu_rq(t);
6523 rq->core = core_rq;
6524 }
6525 }
6526
6527 static inline void sched_core_cpu_dying(unsigned int cpu)
6528 {
6529 struct rq *rq = cpu_rq(cpu);
6530
6531 if (rq->core != rq)
6532 rq->core = rq;
6533 }
6534
6535 #else /* !CONFIG_SCHED_CORE */
6536
6537 static inline void sched_core_cpu_starting(unsigned int cpu) {}
6538 static inline void sched_core_cpu_deactivate(unsigned int cpu) {}
6539 static inline void sched_core_cpu_dying(unsigned int cpu) {}
6540
6541 static struct task_struct *
6542 pick_next_task(struct rq *rq, struct task_struct *prev, struct rq_flags *rf)
6543 {
6544 return __pick_next_task(rq, prev, rf);
6545 }
6546
6547 #endif /* CONFIG_SCHED_CORE */
6548
6549 /*
6550 * Constants for the sched_mode argument of __schedule().
6551 *
6552 * The mode argument allows RT enabled kernels to differentiate a
6553 * preemption from blocking on an 'sleeping' spin/rwlock.
6554 */
6555 #define SM_IDLE (-1)
6556 #define SM_NONE 0
6557 #define SM_PREEMPT 1
6558 #define SM_RTLOCK_WAIT 2
6559
6560 /*
6561 * Helper function for __schedule()
6562 *
6563 * If a task does not have signals pending, deactivate it
6564 * Otherwise marks the task's __state as RUNNING
6565 */
6566 static bool try_to_block_task(struct rq *rq, struct task_struct *p,
6567 unsigned long task_state)
6568 {
6569 int flags = DEQUEUE_NOCLOCK;
6570
6571 if (signal_pending_state(task_state, p)) {
6572 WRITE_ONCE(p->__state, TASK_RUNNING);
6573 return false;
6574 }
6575
6576 p->sched_contributes_to_load =
6577 (task_state & TASK_UNINTERRUPTIBLE) &&
6578 !(task_state & TASK_NOLOAD) &&
6579 !(task_state & TASK_FROZEN);
6580
6581 if (unlikely(is_special_task_state(task_state)))
6582 flags |= DEQUEUE_SPECIAL;
6583
6584 /*
6585 * __schedule() ttwu()
6586 * prev_state = prev->state; if (p->on_rq && ...)
6587 * if (prev_state) goto out;
6588 * p->on_rq = 0; smp_acquire__after_ctrl_dep();
6589 * p->state = TASK_WAKING
6590 *
6591 * Where __schedule() and ttwu() have matching control dependencies.
6592 *
6593 * After this, schedule() must not care about p->state any more.
6594 */
6595 block_task(rq, p, flags);
6596 return true;
6597 }
6598
6599 /*
6600 * __schedule() is the main scheduler function.
6601 *
6602 * The main means of driving the scheduler and thus entering this function are:
6603 *
6604 * 1. Explicit blocking: mutex, semaphore, waitqueue, etc.
6605 *
6606 * 2. TIF_NEED_RESCHED flag is checked on interrupt and userspace return
6607 * paths. For example, see arch/x86/entry_64.S.
6608 *
6609 * To drive preemption between tasks, the scheduler sets the flag in timer
6610 * interrupt handler sched_tick().
6611 *
6612 * 3. Wakeups don't really cause entry into schedule(). They add a
6613 * task to the run-queue and that's it.
6614 *
6615 * Now, if the new task added to the run-queue preempts the current
6616 * task, then the wakeup sets TIF_NEED_RESCHED and schedule() gets
6617 * called on the nearest possible occasion:
6618 *
6619 * - If the kernel is preemptible (CONFIG_PREEMPTION=y):
6620 *
6621 * - in syscall or exception context, at the next outmost
6622 * preempt_enable(). (this might be as soon as the wake_up()'s
6623 * spin_unlock()!)
6624 *
6625 * - in IRQ context, return from interrupt-handler to
6626 * preemptible context
6627 *
6628 * - If the kernel is not preemptible (CONFIG_PREEMPTION is not set)
6629 * then at the next:
6630 *
6631 * - cond_resched() call
6632 * - explicit schedule() call
6633 * - return from syscall or exception to user-space
6634 * - return from interrupt-handler to user-space
6635 *
6636 * WARNING: must be called with preemption disabled!
6637 */
6638 static void __sched notrace __schedule(int sched_mode)
6639 {
6640 struct task_struct *prev, *next;
6641 /*
6642 * On PREEMPT_RT kernel, SM_RTLOCK_WAIT is noted
6643 * as a preemption by schedule_debug() and RCU.
6644 */
6645 bool preempt = sched_mode > SM_NONE;
6646 bool block = false;
6647 unsigned long *switch_count;
6648 unsigned long prev_state;
6649 struct rq_flags rf;
6650 struct rq *rq;
6651 int cpu;
6652
6653 cpu = smp_processor_id();
6654 rq = cpu_rq(cpu);
6655 prev = rq->curr;
6656
6657 schedule_debug(prev, preempt);
6658
6659 if (sched_feat(HRTICK) || sched_feat(HRTICK_DL))
6660 hrtick_clear(rq);
6661
6662 local_irq_disable();
6663 rcu_note_context_switch(preempt);
6664
6665 /*
6666 * Make sure that signal_pending_state()->signal_pending() below
6667 * can't be reordered with __set_current_state(TASK_INTERRUPTIBLE)
6668 * done by the caller to avoid the race with signal_wake_up():
6669 *
6670 * __set_current_state(@state) signal_wake_up()
6671 * schedule() set_tsk_thread_flag(p, TIF_SIGPENDING)
6672 * wake_up_state(p, state)
6673 * LOCK rq->lock LOCK p->pi_state
6674 * smp_mb__after_spinlock() smp_mb__after_spinlock()
6675 * if (signal_pending_state()) if (p->state & @state)
6676 *
6677 * Also, the membarrier system call requires a full memory barrier
6678 * after coming from user-space, before storing to rq->curr; this
6679 * barrier matches a full barrier in the proximity of the membarrier
6680 * system call exit.
6681 */
6682 rq_lock(rq, &rf);
6683 smp_mb__after_spinlock();
6684
6685 /* Promote REQ to ACT */
6686 rq->clock_update_flags <<= 1;
6687 update_rq_clock(rq);
6688 rq->clock_update_flags = RQCF_UPDATED;
6689
6690 switch_count = &prev->nivcsw;
6691
6692 /* Task state changes only considers SM_PREEMPT as preemption */
6693 preempt = sched_mode == SM_PREEMPT;
6694
6695 /*
6696 * We must load prev->state once (task_struct::state is volatile), such
6697 * that we form a control dependency vs deactivate_task() below.
6698 */
6699 prev_state = READ_ONCE(prev->__state);
6700 if (sched_mode == SM_IDLE) {
6701 /* SCX must consult the BPF scheduler to tell if rq is empty */
6702 if (!rq->nr_running && !scx_enabled()) {
6703 next = prev;
6704 goto picked;
6705 }
6706 } else if (!preempt && prev_state) {
6707 block = try_to_block_task(rq, prev, prev_state);
6708 switch_count = &prev->nvcsw;
6709 }
6710
6711 next = pick_next_task(rq, prev, &rf);
6712 rq_set_donor(rq, next);
6713 picked:
6714 clear_tsk_need_resched(prev);
6715 clear_preempt_need_resched();
6716 #ifdef CONFIG_SCHED_DEBUG
6717 rq->last_seen_need_resched_ns = 0;
6718 #endif
6719
6720 if (likely(prev != next)) {
6721 rq->nr_switches++;
6722 /*
6723 * RCU users of rcu_dereference(rq->curr) may not see
6724 * changes to task_struct made by pick_next_task().
6725 */
6726 RCU_INIT_POINTER(rq->curr, next);
6727 /*
6728 * The membarrier system call requires each architecture
6729 * to have a full memory barrier after updating
6730 * rq->curr, before returning to user-space.
6731 *
6732 * Here are the schemes providing that barrier on the
6733 * various architectures:
6734 * - mm ? switch_mm() : mmdrop() for x86, s390, sparc, PowerPC,
6735 * RISC-V. switch_mm() relies on membarrier_arch_switch_mm()
6736 * on PowerPC and on RISC-V.
6737 * - finish_lock_switch() for weakly-ordered
6738 * architectures where spin_unlock is a full barrier,
6739 * - switch_to() for arm64 (weakly-ordered, spin_unlock
6740 * is a RELEASE barrier),
6741 *
6742 * The barrier matches a full barrier in the proximity of
6743 * the membarrier system call entry.
6744 *
6745 * On RISC-V, this barrier pairing is also needed for the
6746 * SYNC_CORE command when switching between processes, cf.
6747 * the inline comments in membarrier_arch_switch_mm().
6748 */
6749 ++*switch_count;
6750
6751 migrate_disable_switch(rq, prev);
6752 psi_account_irqtime(rq, prev, next);
6753 psi_sched_switch(prev, next, block);
6754
6755 trace_sched_switch(preempt, prev, next, prev_state);
6756
6757 /* Also unlocks the rq: */
6758 rq = context_switch(rq, prev, next, &rf);
6759 } else {
6760 rq_unpin_lock(rq, &rf);
6761 __balance_callbacks(rq);
6762 raw_spin_rq_unlock_irq(rq);
6763 }
6764 }
6765
6766 void __noreturn do_task_dead(void)
6767 {
6768 /* Causes final put_task_struct in finish_task_switch(): */
6769 set_special_state(TASK_DEAD);
6770
6771 /* Tell freezer to ignore us: */
6772 current->flags |= PF_NOFREEZE;
6773
6774 __schedule(SM_NONE);
6775 BUG();
6776
6777 /* Avoid "noreturn function does return" - but don't continue if BUG() is a NOP: */
6778 for (;;)
6779 cpu_relax();
6780 }
6781
6782 static inline void sched_submit_work(struct task_struct *tsk)
6783 {
6784 static DEFINE_WAIT_OVERRIDE_MAP(sched_map, LD_WAIT_CONFIG);
6785 unsigned int task_flags;
6786
6787 /*
6788 * Establish LD_WAIT_CONFIG context to ensure none of the code called
6789 * will use a blocking primitive -- which would lead to recursion.
6790 */
6791 lock_map_acquire_try(&sched_map);
6792
6793 task_flags = tsk->flags;
6794 /*
6795 * If a worker goes to sleep, notify and ask workqueue whether it
6796 * wants to wake up a task to maintain concurrency.
6797 */
6798 if (task_flags & PF_WQ_WORKER)
6799 wq_worker_sleeping(tsk);
6800 else if (task_flags & PF_IO_WORKER)
6801 io_wq_worker_sleeping(tsk);
6802
6803 /*
6804 * spinlock and rwlock must not flush block requests. This will
6805 * deadlock if the callback attempts to acquire a lock which is
6806 * already acquired.
6807 */
6808 SCHED_WARN_ON(current->__state & TASK_RTLOCK_WAIT);
6809
6810 /*
6811 * If we are going to sleep and we have plugged IO queued,
6812 * make sure to submit it to avoid deadlocks.
6813 */
6814 blk_flush_plug(tsk->plug, true);
6815
6816 lock_map_release(&sched_map);
6817 }
6818
6819 static void sched_update_worker(struct task_struct *tsk)
6820 {
6821 if (tsk->flags & (PF_WQ_WORKER | PF_IO_WORKER | PF_BLOCK_TS)) {
6822 if (tsk->flags & PF_BLOCK_TS)
6823 blk_plug_invalidate_ts(tsk);
6824 if (tsk->flags & PF_WQ_WORKER)
6825 wq_worker_running(tsk);
6826 else if (tsk->flags & PF_IO_WORKER)
6827 io_wq_worker_running(tsk);
6828 }
6829 }
6830
6831 static __always_inline void __schedule_loop(int sched_mode)
6832 {
6833 do {
6834 preempt_disable();
6835 __schedule(sched_mode);
6836 sched_preempt_enable_no_resched();
6837 } while (need_resched());
6838 }
6839
6840 asmlinkage __visible void __sched schedule(void)
6841 {
6842 struct task_struct *tsk = current;
6843
6844 #ifdef CONFIG_RT_MUTEXES
6845 lockdep_assert(!tsk->sched_rt_mutex);
6846 #endif
6847
6848 if (!task_is_running(tsk))
6849 sched_submit_work(tsk);
6850 __schedule_loop(SM_NONE);
6851 sched_update_worker(tsk);
6852 }
6853 EXPORT_SYMBOL(schedule);
6854
6855 /*
6856 * synchronize_rcu_tasks() makes sure that no task is stuck in preempted
6857 * state (have scheduled out non-voluntarily) by making sure that all
6858 * tasks have either left the run queue or have gone into user space.
6859 * As idle tasks do not do either, they must not ever be preempted
6860 * (schedule out non-voluntarily).
6861 *
6862 * schedule_idle() is similar to schedule_preempt_disable() except that it
6863 * never enables preemption because it does not call sched_submit_work().
6864 */
6865 void __sched schedule_idle(void)
6866 {
6867 /*
6868 * As this skips calling sched_submit_work(), which the idle task does
6869 * regardless because that function is a NOP when the task is in a
6870 * TASK_RUNNING state, make sure this isn't used someplace that the
6871 * current task can be in any other state. Note, idle is always in the
6872 * TASK_RUNNING state.
6873 */
6874 WARN_ON_ONCE(current->__state);
6875 do {
6876 __schedule(SM_IDLE);
6877 } while (need_resched());
6878 }
6879
6880 #if defined(CONFIG_CONTEXT_TRACKING_USER) && !defined(CONFIG_HAVE_CONTEXT_TRACKING_USER_OFFSTACK)
6881 asmlinkage __visible void __sched schedule_user(void)
6882 {
6883 /*
6884 * If we come here after a random call to set_need_resched(),
6885 * or we have been woken up remotely but the IPI has not yet arrived,
6886 * we haven't yet exited the RCU idle mode. Do it here manually until
6887 * we find a better solution.
6888 *
6889 * NB: There are buggy callers of this function. Ideally we
6890 * should warn if prev_state != CT_STATE_USER, but that will trigger
6891 * too frequently to make sense yet.
6892 */
6893 enum ctx_state prev_state = exception_enter();
6894 schedule();
6895 exception_exit(prev_state);
6896 }
6897 #endif
6898
6899 /**
6900 * schedule_preempt_disabled - called with preemption disabled
6901 *
6902 * Returns with preemption disabled. Note: preempt_count must be 1
6903 */
6904 void __sched schedule_preempt_disabled(void)
6905 {
6906 sched_preempt_enable_no_resched();
6907 schedule();
6908 preempt_disable();
6909 }
6910
6911 #ifdef CONFIG_PREEMPT_RT
6912 void __sched notrace schedule_rtlock(void)
6913 {
6914 __schedule_loop(SM_RTLOCK_WAIT);
6915 }
6916 NOKPROBE_SYMBOL(schedule_rtlock);
6917 #endif
6918
6919 static void __sched notrace preempt_schedule_common(void)
6920 {
6921 do {
6922 /*
6923 * Because the function tracer can trace preempt_count_sub()
6924 * and it also uses preempt_enable/disable_notrace(), if
6925 * NEED_RESCHED is set, the preempt_enable_notrace() called
6926 * by the function tracer will call this function again and
6927 * cause infinite recursion.
6928 *
6929 * Preemption must be disabled here before the function
6930 * tracer can trace. Break up preempt_disable() into two
6931 * calls. One to disable preemption without fear of being
6932 * traced. The other to still record the preemption latency,
6933 * which can also be traced by the function tracer.
6934 */
6935 preempt_disable_notrace();
6936 preempt_latency_start(1);
6937 __schedule(SM_PREEMPT);
6938 preempt_latency_stop(1);
6939 preempt_enable_no_resched_notrace();
6940
6941 /*
6942 * Check again in case we missed a preemption opportunity
6943 * between schedule and now.
6944 */
6945 } while (need_resched());
6946 }
6947
6948 #ifdef CONFIG_PREEMPTION
6949 /*
6950 * This is the entry point to schedule() from in-kernel preemption
6951 * off of preempt_enable.
6952 */
6953 asmlinkage __visible void __sched notrace preempt_schedule(void)
6954 {
6955 /*
6956 * If there is a non-zero preempt_count or interrupts are disabled,
6957 * we do not want to preempt the current task. Just return..
6958 */
6959 if (likely(!preemptible()))
6960 return;
6961 preempt_schedule_common();
6962 }
6963 NOKPROBE_SYMBOL(preempt_schedule);
6964 EXPORT_SYMBOL(preempt_schedule);
6965
6966 #ifdef CONFIG_PREEMPT_DYNAMIC
6967 #if defined(CONFIG_HAVE_PREEMPT_DYNAMIC_CALL)
6968 #ifndef preempt_schedule_dynamic_enabled
6969 #define preempt_schedule_dynamic_enabled preempt_schedule
6970 #define preempt_schedule_dynamic_disabled NULL
6971 #endif
6972 DEFINE_STATIC_CALL(preempt_schedule, preempt_schedule_dynamic_enabled);
6973 EXPORT_STATIC_CALL_TRAMP(preempt_schedule);
6974 #elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY)
6975 static DEFINE_STATIC_KEY_TRUE(sk_dynamic_preempt_schedule);
6976 void __sched notrace dynamic_preempt_schedule(void)
6977 {
6978 if (!static_branch_unlikely(&sk_dynamic_preempt_schedule))
6979 return;
6980 preempt_schedule();
6981 }
6982 NOKPROBE_SYMBOL(dynamic_preempt_schedule);
6983 EXPORT_SYMBOL(dynamic_preempt_schedule);
6984 #endif
6985 #endif
6986
6987 /**
6988 * preempt_schedule_notrace - preempt_schedule called by tracing
6989 *
6990 * The tracing infrastructure uses preempt_enable_notrace to prevent
6991 * recursion and tracing preempt enabling caused by the tracing
6992 * infrastructure itself. But as tracing can happen in areas coming
6993 * from userspace or just about to enter userspace, a preempt enable
6994 * can occur before user_exit() is called. This will cause the scheduler
6995 * to be called when the system is still in usermode.
6996 *
6997 * To prevent this, the preempt_enable_notrace will use this function
6998 * instead of preempt_schedule() to exit user context if needed before
6999 * calling the scheduler.
7000 */
7001 asmlinkage __visible void __sched notrace preempt_schedule_notrace(void)
7002 {
7003 enum ctx_state prev_ctx;
7004
7005 if (likely(!preemptible()))
7006 return;
7007
7008 do {
7009 /*
7010 * Because the function tracer can trace preempt_count_sub()
7011 * and it also uses preempt_enable/disable_notrace(), if
7012 * NEED_RESCHED is set, the preempt_enable_notrace() called
7013 * by the function tracer will call this function again and
7014 * cause infinite recursion.
7015 *
7016 * Preemption must be disabled here before the function
7017 * tracer can trace. Break up preempt_disable() into two
7018 * calls. One to disable preemption without fear of being
7019 * traced. The other to still record the preemption latency,
7020 * which can also be traced by the function tracer.
7021 */
7022 preempt_disable_notrace();
7023 preempt_latency_start(1);
7024 /*
7025 * Needs preempt disabled in case user_exit() is traced
7026 * and the tracer calls preempt_enable_notrace() causing
7027 * an infinite recursion.
7028 */
7029 prev_ctx = exception_enter();
7030 __schedule(SM_PREEMPT);
7031 exception_exit(prev_ctx);
7032
7033 preempt_latency_stop(1);
7034 preempt_enable_no_resched_notrace();
7035 } while (need_resched());
7036 }
7037 EXPORT_SYMBOL_GPL(preempt_schedule_notrace);
7038
7039 #ifdef CONFIG_PREEMPT_DYNAMIC
7040 #if defined(CONFIG_HAVE_PREEMPT_DYNAMIC_CALL)
7041 #ifndef preempt_schedule_notrace_dynamic_enabled
7042 #define preempt_schedule_notrace_dynamic_enabled preempt_schedule_notrace
7043 #define preempt_schedule_notrace_dynamic_disabled NULL
7044 #endif
7045 DEFINE_STATIC_CALL(preempt_schedule_notrace, preempt_schedule_notrace_dynamic_enabled);
7046 EXPORT_STATIC_CALL_TRAMP(preempt_schedule_notrace);
7047 #elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY)
7048 static DEFINE_STATIC_KEY_TRUE(sk_dynamic_preempt_schedule_notrace);
7049 void __sched notrace dynamic_preempt_schedule_notrace(void)
7050 {
7051 if (!static_branch_unlikely(&sk_dynamic_preempt_schedule_notrace))
7052 return;
7053 preempt_schedule_notrace();
7054 }
7055 NOKPROBE_SYMBOL(dynamic_preempt_schedule_notrace);
7056 EXPORT_SYMBOL(dynamic_preempt_schedule_notrace);
7057 #endif
7058 #endif
7059
7060 #endif /* CONFIG_PREEMPTION */
7061
7062 /*
7063 * This is the entry point to schedule() from kernel preemption
7064 * off of IRQ context.
7065 * Note, that this is called and return with IRQs disabled. This will
7066 * protect us against recursive calling from IRQ contexts.
7067 */
7068 asmlinkage __visible void __sched preempt_schedule_irq(void)
7069 {
7070 enum ctx_state prev_state;
7071
7072 /* Catch callers which need to be fixed */
7073 BUG_ON(preempt_count() || !irqs_disabled());
7074
7075 prev_state = exception_enter();
7076
7077 do {
7078 preempt_disable();
7079 local_irq_enable();
7080 __schedule(SM_PREEMPT);
7081 local_irq_disable();
7082 sched_preempt_enable_no_resched();
7083 } while (need_resched());
7084
7085 exception_exit(prev_state);
7086 }
7087
7088 int default_wake_function(wait_queue_entry_t *curr, unsigned mode, int wake_flags,
7089 void *key)
7090 {
7091 WARN_ON_ONCE(IS_ENABLED(CONFIG_SCHED_DEBUG) && wake_flags & ~(WF_SYNC|WF_CURRENT_CPU));
7092 return try_to_wake_up(curr->private, mode, wake_flags);
7093 }
7094 EXPORT_SYMBOL(default_wake_function);
7095
7096 const struct sched_class *__setscheduler_class(int policy, int prio)
7097 {
7098 if (dl_prio(prio))
7099 return &dl_sched_class;
7100
7101 if (rt_prio(prio))
7102 return &rt_sched_class;
7103
7104 #ifdef CONFIG_SCHED_CLASS_EXT
7105 if (task_should_scx(policy))
7106 return &ext_sched_class;
7107 #endif
7108
7109 return &fair_sched_class;
7110 }
7111
7112 #ifdef CONFIG_RT_MUTEXES
7113
7114 /*
7115 * Would be more useful with typeof()/auto_type but they don't mix with
7116 * bit-fields. Since it's a local thing, use int. Keep the generic sounding
7117 * name such that if someone were to implement this function we get to compare
7118 * notes.
7119 */
7120 #define fetch_and_set(x, v) ({ int _x = (x); (x) = (v); _x; })
7121
7122 void rt_mutex_pre_schedule(void)
7123 {
7124 lockdep_assert(!fetch_and_set(current->sched_rt_mutex, 1));
7125 sched_submit_work(current);
7126 }
7127
7128 void rt_mutex_schedule(void)
7129 {
7130 lockdep_assert(current->sched_rt_mutex);
7131 __schedule_loop(SM_NONE);
7132 }
7133
7134 void rt_mutex_post_schedule(void)
7135 {
7136 sched_update_worker(current);
7137 lockdep_assert(fetch_and_set(current->sched_rt_mutex, 0));
7138 }
7139
7140 /*
7141 * rt_mutex_setprio - set the current priority of a task
7142 * @p: task to boost
7143 * @pi_task: donor task
7144 *
7145 * This function changes the 'effective' priority of a task. It does
7146 * not touch ->normal_prio like __setscheduler().
7147 *
7148 * Used by the rt_mutex code to implement priority inheritance
7149 * logic. Call site only calls if the priority of the task changed.
7150 */
7151 void rt_mutex_setprio(struct task_struct *p, struct task_struct *pi_task)
7152 {
7153 int prio, oldprio, queued, running, queue_flag =
7154 DEQUEUE_SAVE | DEQUEUE_MOVE | DEQUEUE_NOCLOCK;
7155 const struct sched_class *prev_class, *next_class;
7156 struct rq_flags rf;
7157 struct rq *rq;
7158
7159 /* XXX used to be waiter->prio, not waiter->task->prio */
7160 prio = __rt_effective_prio(pi_task, p->normal_prio);
7161
7162 /*
7163 * If nothing changed; bail early.
7164 */
7165 if (p->pi_top_task == pi_task && prio == p->prio && !dl_prio(prio))
7166 return;
7167
7168 rq = __task_rq_lock(p, &rf);
7169 update_rq_clock(rq);
7170 /*
7171 * Set under pi_lock && rq->lock, such that the value can be used under
7172 * either lock.
7173 *
7174 * Note that there is loads of tricky to make this pointer cache work
7175 * right. rt_mutex_slowunlock()+rt_mutex_postunlock() work together to
7176 * ensure a task is de-boosted (pi_task is set to NULL) before the
7177 * task is allowed to run again (and can exit). This ensures the pointer
7178 * points to a blocked task -- which guarantees the task is present.
7179 */
7180 p->pi_top_task = pi_task;
7181
7182 /*
7183 * For FIFO/RR we only need to set prio, if that matches we're done.
7184 */
7185 if (prio == p->prio && !dl_prio(prio))
7186 goto out_unlock;
7187
7188 /*
7189 * Idle task boosting is a no-no in general. There is one
7190 * exception, when PREEMPT_RT and NOHZ is active:
7191 *
7192 * The idle task calls get_next_timer_interrupt() and holds
7193 * the timer wheel base->lock on the CPU and another CPU wants
7194 * to access the timer (probably to cancel it). We can safely
7195 * ignore the boosting request, as the idle CPU runs this code
7196 * with interrupts disabled and will complete the lock
7197 * protected section without being interrupted. So there is no
7198 * real need to boost.
7199 */
7200 if (unlikely(p == rq->idle)) {
7201 WARN_ON(p != rq->curr);
7202 WARN_ON(p->pi_blocked_on);
7203 goto out_unlock;
7204 }
7205
7206 trace_sched_pi_setprio(p, pi_task);
7207 oldprio = p->prio;
7208
7209 if (oldprio == prio)
7210 queue_flag &= ~DEQUEUE_MOVE;
7211
7212 prev_class = p->sched_class;
7213 next_class = __setscheduler_class(p->policy, prio);
7214
7215 if (prev_class != next_class && p->se.sched_delayed)
7216 dequeue_task(rq, p, DEQUEUE_SLEEP | DEQUEUE_DELAYED | DEQUEUE_NOCLOCK);
7217
7218 queued = task_on_rq_queued(p);
7219 running = task_current_donor(rq, p);
7220 if (queued)
7221 dequeue_task(rq, p, queue_flag);
7222 if (running)
7223 put_prev_task(rq, p);
7224
7225 /*
7226 * Boosting condition are:
7227 * 1. -rt task is running and holds mutex A
7228 * --> -dl task blocks on mutex A
7229 *
7230 * 2. -dl task is running and holds mutex A
7231 * --> -dl task blocks on mutex A and could preempt the
7232 * running task
7233 */
7234 if (dl_prio(prio)) {
7235 if (!dl_prio(p->normal_prio) ||
7236 (pi_task && dl_prio(pi_task->prio) &&
7237 dl_entity_preempt(&pi_task->dl, &p->dl))) {
7238 p->dl.pi_se = pi_task->dl.pi_se;
7239 queue_flag |= ENQUEUE_REPLENISH;
7240 } else {
7241 p->dl.pi_se = &p->dl;
7242 }
7243 } else if (rt_prio(prio)) {
7244 if (dl_prio(oldprio))
7245 p->dl.pi_se = &p->dl;
7246 if (oldprio < prio)
7247 queue_flag |= ENQUEUE_HEAD;
7248 } else {
7249 if (dl_prio(oldprio))
7250 p->dl.pi_se = &p->dl;
7251 if (rt_prio(oldprio))
7252 p->rt.timeout = 0;
7253 }
7254
7255 p->sched_class = next_class;
7256 p->prio = prio;
7257
7258 check_class_changing(rq, p, prev_class);
7259
7260 if (queued)
7261 enqueue_task(rq, p, queue_flag);
7262 if (running)
7263 set_next_task(rq, p);
7264
7265 check_class_changed(rq, p, prev_class, oldprio);
7266 out_unlock:
7267 /* Avoid rq from going away on us: */
7268 preempt_disable();
7269
7270 rq_unpin_lock(rq, &rf);
7271 __balance_callbacks(rq);
7272 raw_spin_rq_unlock(rq);
7273
7274 preempt_enable();
7275 }
7276 #endif
7277
7278 #if !defined(CONFIG_PREEMPTION) || defined(CONFIG_PREEMPT_DYNAMIC)
7279 int __sched __cond_resched(void)
7280 {
7281 if (should_resched(0)) {
7282 preempt_schedule_common();
7283 return 1;
7284 }
7285 /*
7286 * In preemptible kernels, ->rcu_read_lock_nesting tells the tick
7287 * whether the current CPU is in an RCU read-side critical section,
7288 * so the tick can report quiescent states even for CPUs looping
7289 * in kernel context. In contrast, in non-preemptible kernels,
7290 * RCU readers leave no in-memory hints, which means that CPU-bound
7291 * processes executing in kernel context might never report an
7292 * RCU quiescent state. Therefore, the following code causes
7293 * cond_resched() to report a quiescent state, but only when RCU
7294 * is in urgent need of one.
7295 */
7296 #ifndef CONFIG_PREEMPT_RCU
7297 rcu_all_qs();
7298 #endif
7299 return 0;
7300 }
7301 EXPORT_SYMBOL(__cond_resched);
7302 #endif
7303
7304 #ifdef CONFIG_PREEMPT_DYNAMIC
7305 #if defined(CONFIG_HAVE_PREEMPT_DYNAMIC_CALL)
7306 #define cond_resched_dynamic_enabled __cond_resched
7307 #define cond_resched_dynamic_disabled ((void *)&__static_call_return0)
7308 DEFINE_STATIC_CALL_RET0(cond_resched, __cond_resched);
7309 EXPORT_STATIC_CALL_TRAMP(cond_resched);
7310
7311 #define might_resched_dynamic_enabled __cond_resched
7312 #define might_resched_dynamic_disabled ((void *)&__static_call_return0)
7313 DEFINE_STATIC_CALL_RET0(might_resched, __cond_resched);
7314 EXPORT_STATIC_CALL_TRAMP(might_resched);
7315 #elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY)
7316 static DEFINE_STATIC_KEY_FALSE(sk_dynamic_cond_resched);
7317 int __sched dynamic_cond_resched(void)
7318 {
7319 klp_sched_try_switch();
7320 if (!static_branch_unlikely(&sk_dynamic_cond_resched))
7321 return 0;
7322 return __cond_resched();
7323 }
7324 EXPORT_SYMBOL(dynamic_cond_resched);
7325
7326 static DEFINE_STATIC_KEY_FALSE(sk_dynamic_might_resched);
7327 int __sched dynamic_might_resched(void)
7328 {
7329 if (!static_branch_unlikely(&sk_dynamic_might_resched))
7330 return 0;
7331 return __cond_resched();
7332 }
7333 EXPORT_SYMBOL(dynamic_might_resched);
7334 #endif
7335 #endif
7336
7337 /*
7338 * __cond_resched_lock() - if a reschedule is pending, drop the given lock,
7339 * call schedule, and on return reacquire the lock.
7340 *
7341 * This works OK both with and without CONFIG_PREEMPTION. We do strange low-level
7342 * operations here to prevent schedule() from being called twice (once via
7343 * spin_unlock(), once by hand).
7344 */
7345 int __cond_resched_lock(spinlock_t *lock)
7346 {
7347 int resched = should_resched(PREEMPT_LOCK_OFFSET);
7348 int ret = 0;
7349
7350 lockdep_assert_held(lock);
7351
7352 if (spin_needbreak(lock) || resched) {
7353 spin_unlock(lock);
7354 if (!_cond_resched())
7355 cpu_relax();
7356 ret = 1;
7357 spin_lock(lock);
7358 }
7359 return ret;
7360 }
7361 EXPORT_SYMBOL(__cond_resched_lock);
7362
7363 int __cond_resched_rwlock_read(rwlock_t *lock)
7364 {
7365 int resched = should_resched(PREEMPT_LOCK_OFFSET);
7366 int ret = 0;
7367
7368 lockdep_assert_held_read(lock);
7369
7370 if (rwlock_needbreak(lock) || resched) {
7371 read_unlock(lock);
7372 if (!_cond_resched())
7373 cpu_relax();
7374 ret = 1;
7375 read_lock(lock);
7376 }
7377 return ret;
7378 }
7379 EXPORT_SYMBOL(__cond_resched_rwlock_read);
7380
7381 int __cond_resched_rwlock_write(rwlock_t *lock)
7382 {
7383 int resched = should_resched(PREEMPT_LOCK_OFFSET);
7384 int ret = 0;
7385
7386 lockdep_assert_held_write(lock);
7387
7388 if (rwlock_needbreak(lock) || resched) {
7389 write_unlock(lock);
7390 if (!_cond_resched())
7391 cpu_relax();
7392 ret = 1;
7393 write_lock(lock);
7394 }
7395 return ret;
7396 }
7397 EXPORT_SYMBOL(__cond_resched_rwlock_write);
7398
7399 #ifdef CONFIG_PREEMPT_DYNAMIC
7400
7401 #ifdef CONFIG_GENERIC_IRQ_ENTRY
> 7402 #include <linux/irq-entry-common.h>
7403 #endif
7404
--
0-DAY CI Kernel Test Service
https://github.com/intel/lkp-tests/wiki
^ permalink raw reply [flat|nested] 3+ messages in thread* [PATCH -next v5 00/22] arm64: entry: Convert to generic entry
@ 2024-12-06 10:17 Jinjie Ruan
2024-12-06 10:17 ` [PATCH -next v5 09/22] entry: Split generic entry into irq and syscall Jinjie Ruan
0 siblings, 1 reply; 3+ messages in thread
From: Jinjie Ruan @ 2024-12-06 10:17 UTC (permalink / raw)
To: catalin.marinas, will, oleg, sstabellini, tglx, peterz, luto,
mingo, juri.lelli, vincent.guittot, dietmar.eggemann, rostedt,
bsegall, mgorman, vschneid, kees, wad, akpm, samitolvanen,
masahiroy, hca, aliceryhl, rppt, xur, paulmck, arnd, mbenes,
puranjay, mark.rutland, ruanjinjie, pcc, ardb, sudeep.holla,
guohanjun, rafael, liuwei09, dwmw, Jonathan.Cameron, liaochang1,
kristina.martsenko, ptosi, broonie, thiago.bauermann,
kevin.brodsky, joey.gouly, liuyuntao12, leobras, linux-kernel,
linux-arm-kernel, xen-devel
Currently, x86, Riscv, Loongarch use the generic entry. Convert arm64
to use the generic entry infrastructure from kernel/entry/*. The generic
entry makes maintainers' work easier and codes more elegant, which aslo
removed a lot of duplicate code.
The main steps are as follows:
- Make arm64 easier to use irqentry_enter/exit().
- Make arm64 closer to the PREEMPT_DYNAMIC code of generic entry.
- Split generic entry into generic irq entry and generic syscall to
make the single patch more concentrated in switching to one thing.
- Switch to generic irq entry.
- Make arm64 closer to the generic syscall code.
- Switch to generic entry completely.
Changes in v5:
- Not change arm32 and keep inerrupts_enabled() macro for gicv3 driver.
- Move irqentry_state definition into arch/arm64/kernel/entry-common.c.
- Avoid removing the __enter_from_*() and __exit_to_*() wrappers.
- Update "irqentry_state_t ret/irq_state" to "state"
to keep it consistently.
- Use generic irq entry header for PREEMPT_DYNAMIC after split
the generic entry.
- Also refactor the ARM64 syscall code.
- Introduce arch_ptrace_report_syscall_entry/exit(), instead of
arch_pre/post_report_syscall_entry/exit() to simplify code.
- Make the syscall patches clear separation.
- Update the commit message.
Changes in v4:
- Rework/cleanup split into a few patches as Mark suggested.
- Replace interrupts_enabled() macro with regs_irqs_disabled(), instead
of left it here.
- Remove rcu and lockdep state in pt_regs by using temporary
irqentry_state_t as Mark suggested.
- Remove some unnecessary intermediate functions to make it clear.
- Rework preempt irq and PREEMPT_DYNAMIC code
to make the switch more clear.
- arch_prepare_*_entry/exit() -> arch_pre_*_entry/exit().
- Expand the arch functions comment.
- Make arch functions closer to its caller.
- Declare saved_reg in for block.
- Remove arch_exit_to_kernel_mode_prepare(), arch_enter_from_kernel_mode().
- Adjust "Add few arch functions to use generic entry" patch to be
the penultimate.
- Update the commit message.
- Add suggested-by.
Changes in v3:
- Test the MTE test cases.
- Handle forget_syscall() in arch_post_report_syscall_entry()
- Make the arch funcs not use __weak as Thomas suggested, so move
the arch funcs to entry-common.h, and make arch_forget_syscall() folded
in arch_post_report_syscall_entry() as suggested.
- Move report_single_step() to thread_info.h for arm64
- Change __always_inline() to inline, add inline for the other arch funcs.
- Remove unused signal.h for entry-common.h.
- Add Suggested-by.
- Update the commit message.
Changes in v2:
- Add tested-by.
- Fix a bug that not call arch_post_report_syscall_entry() in
syscall_trace_enter() if ptrace_report_syscall_entry() return not zero.
- Refactor report_syscall().
- Add comment for arch_prepare_report_syscall_exit().
- Adjust entry-common.h header file inclusion to alphabetical order.
- Update the commit message.
Jinjie Ruan (22):
arm64: ptrace: Replace interrupts_enabled() with regs_irqs_disabled()
arm64: entry: Refactor the entry and exit for exceptions from EL1
arm64: entry: Move arm64_preempt_schedule_irq() into
__exit_to_kernel_mode()
arm64: entry: Rework arm64_preempt_schedule_irq()
arm64: entry: Use preempt_count() and need_resched() helper
arm64: entry: Expand the need_irq_preemption() macro ahead
arm64: entry: preempt_schedule_irq() only if PREEMPTION enabled
arm64: entry: Use different helpers to check resched for
PREEMPT_DYNAMIC
entry: Split generic entry into irq and syscall
entry: Add arch_irqentry_exit_need_resched() for arm64
arm64: entry: Switch to generic IRQ entry
arm64/ptrace: Split report_syscall() function
arm64/ptrace: Refactor syscall_trace_enter()
arm64/ptrace: Refactor syscall_trace_exit()
arm64/ptrace: Refator el0_svc_common()
entry: Make syscall_exit_to_user_mode_prepare() not static
arm64/ptrace: Return early for ptrace_report_syscall_entry() error
arm64/ptrace: Expand secure_computing() in place
arm64/ptrace: Use syscall_get_arguments() heleper
entry: Add arch_ptrace_report_syscall_entry/exit()
entry: Add has_syscall_work() helepr
arm64: entry: Convert to generic entry
MAINTAINERS | 1 +
arch/Kconfig | 8 +
arch/arm64/Kconfig | 1 +
arch/arm64/include/asm/daifflags.h | 2 +-
arch/arm64/include/asm/entry-common.h | 134 +++++++++
arch/arm64/include/asm/preempt.h | 2 -
arch/arm64/include/asm/ptrace.h | 11 +-
arch/arm64/include/asm/syscall.h | 6 +-
arch/arm64/include/asm/thread_info.h | 23 +-
arch/arm64/include/asm/xen/events.h | 2 +-
arch/arm64/kernel/acpi.c | 2 +-
arch/arm64/kernel/debug-monitors.c | 9 +-
arch/arm64/kernel/entry-common.c | 377 ++++++++-----------------
arch/arm64/kernel/ptrace.c | 90 ------
arch/arm64/kernel/sdei.c | 2 +-
arch/arm64/kernel/signal.c | 3 +-
arch/arm64/kernel/syscall.c | 31 +-
include/linux/entry-common.h | 384 +------------------------
include/linux/irq-entry-common.h | 389 ++++++++++++++++++++++++++
kernel/entry/Makefile | 3 +-
kernel/entry/common.c | 176 ++----------
kernel/entry/syscall-common.c | 198 +++++++++++++
kernel/sched/core.c | 8 +-
23 files changed, 909 insertions(+), 953 deletions(-)
create mode 100644 arch/arm64/include/asm/entry-common.h
create mode 100644 include/linux/irq-entry-common.h
create mode 100644 kernel/entry/syscall-common.c
--
2.34.1
^ permalink raw reply [flat|nested] 3+ messages in thread* [PATCH -next v5 09/22] entry: Split generic entry into irq and syscall
2024-12-06 10:17 [PATCH -next v5 00/22] arm64: entry: Convert to generic entry Jinjie Ruan
@ 2024-12-06 10:17 ` Jinjie Ruan
2025-02-10 12:04 ` Mark Rutland
0 siblings, 1 reply; 3+ messages in thread
From: Jinjie Ruan @ 2024-12-06 10:17 UTC (permalink / raw)
To: catalin.marinas, will, oleg, sstabellini, tglx, peterz, luto,
mingo, juri.lelli, vincent.guittot, dietmar.eggemann, rostedt,
bsegall, mgorman, vschneid, kees, wad, akpm, samitolvanen,
masahiroy, hca, aliceryhl, rppt, xur, paulmck, arnd, mbenes,
puranjay, mark.rutland, ruanjinjie, pcc, ardb, sudeep.holla,
guohanjun, rafael, liuwei09, dwmw, Jonathan.Cameron, liaochang1,
kristina.martsenko, ptosi, broonie, thiago.bauermann,
kevin.brodsky, joey.gouly, liuyuntao12, leobras, linux-kernel,
linux-arm-kernel, xen-devel
As Mark pointed out, do not try to switch to *all* the
generic entry code in one go. The regular entry state management
(e.g. enter_from_user_mode() and exit_to_user_mode()) is largely
separate from the syscall state management. Move arm64 over to
enter_from_user_mode() and exit_to_user_mode() without needing to use
any of the generic syscall logic. Doing that first, *then* moving over
to the generic syscall handling would be much easier to
review/test/bisect, and if there are any ABI issues with the syscall
handling in particular, it will be easier to handle those in isolation.
So split generic entry into irq entry and syscall code, which will
make review work easier and switch to generic entry clear.
Introdue two configs called GENERIC_SYSCALL and GENERIC_IRQ_ENTRY,
which control the irq entry and syscall parts of the generic code
respectively. And split the header file irq-entry-common.h from
entry-common.h for GENERIC_IRQ_ENTRY.
Suggested-by: Mark Rutland <mark.rutland@arm.com>
Signed-off-by: Jinjie Ruan <ruanjinjie@huawei.com>
---
MAINTAINERS | 1 +
arch/Kconfig | 8 +
include/linux/entry-common.h | 382 +-----------------------------
include/linux/irq-entry-common.h | 389 +++++++++++++++++++++++++++++++
kernel/entry/Makefile | 3 +-
kernel/entry/common.c | 160 +------------
kernel/entry/syscall-common.c | 159 +++++++++++++
kernel/sched/core.c | 8 +-
8 files changed, 565 insertions(+), 545 deletions(-)
create mode 100644 include/linux/irq-entry-common.h
create mode 100644 kernel/entry/syscall-common.c
diff --git a/MAINTAINERS b/MAINTAINERS
index 21f855fe468b..7a6e87587101 100644
--- a/MAINTAINERS
+++ b/MAINTAINERS
@@ -9585,6 +9585,7 @@ S: Maintained
T: git git://git.kernel.org/pub/scm/linux/kernel/git/tip/tip.git core/entry
F: include/linux/entry-common.h
F: include/linux/entry-kvm.h
+F: include/linux/irq-entry-common.h
F: kernel/entry/
GENERIC GPIO I2C DRIVER
diff --git a/arch/Kconfig b/arch/Kconfig
index 6682b2a53e34..5a454eff780b 100644
--- a/arch/Kconfig
+++ b/arch/Kconfig
@@ -64,8 +64,16 @@ config HOTPLUG_PARALLEL
bool
select HOTPLUG_SPLIT_STARTUP
+config GENERIC_IRQ_ENTRY
+ bool
+
+config GENERIC_SYSCALL
+ bool
+
config GENERIC_ENTRY
bool
+ select GENERIC_IRQ_ENTRY
+ select GENERIC_SYSCALL
config KPROBES
bool "Kprobes"
diff --git a/include/linux/entry-common.h b/include/linux/entry-common.h
index fc61d0205c97..b3233e8328c5 100644
--- a/include/linux/entry-common.h
+++ b/include/linux/entry-common.h
@@ -2,27 +2,15 @@
#ifndef __LINUX_ENTRYCOMMON_H
#define __LINUX_ENTRYCOMMON_H
-#include <linux/static_call_types.h>
+#include <linux/irq-entry-common.h>
#include <linux/ptrace.h>
-#include <linux/syscalls.h>
#include <linux/seccomp.h>
#include <linux/sched.h>
-#include <linux/context_tracking.h>
#include <linux/livepatch.h>
#include <linux/resume_user_mode.h>
-#include <linux/tick.h>
-#include <linux/kmsan.h>
#include <asm/entry-common.h>
-/*
- * Define dummy _TIF work flags if not defined by the architecture or for
- * disabled functionality.
- */
-#ifndef _TIF_PATCH_PENDING
-# define _TIF_PATCH_PENDING (0)
-#endif
-
#ifndef _TIF_UPROBE
# define _TIF_UPROBE (0)
#endif
@@ -55,69 +43,6 @@
SYSCALL_WORK_SYSCALL_EXIT_TRAP | \
ARCH_SYSCALL_WORK_EXIT)
-/*
- * TIF flags handled in exit_to_user_mode_loop()
- */
-#ifndef ARCH_EXIT_TO_USER_MODE_WORK
-# define ARCH_EXIT_TO_USER_MODE_WORK (0)
-#endif
-
-#define EXIT_TO_USER_MODE_WORK \
- (_TIF_SIGPENDING | _TIF_NOTIFY_RESUME | _TIF_UPROBE | \
- _TIF_NEED_RESCHED | _TIF_NEED_RESCHED_LAZY | \
- _TIF_PATCH_PENDING | _TIF_NOTIFY_SIGNAL | \
- ARCH_EXIT_TO_USER_MODE_WORK)
-
-/**
- * arch_enter_from_user_mode - Architecture specific sanity check for user mode regs
- * @regs: Pointer to currents pt_regs
- *
- * Defaults to an empty implementation. Can be replaced by architecture
- * specific code.
- *
- * Invoked from syscall_enter_from_user_mode() in the non-instrumentable
- * section. Use __always_inline so the compiler cannot push it out of line
- * and make it instrumentable.
- */
-static __always_inline void arch_enter_from_user_mode(struct pt_regs *regs);
-
-#ifndef arch_enter_from_user_mode
-static __always_inline void arch_enter_from_user_mode(struct pt_regs *regs) {}
-#endif
-
-/**
- * enter_from_user_mode - Establish state when coming from user mode
- *
- * Syscall/interrupt entry disables interrupts, but user mode is traced as
- * interrupts enabled. Also with NO_HZ_FULL RCU might be idle.
- *
- * 1) Tell lockdep that interrupts are disabled
- * 2) Invoke context tracking if enabled to reactivate RCU
- * 3) Trace interrupts off state
- *
- * Invoked from architecture specific syscall entry code with interrupts
- * disabled. The calling code has to be non-instrumentable. When the
- * function returns all state is correct and interrupts are still
- * disabled. The subsequent functions can be instrumented.
- *
- * This is invoked when there is architecture specific functionality to be
- * done between establishing state and enabling interrupts. The caller must
- * enable interrupts before invoking syscall_enter_from_user_mode_work().
- */
-static __always_inline void enter_from_user_mode(struct pt_regs *regs)
-{
- arch_enter_from_user_mode(regs);
- lockdep_hardirqs_off(CALLER_ADDR0);
-
- CT_WARN_ON(__ct_state() != CT_STATE_USER);
- user_exit_irqoff();
-
- instrumentation_begin();
- kmsan_unpoison_entry_regs(regs);
- trace_hardirqs_off_finish();
- instrumentation_end();
-}
-
/**
* syscall_enter_from_user_mode_prepare - Establish state and enable interrupts
* @regs: Pointer to currents pt_regs
@@ -202,170 +127,6 @@ static __always_inline long syscall_enter_from_user_mode(struct pt_regs *regs, l
return ret;
}
-/**
- * local_irq_enable_exit_to_user - Exit to user variant of local_irq_enable()
- * @ti_work: Cached TIF flags gathered with interrupts disabled
- *
- * Defaults to local_irq_enable(). Can be supplied by architecture specific
- * code.
- */
-static inline void local_irq_enable_exit_to_user(unsigned long ti_work);
-
-#ifndef local_irq_enable_exit_to_user
-static inline void local_irq_enable_exit_to_user(unsigned long ti_work)
-{
- local_irq_enable();
-}
-#endif
-
-/**
- * local_irq_disable_exit_to_user - Exit to user variant of local_irq_disable()
- *
- * Defaults to local_irq_disable(). Can be supplied by architecture specific
- * code.
- */
-static inline void local_irq_disable_exit_to_user(void);
-
-#ifndef local_irq_disable_exit_to_user
-static inline void local_irq_disable_exit_to_user(void)
-{
- local_irq_disable();
-}
-#endif
-
-/**
- * arch_exit_to_user_mode_work - Architecture specific TIF work for exit
- * to user mode.
- * @regs: Pointer to currents pt_regs
- * @ti_work: Cached TIF flags gathered with interrupts disabled
- *
- * Invoked from exit_to_user_mode_loop() with interrupt enabled
- *
- * Defaults to NOOP. Can be supplied by architecture specific code.
- */
-static inline void arch_exit_to_user_mode_work(struct pt_regs *regs,
- unsigned long ti_work);
-
-#ifndef arch_exit_to_user_mode_work
-static inline void arch_exit_to_user_mode_work(struct pt_regs *regs,
- unsigned long ti_work)
-{
-}
-#endif
-
-/**
- * arch_exit_to_user_mode_prepare - Architecture specific preparation for
- * exit to user mode.
- * @regs: Pointer to currents pt_regs
- * @ti_work: Cached TIF flags gathered with interrupts disabled
- *
- * Invoked from exit_to_user_mode_prepare() with interrupt disabled as the last
- * function before return. Defaults to NOOP.
- */
-static inline void arch_exit_to_user_mode_prepare(struct pt_regs *regs,
- unsigned long ti_work);
-
-#ifndef arch_exit_to_user_mode_prepare
-static inline void arch_exit_to_user_mode_prepare(struct pt_regs *regs,
- unsigned long ti_work)
-{
-}
-#endif
-
-/**
- * arch_exit_to_user_mode - Architecture specific final work before
- * exit to user mode.
- *
- * Invoked from exit_to_user_mode() with interrupt disabled as the last
- * function before return. Defaults to NOOP.
- *
- * This needs to be __always_inline because it is non-instrumentable code
- * invoked after context tracking switched to user mode.
- *
- * An architecture implementation must not do anything complex, no locking
- * etc. The main purpose is for speculation mitigations.
- */
-static __always_inline void arch_exit_to_user_mode(void);
-
-#ifndef arch_exit_to_user_mode
-static __always_inline void arch_exit_to_user_mode(void) { }
-#endif
-
-/**
- * arch_do_signal_or_restart - Architecture specific signal delivery function
- * @regs: Pointer to currents pt_regs
- *
- * Invoked from exit_to_user_mode_loop().
- */
-void arch_do_signal_or_restart(struct pt_regs *regs);
-
-/**
- * exit_to_user_mode_loop - do any pending work before leaving to user space
- */
-unsigned long exit_to_user_mode_loop(struct pt_regs *regs,
- unsigned long ti_work);
-
-/**
- * exit_to_user_mode_prepare - call exit_to_user_mode_loop() if required
- * @regs: Pointer to pt_regs on entry stack
- *
- * 1) check that interrupts are disabled
- * 2) call tick_nohz_user_enter_prepare()
- * 3) call exit_to_user_mode_loop() if any flags from
- * EXIT_TO_USER_MODE_WORK are set
- * 4) check that interrupts are still disabled
- */
-static __always_inline void exit_to_user_mode_prepare(struct pt_regs *regs)
-{
- unsigned long ti_work;
-
- lockdep_assert_irqs_disabled();
-
- /* Flush pending rcuog wakeup before the last need_resched() check */
- tick_nohz_user_enter_prepare();
-
- ti_work = read_thread_flags();
- if (unlikely(ti_work & EXIT_TO_USER_MODE_WORK))
- ti_work = exit_to_user_mode_loop(regs, ti_work);
-
- arch_exit_to_user_mode_prepare(regs, ti_work);
-
- /* Ensure that kernel state is sane for a return to userspace */
- kmap_assert_nomap();
- lockdep_assert_irqs_disabled();
- lockdep_sys_exit();
-}
-
-/**
- * exit_to_user_mode - Fixup state when exiting to user mode
- *
- * Syscall/interrupt exit enables interrupts, but the kernel state is
- * interrupts disabled when this is invoked. Also tell RCU about it.
- *
- * 1) Trace interrupts on state
- * 2) Invoke context tracking if enabled to adjust RCU state
- * 3) Invoke architecture specific last minute exit code, e.g. speculation
- * mitigations, etc.: arch_exit_to_user_mode()
- * 4) Tell lockdep that interrupts are enabled
- *
- * Invoked from architecture specific code when syscall_exit_to_user_mode()
- * is not suitable as the last step before returning to userspace. Must be
- * invoked with interrupts disabled and the caller must be
- * non-instrumentable.
- * The caller has to invoke syscall_exit_to_user_mode_work() before this.
- */
-static __always_inline void exit_to_user_mode(void)
-{
- instrumentation_begin();
- trace_hardirqs_on_prepare();
- lockdep_hardirqs_on_prepare();
- instrumentation_end();
-
- user_enter_irqoff();
- arch_exit_to_user_mode();
- lockdep_hardirqs_on(CALLER_ADDR0);
-}
-
/**
* syscall_exit_to_user_mode_work - Handle work before returning to user mode
* @regs: Pointer to currents pt_regs
@@ -412,145 +173,4 @@ void syscall_exit_to_user_mode_work(struct pt_regs *regs);
*/
void syscall_exit_to_user_mode(struct pt_regs *regs);
-/**
- * irqentry_enter_from_user_mode - Establish state before invoking the irq handler
- * @regs: Pointer to currents pt_regs
- *
- * Invoked from architecture specific entry code with interrupts disabled.
- * Can only be called when the interrupt entry came from user mode. The
- * calling code must be non-instrumentable. When the function returns all
- * state is correct and the subsequent functions can be instrumented.
- *
- * The function establishes state (lockdep, RCU (context tracking), tracing)
- */
-void irqentry_enter_from_user_mode(struct pt_regs *regs);
-
-/**
- * irqentry_exit_to_user_mode - Interrupt exit work
- * @regs: Pointer to current's pt_regs
- *
- * Invoked with interrupts disabled and fully valid regs. Returns with all
- * work handled, interrupts disabled such that the caller can immediately
- * switch to user mode. Called from architecture specific interrupt
- * handling code.
- *
- * The call order is #2 and #3 as described in syscall_exit_to_user_mode().
- * Interrupt exit is not invoking #1 which is the syscall specific one time
- * work.
- */
-void irqentry_exit_to_user_mode(struct pt_regs *regs);
-
-#ifndef irqentry_state
-/**
- * struct irqentry_state - Opaque object for exception state storage
- * @exit_rcu: Used exclusively in the irqentry_*() calls; signals whether the
- * exit path has to invoke ct_irq_exit().
- * @lockdep: Used exclusively in the irqentry_nmi_*() calls; ensures that
- * lockdep state is restored correctly on exit from nmi.
- *
- * This opaque object is filled in by the irqentry_*_enter() functions and
- * must be passed back into the corresponding irqentry_*_exit() functions
- * when the exception is complete.
- *
- * Callers of irqentry_*_[enter|exit]() must consider this structure opaque
- * and all members private. Descriptions of the members are provided to aid in
- * the maintenance of the irqentry_*() functions.
- */
-typedef struct irqentry_state {
- union {
- bool exit_rcu;
- bool lockdep;
- };
-} irqentry_state_t;
-#endif
-
-/**
- * irqentry_enter - Handle state tracking on ordinary interrupt entries
- * @regs: Pointer to pt_regs of interrupted context
- *
- * Invokes:
- * - lockdep irqflag state tracking as low level ASM entry disabled
- * interrupts.
- *
- * - Context tracking if the exception hit user mode.
- *
- * - The hardirq tracer to keep the state consistent as low level ASM
- * entry disabled interrupts.
- *
- * As a precondition, this requires that the entry came from user mode,
- * idle, or a kernel context in which RCU is watching.
- *
- * For kernel mode entries RCU handling is done conditional. If RCU is
- * watching then the only RCU requirement is to check whether the tick has
- * to be restarted. If RCU is not watching then ct_irq_enter() has to be
- * invoked on entry and ct_irq_exit() on exit.
- *
- * Avoiding the ct_irq_enter/exit() calls is an optimization but also
- * solves the problem of kernel mode pagefaults which can schedule, which
- * is not possible after invoking ct_irq_enter() without undoing it.
- *
- * For user mode entries irqentry_enter_from_user_mode() is invoked to
- * establish the proper context for NOHZ_FULL. Otherwise scheduling on exit
- * would not be possible.
- *
- * Returns: An opaque object that must be passed to idtentry_exit()
- */
-irqentry_state_t noinstr irqentry_enter(struct pt_regs *regs);
-
-/**
- * irqentry_exit_cond_resched - Conditionally reschedule on return from interrupt
- *
- * Conditional reschedule with additional sanity checks.
- */
-void raw_irqentry_exit_cond_resched(void);
-#ifdef CONFIG_PREEMPT_DYNAMIC
-#if defined(CONFIG_HAVE_PREEMPT_DYNAMIC_CALL)
-#define irqentry_exit_cond_resched_dynamic_enabled raw_irqentry_exit_cond_resched
-#define irqentry_exit_cond_resched_dynamic_disabled NULL
-DECLARE_STATIC_CALL(irqentry_exit_cond_resched, raw_irqentry_exit_cond_resched);
-#define irqentry_exit_cond_resched() static_call(irqentry_exit_cond_resched)()
-#elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY)
-DECLARE_STATIC_KEY_TRUE(sk_dynamic_irqentry_exit_cond_resched);
-void dynamic_irqentry_exit_cond_resched(void);
-#define irqentry_exit_cond_resched() dynamic_irqentry_exit_cond_resched()
-#endif
-#else /* CONFIG_PREEMPT_DYNAMIC */
-#define irqentry_exit_cond_resched() raw_irqentry_exit_cond_resched()
-#endif /* CONFIG_PREEMPT_DYNAMIC */
-
-/**
- * irqentry_exit - Handle return from exception that used irqentry_enter()
- * @regs: Pointer to pt_regs (exception entry regs)
- * @state: Return value from matching call to irqentry_enter()
- *
- * Depending on the return target (kernel/user) this runs the necessary
- * preemption and work checks if possible and required and returns to
- * the caller with interrupts disabled and no further work pending.
- *
- * This is the last action before returning to the low level ASM code which
- * just needs to return to the appropriate context.
- *
- * Counterpart to irqentry_enter().
- */
-void noinstr irqentry_exit(struct pt_regs *regs, irqentry_state_t state);
-
-/**
- * irqentry_nmi_enter - Handle NMI entry
- * @regs: Pointer to currents pt_regs
- *
- * Similar to irqentry_enter() but taking care of the NMI constraints.
- */
-irqentry_state_t noinstr irqentry_nmi_enter(struct pt_regs *regs);
-
-/**
- * irqentry_nmi_exit - Handle return from NMI handling
- * @regs: Pointer to pt_regs (NMI entry regs)
- * @irq_state: Return value from matching call to irqentry_nmi_enter()
- *
- * Last action before returning to the low level assembly code.
- *
- * Counterpart to irqentry_nmi_enter().
- */
-void noinstr irqentry_nmi_exit(struct pt_regs *regs, irqentry_state_t irq_state);
-
#endif
diff --git a/include/linux/irq-entry-common.h b/include/linux/irq-entry-common.h
new file mode 100644
index 000000000000..8af374331900
--- /dev/null
+++ b/include/linux/irq-entry-common.h
@@ -0,0 +1,389 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+#ifndef __LINUX_IRQENTRYCOMMON_H
+#define __LINUX_IRQENTRYCOMMON_H
+
+#include <linux/static_call_types.h>
+#include <linux/syscalls.h>
+#include <linux/context_tracking.h>
+#include <linux/tick.h>
+#include <linux/kmsan.h>
+
+#include <asm/entry-common.h>
+
+/*
+ * Define dummy _TIF work flags if not defined by the architecture or for
+ * disabled functionality.
+ */
+#ifndef _TIF_PATCH_PENDING
+# define _TIF_PATCH_PENDING (0)
+#endif
+
+/*
+ * TIF flags handled in exit_to_user_mode_loop()
+ */
+#ifndef ARCH_EXIT_TO_USER_MODE_WORK
+# define ARCH_EXIT_TO_USER_MODE_WORK (0)
+#endif
+
+#define EXIT_TO_USER_MODE_WORK \
+ (_TIF_SIGPENDING | _TIF_NOTIFY_RESUME | _TIF_UPROBE | \
+ _TIF_NEED_RESCHED | _TIF_NEED_RESCHED_LAZY | \
+ _TIF_PATCH_PENDING | _TIF_NOTIFY_SIGNAL | \
+ ARCH_EXIT_TO_USER_MODE_WORK)
+
+/**
+ * arch_enter_from_user_mode - Architecture specific sanity check for user mode regs
+ * @regs: Pointer to currents pt_regs
+ *
+ * Defaults to an empty implementation. Can be replaced by architecture
+ * specific code.
+ *
+ * Invoked from syscall_enter_from_user_mode() in the non-instrumentable
+ * section. Use __always_inline so the compiler cannot push it out of line
+ * and make it instrumentable.
+ */
+static __always_inline void arch_enter_from_user_mode(struct pt_regs *regs);
+
+#ifndef arch_enter_from_user_mode
+static __always_inline void arch_enter_from_user_mode(struct pt_regs *regs) {}
+#endif
+
+/**
+ * enter_from_user_mode - Establish state when coming from user mode
+ *
+ * Syscall/interrupt entry disables interrupts, but user mode is traced as
+ * interrupts enabled. Also with NO_HZ_FULL RCU might be idle.
+ *
+ * 1) Tell lockdep that interrupts are disabled
+ * 2) Invoke context tracking if enabled to reactivate RCU
+ * 3) Trace interrupts off state
+ *
+ * Invoked from architecture specific syscall entry code with interrupts
+ * disabled. The calling code has to be non-instrumentable. When the
+ * function returns all state is correct and interrupts are still
+ * disabled. The subsequent functions can be instrumented.
+ *
+ * This is invoked when there is architecture specific functionality to be
+ * done between establishing state and enabling interrupts. The caller must
+ * enable interrupts before invoking syscall_enter_from_user_mode_work().
+ */
+static __always_inline void enter_from_user_mode(struct pt_regs *regs)
+{
+ arch_enter_from_user_mode(regs);
+ lockdep_hardirqs_off(CALLER_ADDR0);
+
+ CT_WARN_ON(__ct_state() != CT_STATE_USER);
+ user_exit_irqoff();
+
+ instrumentation_begin();
+ kmsan_unpoison_entry_regs(regs);
+ trace_hardirqs_off_finish();
+ instrumentation_end();
+}
+
+/**
+ * local_irq_enable_exit_to_user - Exit to user variant of local_irq_enable()
+ * @ti_work: Cached TIF flags gathered with interrupts disabled
+ *
+ * Defaults to local_irq_enable(). Can be supplied by architecture specific
+ * code.
+ */
+static inline void local_irq_enable_exit_to_user(unsigned long ti_work);
+
+#ifndef local_irq_enable_exit_to_user
+static inline void local_irq_enable_exit_to_user(unsigned long ti_work)
+{
+ local_irq_enable();
+}
+#endif
+
+/**
+ * local_irq_disable_exit_to_user - Exit to user variant of local_irq_disable()
+ *
+ * Defaults to local_irq_disable(). Can be supplied by architecture specific
+ * code.
+ */
+static inline void local_irq_disable_exit_to_user(void);
+
+#ifndef local_irq_disable_exit_to_user
+static inline void local_irq_disable_exit_to_user(void)
+{
+ local_irq_disable();
+}
+#endif
+
+/**
+ * arch_exit_to_user_mode_work - Architecture specific TIF work for exit
+ * to user mode.
+ * @regs: Pointer to currents pt_regs
+ * @ti_work: Cached TIF flags gathered with interrupts disabled
+ *
+ * Invoked from exit_to_user_mode_loop() with interrupt enabled
+ *
+ * Defaults to NOOP. Can be supplied by architecture specific code.
+ */
+static inline void arch_exit_to_user_mode_work(struct pt_regs *regs,
+ unsigned long ti_work);
+
+#ifndef arch_exit_to_user_mode_work
+static inline void arch_exit_to_user_mode_work(struct pt_regs *regs,
+ unsigned long ti_work)
+{
+}
+#endif
+
+/**
+ * arch_exit_to_user_mode_prepare - Architecture specific preparation for
+ * exit to user mode.
+ * @regs: Pointer to currents pt_regs
+ * @ti_work: Cached TIF flags gathered with interrupts disabled
+ *
+ * Invoked from exit_to_user_mode_prepare() with interrupt disabled as the last
+ * function before return. Defaults to NOOP.
+ */
+static inline void arch_exit_to_user_mode_prepare(struct pt_regs *regs,
+ unsigned long ti_work);
+
+#ifndef arch_exit_to_user_mode_prepare
+static inline void arch_exit_to_user_mode_prepare(struct pt_regs *regs,
+ unsigned long ti_work)
+{
+}
+#endif
+
+/**
+ * arch_exit_to_user_mode - Architecture specific final work before
+ * exit to user mode.
+ *
+ * Invoked from exit_to_user_mode() with interrupt disabled as the last
+ * function before return. Defaults to NOOP.
+ *
+ * This needs to be __always_inline because it is non-instrumentable code
+ * invoked after context tracking switched to user mode.
+ *
+ * An architecture implementation must not do anything complex, no locking
+ * etc. The main purpose is for speculation mitigations.
+ */
+static __always_inline void arch_exit_to_user_mode(void);
+
+#ifndef arch_exit_to_user_mode
+static __always_inline void arch_exit_to_user_mode(void) { }
+#endif
+
+/**
+ * arch_do_signal_or_restart - Architecture specific signal delivery function
+ * @regs: Pointer to currents pt_regs
+ *
+ * Invoked from exit_to_user_mode_loop().
+ */
+void arch_do_signal_or_restart(struct pt_regs *regs);
+
+/**
+ * exit_to_user_mode_loop - do any pending work before leaving to user space
+ */
+unsigned long exit_to_user_mode_loop(struct pt_regs *regs,
+ unsigned long ti_work);
+
+/**
+ * exit_to_user_mode_prepare - call exit_to_user_mode_loop() if required
+ * @regs: Pointer to pt_regs on entry stack
+ *
+ * 1) check that interrupts are disabled
+ * 2) call tick_nohz_user_enter_prepare()
+ * 3) call exit_to_user_mode_loop() if any flags from
+ * EXIT_TO_USER_MODE_WORK are set
+ * 4) check that interrupts are still disabled
+ */
+static __always_inline void exit_to_user_mode_prepare(struct pt_regs *regs)
+{
+ unsigned long ti_work;
+
+ lockdep_assert_irqs_disabled();
+
+ /* Flush pending rcuog wakeup before the last need_resched() check */
+ tick_nohz_user_enter_prepare();
+
+ ti_work = read_thread_flags();
+ if (unlikely(ti_work & EXIT_TO_USER_MODE_WORK))
+ ti_work = exit_to_user_mode_loop(regs, ti_work);
+
+ arch_exit_to_user_mode_prepare(regs, ti_work);
+
+ /* Ensure that kernel state is sane for a return to userspace */
+ kmap_assert_nomap();
+ lockdep_assert_irqs_disabled();
+ lockdep_sys_exit();
+}
+
+/**
+ * exit_to_user_mode - Fixup state when exiting to user mode
+ *
+ * Syscall/interrupt exit enables interrupts, but the kernel state is
+ * interrupts disabled when this is invoked. Also tell RCU about it.
+ *
+ * 1) Trace interrupts on state
+ * 2) Invoke context tracking if enabled to adjust RCU state
+ * 3) Invoke architecture specific last minute exit code, e.g. speculation
+ * mitigations, etc.: arch_exit_to_user_mode()
+ * 4) Tell lockdep that interrupts are enabled
+ *
+ * Invoked from architecture specific code when syscall_exit_to_user_mode()
+ * is not suitable as the last step before returning to userspace. Must be
+ * invoked with interrupts disabled and the caller must be
+ * non-instrumentable.
+ * The caller has to invoke syscall_exit_to_user_mode_work() before this.
+ */
+static __always_inline void exit_to_user_mode(void)
+{
+ instrumentation_begin();
+ trace_hardirqs_on_prepare();
+ lockdep_hardirqs_on_prepare();
+ instrumentation_end();
+
+ user_enter_irqoff();
+ arch_exit_to_user_mode();
+ lockdep_hardirqs_on(CALLER_ADDR0);
+}
+
+/**
+ * irqentry_enter_from_user_mode - Establish state before invoking the irq handler
+ * @regs: Pointer to currents pt_regs
+ *
+ * Invoked from architecture specific entry code with interrupts disabled.
+ * Can only be called when the interrupt entry came from user mode. The
+ * calling code must be non-instrumentable. When the function returns all
+ * state is correct and the subsequent functions can be instrumented.
+ *
+ * The function establishes state (lockdep, RCU (context tracking), tracing)
+ */
+void irqentry_enter_from_user_mode(struct pt_regs *regs);
+
+/**
+ * irqentry_exit_to_user_mode - Interrupt exit work
+ * @regs: Pointer to current's pt_regs
+ *
+ * Invoked with interrupts disabled and fully valid regs. Returns with all
+ * work handled, interrupts disabled such that the caller can immediately
+ * switch to user mode. Called from architecture specific interrupt
+ * handling code.
+ *
+ * The call order is #2 and #3 as described in syscall_exit_to_user_mode().
+ * Interrupt exit is not invoking #1 which is the syscall specific one time
+ * work.
+ */
+void irqentry_exit_to_user_mode(struct pt_regs *regs);
+
+#ifndef irqentry_state
+/**
+ * struct irqentry_state - Opaque object for exception state storage
+ * @exit_rcu: Used exclusively in the irqentry_*() calls; signals whether the
+ * exit path has to invoke ct_irq_exit().
+ * @lockdep: Used exclusively in the irqentry_nmi_*() calls; ensures that
+ * lockdep state is restored correctly on exit from nmi.
+ *
+ * This opaque object is filled in by the irqentry_*_enter() functions and
+ * must be passed back into the corresponding irqentry_*_exit() functions
+ * when the exception is complete.
+ *
+ * Callers of irqentry_*_[enter|exit]() must consider this structure opaque
+ * and all members private. Descriptions of the members are provided to aid in
+ * the maintenance of the irqentry_*() functions.
+ */
+typedef struct irqentry_state {
+ union {
+ bool exit_rcu;
+ bool lockdep;
+ };
+} irqentry_state_t;
+#endif
+
+/**
+ * irqentry_enter - Handle state tracking on ordinary interrupt entries
+ * @regs: Pointer to pt_regs of interrupted context
+ *
+ * Invokes:
+ * - lockdep irqflag state tracking as low level ASM entry disabled
+ * interrupts.
+ *
+ * - Context tracking if the exception hit user mode.
+ *
+ * - The hardirq tracer to keep the state consistent as low level ASM
+ * entry disabled interrupts.
+ *
+ * As a precondition, this requires that the entry came from user mode,
+ * idle, or a kernel context in which RCU is watching.
+ *
+ * For kernel mode entries RCU handling is done conditional. If RCU is
+ * watching then the only RCU requirement is to check whether the tick has
+ * to be restarted. If RCU is not watching then ct_irq_enter() has to be
+ * invoked on entry and ct_irq_exit() on exit.
+ *
+ * Avoiding the ct_irq_enter/exit() calls is an optimization but also
+ * solves the problem of kernel mode pagefaults which can schedule, which
+ * is not possible after invoking ct_irq_enter() without undoing it.
+ *
+ * For user mode entries irqentry_enter_from_user_mode() is invoked to
+ * establish the proper context for NOHZ_FULL. Otherwise scheduling on exit
+ * would not be possible.
+ *
+ * Returns: An opaque object that must be passed to idtentry_exit()
+ */
+irqentry_state_t noinstr irqentry_enter(struct pt_regs *regs);
+
+/**
+ * irqentry_exit_cond_resched - Conditionally reschedule on return from interrupt
+ *
+ * Conditional reschedule with additional sanity checks.
+ */
+void raw_irqentry_exit_cond_resched(void);
+#ifdef CONFIG_PREEMPT_DYNAMIC
+#if defined(CONFIG_HAVE_PREEMPT_DYNAMIC_CALL)
+#define irqentry_exit_cond_resched_dynamic_enabled raw_irqentry_exit_cond_resched
+#define irqentry_exit_cond_resched_dynamic_disabled NULL
+DECLARE_STATIC_CALL(irqentry_exit_cond_resched, raw_irqentry_exit_cond_resched);
+#define irqentry_exit_cond_resched() static_call(irqentry_exit_cond_resched)()
+#elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY)
+DECLARE_STATIC_KEY_TRUE(sk_dynamic_irqentry_exit_cond_resched);
+void dynamic_irqentry_exit_cond_resched(void);
+#define irqentry_exit_cond_resched() dynamic_irqentry_exit_cond_resched()
+#endif
+#else /* CONFIG_PREEMPT_DYNAMIC */
+#define irqentry_exit_cond_resched() raw_irqentry_exit_cond_resched()
+#endif /* CONFIG_PREEMPT_DYNAMIC */
+
+/**
+ * irqentry_exit - Handle return from exception that used irqentry_enter()
+ * @regs: Pointer to pt_regs (exception entry regs)
+ * @state: Return value from matching call to irqentry_enter()
+ *
+ * Depending on the return target (kernel/user) this runs the necessary
+ * preemption and work checks if possible and required and returns to
+ * the caller with interrupts disabled and no further work pending.
+ *
+ * This is the last action before returning to the low level ASM code which
+ * just needs to return to the appropriate context.
+ *
+ * Counterpart to irqentry_enter().
+ */
+void noinstr irqentry_exit(struct pt_regs *regs, irqentry_state_t state);
+
+/**
+ * irqentry_nmi_enter - Handle NMI entry
+ * @regs: Pointer to currents pt_regs
+ *
+ * Similar to irqentry_enter() but taking care of the NMI constraints.
+ */
+irqentry_state_t noinstr irqentry_nmi_enter(struct pt_regs *regs);
+
+/**
+ * irqentry_nmi_exit - Handle return from NMI handling
+ * @regs: Pointer to pt_regs (NMI entry regs)
+ * @irq_state: Return value from matching call to irqentry_nmi_enter()
+ *
+ * Last action before returning to the low level assembly code.
+ *
+ * Counterpart to irqentry_nmi_enter().
+ */
+void noinstr irqentry_nmi_exit(struct pt_regs *regs, irqentry_state_t irq_state);
+
+#endif
diff --git a/kernel/entry/Makefile b/kernel/entry/Makefile
index 095c775e001e..d38f3a7e7396 100644
--- a/kernel/entry/Makefile
+++ b/kernel/entry/Makefile
@@ -9,5 +9,6 @@ KCOV_INSTRUMENT := n
CFLAGS_REMOVE_common.o = -fstack-protector -fstack-protector-strong
CFLAGS_common.o += -fno-stack-protector
-obj-$(CONFIG_GENERIC_ENTRY) += common.o syscall_user_dispatch.o
+obj-$(CONFIG_GENERIC_IRQ_ENTRY) += common.o
+obj-$(CONFIG_GENERIC_SYSCALL) += syscall-common.o syscall_user_dispatch.o
obj-$(CONFIG_KVM_XFER_TO_GUEST_WORK) += kvm.o
diff --git a/kernel/entry/common.c b/kernel/entry/common.c
index e33691d5adf7..b82032777310 100644
--- a/kernel/entry/common.c
+++ b/kernel/entry/common.c
@@ -1,84 +1,13 @@
// SPDX-License-Identifier: GPL-2.0
-#include <linux/context_tracking.h>
-#include <linux/entry-common.h>
+#include <linux/irq-entry-common.h>
#include <linux/resume_user_mode.h>
#include <linux/highmem.h>
#include <linux/jump_label.h>
#include <linux/kmsan.h>
#include <linux/livepatch.h>
-#include <linux/audit.h>
#include <linux/tick.h>
-#include "common.h"
-
-#define CREATE_TRACE_POINTS
-#include <trace/events/syscalls.h>
-
-static inline void syscall_enter_audit(struct pt_regs *regs, long syscall)
-{
- if (unlikely(audit_context())) {
- unsigned long args[6];
-
- syscall_get_arguments(current, regs, args);
- audit_syscall_entry(syscall, args[0], args[1], args[2], args[3]);
- }
-}
-
-long syscall_trace_enter(struct pt_regs *regs, long syscall,
- unsigned long work)
-{
- long ret = 0;
-
- /*
- * Handle Syscall User Dispatch. This must comes first, since
- * the ABI here can be something that doesn't make sense for
- * other syscall_work features.
- */
- if (work & SYSCALL_WORK_SYSCALL_USER_DISPATCH) {
- if (syscall_user_dispatch(regs))
- return -1L;
- }
-
- /* Handle ptrace */
- if (work & (SYSCALL_WORK_SYSCALL_TRACE | SYSCALL_WORK_SYSCALL_EMU)) {
- ret = ptrace_report_syscall_entry(regs);
- if (ret || (work & SYSCALL_WORK_SYSCALL_EMU))
- return -1L;
- }
-
- /* Do seccomp after ptrace, to catch any tracer changes. */
- if (work & SYSCALL_WORK_SECCOMP) {
- ret = __secure_computing(NULL);
- if (ret == -1L)
- return ret;
- }
-
- /* Either of the above might have changed the syscall number */
- syscall = syscall_get_nr(current, regs);
-
- if (unlikely(work & SYSCALL_WORK_SYSCALL_TRACEPOINT)) {
- trace_sys_enter(regs, syscall);
- /*
- * Probes or BPF hooks in the tracepoint may have changed the
- * system call number as well.
- */
- syscall = syscall_get_nr(current, regs);
- }
-
- syscall_enter_audit(regs, syscall);
-
- return ret ? : syscall;
-}
-
-noinstr void syscall_enter_from_user_mode_prepare(struct pt_regs *regs)
-{
- enter_from_user_mode(regs);
- instrumentation_begin();
- local_irq_enable();
- instrumentation_end();
-}
-
/* Workaround to allow gradual conversion of architecture code */
void __weak arch_do_signal_or_restart(struct pt_regs *regs) { }
@@ -133,93 +62,6 @@ __always_inline unsigned long exit_to_user_mode_loop(struct pt_regs *regs,
return ti_work;
}
-/*
- * If SYSCALL_EMU is set, then the only reason to report is when
- * SINGLESTEP is set (i.e. PTRACE_SYSEMU_SINGLESTEP). This syscall
- * instruction has been already reported in syscall_enter_from_user_mode().
- */
-static inline bool report_single_step(unsigned long work)
-{
- if (work & SYSCALL_WORK_SYSCALL_EMU)
- return false;
-
- return work & SYSCALL_WORK_SYSCALL_EXIT_TRAP;
-}
-
-static void syscall_exit_work(struct pt_regs *regs, unsigned long work)
-{
- bool step;
-
- /*
- * If the syscall was rolled back due to syscall user dispatching,
- * then the tracers below are not invoked for the same reason as
- * the entry side was not invoked in syscall_trace_enter(): The ABI
- * of these syscalls is unknown.
- */
- if (work & SYSCALL_WORK_SYSCALL_USER_DISPATCH) {
- if (unlikely(current->syscall_dispatch.on_dispatch)) {
- current->syscall_dispatch.on_dispatch = false;
- return;
- }
- }
-
- audit_syscall_exit(regs);
-
- if (work & SYSCALL_WORK_SYSCALL_TRACEPOINT)
- trace_sys_exit(regs, syscall_get_return_value(current, regs));
-
- step = report_single_step(work);
- if (step || work & SYSCALL_WORK_SYSCALL_TRACE)
- ptrace_report_syscall_exit(regs, step);
-}
-
-/*
- * Syscall specific exit to user mode preparation. Runs with interrupts
- * enabled.
- */
-static void syscall_exit_to_user_mode_prepare(struct pt_regs *regs)
-{
- unsigned long work = READ_ONCE(current_thread_info()->syscall_work);
- unsigned long nr = syscall_get_nr(current, regs);
-
- CT_WARN_ON(ct_state() != CT_STATE_KERNEL);
-
- if (IS_ENABLED(CONFIG_PROVE_LOCKING)) {
- if (WARN(irqs_disabled(), "syscall %lu left IRQs disabled", nr))
- local_irq_enable();
- }
-
- rseq_syscall(regs);
-
- /*
- * Do one-time syscall specific work. If these work items are
- * enabled, we want to run them exactly once per syscall exit with
- * interrupts enabled.
- */
- if (unlikely(work & SYSCALL_WORK_EXIT))
- syscall_exit_work(regs, work);
-}
-
-static __always_inline void __syscall_exit_to_user_mode_work(struct pt_regs *regs)
-{
- syscall_exit_to_user_mode_prepare(regs);
- local_irq_disable_exit_to_user();
- exit_to_user_mode_prepare(regs);
-}
-
-void syscall_exit_to_user_mode_work(struct pt_regs *regs)
-{
- __syscall_exit_to_user_mode_work(regs);
-}
-
-__visible noinstr void syscall_exit_to_user_mode(struct pt_regs *regs)
-{
- instrumentation_begin();
- __syscall_exit_to_user_mode_work(regs);
- instrumentation_end();
- exit_to_user_mode();
-}
-
noinstr void irqentry_enter_from_user_mode(struct pt_regs *regs)
{
enter_from_user_mode(regs);
diff --git a/kernel/entry/syscall-common.c b/kernel/entry/syscall-common.c
new file mode 100644
index 000000000000..0eb036986ad4
--- /dev/null
+++ b/kernel/entry/syscall-common.c
@@ -0,0 +1,159 @@
+// SPDX-License-Identifier: GPL-2.0
+
+#include <linux/audit.h>
+#include <linux/entry-common.h>
+#include "common.h"
+
+#define CREATE_TRACE_POINTS
+#include <trace/events/syscalls.h>
+
+static inline void syscall_enter_audit(struct pt_regs *regs, long syscall)
+{
+ if (unlikely(audit_context())) {
+ unsigned long args[6];
+
+ syscall_get_arguments(current, regs, args);
+ audit_syscall_entry(syscall, args[0], args[1], args[2], args[3]);
+ }
+}
+
+long syscall_trace_enter(struct pt_regs *regs, long syscall,
+ unsigned long work)
+{
+ long ret = 0;
+
+ /*
+ * Handle Syscall User Dispatch. This must comes first, since
+ * the ABI here can be something that doesn't make sense for
+ * other syscall_work features.
+ */
+ if (work & SYSCALL_WORK_SYSCALL_USER_DISPATCH) {
+ if (syscall_user_dispatch(regs))
+ return -1L;
+ }
+
+ /* Handle ptrace */
+ if (work & (SYSCALL_WORK_SYSCALL_TRACE | SYSCALL_WORK_SYSCALL_EMU)) {
+ ret = ptrace_report_syscall_entry(regs);
+ if (ret || (work & SYSCALL_WORK_SYSCALL_EMU))
+ return -1L;
+ }
+
+ /* Do seccomp after ptrace, to catch any tracer changes. */
+ if (work & SYSCALL_WORK_SECCOMP) {
+ ret = __secure_computing(NULL);
+ if (ret == -1L)
+ return ret;
+ }
+
+ /* Either of the above might have changed the syscall number */
+ syscall = syscall_get_nr(current, regs);
+
+ if (unlikely(work & SYSCALL_WORK_SYSCALL_TRACEPOINT)) {
+ trace_sys_enter(regs, syscall);
+ /*
+ * Probes or BPF hooks in the tracepoint may have changed the
+ * system call number as well.
+ */
+ syscall = syscall_get_nr(current, regs);
+ }
+
+ syscall_enter_audit(regs, syscall);
+
+ return ret ? : syscall;
+}
+
+noinstr void syscall_enter_from_user_mode_prepare(struct pt_regs *regs)
+{
+ enter_from_user_mode(regs);
+ instrumentation_begin();
+ local_irq_enable();
+ instrumentation_end();
+}
+
+/*
+ * If SYSCALL_EMU is set, then the only reason to report is when
+ * SINGLESTEP is set (i.e. PTRACE_SYSEMU_SINGLESTEP). This syscall
+ * instruction has been already reported in syscall_enter_from_user_mode().
+ */
+static inline bool report_single_step(unsigned long work)
+{
+ if (work & SYSCALL_WORK_SYSCALL_EMU)
+ return false;
+
+ return work & SYSCALL_WORK_SYSCALL_EXIT_TRAP;
+}
+
+static void syscall_exit_work(struct pt_regs *regs, unsigned long work)
+{
+ bool step;
+
+ /*
+ * If the syscall was rolled back due to syscall user dispatching,
+ * then the tracers below are not invoked for the same reason as
+ * the entry side was not invoked in syscall_trace_enter(): The ABI
+ * of these syscalls is unknown.
+ */
+ if (work & SYSCALL_WORK_SYSCALL_USER_DISPATCH) {
+ if (unlikely(current->syscall_dispatch.on_dispatch)) {
+ current->syscall_dispatch.on_dispatch = false;
+ return;
+ }
+ }
+
+ audit_syscall_exit(regs);
+
+ if (work & SYSCALL_WORK_SYSCALL_TRACEPOINT)
+ trace_sys_exit(regs, syscall_get_return_value(current, regs));
+
+ step = report_single_step(work);
+ if (step || work & SYSCALL_WORK_SYSCALL_TRACE)
+ ptrace_report_syscall_exit(regs, step);
+}
+
+/*
+ * Syscall specific exit to user mode preparation. Runs with interrupts
+ * enabled.
+ */
+static void syscall_exit_to_user_mode_prepare(struct pt_regs *regs)
+{
+ unsigned long work = READ_ONCE(current_thread_info()->syscall_work);
+ unsigned long nr = syscall_get_nr(current, regs);
+
+ CT_WARN_ON(ct_state() != CT_STATE_KERNEL);
+
+ if (IS_ENABLED(CONFIG_PROVE_LOCKING)) {
+ if (WARN(irqs_disabled(), "syscall %lu left IRQs disabled", nr))
+ local_irq_enable();
+ }
+
+ rseq_syscall(regs);
+
+ /*
+ * Do one-time syscall specific work. If these work items are
+ * enabled, we want to run them exactly once per syscall exit with
+ * interrupts enabled.
+ */
+ if (unlikely(work & SYSCALL_WORK_EXIT))
+ syscall_exit_work(regs, work);
+}
+
+static __always_inline void __syscall_exit_to_user_mode_work(struct pt_regs *regs)
+{
+ syscall_exit_to_user_mode_prepare(regs);
+ local_irq_disable_exit_to_user();
+ exit_to_user_mode_prepare(regs);
+}
+
+void syscall_exit_to_user_mode_work(struct pt_regs *regs)
+{
+ __syscall_exit_to_user_mode_work(regs);
+}
+
+__visible noinstr void syscall_exit_to_user_mode(struct pt_regs *regs)
+{
+ instrumentation_begin();
+ __syscall_exit_to_user_mode_work(regs);
+ instrumentation_end();
+ exit_to_user_mode();
+}
diff --git a/kernel/sched/core.c b/kernel/sched/core.c
index 27a8fbd58091..2d560bb3efaa 100644
--- a/kernel/sched/core.c
+++ b/kernel/sched/core.c
@@ -68,8 +68,8 @@
#include <linux/workqueue_api.h>
#ifdef CONFIG_PREEMPT_DYNAMIC
-# ifdef CONFIG_GENERIC_ENTRY
-# include <linux/entry-common.h>
+# ifdef CONFIG_GENERIC_IRQ_ENTRY
+# include <linux/irq-entry-common.h>
# endif
#endif
@@ -7398,8 +7398,8 @@ EXPORT_SYMBOL(__cond_resched_rwlock_write);
#ifdef CONFIG_PREEMPT_DYNAMIC
-#ifdef CONFIG_GENERIC_ENTRY
-#include <linux/entry-common.h>
+#ifdef CONFIG_GENERIC_IRQ_ENTRY
+#include <linux/irq-entry-common.h>
#endif
/*
--
2.34.1
^ permalink raw reply related [flat|nested] 3+ messages in thread* Re: [PATCH -next v5 09/22] entry: Split generic entry into irq and syscall
2024-12-06 10:17 ` [PATCH -next v5 09/22] entry: Split generic entry into irq and syscall Jinjie Ruan
@ 2025-02-10 12:04 ` Mark Rutland
0 siblings, 0 replies; 3+ messages in thread
From: Mark Rutland @ 2025-02-10 12:04 UTC (permalink / raw)
To: Jinjie Ruan, tglx
Cc: catalin.marinas, will, oleg, sstabellini, peterz, luto, mingo,
juri.lelli, vincent.guittot, dietmar.eggemann, rostedt, bsegall,
mgorman, vschneid, kees, wad, akpm, samitolvanen, masahiroy, hca,
aliceryhl, rppt, xur, paulmck, arnd, mbenes, puranjay, pcc, ardb,
sudeep.holla, guohanjun, rafael, liuwei09, dwmw, Jonathan.Cameron,
liaochang1, kristina.martsenko, ptosi, broonie, thiago.bauermann,
kevin.brodsky, joey.gouly, liuyuntao12, leobras, linux-kernel,
linux-arm-kernel, xen-devel
On Fri, Dec 06, 2024 at 06:17:31PM +0800, Jinjie Ruan wrote:
> As Mark pointed out, do not try to switch to *all* the
> generic entry code in one go. The regular entry state management
> (e.g. enter_from_user_mode() and exit_to_user_mode()) is largely
> separate from the syscall state management. Move arm64 over to
> enter_from_user_mode() and exit_to_user_mode() without needing to use
> any of the generic syscall logic. Doing that first, *then* moving over
> to the generic syscall handling would be much easier to
> review/test/bisect, and if there are any ABI issues with the syscall
> handling in particular, it will be easier to handle those in isolation.
>
> So split generic entry into irq entry and syscall code, which will
> make review work easier and switch to generic entry clear.
> Introdue two configs called GENERIC_SYSCALL and GENERIC_IRQ_ENTRY,
> which control the irq entry and syscall parts of the generic code
> respectively. And split the header file irq-entry-common.h from
> entry-common.h for GENERIC_IRQ_ENTRY.
I think this would be simpler and clearer as:
| Currently CONFIG_GENERIC_ENTRY enables both the generic exception
| entry logic and the generic syscall entry logic, which are otherwise
| loosely coupled.
|
| Introduce separate config options for these so that archtiectures can
| select the two independently. This will make it easier for
| architectures to migrate to generic entry code.
It would be good to have this *before* the arm64 changes, either at the
start of the series or upstreamed earlier.
Thomas, can you confirm whether you're happy with splitting this up?
As above, the thinking is that we can easily/quickly move arm64 over to
the generic exception/irq entry code, but the syscall changes have a
much bigger potential impact (e.g. we've had lots of fun historically
with the ptrace state machine), and I'd like to handle the syscall
changes as a follow-up.
Mark.
> Suggested-by: Mark Rutland <mark.rutland@arm.com>
> Signed-off-by: Jinjie Ruan <ruanjinjie@huawei.com>
> ---
> MAINTAINERS | 1 +
> arch/Kconfig | 8 +
> include/linux/entry-common.h | 382 +-----------------------------
> include/linux/irq-entry-common.h | 389 +++++++++++++++++++++++++++++++
> kernel/entry/Makefile | 3 +-
> kernel/entry/common.c | 160 +------------
> kernel/entry/syscall-common.c | 159 +++++++++++++
> kernel/sched/core.c | 8 +-
> 8 files changed, 565 insertions(+), 545 deletions(-)
> create mode 100644 include/linux/irq-entry-common.h
> create mode 100644 kernel/entry/syscall-common.c
>
> diff --git a/MAINTAINERS b/MAINTAINERS
> index 21f855fe468b..7a6e87587101 100644
> --- a/MAINTAINERS
> +++ b/MAINTAINERS
> @@ -9585,6 +9585,7 @@ S: Maintained
> T: git git://git.kernel.org/pub/scm/linux/kernel/git/tip/tip.git core/entry
> F: include/linux/entry-common.h
> F: include/linux/entry-kvm.h
> +F: include/linux/irq-entry-common.h
> F: kernel/entry/
>
> GENERIC GPIO I2C DRIVER
> diff --git a/arch/Kconfig b/arch/Kconfig
> index 6682b2a53e34..5a454eff780b 100644
> --- a/arch/Kconfig
> +++ b/arch/Kconfig
> @@ -64,8 +64,16 @@ config HOTPLUG_PARALLEL
> bool
> select HOTPLUG_SPLIT_STARTUP
>
> +config GENERIC_IRQ_ENTRY
> + bool
> +
> +config GENERIC_SYSCALL
> + bool
> +
> config GENERIC_ENTRY
> bool
> + select GENERIC_IRQ_ENTRY
> + select GENERIC_SYSCALL
>
> config KPROBES
> bool "Kprobes"
> diff --git a/include/linux/entry-common.h b/include/linux/entry-common.h
> index fc61d0205c97..b3233e8328c5 100644
> --- a/include/linux/entry-common.h
> +++ b/include/linux/entry-common.h
> @@ -2,27 +2,15 @@
> #ifndef __LINUX_ENTRYCOMMON_H
> #define __LINUX_ENTRYCOMMON_H
>
> -#include <linux/static_call_types.h>
> +#include <linux/irq-entry-common.h>
> #include <linux/ptrace.h>
> -#include <linux/syscalls.h>
> #include <linux/seccomp.h>
> #include <linux/sched.h>
> -#include <linux/context_tracking.h>
> #include <linux/livepatch.h>
> #include <linux/resume_user_mode.h>
> -#include <linux/tick.h>
> -#include <linux/kmsan.h>
>
> #include <asm/entry-common.h>
>
> -/*
> - * Define dummy _TIF work flags if not defined by the architecture or for
> - * disabled functionality.
> - */
> -#ifndef _TIF_PATCH_PENDING
> -# define _TIF_PATCH_PENDING (0)
> -#endif
> -
> #ifndef _TIF_UPROBE
> # define _TIF_UPROBE (0)
> #endif
> @@ -55,69 +43,6 @@
> SYSCALL_WORK_SYSCALL_EXIT_TRAP | \
> ARCH_SYSCALL_WORK_EXIT)
>
> -/*
> - * TIF flags handled in exit_to_user_mode_loop()
> - */
> -#ifndef ARCH_EXIT_TO_USER_MODE_WORK
> -# define ARCH_EXIT_TO_USER_MODE_WORK (0)
> -#endif
> -
> -#define EXIT_TO_USER_MODE_WORK \
> - (_TIF_SIGPENDING | _TIF_NOTIFY_RESUME | _TIF_UPROBE | \
> - _TIF_NEED_RESCHED | _TIF_NEED_RESCHED_LAZY | \
> - _TIF_PATCH_PENDING | _TIF_NOTIFY_SIGNAL | \
> - ARCH_EXIT_TO_USER_MODE_WORK)
> -
> -/**
> - * arch_enter_from_user_mode - Architecture specific sanity check for user mode regs
> - * @regs: Pointer to currents pt_regs
> - *
> - * Defaults to an empty implementation. Can be replaced by architecture
> - * specific code.
> - *
> - * Invoked from syscall_enter_from_user_mode() in the non-instrumentable
> - * section. Use __always_inline so the compiler cannot push it out of line
> - * and make it instrumentable.
> - */
> -static __always_inline void arch_enter_from_user_mode(struct pt_regs *regs);
> -
> -#ifndef arch_enter_from_user_mode
> -static __always_inline void arch_enter_from_user_mode(struct pt_regs *regs) {}
> -#endif
> -
> -/**
> - * enter_from_user_mode - Establish state when coming from user mode
> - *
> - * Syscall/interrupt entry disables interrupts, but user mode is traced as
> - * interrupts enabled. Also with NO_HZ_FULL RCU might be idle.
> - *
> - * 1) Tell lockdep that interrupts are disabled
> - * 2) Invoke context tracking if enabled to reactivate RCU
> - * 3) Trace interrupts off state
> - *
> - * Invoked from architecture specific syscall entry code with interrupts
> - * disabled. The calling code has to be non-instrumentable. When the
> - * function returns all state is correct and interrupts are still
> - * disabled. The subsequent functions can be instrumented.
> - *
> - * This is invoked when there is architecture specific functionality to be
> - * done between establishing state and enabling interrupts. The caller must
> - * enable interrupts before invoking syscall_enter_from_user_mode_work().
> - */
> -static __always_inline void enter_from_user_mode(struct pt_regs *regs)
> -{
> - arch_enter_from_user_mode(regs);
> - lockdep_hardirqs_off(CALLER_ADDR0);
> -
> - CT_WARN_ON(__ct_state() != CT_STATE_USER);
> - user_exit_irqoff();
> -
> - instrumentation_begin();
> - kmsan_unpoison_entry_regs(regs);
> - trace_hardirqs_off_finish();
> - instrumentation_end();
> -}
> -
> /**
> * syscall_enter_from_user_mode_prepare - Establish state and enable interrupts
> * @regs: Pointer to currents pt_regs
> @@ -202,170 +127,6 @@ static __always_inline long syscall_enter_from_user_mode(struct pt_regs *regs, l
> return ret;
> }
>
> -/**
> - * local_irq_enable_exit_to_user - Exit to user variant of local_irq_enable()
> - * @ti_work: Cached TIF flags gathered with interrupts disabled
> - *
> - * Defaults to local_irq_enable(). Can be supplied by architecture specific
> - * code.
> - */
> -static inline void local_irq_enable_exit_to_user(unsigned long ti_work);
> -
> -#ifndef local_irq_enable_exit_to_user
> -static inline void local_irq_enable_exit_to_user(unsigned long ti_work)
> -{
> - local_irq_enable();
> -}
> -#endif
> -
> -/**
> - * local_irq_disable_exit_to_user - Exit to user variant of local_irq_disable()
> - *
> - * Defaults to local_irq_disable(). Can be supplied by architecture specific
> - * code.
> - */
> -static inline void local_irq_disable_exit_to_user(void);
> -
> -#ifndef local_irq_disable_exit_to_user
> -static inline void local_irq_disable_exit_to_user(void)
> -{
> - local_irq_disable();
> -}
> -#endif
> -
> -/**
> - * arch_exit_to_user_mode_work - Architecture specific TIF work for exit
> - * to user mode.
> - * @regs: Pointer to currents pt_regs
> - * @ti_work: Cached TIF flags gathered with interrupts disabled
> - *
> - * Invoked from exit_to_user_mode_loop() with interrupt enabled
> - *
> - * Defaults to NOOP. Can be supplied by architecture specific code.
> - */
> -static inline void arch_exit_to_user_mode_work(struct pt_regs *regs,
> - unsigned long ti_work);
> -
> -#ifndef arch_exit_to_user_mode_work
> -static inline void arch_exit_to_user_mode_work(struct pt_regs *regs,
> - unsigned long ti_work)
> -{
> -}
> -#endif
> -
> -/**
> - * arch_exit_to_user_mode_prepare - Architecture specific preparation for
> - * exit to user mode.
> - * @regs: Pointer to currents pt_regs
> - * @ti_work: Cached TIF flags gathered with interrupts disabled
> - *
> - * Invoked from exit_to_user_mode_prepare() with interrupt disabled as the last
> - * function before return. Defaults to NOOP.
> - */
> -static inline void arch_exit_to_user_mode_prepare(struct pt_regs *regs,
> - unsigned long ti_work);
> -
> -#ifndef arch_exit_to_user_mode_prepare
> -static inline void arch_exit_to_user_mode_prepare(struct pt_regs *regs,
> - unsigned long ti_work)
> -{
> -}
> -#endif
> -
> -/**
> - * arch_exit_to_user_mode - Architecture specific final work before
> - * exit to user mode.
> - *
> - * Invoked from exit_to_user_mode() with interrupt disabled as the last
> - * function before return. Defaults to NOOP.
> - *
> - * This needs to be __always_inline because it is non-instrumentable code
> - * invoked after context tracking switched to user mode.
> - *
> - * An architecture implementation must not do anything complex, no locking
> - * etc. The main purpose is for speculation mitigations.
> - */
> -static __always_inline void arch_exit_to_user_mode(void);
> -
> -#ifndef arch_exit_to_user_mode
> -static __always_inline void arch_exit_to_user_mode(void) { }
> -#endif
> -
> -/**
> - * arch_do_signal_or_restart - Architecture specific signal delivery function
> - * @regs: Pointer to currents pt_regs
> - *
> - * Invoked from exit_to_user_mode_loop().
> - */
> -void arch_do_signal_or_restart(struct pt_regs *regs);
> -
> -/**
> - * exit_to_user_mode_loop - do any pending work before leaving to user space
> - */
> -unsigned long exit_to_user_mode_loop(struct pt_regs *regs,
> - unsigned long ti_work);
> -
> -/**
> - * exit_to_user_mode_prepare - call exit_to_user_mode_loop() if required
> - * @regs: Pointer to pt_regs on entry stack
> - *
> - * 1) check that interrupts are disabled
> - * 2) call tick_nohz_user_enter_prepare()
> - * 3) call exit_to_user_mode_loop() if any flags from
> - * EXIT_TO_USER_MODE_WORK are set
> - * 4) check that interrupts are still disabled
> - */
> -static __always_inline void exit_to_user_mode_prepare(struct pt_regs *regs)
> -{
> - unsigned long ti_work;
> -
> - lockdep_assert_irqs_disabled();
> -
> - /* Flush pending rcuog wakeup before the last need_resched() check */
> - tick_nohz_user_enter_prepare();
> -
> - ti_work = read_thread_flags();
> - if (unlikely(ti_work & EXIT_TO_USER_MODE_WORK))
> - ti_work = exit_to_user_mode_loop(regs, ti_work);
> -
> - arch_exit_to_user_mode_prepare(regs, ti_work);
> -
> - /* Ensure that kernel state is sane for a return to userspace */
> - kmap_assert_nomap();
> - lockdep_assert_irqs_disabled();
> - lockdep_sys_exit();
> -}
> -
> -/**
> - * exit_to_user_mode - Fixup state when exiting to user mode
> - *
> - * Syscall/interrupt exit enables interrupts, but the kernel state is
> - * interrupts disabled when this is invoked. Also tell RCU about it.
> - *
> - * 1) Trace interrupts on state
> - * 2) Invoke context tracking if enabled to adjust RCU state
> - * 3) Invoke architecture specific last minute exit code, e.g. speculation
> - * mitigations, etc.: arch_exit_to_user_mode()
> - * 4) Tell lockdep that interrupts are enabled
> - *
> - * Invoked from architecture specific code when syscall_exit_to_user_mode()
> - * is not suitable as the last step before returning to userspace. Must be
> - * invoked with interrupts disabled and the caller must be
> - * non-instrumentable.
> - * The caller has to invoke syscall_exit_to_user_mode_work() before this.
> - */
> -static __always_inline void exit_to_user_mode(void)
> -{
> - instrumentation_begin();
> - trace_hardirqs_on_prepare();
> - lockdep_hardirqs_on_prepare();
> - instrumentation_end();
> -
> - user_enter_irqoff();
> - arch_exit_to_user_mode();
> - lockdep_hardirqs_on(CALLER_ADDR0);
> -}
> -
> /**
> * syscall_exit_to_user_mode_work - Handle work before returning to user mode
> * @regs: Pointer to currents pt_regs
> @@ -412,145 +173,4 @@ void syscall_exit_to_user_mode_work(struct pt_regs *regs);
> */
> void syscall_exit_to_user_mode(struct pt_regs *regs);
>
> -/**
> - * irqentry_enter_from_user_mode - Establish state before invoking the irq handler
> - * @regs: Pointer to currents pt_regs
> - *
> - * Invoked from architecture specific entry code with interrupts disabled.
> - * Can only be called when the interrupt entry came from user mode. The
> - * calling code must be non-instrumentable. When the function returns all
> - * state is correct and the subsequent functions can be instrumented.
> - *
> - * The function establishes state (lockdep, RCU (context tracking), tracing)
> - */
> -void irqentry_enter_from_user_mode(struct pt_regs *regs);
> -
> -/**
> - * irqentry_exit_to_user_mode - Interrupt exit work
> - * @regs: Pointer to current's pt_regs
> - *
> - * Invoked with interrupts disabled and fully valid regs. Returns with all
> - * work handled, interrupts disabled such that the caller can immediately
> - * switch to user mode. Called from architecture specific interrupt
> - * handling code.
> - *
> - * The call order is #2 and #3 as described in syscall_exit_to_user_mode().
> - * Interrupt exit is not invoking #1 which is the syscall specific one time
> - * work.
> - */
> -void irqentry_exit_to_user_mode(struct pt_regs *regs);
> -
> -#ifndef irqentry_state
> -/**
> - * struct irqentry_state - Opaque object for exception state storage
> - * @exit_rcu: Used exclusively in the irqentry_*() calls; signals whether the
> - * exit path has to invoke ct_irq_exit().
> - * @lockdep: Used exclusively in the irqentry_nmi_*() calls; ensures that
> - * lockdep state is restored correctly on exit from nmi.
> - *
> - * This opaque object is filled in by the irqentry_*_enter() functions and
> - * must be passed back into the corresponding irqentry_*_exit() functions
> - * when the exception is complete.
> - *
> - * Callers of irqentry_*_[enter|exit]() must consider this structure opaque
> - * and all members private. Descriptions of the members are provided to aid in
> - * the maintenance of the irqentry_*() functions.
> - */
> -typedef struct irqentry_state {
> - union {
> - bool exit_rcu;
> - bool lockdep;
> - };
> -} irqentry_state_t;
> -#endif
> -
> -/**
> - * irqentry_enter - Handle state tracking on ordinary interrupt entries
> - * @regs: Pointer to pt_regs of interrupted context
> - *
> - * Invokes:
> - * - lockdep irqflag state tracking as low level ASM entry disabled
> - * interrupts.
> - *
> - * - Context tracking if the exception hit user mode.
> - *
> - * - The hardirq tracer to keep the state consistent as low level ASM
> - * entry disabled interrupts.
> - *
> - * As a precondition, this requires that the entry came from user mode,
> - * idle, or a kernel context in which RCU is watching.
> - *
> - * For kernel mode entries RCU handling is done conditional. If RCU is
> - * watching then the only RCU requirement is to check whether the tick has
> - * to be restarted. If RCU is not watching then ct_irq_enter() has to be
> - * invoked on entry and ct_irq_exit() on exit.
> - *
> - * Avoiding the ct_irq_enter/exit() calls is an optimization but also
> - * solves the problem of kernel mode pagefaults which can schedule, which
> - * is not possible after invoking ct_irq_enter() without undoing it.
> - *
> - * For user mode entries irqentry_enter_from_user_mode() is invoked to
> - * establish the proper context for NOHZ_FULL. Otherwise scheduling on exit
> - * would not be possible.
> - *
> - * Returns: An opaque object that must be passed to idtentry_exit()
> - */
> -irqentry_state_t noinstr irqentry_enter(struct pt_regs *regs);
> -
> -/**
> - * irqentry_exit_cond_resched - Conditionally reschedule on return from interrupt
> - *
> - * Conditional reschedule with additional sanity checks.
> - */
> -void raw_irqentry_exit_cond_resched(void);
> -#ifdef CONFIG_PREEMPT_DYNAMIC
> -#if defined(CONFIG_HAVE_PREEMPT_DYNAMIC_CALL)
> -#define irqentry_exit_cond_resched_dynamic_enabled raw_irqentry_exit_cond_resched
> -#define irqentry_exit_cond_resched_dynamic_disabled NULL
> -DECLARE_STATIC_CALL(irqentry_exit_cond_resched, raw_irqentry_exit_cond_resched);
> -#define irqentry_exit_cond_resched() static_call(irqentry_exit_cond_resched)()
> -#elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY)
> -DECLARE_STATIC_KEY_TRUE(sk_dynamic_irqentry_exit_cond_resched);
> -void dynamic_irqentry_exit_cond_resched(void);
> -#define irqentry_exit_cond_resched() dynamic_irqentry_exit_cond_resched()
> -#endif
> -#else /* CONFIG_PREEMPT_DYNAMIC */
> -#define irqentry_exit_cond_resched() raw_irqentry_exit_cond_resched()
> -#endif /* CONFIG_PREEMPT_DYNAMIC */
> -
> -/**
> - * irqentry_exit - Handle return from exception that used irqentry_enter()
> - * @regs: Pointer to pt_regs (exception entry regs)
> - * @state: Return value from matching call to irqentry_enter()
> - *
> - * Depending on the return target (kernel/user) this runs the necessary
> - * preemption and work checks if possible and required and returns to
> - * the caller with interrupts disabled and no further work pending.
> - *
> - * This is the last action before returning to the low level ASM code which
> - * just needs to return to the appropriate context.
> - *
> - * Counterpart to irqentry_enter().
> - */
> -void noinstr irqentry_exit(struct pt_regs *regs, irqentry_state_t state);
> -
> -/**
> - * irqentry_nmi_enter - Handle NMI entry
> - * @regs: Pointer to currents pt_regs
> - *
> - * Similar to irqentry_enter() but taking care of the NMI constraints.
> - */
> -irqentry_state_t noinstr irqentry_nmi_enter(struct pt_regs *regs);
> -
> -/**
> - * irqentry_nmi_exit - Handle return from NMI handling
> - * @regs: Pointer to pt_regs (NMI entry regs)
> - * @irq_state: Return value from matching call to irqentry_nmi_enter()
> - *
> - * Last action before returning to the low level assembly code.
> - *
> - * Counterpart to irqentry_nmi_enter().
> - */
> -void noinstr irqentry_nmi_exit(struct pt_regs *regs, irqentry_state_t irq_state);
> -
> #endif
> diff --git a/include/linux/irq-entry-common.h b/include/linux/irq-entry-common.h
> new file mode 100644
> index 000000000000..8af374331900
> --- /dev/null
> +++ b/include/linux/irq-entry-common.h
> @@ -0,0 +1,389 @@
> +/* SPDX-License-Identifier: GPL-2.0 */
> +#ifndef __LINUX_IRQENTRYCOMMON_H
> +#define __LINUX_IRQENTRYCOMMON_H
> +
> +#include <linux/static_call_types.h>
> +#include <linux/syscalls.h>
> +#include <linux/context_tracking.h>
> +#include <linux/tick.h>
> +#include <linux/kmsan.h>
> +
> +#include <asm/entry-common.h>
> +
> +/*
> + * Define dummy _TIF work flags if not defined by the architecture or for
> + * disabled functionality.
> + */
> +#ifndef _TIF_PATCH_PENDING
> +# define _TIF_PATCH_PENDING (0)
> +#endif
> +
> +/*
> + * TIF flags handled in exit_to_user_mode_loop()
> + */
> +#ifndef ARCH_EXIT_TO_USER_MODE_WORK
> +# define ARCH_EXIT_TO_USER_MODE_WORK (0)
> +#endif
> +
> +#define EXIT_TO_USER_MODE_WORK \
> + (_TIF_SIGPENDING | _TIF_NOTIFY_RESUME | _TIF_UPROBE | \
> + _TIF_NEED_RESCHED | _TIF_NEED_RESCHED_LAZY | \
> + _TIF_PATCH_PENDING | _TIF_NOTIFY_SIGNAL | \
> + ARCH_EXIT_TO_USER_MODE_WORK)
> +
> +/**
> + * arch_enter_from_user_mode - Architecture specific sanity check for user mode regs
> + * @regs: Pointer to currents pt_regs
> + *
> + * Defaults to an empty implementation. Can be replaced by architecture
> + * specific code.
> + *
> + * Invoked from syscall_enter_from_user_mode() in the non-instrumentable
> + * section. Use __always_inline so the compiler cannot push it out of line
> + * and make it instrumentable.
> + */
> +static __always_inline void arch_enter_from_user_mode(struct pt_regs *regs);
> +
> +#ifndef arch_enter_from_user_mode
> +static __always_inline void arch_enter_from_user_mode(struct pt_regs *regs) {}
> +#endif
> +
> +/**
> + * enter_from_user_mode - Establish state when coming from user mode
> + *
> + * Syscall/interrupt entry disables interrupts, but user mode is traced as
> + * interrupts enabled. Also with NO_HZ_FULL RCU might be idle.
> + *
> + * 1) Tell lockdep that interrupts are disabled
> + * 2) Invoke context tracking if enabled to reactivate RCU
> + * 3) Trace interrupts off state
> + *
> + * Invoked from architecture specific syscall entry code with interrupts
> + * disabled. The calling code has to be non-instrumentable. When the
> + * function returns all state is correct and interrupts are still
> + * disabled. The subsequent functions can be instrumented.
> + *
> + * This is invoked when there is architecture specific functionality to be
> + * done between establishing state and enabling interrupts. The caller must
> + * enable interrupts before invoking syscall_enter_from_user_mode_work().
> + */
> +static __always_inline void enter_from_user_mode(struct pt_regs *regs)
> +{
> + arch_enter_from_user_mode(regs);
> + lockdep_hardirqs_off(CALLER_ADDR0);
> +
> + CT_WARN_ON(__ct_state() != CT_STATE_USER);
> + user_exit_irqoff();
> +
> + instrumentation_begin();
> + kmsan_unpoison_entry_regs(regs);
> + trace_hardirqs_off_finish();
> + instrumentation_end();
> +}
> +
> +/**
> + * local_irq_enable_exit_to_user - Exit to user variant of local_irq_enable()
> + * @ti_work: Cached TIF flags gathered with interrupts disabled
> + *
> + * Defaults to local_irq_enable(). Can be supplied by architecture specific
> + * code.
> + */
> +static inline void local_irq_enable_exit_to_user(unsigned long ti_work);
> +
> +#ifndef local_irq_enable_exit_to_user
> +static inline void local_irq_enable_exit_to_user(unsigned long ti_work)
> +{
> + local_irq_enable();
> +}
> +#endif
> +
> +/**
> + * local_irq_disable_exit_to_user - Exit to user variant of local_irq_disable()
> + *
> + * Defaults to local_irq_disable(). Can be supplied by architecture specific
> + * code.
> + */
> +static inline void local_irq_disable_exit_to_user(void);
> +
> +#ifndef local_irq_disable_exit_to_user
> +static inline void local_irq_disable_exit_to_user(void)
> +{
> + local_irq_disable();
> +}
> +#endif
> +
> +/**
> + * arch_exit_to_user_mode_work - Architecture specific TIF work for exit
> + * to user mode.
> + * @regs: Pointer to currents pt_regs
> + * @ti_work: Cached TIF flags gathered with interrupts disabled
> + *
> + * Invoked from exit_to_user_mode_loop() with interrupt enabled
> + *
> + * Defaults to NOOP. Can be supplied by architecture specific code.
> + */
> +static inline void arch_exit_to_user_mode_work(struct pt_regs *regs,
> + unsigned long ti_work);
> +
> +#ifndef arch_exit_to_user_mode_work
> +static inline void arch_exit_to_user_mode_work(struct pt_regs *regs,
> + unsigned long ti_work)
> +{
> +}
> +#endif
> +
> +/**
> + * arch_exit_to_user_mode_prepare - Architecture specific preparation for
> + * exit to user mode.
> + * @regs: Pointer to currents pt_regs
> + * @ti_work: Cached TIF flags gathered with interrupts disabled
> + *
> + * Invoked from exit_to_user_mode_prepare() with interrupt disabled as the last
> + * function before return. Defaults to NOOP.
> + */
> +static inline void arch_exit_to_user_mode_prepare(struct pt_regs *regs,
> + unsigned long ti_work);
> +
> +#ifndef arch_exit_to_user_mode_prepare
> +static inline void arch_exit_to_user_mode_prepare(struct pt_regs *regs,
> + unsigned long ti_work)
> +{
> +}
> +#endif
> +
> +/**
> + * arch_exit_to_user_mode - Architecture specific final work before
> + * exit to user mode.
> + *
> + * Invoked from exit_to_user_mode() with interrupt disabled as the last
> + * function before return. Defaults to NOOP.
> + *
> + * This needs to be __always_inline because it is non-instrumentable code
> + * invoked after context tracking switched to user mode.
> + *
> + * An architecture implementation must not do anything complex, no locking
> + * etc. The main purpose is for speculation mitigations.
> + */
> +static __always_inline void arch_exit_to_user_mode(void);
> +
> +#ifndef arch_exit_to_user_mode
> +static __always_inline void arch_exit_to_user_mode(void) { }
> +#endif
> +
> +/**
> + * arch_do_signal_or_restart - Architecture specific signal delivery function
> + * @regs: Pointer to currents pt_regs
> + *
> + * Invoked from exit_to_user_mode_loop().
> + */
> +void arch_do_signal_or_restart(struct pt_regs *regs);
> +
> +/**
> + * exit_to_user_mode_loop - do any pending work before leaving to user space
> + */
> +unsigned long exit_to_user_mode_loop(struct pt_regs *regs,
> + unsigned long ti_work);
> +
> +/**
> + * exit_to_user_mode_prepare - call exit_to_user_mode_loop() if required
> + * @regs: Pointer to pt_regs on entry stack
> + *
> + * 1) check that interrupts are disabled
> + * 2) call tick_nohz_user_enter_prepare()
> + * 3) call exit_to_user_mode_loop() if any flags from
> + * EXIT_TO_USER_MODE_WORK are set
> + * 4) check that interrupts are still disabled
> + */
> +static __always_inline void exit_to_user_mode_prepare(struct pt_regs *regs)
> +{
> + unsigned long ti_work;
> +
> + lockdep_assert_irqs_disabled();
> +
> + /* Flush pending rcuog wakeup before the last need_resched() check */
> + tick_nohz_user_enter_prepare();
> +
> + ti_work = read_thread_flags();
> + if (unlikely(ti_work & EXIT_TO_USER_MODE_WORK))
> + ti_work = exit_to_user_mode_loop(regs, ti_work);
> +
> + arch_exit_to_user_mode_prepare(regs, ti_work);
> +
> + /* Ensure that kernel state is sane for a return to userspace */
> + kmap_assert_nomap();
> + lockdep_assert_irqs_disabled();
> + lockdep_sys_exit();
> +}
> +
> +/**
> + * exit_to_user_mode - Fixup state when exiting to user mode
> + *
> + * Syscall/interrupt exit enables interrupts, but the kernel state is
> + * interrupts disabled when this is invoked. Also tell RCU about it.
> + *
> + * 1) Trace interrupts on state
> + * 2) Invoke context tracking if enabled to adjust RCU state
> + * 3) Invoke architecture specific last minute exit code, e.g. speculation
> + * mitigations, etc.: arch_exit_to_user_mode()
> + * 4) Tell lockdep that interrupts are enabled
> + *
> + * Invoked from architecture specific code when syscall_exit_to_user_mode()
> + * is not suitable as the last step before returning to userspace. Must be
> + * invoked with interrupts disabled and the caller must be
> + * non-instrumentable.
> + * The caller has to invoke syscall_exit_to_user_mode_work() before this.
> + */
> +static __always_inline void exit_to_user_mode(void)
> +{
> + instrumentation_begin();
> + trace_hardirqs_on_prepare();
> + lockdep_hardirqs_on_prepare();
> + instrumentation_end();
> +
> + user_enter_irqoff();
> + arch_exit_to_user_mode();
> + lockdep_hardirqs_on(CALLER_ADDR0);
> +}
> +
> +/**
> + * irqentry_enter_from_user_mode - Establish state before invoking the irq handler
> + * @regs: Pointer to currents pt_regs
> + *
> + * Invoked from architecture specific entry code with interrupts disabled.
> + * Can only be called when the interrupt entry came from user mode. The
> + * calling code must be non-instrumentable. When the function returns all
> + * state is correct and the subsequent functions can be instrumented.
> + *
> + * The function establishes state (lockdep, RCU (context tracking), tracing)
> + */
> +void irqentry_enter_from_user_mode(struct pt_regs *regs);
> +
> +/**
> + * irqentry_exit_to_user_mode - Interrupt exit work
> + * @regs: Pointer to current's pt_regs
> + *
> + * Invoked with interrupts disabled and fully valid regs. Returns with all
> + * work handled, interrupts disabled such that the caller can immediately
> + * switch to user mode. Called from architecture specific interrupt
> + * handling code.
> + *
> + * The call order is #2 and #3 as described in syscall_exit_to_user_mode().
> + * Interrupt exit is not invoking #1 which is the syscall specific one time
> + * work.
> + */
> +void irqentry_exit_to_user_mode(struct pt_regs *regs);
> +
> +#ifndef irqentry_state
> +/**
> + * struct irqentry_state - Opaque object for exception state storage
> + * @exit_rcu: Used exclusively in the irqentry_*() calls; signals whether the
> + * exit path has to invoke ct_irq_exit().
> + * @lockdep: Used exclusively in the irqentry_nmi_*() calls; ensures that
> + * lockdep state is restored correctly on exit from nmi.
> + *
> + * This opaque object is filled in by the irqentry_*_enter() functions and
> + * must be passed back into the corresponding irqentry_*_exit() functions
> + * when the exception is complete.
> + *
> + * Callers of irqentry_*_[enter|exit]() must consider this structure opaque
> + * and all members private. Descriptions of the members are provided to aid in
> + * the maintenance of the irqentry_*() functions.
> + */
> +typedef struct irqentry_state {
> + union {
> + bool exit_rcu;
> + bool lockdep;
> + };
> +} irqentry_state_t;
> +#endif
> +
> +/**
> + * irqentry_enter - Handle state tracking on ordinary interrupt entries
> + * @regs: Pointer to pt_regs of interrupted context
> + *
> + * Invokes:
> + * - lockdep irqflag state tracking as low level ASM entry disabled
> + * interrupts.
> + *
> + * - Context tracking if the exception hit user mode.
> + *
> + * - The hardirq tracer to keep the state consistent as low level ASM
> + * entry disabled interrupts.
> + *
> + * As a precondition, this requires that the entry came from user mode,
> + * idle, or a kernel context in which RCU is watching.
> + *
> + * For kernel mode entries RCU handling is done conditional. If RCU is
> + * watching then the only RCU requirement is to check whether the tick has
> + * to be restarted. If RCU is not watching then ct_irq_enter() has to be
> + * invoked on entry and ct_irq_exit() on exit.
> + *
> + * Avoiding the ct_irq_enter/exit() calls is an optimization but also
> + * solves the problem of kernel mode pagefaults which can schedule, which
> + * is not possible after invoking ct_irq_enter() without undoing it.
> + *
> + * For user mode entries irqentry_enter_from_user_mode() is invoked to
> + * establish the proper context for NOHZ_FULL. Otherwise scheduling on exit
> + * would not be possible.
> + *
> + * Returns: An opaque object that must be passed to idtentry_exit()
> + */
> +irqentry_state_t noinstr irqentry_enter(struct pt_regs *regs);
> +
> +/**
> + * irqentry_exit_cond_resched - Conditionally reschedule on return from interrupt
> + *
> + * Conditional reschedule with additional sanity checks.
> + */
> +void raw_irqentry_exit_cond_resched(void);
> +#ifdef CONFIG_PREEMPT_DYNAMIC
> +#if defined(CONFIG_HAVE_PREEMPT_DYNAMIC_CALL)
> +#define irqentry_exit_cond_resched_dynamic_enabled raw_irqentry_exit_cond_resched
> +#define irqentry_exit_cond_resched_dynamic_disabled NULL
> +DECLARE_STATIC_CALL(irqentry_exit_cond_resched, raw_irqentry_exit_cond_resched);
> +#define irqentry_exit_cond_resched() static_call(irqentry_exit_cond_resched)()
> +#elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY)
> +DECLARE_STATIC_KEY_TRUE(sk_dynamic_irqentry_exit_cond_resched);
> +void dynamic_irqentry_exit_cond_resched(void);
> +#define irqentry_exit_cond_resched() dynamic_irqentry_exit_cond_resched()
> +#endif
> +#else /* CONFIG_PREEMPT_DYNAMIC */
> +#define irqentry_exit_cond_resched() raw_irqentry_exit_cond_resched()
> +#endif /* CONFIG_PREEMPT_DYNAMIC */
> +
> +/**
> + * irqentry_exit - Handle return from exception that used irqentry_enter()
> + * @regs: Pointer to pt_regs (exception entry regs)
> + * @state: Return value from matching call to irqentry_enter()
> + *
> + * Depending on the return target (kernel/user) this runs the necessary
> + * preemption and work checks if possible and required and returns to
> + * the caller with interrupts disabled and no further work pending.
> + *
> + * This is the last action before returning to the low level ASM code which
> + * just needs to return to the appropriate context.
> + *
> + * Counterpart to irqentry_enter().
> + */
> +void noinstr irqentry_exit(struct pt_regs *regs, irqentry_state_t state);
> +
> +/**
> + * irqentry_nmi_enter - Handle NMI entry
> + * @regs: Pointer to currents pt_regs
> + *
> + * Similar to irqentry_enter() but taking care of the NMI constraints.
> + */
> +irqentry_state_t noinstr irqentry_nmi_enter(struct pt_regs *regs);
> +
> +/**
> + * irqentry_nmi_exit - Handle return from NMI handling
> + * @regs: Pointer to pt_regs (NMI entry regs)
> + * @irq_state: Return value from matching call to irqentry_nmi_enter()
> + *
> + * Last action before returning to the low level assembly code.
> + *
> + * Counterpart to irqentry_nmi_enter().
> + */
> +void noinstr irqentry_nmi_exit(struct pt_regs *regs, irqentry_state_t irq_state);
> +
> +#endif
> diff --git a/kernel/entry/Makefile b/kernel/entry/Makefile
> index 095c775e001e..d38f3a7e7396 100644
> --- a/kernel/entry/Makefile
> +++ b/kernel/entry/Makefile
> @@ -9,5 +9,6 @@ KCOV_INSTRUMENT := n
> CFLAGS_REMOVE_common.o = -fstack-protector -fstack-protector-strong
> CFLAGS_common.o += -fno-stack-protector
>
> -obj-$(CONFIG_GENERIC_ENTRY) += common.o syscall_user_dispatch.o
> +obj-$(CONFIG_GENERIC_IRQ_ENTRY) += common.o
> +obj-$(CONFIG_GENERIC_SYSCALL) += syscall-common.o syscall_user_dispatch.o
> obj-$(CONFIG_KVM_XFER_TO_GUEST_WORK) += kvm.o
> diff --git a/kernel/entry/common.c b/kernel/entry/common.c
> index e33691d5adf7..b82032777310 100644
> --- a/kernel/entry/common.c
> +++ b/kernel/entry/common.c
> @@ -1,84 +1,13 @@
> // SPDX-License-Identifier: GPL-2.0
>
> -#include <linux/context_tracking.h>
> -#include <linux/entry-common.h>
> +#include <linux/irq-entry-common.h>
> #include <linux/resume_user_mode.h>
> #include <linux/highmem.h>
> #include <linux/jump_label.h>
> #include <linux/kmsan.h>
> #include <linux/livepatch.h>
> -#include <linux/audit.h>
> #include <linux/tick.h>
>
> -#include "common.h"
> -
> -#define CREATE_TRACE_POINTS
> -#include <trace/events/syscalls.h>
> -
> -static inline void syscall_enter_audit(struct pt_regs *regs, long syscall)
> -{
> - if (unlikely(audit_context())) {
> - unsigned long args[6];
> -
> - syscall_get_arguments(current, regs, args);
> - audit_syscall_entry(syscall, args[0], args[1], args[2], args[3]);
> - }
> -}
> -
> -long syscall_trace_enter(struct pt_regs *regs, long syscall,
> - unsigned long work)
> -{
> - long ret = 0;
> -
> - /*
> - * Handle Syscall User Dispatch. This must comes first, since
> - * the ABI here can be something that doesn't make sense for
> - * other syscall_work features.
> - */
> - if (work & SYSCALL_WORK_SYSCALL_USER_DISPATCH) {
> - if (syscall_user_dispatch(regs))
> - return -1L;
> - }
> -
> - /* Handle ptrace */
> - if (work & (SYSCALL_WORK_SYSCALL_TRACE | SYSCALL_WORK_SYSCALL_EMU)) {
> - ret = ptrace_report_syscall_entry(regs);
> - if (ret || (work & SYSCALL_WORK_SYSCALL_EMU))
> - return -1L;
> - }
> -
> - /* Do seccomp after ptrace, to catch any tracer changes. */
> - if (work & SYSCALL_WORK_SECCOMP) {
> - ret = __secure_computing(NULL);
> - if (ret == -1L)
> - return ret;
> - }
> -
> - /* Either of the above might have changed the syscall number */
> - syscall = syscall_get_nr(current, regs);
> -
> - if (unlikely(work & SYSCALL_WORK_SYSCALL_TRACEPOINT)) {
> - trace_sys_enter(regs, syscall);
> - /*
> - * Probes or BPF hooks in the tracepoint may have changed the
> - * system call number as well.
> - */
> - syscall = syscall_get_nr(current, regs);
> - }
> -
> - syscall_enter_audit(regs, syscall);
> -
> - return ret ? : syscall;
> -}
> -
> -noinstr void syscall_enter_from_user_mode_prepare(struct pt_regs *regs)
> -{
> - enter_from_user_mode(regs);
> - instrumentation_begin();
> - local_irq_enable();
> - instrumentation_end();
> -}
> -
> /* Workaround to allow gradual conversion of architecture code */
> void __weak arch_do_signal_or_restart(struct pt_regs *regs) { }
>
> @@ -133,93 +62,6 @@ __always_inline unsigned long exit_to_user_mode_loop(struct pt_regs *regs,
> return ti_work;
> }
>
> -/*
> - * If SYSCALL_EMU is set, then the only reason to report is when
> - * SINGLESTEP is set (i.e. PTRACE_SYSEMU_SINGLESTEP). This syscall
> - * instruction has been already reported in syscall_enter_from_user_mode().
> - */
> -static inline bool report_single_step(unsigned long work)
> -{
> - if (work & SYSCALL_WORK_SYSCALL_EMU)
> - return false;
> -
> - return work & SYSCALL_WORK_SYSCALL_EXIT_TRAP;
> -}
> -
> -static void syscall_exit_work(struct pt_regs *regs, unsigned long work)
> -{
> - bool step;
> -
> - /*
> - * If the syscall was rolled back due to syscall user dispatching,
> - * then the tracers below are not invoked for the same reason as
> - * the entry side was not invoked in syscall_trace_enter(): The ABI
> - * of these syscalls is unknown.
> - */
> - if (work & SYSCALL_WORK_SYSCALL_USER_DISPATCH) {
> - if (unlikely(current->syscall_dispatch.on_dispatch)) {
> - current->syscall_dispatch.on_dispatch = false;
> - return;
> - }
> - }
> -
> - audit_syscall_exit(regs);
> -
> - if (work & SYSCALL_WORK_SYSCALL_TRACEPOINT)
> - trace_sys_exit(regs, syscall_get_return_value(current, regs));
> -
> - step = report_single_step(work);
> - if (step || work & SYSCALL_WORK_SYSCALL_TRACE)
> - ptrace_report_syscall_exit(regs, step);
> -}
> -
> -/*
> - * Syscall specific exit to user mode preparation. Runs with interrupts
> - * enabled.
> - */
> -static void syscall_exit_to_user_mode_prepare(struct pt_regs *regs)
> -{
> - unsigned long work = READ_ONCE(current_thread_info()->syscall_work);
> - unsigned long nr = syscall_get_nr(current, regs);
> -
> - CT_WARN_ON(ct_state() != CT_STATE_KERNEL);
> -
> - if (IS_ENABLED(CONFIG_PROVE_LOCKING)) {
> - if (WARN(irqs_disabled(), "syscall %lu left IRQs disabled", nr))
> - local_irq_enable();
> - }
> -
> - rseq_syscall(regs);
> -
> - /*
> - * Do one-time syscall specific work. If these work items are
> - * enabled, we want to run them exactly once per syscall exit with
> - * interrupts enabled.
> - */
> - if (unlikely(work & SYSCALL_WORK_EXIT))
> - syscall_exit_work(regs, work);
> -}
> -
> -static __always_inline void __syscall_exit_to_user_mode_work(struct pt_regs *regs)
> -{
> - syscall_exit_to_user_mode_prepare(regs);
> - local_irq_disable_exit_to_user();
> - exit_to_user_mode_prepare(regs);
> -}
> -
> -void syscall_exit_to_user_mode_work(struct pt_regs *regs)
> -{
> - __syscall_exit_to_user_mode_work(regs);
> -}
> -
> -__visible noinstr void syscall_exit_to_user_mode(struct pt_regs *regs)
> -{
> - instrumentation_begin();
> - __syscall_exit_to_user_mode_work(regs);
> - instrumentation_end();
> - exit_to_user_mode();
> -}
> -
> noinstr void irqentry_enter_from_user_mode(struct pt_regs *regs)
> {
> enter_from_user_mode(regs);
> diff --git a/kernel/entry/syscall-common.c b/kernel/entry/syscall-common.c
> new file mode 100644
> index 000000000000..0eb036986ad4
> --- /dev/null
> +++ b/kernel/entry/syscall-common.c
> @@ -0,0 +1,159 @@
> +// SPDX-License-Identifier: GPL-2.0
> +
> +#include <linux/audit.h>
> +#include <linux/entry-common.h>
> +#include "common.h"
> +
> +#define CREATE_TRACE_POINTS
> +#include <trace/events/syscalls.h>
> +
> +static inline void syscall_enter_audit(struct pt_regs *regs, long syscall)
> +{
> + if (unlikely(audit_context())) {
> + unsigned long args[6];
> +
> + syscall_get_arguments(current, regs, args);
> + audit_syscall_entry(syscall, args[0], args[1], args[2], args[3]);
> + }
> +}
> +
> +long syscall_trace_enter(struct pt_regs *regs, long syscall,
> + unsigned long work)
> +{
> + long ret = 0;
> +
> + /*
> + * Handle Syscall User Dispatch. This must comes first, since
> + * the ABI here can be something that doesn't make sense for
> + * other syscall_work features.
> + */
> + if (work & SYSCALL_WORK_SYSCALL_USER_DISPATCH) {
> + if (syscall_user_dispatch(regs))
> + return -1L;
> + }
> +
> + /* Handle ptrace */
> + if (work & (SYSCALL_WORK_SYSCALL_TRACE | SYSCALL_WORK_SYSCALL_EMU)) {
> + ret = ptrace_report_syscall_entry(regs);
> + if (ret || (work & SYSCALL_WORK_SYSCALL_EMU))
> + return -1L;
> + }
> +
> + /* Do seccomp after ptrace, to catch any tracer changes. */
> + if (work & SYSCALL_WORK_SECCOMP) {
> + ret = __secure_computing(NULL);
> + if (ret == -1L)
> + return ret;
> + }
> +
> + /* Either of the above might have changed the syscall number */
> + syscall = syscall_get_nr(current, regs);
> +
> + if (unlikely(work & SYSCALL_WORK_SYSCALL_TRACEPOINT)) {
> + trace_sys_enter(regs, syscall);
> + /*
> + * Probes or BPF hooks in the tracepoint may have changed the
> + * system call number as well.
> + */
> + syscall = syscall_get_nr(current, regs);
> + }
> +
> + syscall_enter_audit(regs, syscall);
> +
> + return ret ? : syscall;
> +}
> +
> +noinstr void syscall_enter_from_user_mode_prepare(struct pt_regs *regs)
> +{
> + enter_from_user_mode(regs);
> + instrumentation_begin();
> + local_irq_enable();
> + instrumentation_end();
> +}
> +
> +/*
> + * If SYSCALL_EMU is set, then the only reason to report is when
> + * SINGLESTEP is set (i.e. PTRACE_SYSEMU_SINGLESTEP). This syscall
> + * instruction has been already reported in syscall_enter_from_user_mode().
> + */
> +static inline bool report_single_step(unsigned long work)
> +{
> + if (work & SYSCALL_WORK_SYSCALL_EMU)
> + return false;
> +
> + return work & SYSCALL_WORK_SYSCALL_EXIT_TRAP;
> +}
> +
> +static void syscall_exit_work(struct pt_regs *regs, unsigned long work)
> +{
> + bool step;
> +
> + /*
> + * If the syscall was rolled back due to syscall user dispatching,
> + * then the tracers below are not invoked for the same reason as
> + * the entry side was not invoked in syscall_trace_enter(): The ABI
> + * of these syscalls is unknown.
> + */
> + if (work & SYSCALL_WORK_SYSCALL_USER_DISPATCH) {
> + if (unlikely(current->syscall_dispatch.on_dispatch)) {
> + current->syscall_dispatch.on_dispatch = false;
> + return;
> + }
> + }
> +
> + audit_syscall_exit(regs);
> +
> + if (work & SYSCALL_WORK_SYSCALL_TRACEPOINT)
> + trace_sys_exit(regs, syscall_get_return_value(current, regs));
> +
> + step = report_single_step(work);
> + if (step || work & SYSCALL_WORK_SYSCALL_TRACE)
> + ptrace_report_syscall_exit(regs, step);
> +}
> +
> +/*
> + * Syscall specific exit to user mode preparation. Runs with interrupts
> + * enabled.
> + */
> +static void syscall_exit_to_user_mode_prepare(struct pt_regs *regs)
> +{
> + unsigned long work = READ_ONCE(current_thread_info()->syscall_work);
> + unsigned long nr = syscall_get_nr(current, regs);
> +
> + CT_WARN_ON(ct_state() != CT_STATE_KERNEL);
> +
> + if (IS_ENABLED(CONFIG_PROVE_LOCKING)) {
> + if (WARN(irqs_disabled(), "syscall %lu left IRQs disabled", nr))
> + local_irq_enable();
> + }
> +
> + rseq_syscall(regs);
> +
> + /*
> + * Do one-time syscall specific work. If these work items are
> + * enabled, we want to run them exactly once per syscall exit with
> + * interrupts enabled.
> + */
> + if (unlikely(work & SYSCALL_WORK_EXIT))
> + syscall_exit_work(regs, work);
> +}
> +
> +static __always_inline void __syscall_exit_to_user_mode_work(struct pt_regs *regs)
> +{
> + syscall_exit_to_user_mode_prepare(regs);
> + local_irq_disable_exit_to_user();
> + exit_to_user_mode_prepare(regs);
> +}
> +
> +void syscall_exit_to_user_mode_work(struct pt_regs *regs)
> +{
> + __syscall_exit_to_user_mode_work(regs);
> +}
> +
> +__visible noinstr void syscall_exit_to_user_mode(struct pt_regs *regs)
> +{
> + instrumentation_begin();
> + __syscall_exit_to_user_mode_work(regs);
> + instrumentation_end();
> + exit_to_user_mode();
> +}
> diff --git a/kernel/sched/core.c b/kernel/sched/core.c
> index 27a8fbd58091..2d560bb3efaa 100644
> --- a/kernel/sched/core.c
> +++ b/kernel/sched/core.c
> @@ -68,8 +68,8 @@
> #include <linux/workqueue_api.h>
>
> #ifdef CONFIG_PREEMPT_DYNAMIC
> -# ifdef CONFIG_GENERIC_ENTRY
> -# include <linux/entry-common.h>
> +# ifdef CONFIG_GENERIC_IRQ_ENTRY
> +# include <linux/irq-entry-common.h>
> # endif
> #endif
>
> @@ -7398,8 +7398,8 @@ EXPORT_SYMBOL(__cond_resched_rwlock_write);
>
> #ifdef CONFIG_PREEMPT_DYNAMIC
>
> -#ifdef CONFIG_GENERIC_ENTRY
> -#include <linux/entry-common.h>
> +#ifdef CONFIG_GENERIC_IRQ_ENTRY
> +#include <linux/irq-entry-common.h>
> #endif
>
> /*
> --
> 2.34.1
>
^ permalink raw reply [flat|nested] 3+ messages in thread
end of thread, other threads:[~2025-02-10 12:17 UTC | newest]
Thread overview: 3+ messages (download: mbox.gz follow: Atom feed
-- links below jump to the message on this page --
2024-12-09 4:04 [PATCH -next v5 09/22] entry: Split generic entry into irq and syscall kernel test robot
-- strict thread matches above, loose matches on Subject: below --
2024-12-06 10:17 [PATCH -next v5 00/22] arm64: entry: Convert to generic entry Jinjie Ruan
2024-12-06 10:17 ` [PATCH -next v5 09/22] entry: Split generic entry into irq and syscall Jinjie Ruan
2025-02-10 12:04 ` Mark Rutland
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.