All of lore.kernel.org
 help / color / mirror / Atom feed
* Re: [PATCH -next v5 09/22] entry: Split generic entry into irq and syscall
@ 2024-12-09  4:04 kernel test robot
  0 siblings, 0 replies; 3+ messages in thread
From: kernel test robot @ 2024-12-09  4:04 UTC (permalink / raw)
  To: oe-kbuild; +Cc: lkp

:::::: 
:::::: Manual check reason: "low confidence bisect report"
:::::: 

BCC: lkp@intel.com
CC: oe-kbuild-all@lists.linux.dev
In-Reply-To: <20241206101744.4161990-10-ruanjinjie@huawei.com>
References: <20241206101744.4161990-10-ruanjinjie@huawei.com>
TO: Jinjie Ruan <ruanjinjie@huawei.com>
TO: catalin.marinas@arm.com
TO: will@kernel.org
TO: oleg@redhat.com
TO: sstabellini@kernel.org
TO: tglx@linutronix.de
TO: peterz@infradead.org
TO: luto@kernel.org
TO: mingo@redhat.com
TO: juri.lelli@redhat.com
TO: vincent.guittot@linaro.org
TO: dietmar.eggemann@arm.com
TO: rostedt@goodmis.org
TO: bsegall@google.com
TO: mgorman@suse.de
TO: vschneid@redhat.com
TO: kees@kernel.org
TO: wad@chromium.org
TO: akpm@linux-foundation.org
TO: samitolvanen@google.com
TO: masahiroy@kernel.org
TO: hca@linux.ibm.com
TO: aliceryhl@google.com
TO: rppt@kernel.org
TO: xur@google.com
TO: paulmck@kernel.org
TO: arnd@arndb.de
TO: mbenes@suse.cz
TO: puranjay@kernel.org
TO: mark.rutland@arm.com
TO: ruanjinjie@huawei.com

Hi Jinjie,

kernel test robot noticed the following build warnings:

[auto build test WARNING on next-20241205]

url:    https://github.com/intel-lab-lkp/linux/commits/Jinjie-Ruan/arm64-ptrace-Replace-interrupts_enabled-with-regs_irqs_disabled/20241206-183134
base:   next-20241205
patch link:    https://lore.kernel.org/r/20241206101744.4161990-10-ruanjinjie%40huawei.com
patch subject: [PATCH -next v5 09/22] entry: Split generic entry into irq and syscall
:::::: branch date: 17 hours ago
:::::: commit date: 17 hours ago
compiler: clang version 19.1.3 (https://github.com/llvm/llvm-project ab51eccf88f5321e7c60591c5546b254b6afab99)

If you fix the issue in a separate patch/commit (i.e. not just a new version of
the same patch/commit), kindly add following tags
| Reported-by: kernel test robot <lkp@intel.com>
| Closes: https://lore.kernel.org/r/202412071107.6IQ3yz7r-lkp@intel.com/

includecheck warnings: (new ones prefixed by >>)
   kernel/sched/core.c: linux/sched/rseq_api.h is included more than once.
   kernel/sched/core.c: stats.h is included more than once.
>> kernel/sched/core.c: linux/irq-entry-common.h is included more than once.

vim +72 kernel/sched/core.c

    69	
    70	#ifdef CONFIG_PREEMPT_DYNAMIC
    71	# ifdef CONFIG_GENERIC_IRQ_ENTRY
  > 72	#  include <linux/irq-entry-common.h>
    73	# endif
    74	#endif
    75	
    76	#include <uapi/linux/sched/types.h>
    77	
    78	#include <asm/irq_regs.h>
    79	#include <asm/switch_to.h>
    80	#include <asm/tlb.h>
    81	
    82	#define CREATE_TRACE_POINTS
    83	#include <linux/sched/rseq_api.h>
    84	#include <trace/events/sched.h>
    85	#include <trace/events/ipi.h>
    86	#undef CREATE_TRACE_POINTS
    87	
    88	#include "sched.h"
    89	#include "stats.h"
    90	
    91	#include "autogroup.h"
    92	#include "pelt.h"
    93	#include "smp.h"
    94	#include "stats.h"
    95	
    96	#include "../workqueue_internal.h"
    97	#include "../../io_uring/io-wq.h"
    98	#include "../smpboot.h"
    99	
   100	EXPORT_TRACEPOINT_SYMBOL_GPL(ipi_send_cpu);
   101	EXPORT_TRACEPOINT_SYMBOL_GPL(ipi_send_cpumask);
   102	
   103	/*
   104	 * Export tracepoints that act as a bare tracehook (ie: have no trace event
   105	 * associated with them) to allow external modules to probe them.
   106	 */
   107	EXPORT_TRACEPOINT_SYMBOL_GPL(pelt_cfs_tp);
   108	EXPORT_TRACEPOINT_SYMBOL_GPL(pelt_rt_tp);
   109	EXPORT_TRACEPOINT_SYMBOL_GPL(pelt_dl_tp);
   110	EXPORT_TRACEPOINT_SYMBOL_GPL(pelt_irq_tp);
   111	EXPORT_TRACEPOINT_SYMBOL_GPL(pelt_se_tp);
   112	EXPORT_TRACEPOINT_SYMBOL_GPL(pelt_hw_tp);
   113	EXPORT_TRACEPOINT_SYMBOL_GPL(sched_cpu_capacity_tp);
   114	EXPORT_TRACEPOINT_SYMBOL_GPL(sched_overutilized_tp);
   115	EXPORT_TRACEPOINT_SYMBOL_GPL(sched_util_est_cfs_tp);
   116	EXPORT_TRACEPOINT_SYMBOL_GPL(sched_util_est_se_tp);
   117	EXPORT_TRACEPOINT_SYMBOL_GPL(sched_update_nr_running_tp);
   118	EXPORT_TRACEPOINT_SYMBOL_GPL(sched_compute_energy_tp);
   119	
   120	DEFINE_PER_CPU_SHARED_ALIGNED(struct rq, runqueues);
   121	
   122	#ifdef CONFIG_SCHED_DEBUG
   123	/*
   124	 * Debugging: various feature bits
   125	 *
   126	 * If SCHED_DEBUG is disabled, each compilation unit has its own copy of
   127	 * sysctl_sched_features, defined in sched.h, to allow constants propagation
   128	 * at compile time and compiler optimization based on features default.
   129	 */
   130	#define SCHED_FEAT(name, enabled)	\
   131		(1UL << __SCHED_FEAT_##name) * enabled |
   132	const_debug unsigned int sysctl_sched_features =
   133	#include "features.h"
   134		0;
   135	#undef SCHED_FEAT
   136	
   137	/*
   138	 * Print a warning if need_resched is set for the given duration (if
   139	 * LATENCY_WARN is enabled).
   140	 *
   141	 * If sysctl_resched_latency_warn_once is set, only one warning will be shown
   142	 * per boot.
   143	 */
   144	__read_mostly int sysctl_resched_latency_warn_ms = 100;
   145	__read_mostly int sysctl_resched_latency_warn_once = 1;
   146	#endif /* CONFIG_SCHED_DEBUG */
   147	
   148	/*
   149	 * Number of tasks to iterate in a single balance run.
   150	 * Limited because this is done with IRQs disabled.
   151	 */
   152	const_debug unsigned int sysctl_sched_nr_migrate = SCHED_NR_MIGRATE_BREAK;
   153	
   154	__read_mostly int scheduler_running;
   155	
   156	#ifdef CONFIG_SCHED_CORE
   157	
   158	DEFINE_STATIC_KEY_FALSE(__sched_core_enabled);
   159	
   160	/* kernel prio, less is more */
   161	static inline int __task_prio(const struct task_struct *p)
   162	{
   163		if (p->sched_class == &stop_sched_class) /* trumps deadline */
   164			return -2;
   165	
   166		if (p->dl_server)
   167			return -1; /* deadline */
   168	
   169		if (rt_or_dl_prio(p->prio))
   170			return p->prio; /* [-1, 99] */
   171	
   172		if (p->sched_class == &idle_sched_class)
   173			return MAX_RT_PRIO + NICE_WIDTH; /* 140 */
   174	
   175		if (task_on_scx(p))
   176			return MAX_RT_PRIO + MAX_NICE + 1; /* 120, squash ext */
   177	
   178		return MAX_RT_PRIO + MAX_NICE; /* 119, squash fair */
   179	}
   180	
   181	/*
   182	 * l(a,b)
   183	 * le(a,b) := !l(b,a)
   184	 * g(a,b)  := l(b,a)
   185	 * ge(a,b) := !l(a,b)
   186	 */
   187	
   188	/* real prio, less is less */
   189	static inline bool prio_less(const struct task_struct *a,
   190				     const struct task_struct *b, bool in_fi)
   191	{
   192	
   193		int pa = __task_prio(a), pb = __task_prio(b);
   194	
   195		if (-pa < -pb)
   196			return true;
   197	
   198		if (-pb < -pa)
   199			return false;
   200	
   201		if (pa == -1) { /* dl_prio() doesn't work because of stop_class above */
   202			const struct sched_dl_entity *a_dl, *b_dl;
   203	
   204			a_dl = &a->dl;
   205			/*
   206			 * Since,'a' and 'b' can be CFS tasks served by DL server,
   207			 * __task_prio() can return -1 (for DL) even for those. In that
   208			 * case, get to the dl_server's DL entity.
   209			 */
   210			if (a->dl_server)
   211				a_dl = a->dl_server;
   212	
   213			b_dl = &b->dl;
   214			if (b->dl_server)
   215				b_dl = b->dl_server;
   216	
   217			return !dl_time_before(a_dl->deadline, b_dl->deadline);
   218		}
   219	
   220		if (pa == MAX_RT_PRIO + MAX_NICE)	/* fair */
   221			return cfs_prio_less(a, b, in_fi);
   222	
   223	#ifdef CONFIG_SCHED_CLASS_EXT
   224		if (pa == MAX_RT_PRIO + MAX_NICE + 1)	/* ext */
   225			return scx_prio_less(a, b, in_fi);
   226	#endif
   227	
   228		return false;
   229	}
   230	
   231	static inline bool __sched_core_less(const struct task_struct *a,
   232					     const struct task_struct *b)
   233	{
   234		if (a->core_cookie < b->core_cookie)
   235			return true;
   236	
   237		if (a->core_cookie > b->core_cookie)
   238			return false;
   239	
   240		/* flip prio, so high prio is leftmost */
   241		if (prio_less(b, a, !!task_rq(a)->core->core_forceidle_count))
   242			return true;
   243	
   244		return false;
   245	}
   246	
   247	#define __node_2_sc(node) rb_entry((node), struct task_struct, core_node)
   248	
   249	static inline bool rb_sched_core_less(struct rb_node *a, const struct rb_node *b)
   250	{
   251		return __sched_core_less(__node_2_sc(a), __node_2_sc(b));
   252	}
   253	
   254	static inline int rb_sched_core_cmp(const void *key, const struct rb_node *node)
   255	{
   256		const struct task_struct *p = __node_2_sc(node);
   257		unsigned long cookie = (unsigned long)key;
   258	
   259		if (cookie < p->core_cookie)
   260			return -1;
   261	
   262		if (cookie > p->core_cookie)
   263			return 1;
   264	
   265		return 0;
   266	}
   267	
   268	void sched_core_enqueue(struct rq *rq, struct task_struct *p)
   269	{
   270		if (p->se.sched_delayed)
   271			return;
   272	
   273		rq->core->core_task_seq++;
   274	
   275		if (!p->core_cookie)
   276			return;
   277	
   278		rb_add(&p->core_node, &rq->core_tree, rb_sched_core_less);
   279	}
   280	
   281	void sched_core_dequeue(struct rq *rq, struct task_struct *p, int flags)
   282	{
   283		if (p->se.sched_delayed)
   284			return;
   285	
   286		rq->core->core_task_seq++;
   287	
   288		if (sched_core_enqueued(p)) {
   289			rb_erase(&p->core_node, &rq->core_tree);
   290			RB_CLEAR_NODE(&p->core_node);
   291		}
   292	
   293		/*
   294		 * Migrating the last task off the cpu, with the cpu in forced idle
   295		 * state. Reschedule to create an accounting edge for forced idle,
   296		 * and re-examine whether the core is still in forced idle state.
   297		 */
   298		if (!(flags & DEQUEUE_SAVE) && rq->nr_running == 1 &&
   299		    rq->core->core_forceidle_count && rq->curr == rq->idle)
   300			resched_curr(rq);
   301	}
   302	
   303	static int sched_task_is_throttled(struct task_struct *p, int cpu)
   304	{
   305		if (p->sched_class->task_is_throttled)
   306			return p->sched_class->task_is_throttled(p, cpu);
   307	
   308		return 0;
   309	}
   310	
   311	static struct task_struct *sched_core_next(struct task_struct *p, unsigned long cookie)
   312	{
   313		struct rb_node *node = &p->core_node;
   314		int cpu = task_cpu(p);
   315	
   316		do {
   317			node = rb_next(node);
   318			if (!node)
   319				return NULL;
   320	
   321			p = __node_2_sc(node);
   322			if (p->core_cookie != cookie)
   323				return NULL;
   324	
   325		} while (sched_task_is_throttled(p, cpu));
   326	
   327		return p;
   328	}
   329	
   330	/*
   331	 * Find left-most (aka, highest priority) and unthrottled task matching @cookie.
   332	 * If no suitable task is found, NULL will be returned.
   333	 */
   334	static struct task_struct *sched_core_find(struct rq *rq, unsigned long cookie)
   335	{
   336		struct task_struct *p;
   337		struct rb_node *node;
   338	
   339		node = rb_find_first((void *)cookie, &rq->core_tree, rb_sched_core_cmp);
   340		if (!node)
   341			return NULL;
   342	
   343		p = __node_2_sc(node);
   344		if (!sched_task_is_throttled(p, rq->cpu))
   345			return p;
   346	
   347		return sched_core_next(p, cookie);
   348	}
   349	
   350	/*
   351	 * Magic required such that:
   352	 *
   353	 *	raw_spin_rq_lock(rq);
   354	 *	...
   355	 *	raw_spin_rq_unlock(rq);
   356	 *
   357	 * ends up locking and unlocking the _same_ lock, and all CPUs
   358	 * always agree on what rq has what lock.
   359	 *
   360	 * XXX entirely possible to selectively enable cores, don't bother for now.
   361	 */
   362	
   363	static DEFINE_MUTEX(sched_core_mutex);
   364	static atomic_t sched_core_count;
   365	static struct cpumask sched_core_mask;
   366	
   367	static void sched_core_lock(int cpu, unsigned long *flags)
   368	{
   369		const struct cpumask *smt_mask = cpu_smt_mask(cpu);
   370		int t, i = 0;
   371	
   372		local_irq_save(*flags);
   373		for_each_cpu(t, smt_mask)
   374			raw_spin_lock_nested(&cpu_rq(t)->__lock, i++);
   375	}
   376	
   377	static void sched_core_unlock(int cpu, unsigned long *flags)
   378	{
   379		const struct cpumask *smt_mask = cpu_smt_mask(cpu);
   380		int t;
   381	
   382		for_each_cpu(t, smt_mask)
   383			raw_spin_unlock(&cpu_rq(t)->__lock);
   384		local_irq_restore(*flags);
   385	}
   386	
   387	static void __sched_core_flip(bool enabled)
   388	{
   389		unsigned long flags;
   390		int cpu, t;
   391	
   392		cpus_read_lock();
   393	
   394		/*
   395		 * Toggle the online cores, one by one.
   396		 */
   397		cpumask_copy(&sched_core_mask, cpu_online_mask);
   398		for_each_cpu(cpu, &sched_core_mask) {
   399			const struct cpumask *smt_mask = cpu_smt_mask(cpu);
   400	
   401			sched_core_lock(cpu, &flags);
   402	
   403			for_each_cpu(t, smt_mask)
   404				cpu_rq(t)->core_enabled = enabled;
   405	
   406			cpu_rq(cpu)->core->core_forceidle_start = 0;
   407	
   408			sched_core_unlock(cpu, &flags);
   409	
   410			cpumask_andnot(&sched_core_mask, &sched_core_mask, smt_mask);
   411		}
   412	
   413		/*
   414		 * Toggle the offline CPUs.
   415		 */
   416		for_each_cpu_andnot(cpu, cpu_possible_mask, cpu_online_mask)
   417			cpu_rq(cpu)->core_enabled = enabled;
   418	
   419		cpus_read_unlock();
   420	}
   421	
   422	static void sched_core_assert_empty(void)
   423	{
   424		int cpu;
   425	
   426		for_each_possible_cpu(cpu)
   427			WARN_ON_ONCE(!RB_EMPTY_ROOT(&cpu_rq(cpu)->core_tree));
   428	}
   429	
   430	static void __sched_core_enable(void)
   431	{
   432		static_branch_enable(&__sched_core_enabled);
   433		/*
   434		 * Ensure all previous instances of raw_spin_rq_*lock() have finished
   435		 * and future ones will observe !sched_core_disabled().
   436		 */
   437		synchronize_rcu();
   438		__sched_core_flip(true);
   439		sched_core_assert_empty();
   440	}
   441	
   442	static void __sched_core_disable(void)
   443	{
   444		sched_core_assert_empty();
   445		__sched_core_flip(false);
   446		static_branch_disable(&__sched_core_enabled);
   447	}
   448	
   449	void sched_core_get(void)
   450	{
   451		if (atomic_inc_not_zero(&sched_core_count))
   452			return;
   453	
   454		mutex_lock(&sched_core_mutex);
   455		if (!atomic_read(&sched_core_count))
   456			__sched_core_enable();
   457	
   458		smp_mb__before_atomic();
   459		atomic_inc(&sched_core_count);
   460		mutex_unlock(&sched_core_mutex);
   461	}
   462	
   463	static void __sched_core_put(struct work_struct *work)
   464	{
   465		if (atomic_dec_and_mutex_lock(&sched_core_count, &sched_core_mutex)) {
   466			__sched_core_disable();
   467			mutex_unlock(&sched_core_mutex);
   468		}
   469	}
   470	
   471	void sched_core_put(void)
   472	{
   473		static DECLARE_WORK(_work, __sched_core_put);
   474	
   475		/*
   476		 * "There can be only one"
   477		 *
   478		 * Either this is the last one, or we don't actually need to do any
   479		 * 'work'. If it is the last *again*, we rely on
   480		 * WORK_STRUCT_PENDING_BIT.
   481		 */
   482		if (!atomic_add_unless(&sched_core_count, -1, 1))
   483			schedule_work(&_work);
   484	}
   485	
   486	#else /* !CONFIG_SCHED_CORE */
   487	
   488	static inline void sched_core_enqueue(struct rq *rq, struct task_struct *p) { }
   489	static inline void
   490	sched_core_dequeue(struct rq *rq, struct task_struct *p, int flags) { }
   491	
   492	#endif /* CONFIG_SCHED_CORE */
   493	
   494	/*
   495	 * Serialization rules:
   496	 *
   497	 * Lock order:
   498	 *
   499	 *   p->pi_lock
   500	 *     rq->lock
   501	 *       hrtimer_cpu_base->lock (hrtimer_start() for bandwidth controls)
   502	 *
   503	 *  rq1->lock
   504	 *    rq2->lock  where: rq1 < rq2
   505	 *
   506	 * Regular state:
   507	 *
   508	 * Normal scheduling state is serialized by rq->lock. __schedule() takes the
   509	 * local CPU's rq->lock, it optionally removes the task from the runqueue and
   510	 * always looks at the local rq data structures to find the most eligible task
   511	 * to run next.
   512	 *
   513	 * Task enqueue is also under rq->lock, possibly taken from another CPU.
   514	 * Wakeups from another LLC domain might use an IPI to transfer the enqueue to
   515	 * the local CPU to avoid bouncing the runqueue state around [ see
   516	 * ttwu_queue_wakelist() ]
   517	 *
   518	 * Task wakeup, specifically wakeups that involve migration, are horribly
   519	 * complicated to avoid having to take two rq->locks.
   520	 *
   521	 * Special state:
   522	 *
   523	 * System-calls and anything external will use task_rq_lock() which acquires
   524	 * both p->pi_lock and rq->lock. As a consequence the state they change is
   525	 * stable while holding either lock:
   526	 *
   527	 *  - sched_setaffinity()/
   528	 *    set_cpus_allowed_ptr():	p->cpus_ptr, p->nr_cpus_allowed
   529	 *  - set_user_nice():		p->se.load, p->*prio
   530	 *  - __sched_setscheduler():	p->sched_class, p->policy, p->*prio,
   531	 *				p->se.load, p->rt_priority,
   532	 *				p->dl.dl_{runtime, deadline, period, flags, bw, density}
   533	 *  - sched_setnuma():		p->numa_preferred_nid
   534	 *  - sched_move_task():	p->sched_task_group
   535	 *  - uclamp_update_active()	p->uclamp*
   536	 *
   537	 * p->state <- TASK_*:
   538	 *
   539	 *   is changed locklessly using set_current_state(), __set_current_state() or
   540	 *   set_special_state(), see their respective comments, or by
   541	 *   try_to_wake_up(). This latter uses p->pi_lock to serialize against
   542	 *   concurrent self.
   543	 *
   544	 * p->on_rq <- { 0, 1 = TASK_ON_RQ_QUEUED, 2 = TASK_ON_RQ_MIGRATING }:
   545	 *
   546	 *   is set by activate_task() and cleared by deactivate_task(), under
   547	 *   rq->lock. Non-zero indicates the task is runnable, the special
   548	 *   ON_RQ_MIGRATING state is used for migration without holding both
   549	 *   rq->locks. It indicates task_cpu() is not stable, see task_rq_lock().
   550	 *
   551	 *   Additionally it is possible to be ->on_rq but still be considered not
   552	 *   runnable when p->se.sched_delayed is true. These tasks are on the runqueue
   553	 *   but will be dequeued as soon as they get picked again. See the
   554	 *   task_is_runnable() helper.
   555	 *
   556	 * p->on_cpu <- { 0, 1 }:
   557	 *
   558	 *   is set by prepare_task() and cleared by finish_task() such that it will be
   559	 *   set before p is scheduled-in and cleared after p is scheduled-out, both
   560	 *   under rq->lock. Non-zero indicates the task is running on its CPU.
   561	 *
   562	 *   [ The astute reader will observe that it is possible for two tasks on one
   563	 *     CPU to have ->on_cpu = 1 at the same time. ]
   564	 *
   565	 * task_cpu(p): is changed by set_task_cpu(), the rules are:
   566	 *
   567	 *  - Don't call set_task_cpu() on a blocked task:
   568	 *
   569	 *    We don't care what CPU we're not running on, this simplifies hotplug,
   570	 *    the CPU assignment of blocked tasks isn't required to be valid.
   571	 *
   572	 *  - for try_to_wake_up(), called under p->pi_lock:
   573	 *
   574	 *    This allows try_to_wake_up() to only take one rq->lock, see its comment.
   575	 *
   576	 *  - for migration called under rq->lock:
   577	 *    [ see task_on_rq_migrating() in task_rq_lock() ]
   578	 *
   579	 *    o move_queued_task()
   580	 *    o detach_task()
   581	 *
   582	 *  - for migration called under double_rq_lock():
   583	 *
   584	 *    o __migrate_swap_task()
   585	 *    o push_rt_task() / pull_rt_task()
   586	 *    o push_dl_task() / pull_dl_task()
   587	 *    o dl_task_offline_migration()
   588	 *
   589	 */
   590	
   591	void raw_spin_rq_lock_nested(struct rq *rq, int subclass)
   592	{
   593		raw_spinlock_t *lock;
   594	
   595		/* Matches synchronize_rcu() in __sched_core_enable() */
   596		preempt_disable();
   597		if (sched_core_disabled()) {
   598			raw_spin_lock_nested(&rq->__lock, subclass);
   599			/* preempt_count *MUST* be > 1 */
   600			preempt_enable_no_resched();
   601			return;
   602		}
   603	
   604		for (;;) {
   605			lock = __rq_lockp(rq);
   606			raw_spin_lock_nested(lock, subclass);
   607			if (likely(lock == __rq_lockp(rq))) {
   608				/* preempt_count *MUST* be > 1 */
   609				preempt_enable_no_resched();
   610				return;
   611			}
   612			raw_spin_unlock(lock);
   613		}
   614	}
   615	
   616	bool raw_spin_rq_trylock(struct rq *rq)
   617	{
   618		raw_spinlock_t *lock;
   619		bool ret;
   620	
   621		/* Matches synchronize_rcu() in __sched_core_enable() */
   622		preempt_disable();
   623		if (sched_core_disabled()) {
   624			ret = raw_spin_trylock(&rq->__lock);
   625			preempt_enable();
   626			return ret;
   627		}
   628	
   629		for (;;) {
   630			lock = __rq_lockp(rq);
   631			ret = raw_spin_trylock(lock);
   632			if (!ret || (likely(lock == __rq_lockp(rq)))) {
   633				preempt_enable();
   634				return ret;
   635			}
   636			raw_spin_unlock(lock);
   637		}
   638	}
   639	
   640	void raw_spin_rq_unlock(struct rq *rq)
   641	{
   642		raw_spin_unlock(rq_lockp(rq));
   643	}
   644	
   645	#ifdef CONFIG_SMP
   646	/*
   647	 * double_rq_lock - safely lock two runqueues
   648	 */
   649	void double_rq_lock(struct rq *rq1, struct rq *rq2)
   650	{
   651		lockdep_assert_irqs_disabled();
   652	
   653		if (rq_order_less(rq2, rq1))
   654			swap(rq1, rq2);
   655	
   656		raw_spin_rq_lock(rq1);
   657		if (__rq_lockp(rq1) != __rq_lockp(rq2))
   658			raw_spin_rq_lock_nested(rq2, SINGLE_DEPTH_NESTING);
   659	
   660		double_rq_clock_clear_update(rq1, rq2);
   661	}
   662	#endif
   663	
   664	/*
   665	 * __task_rq_lock - lock the rq @p resides on.
   666	 */
   667	struct rq *__task_rq_lock(struct task_struct *p, struct rq_flags *rf)
   668		__acquires(rq->lock)
   669	{
   670		struct rq *rq;
   671	
   672		lockdep_assert_held(&p->pi_lock);
   673	
   674		for (;;) {
   675			rq = task_rq(p);
   676			raw_spin_rq_lock(rq);
   677			if (likely(rq == task_rq(p) && !task_on_rq_migrating(p))) {
   678				rq_pin_lock(rq, rf);
   679				return rq;
   680			}
   681			raw_spin_rq_unlock(rq);
   682	
   683			while (unlikely(task_on_rq_migrating(p)))
   684				cpu_relax();
   685		}
   686	}
   687	
   688	/*
   689	 * task_rq_lock - lock p->pi_lock and lock the rq @p resides on.
   690	 */
   691	struct rq *task_rq_lock(struct task_struct *p, struct rq_flags *rf)
   692		__acquires(p->pi_lock)
   693		__acquires(rq->lock)
   694	{
   695		struct rq *rq;
   696	
   697		for (;;) {
   698			raw_spin_lock_irqsave(&p->pi_lock, rf->flags);
   699			rq = task_rq(p);
   700			raw_spin_rq_lock(rq);
   701			/*
   702			 *	move_queued_task()		task_rq_lock()
   703			 *
   704			 *	ACQUIRE (rq->lock)
   705			 *	[S] ->on_rq = MIGRATING		[L] rq = task_rq()
   706			 *	WMB (__set_task_cpu())		ACQUIRE (rq->lock);
   707			 *	[S] ->cpu = new_cpu		[L] task_rq()
   708			 *					[L] ->on_rq
   709			 *	RELEASE (rq->lock)
   710			 *
   711			 * If we observe the old CPU in task_rq_lock(), the acquire of
   712			 * the old rq->lock will fully serialize against the stores.
   713			 *
   714			 * If we observe the new CPU in task_rq_lock(), the address
   715			 * dependency headed by '[L] rq = task_rq()' and the acquire
   716			 * will pair with the WMB to ensure we then also see migrating.
   717			 */
   718			if (likely(rq == task_rq(p) && !task_on_rq_migrating(p))) {
   719				rq_pin_lock(rq, rf);
   720				return rq;
   721			}
   722			raw_spin_rq_unlock(rq);
   723			raw_spin_unlock_irqrestore(&p->pi_lock, rf->flags);
   724	
   725			while (unlikely(task_on_rq_migrating(p)))
   726				cpu_relax();
   727		}
   728	}
   729	
   730	/*
   731	 * RQ-clock updating methods:
   732	 */
   733	
   734	static void update_rq_clock_task(struct rq *rq, s64 delta)
   735	{
   736	/*
   737	 * In theory, the compile should just see 0 here, and optimize out the call
   738	 * to sched_rt_avg_update. But I don't trust it...
   739	 */
   740		s64 __maybe_unused steal = 0, irq_delta = 0;
   741	
   742	#ifdef CONFIG_IRQ_TIME_ACCOUNTING
   743		irq_delta = irq_time_read(cpu_of(rq)) - rq->prev_irq_time;
   744	
   745		/*
   746		 * Since irq_time is only updated on {soft,}irq_exit, we might run into
   747		 * this case when a previous update_rq_clock() happened inside a
   748		 * {soft,}IRQ region.
   749		 *
   750		 * When this happens, we stop ->clock_task and only update the
   751		 * prev_irq_time stamp to account for the part that fit, so that a next
   752		 * update will consume the rest. This ensures ->clock_task is
   753		 * monotonic.
   754		 *
   755		 * It does however cause some slight miss-attribution of {soft,}IRQ
   756		 * time, a more accurate solution would be to update the irq_time using
   757		 * the current rq->clock timestamp, except that would require using
   758		 * atomic ops.
   759		 */
   760		if (irq_delta > delta)
   761			irq_delta = delta;
   762	
   763		rq->prev_irq_time += irq_delta;
   764		delta -= irq_delta;
   765		delayacct_irq(rq->curr, irq_delta);
   766	#endif
   767	#ifdef CONFIG_PARAVIRT_TIME_ACCOUNTING
   768		if (static_key_false((&paravirt_steal_rq_enabled))) {
   769			u64 prev_steal;
   770	
   771			steal = prev_steal = paravirt_steal_clock(cpu_of(rq));
   772			steal -= rq->prev_steal_time_rq;
   773	
   774			if (unlikely(steal > delta))
   775				steal = delta;
   776	
   777			rq->prev_steal_time_rq = prev_steal;
   778			delta -= steal;
   779		}
   780	#endif
   781	
   782		rq->clock_task += delta;
   783	
   784	#ifdef CONFIG_HAVE_SCHED_AVG_IRQ
   785		if ((irq_delta + steal) && sched_feat(NONTASK_CAPACITY))
   786			update_irq_load_avg(rq, irq_delta + steal);
   787	#endif
   788		update_rq_clock_pelt(rq, delta);
   789	}
   790	
   791	void update_rq_clock(struct rq *rq)
   792	{
   793		s64 delta;
   794	
   795		lockdep_assert_rq_held(rq);
   796	
   797		if (rq->clock_update_flags & RQCF_ACT_SKIP)
   798			return;
   799	
   800	#ifdef CONFIG_SCHED_DEBUG
   801		if (sched_feat(WARN_DOUBLE_CLOCK))
   802			SCHED_WARN_ON(rq->clock_update_flags & RQCF_UPDATED);
   803		rq->clock_update_flags |= RQCF_UPDATED;
   804	#endif
   805	
   806		delta = sched_clock_cpu(cpu_of(rq)) - rq->clock;
   807		if (delta < 0)
   808			return;
   809		rq->clock += delta;
   810		update_rq_clock_task(rq, delta);
   811	}
   812	
   813	#ifdef CONFIG_SCHED_HRTICK
   814	/*
   815	 * Use HR-timers to deliver accurate preemption points.
   816	 */
   817	
   818	static void hrtick_clear(struct rq *rq)
   819	{
   820		if (hrtimer_active(&rq->hrtick_timer))
   821			hrtimer_cancel(&rq->hrtick_timer);
   822	}
   823	
   824	/*
   825	 * High-resolution timer tick.
   826	 * Runs from hardirq context with interrupts disabled.
   827	 */
   828	static enum hrtimer_restart hrtick(struct hrtimer *timer)
   829	{
   830		struct rq *rq = container_of(timer, struct rq, hrtick_timer);
   831		struct rq_flags rf;
   832	
   833		WARN_ON_ONCE(cpu_of(rq) != smp_processor_id());
   834	
   835		rq_lock(rq, &rf);
   836		update_rq_clock(rq);
   837		rq->donor->sched_class->task_tick(rq, rq->curr, 1);
   838		rq_unlock(rq, &rf);
   839	
   840		return HRTIMER_NORESTART;
   841	}
   842	
   843	#ifdef CONFIG_SMP
   844	
   845	static void __hrtick_restart(struct rq *rq)
   846	{
   847		struct hrtimer *timer = &rq->hrtick_timer;
   848		ktime_t time = rq->hrtick_time;
   849	
   850		hrtimer_start(timer, time, HRTIMER_MODE_ABS_PINNED_HARD);
   851	}
   852	
   853	/*
   854	 * called from hardirq (IPI) context
   855	 */
   856	static void __hrtick_start(void *arg)
   857	{
   858		struct rq *rq = arg;
   859		struct rq_flags rf;
   860	
   861		rq_lock(rq, &rf);
   862		__hrtick_restart(rq);
   863		rq_unlock(rq, &rf);
   864	}
   865	
   866	/*
   867	 * Called to set the hrtick timer state.
   868	 *
   869	 * called with rq->lock held and IRQs disabled
   870	 */
   871	void hrtick_start(struct rq *rq, u64 delay)
   872	{
   873		struct hrtimer *timer = &rq->hrtick_timer;
   874		s64 delta;
   875	
   876		/*
   877		 * Don't schedule slices shorter than 10000ns, that just
   878		 * doesn't make sense and can cause timer DoS.
   879		 */
   880		delta = max_t(s64, delay, 10000LL);
   881		rq->hrtick_time = ktime_add_ns(timer->base->get_time(), delta);
   882	
   883		if (rq == this_rq())
   884			__hrtick_restart(rq);
   885		else
   886			smp_call_function_single_async(cpu_of(rq), &rq->hrtick_csd);
   887	}
   888	
   889	#else
   890	/*
   891	 * Called to set the hrtick timer state.
   892	 *
   893	 * called with rq->lock held and IRQs disabled
   894	 */
   895	void hrtick_start(struct rq *rq, u64 delay)
   896	{
   897		/*
   898		 * Don't schedule slices shorter than 10000ns, that just
   899		 * doesn't make sense. Rely on vruntime for fairness.
   900		 */
   901		delay = max_t(u64, delay, 10000LL);
   902		hrtimer_start(&rq->hrtick_timer, ns_to_ktime(delay),
   903			      HRTIMER_MODE_REL_PINNED_HARD);
   904	}
   905	
   906	#endif /* CONFIG_SMP */
   907	
   908	static void hrtick_rq_init(struct rq *rq)
   909	{
   910	#ifdef CONFIG_SMP
   911		INIT_CSD(&rq->hrtick_csd, __hrtick_start, rq);
   912	#endif
   913		hrtimer_init(&rq->hrtick_timer, CLOCK_MONOTONIC, HRTIMER_MODE_REL_HARD);
   914		rq->hrtick_timer.function = hrtick;
   915	}
   916	#else	/* CONFIG_SCHED_HRTICK */
   917	static inline void hrtick_clear(struct rq *rq)
   918	{
   919	}
   920	
   921	static inline void hrtick_rq_init(struct rq *rq)
   922	{
   923	}
   924	#endif	/* CONFIG_SCHED_HRTICK */
   925	
   926	/*
   927	 * try_cmpxchg based fetch_or() macro so it works for different integer types:
   928	 */
   929	#define fetch_or(ptr, mask)						\
   930		({								\
   931			typeof(ptr) _ptr = (ptr);				\
   932			typeof(mask) _mask = (mask);				\
   933			typeof(*_ptr) _val = *_ptr;				\
   934										\
   935			do {							\
   936			} while (!try_cmpxchg(_ptr, &_val, _val | _mask));	\
   937		_val;								\
   938	})
   939	
   940	#if defined(CONFIG_SMP) && defined(TIF_POLLING_NRFLAG)
   941	/*
   942	 * Atomically set TIF_NEED_RESCHED and test for TIF_POLLING_NRFLAG,
   943	 * this avoids any races wrt polling state changes and thereby avoids
   944	 * spurious IPIs.
   945	 */
   946	static inline bool set_nr_and_not_polling(struct thread_info *ti, int tif)
   947	{
   948		return !(fetch_or(&ti->flags, 1 << tif) & _TIF_POLLING_NRFLAG);
   949	}
   950	
   951	/*
   952	 * Atomically set TIF_NEED_RESCHED if TIF_POLLING_NRFLAG is set.
   953	 *
   954	 * If this returns true, then the idle task promises to call
   955	 * sched_ttwu_pending() and reschedule soon.
   956	 */
   957	static bool set_nr_if_polling(struct task_struct *p)
   958	{
   959		struct thread_info *ti = task_thread_info(p);
   960		typeof(ti->flags) val = READ_ONCE(ti->flags);
   961	
   962		do {
   963			if (!(val & _TIF_POLLING_NRFLAG))
   964				return false;
   965			if (val & _TIF_NEED_RESCHED)
   966				return true;
   967		} while (!try_cmpxchg(&ti->flags, &val, val | _TIF_NEED_RESCHED));
   968	
   969		return true;
   970	}
   971	
   972	#else
   973	static inline bool set_nr_and_not_polling(struct thread_info *ti, int tif)
   974	{
   975		set_ti_thread_flag(ti, tif);
   976		return true;
   977	}
   978	
   979	#ifdef CONFIG_SMP
   980	static inline bool set_nr_if_polling(struct task_struct *p)
   981	{
   982		return false;
   983	}
   984	#endif
   985	#endif
   986	
   987	static bool __wake_q_add(struct wake_q_head *head, struct task_struct *task)
   988	{
   989		struct wake_q_node *node = &task->wake_q;
   990	
   991		/*
   992		 * Atomically grab the task, if ->wake_q is !nil already it means
   993		 * it's already queued (either by us or someone else) and will get the
   994		 * wakeup due to that.
   995		 *
   996		 * In order to ensure that a pending wakeup will observe our pending
   997		 * state, even in the failed case, an explicit smp_mb() must be used.
   998		 */
   999		smp_mb__before_atomic();
  1000		if (unlikely(cmpxchg_relaxed(&node->next, NULL, WAKE_Q_TAIL)))
  1001			return false;
  1002	
  1003		/*
  1004		 * The head is context local, there can be no concurrency.
  1005		 */
  1006		*head->lastp = node;
  1007		head->lastp = &node->next;
  1008		return true;
  1009	}
  1010	
  1011	/**
  1012	 * wake_q_add() - queue a wakeup for 'later' waking.
  1013	 * @head: the wake_q_head to add @task to
  1014	 * @task: the task to queue for 'later' wakeup
  1015	 *
  1016	 * Queue a task for later wakeup, most likely by the wake_up_q() call in the
  1017	 * same context, _HOWEVER_ this is not guaranteed, the wakeup can come
  1018	 * instantly.
  1019	 *
  1020	 * This function must be used as-if it were wake_up_process(); IOW the task
  1021	 * must be ready to be woken at this location.
  1022	 */
  1023	void wake_q_add(struct wake_q_head *head, struct task_struct *task)
  1024	{
  1025		if (__wake_q_add(head, task))
  1026			get_task_struct(task);
  1027	}
  1028	
  1029	/**
  1030	 * wake_q_add_safe() - safely queue a wakeup for 'later' waking.
  1031	 * @head: the wake_q_head to add @task to
  1032	 * @task: the task to queue for 'later' wakeup
  1033	 *
  1034	 * Queue a task for later wakeup, most likely by the wake_up_q() call in the
  1035	 * same context, _HOWEVER_ this is not guaranteed, the wakeup can come
  1036	 * instantly.
  1037	 *
  1038	 * This function must be used as-if it were wake_up_process(); IOW the task
  1039	 * must be ready to be woken at this location.
  1040	 *
  1041	 * This function is essentially a task-safe equivalent to wake_q_add(). Callers
  1042	 * that already hold reference to @task can call the 'safe' version and trust
  1043	 * wake_q to do the right thing depending whether or not the @task is already
  1044	 * queued for wakeup.
  1045	 */
  1046	void wake_q_add_safe(struct wake_q_head *head, struct task_struct *task)
  1047	{
  1048		if (!__wake_q_add(head, task))
  1049			put_task_struct(task);
  1050	}
  1051	
  1052	void wake_up_q(struct wake_q_head *head)
  1053	{
  1054		struct wake_q_node *node = head->first;
  1055	
  1056		while (node != WAKE_Q_TAIL) {
  1057			struct task_struct *task;
  1058	
  1059			task = container_of(node, struct task_struct, wake_q);
  1060			/* Task can safely be re-inserted now: */
  1061			node = node->next;
  1062			task->wake_q.next = NULL;
  1063	
  1064			/*
  1065			 * wake_up_process() executes a full barrier, which pairs with
  1066			 * the queueing in wake_q_add() so as not to miss wakeups.
  1067			 */
  1068			wake_up_process(task);
  1069			put_task_struct(task);
  1070		}
  1071	}
  1072	
  1073	/*
  1074	 * resched_curr - mark rq's current task 'to be rescheduled now'.
  1075	 *
  1076	 * On UP this means the setting of the need_resched flag, on SMP it
  1077	 * might also involve a cross-CPU call to trigger the scheduler on
  1078	 * the target CPU.
  1079	 */
  1080	static void __resched_curr(struct rq *rq, int tif)
  1081	{
  1082		struct task_struct *curr = rq->curr;
  1083		struct thread_info *cti = task_thread_info(curr);
  1084		int cpu;
  1085	
  1086		lockdep_assert_rq_held(rq);
  1087	
  1088		/*
  1089		 * Always immediately preempt the idle task; no point in delaying doing
  1090		 * actual work.
  1091		 */
  1092		if (is_idle_task(curr) && tif == TIF_NEED_RESCHED_LAZY)
  1093			tif = TIF_NEED_RESCHED;
  1094	
  1095		if (cti->flags & ((1 << tif) | _TIF_NEED_RESCHED))
  1096			return;
  1097	
  1098		cpu = cpu_of(rq);
  1099	
  1100		if (cpu == smp_processor_id()) {
  1101			set_ti_thread_flag(cti, tif);
  1102			if (tif == TIF_NEED_RESCHED)
  1103				set_preempt_need_resched();
  1104			return;
  1105		}
  1106	
  1107		if (set_nr_and_not_polling(cti, tif)) {
  1108			if (tif == TIF_NEED_RESCHED)
  1109				smp_send_reschedule(cpu);
  1110		} else {
  1111			trace_sched_wake_idle_without_ipi(cpu);
  1112		}
  1113	}
  1114	
  1115	void resched_curr(struct rq *rq)
  1116	{
  1117		__resched_curr(rq, TIF_NEED_RESCHED);
  1118	}
  1119	
  1120	#ifdef CONFIG_PREEMPT_DYNAMIC
  1121	static DEFINE_STATIC_KEY_FALSE(sk_dynamic_preempt_lazy);
  1122	static __always_inline bool dynamic_preempt_lazy(void)
  1123	{
  1124		return static_branch_unlikely(&sk_dynamic_preempt_lazy);
  1125	}
  1126	#else
  1127	static __always_inline bool dynamic_preempt_lazy(void)
  1128	{
  1129		return IS_ENABLED(CONFIG_PREEMPT_LAZY);
  1130	}
  1131	#endif
  1132	
  1133	static __always_inline int get_lazy_tif_bit(void)
  1134	{
  1135		if (dynamic_preempt_lazy())
  1136			return TIF_NEED_RESCHED_LAZY;
  1137	
  1138		return TIF_NEED_RESCHED;
  1139	}
  1140	
  1141	void resched_curr_lazy(struct rq *rq)
  1142	{
  1143		__resched_curr(rq, get_lazy_tif_bit());
  1144	}
  1145	
  1146	void resched_cpu(int cpu)
  1147	{
  1148		struct rq *rq = cpu_rq(cpu);
  1149		unsigned long flags;
  1150	
  1151		raw_spin_rq_lock_irqsave(rq, flags);
  1152		if (cpu_online(cpu) || cpu == smp_processor_id())
  1153			resched_curr(rq);
  1154		raw_spin_rq_unlock_irqrestore(rq, flags);
  1155	}
  1156	
  1157	#ifdef CONFIG_SMP
  1158	#ifdef CONFIG_NO_HZ_COMMON
  1159	/*
  1160	 * In the semi idle case, use the nearest busy CPU for migrating timers
  1161	 * from an idle CPU.  This is good for power-savings.
  1162	 *
  1163	 * We don't do similar optimization for completely idle system, as
  1164	 * selecting an idle CPU will add more delays to the timers than intended
  1165	 * (as that CPU's timer base may not be up to date wrt jiffies etc).
  1166	 */
  1167	int get_nohz_timer_target(void)
  1168	{
  1169		int i, cpu = smp_processor_id(), default_cpu = -1;
  1170		struct sched_domain *sd;
  1171		const struct cpumask *hk_mask;
  1172	
  1173		if (housekeeping_cpu(cpu, HK_TYPE_KERNEL_NOISE)) {
  1174			if (!idle_cpu(cpu))
  1175				return cpu;
  1176			default_cpu = cpu;
  1177		}
  1178	
  1179		hk_mask = housekeeping_cpumask(HK_TYPE_KERNEL_NOISE);
  1180	
  1181		guard(rcu)();
  1182	
  1183		for_each_domain(cpu, sd) {
  1184			for_each_cpu_and(i, sched_domain_span(sd), hk_mask) {
  1185				if (cpu == i)
  1186					continue;
  1187	
  1188				if (!idle_cpu(i))
  1189					return i;
  1190			}
  1191		}
  1192	
  1193		if (default_cpu == -1)
  1194			default_cpu = housekeeping_any_cpu(HK_TYPE_KERNEL_NOISE);
  1195	
  1196		return default_cpu;
  1197	}
  1198	
  1199	/*
  1200	 * When add_timer_on() enqueues a timer into the timer wheel of an
  1201	 * idle CPU then this timer might expire before the next timer event
  1202	 * which is scheduled to wake up that CPU. In case of a completely
  1203	 * idle system the next event might even be infinite time into the
  1204	 * future. wake_up_idle_cpu() ensures that the CPU is woken up and
  1205	 * leaves the inner idle loop so the newly added timer is taken into
  1206	 * account when the CPU goes back to idle and evaluates the timer
  1207	 * wheel for the next timer event.
  1208	 */
  1209	static void wake_up_idle_cpu(int cpu)
  1210	{
  1211		struct rq *rq = cpu_rq(cpu);
  1212	
  1213		if (cpu == smp_processor_id())
  1214			return;
  1215	
  1216		/*
  1217		 * Set TIF_NEED_RESCHED and send an IPI if in the non-polling
  1218		 * part of the idle loop. This forces an exit from the idle loop
  1219		 * and a round trip to schedule(). Now this could be optimized
  1220		 * because a simple new idle loop iteration is enough to
  1221		 * re-evaluate the next tick. Provided some re-ordering of tick
  1222		 * nohz functions that would need to follow TIF_NR_POLLING
  1223		 * clearing:
  1224		 *
  1225		 * - On most architectures, a simple fetch_or on ti::flags with a
  1226		 *   "0" value would be enough to know if an IPI needs to be sent.
  1227		 *
  1228		 * - x86 needs to perform a last need_resched() check between
  1229		 *   monitor and mwait which doesn't take timers into account.
  1230		 *   There a dedicated TIF_TIMER flag would be required to
  1231		 *   fetch_or here and be checked along with TIF_NEED_RESCHED
  1232		 *   before mwait().
  1233		 *
  1234		 * However, remote timer enqueue is not such a frequent event
  1235		 * and testing of the above solutions didn't appear to report
  1236		 * much benefits.
  1237		 */
  1238		if (set_nr_and_not_polling(task_thread_info(rq->idle), TIF_NEED_RESCHED))
  1239			smp_send_reschedule(cpu);
  1240		else
  1241			trace_sched_wake_idle_without_ipi(cpu);
  1242	}
  1243	
  1244	static bool wake_up_full_nohz_cpu(int cpu)
  1245	{
  1246		/*
  1247		 * We just need the target to call irq_exit() and re-evaluate
  1248		 * the next tick. The nohz full kick at least implies that.
  1249		 * If needed we can still optimize that later with an
  1250		 * empty IRQ.
  1251		 */
  1252		if (cpu_is_offline(cpu))
  1253			return true;  /* Don't try to wake offline CPUs. */
  1254		if (tick_nohz_full_cpu(cpu)) {
  1255			if (cpu != smp_processor_id() ||
  1256			    tick_nohz_tick_stopped())
  1257				tick_nohz_full_kick_cpu(cpu);
  1258			return true;
  1259		}
  1260	
  1261		return false;
  1262	}
  1263	
  1264	/*
  1265	 * Wake up the specified CPU.  If the CPU is going offline, it is the
  1266	 * caller's responsibility to deal with the lost wakeup, for example,
  1267	 * by hooking into the CPU_DEAD notifier like timers and hrtimers do.
  1268	 */
  1269	void wake_up_nohz_cpu(int cpu)
  1270	{
  1271		if (!wake_up_full_nohz_cpu(cpu))
  1272			wake_up_idle_cpu(cpu);
  1273	}
  1274	
  1275	static void nohz_csd_func(void *info)
  1276	{
  1277		struct rq *rq = info;
  1278		int cpu = cpu_of(rq);
  1279		unsigned int flags;
  1280	
  1281		/*
  1282		 * Release the rq::nohz_csd.
  1283		 */
  1284		flags = atomic_fetch_andnot(NOHZ_KICK_MASK | NOHZ_NEWILB_KICK, nohz_flags(cpu));
  1285		WARN_ON(!(flags & NOHZ_KICK_MASK));
  1286	
  1287		rq->idle_balance = idle_cpu(cpu);
  1288		if (rq->idle_balance) {
  1289			rq->nohz_idle_balance = flags;
  1290			__raise_softirq_irqoff(SCHED_SOFTIRQ);
  1291		}
  1292	}
  1293	
  1294	#endif /* CONFIG_NO_HZ_COMMON */
  1295	
  1296	#ifdef CONFIG_NO_HZ_FULL
  1297	static inline bool __need_bw_check(struct rq *rq, struct task_struct *p)
  1298	{
  1299		if (rq->nr_running != 1)
  1300			return false;
  1301	
  1302		if (p->sched_class != &fair_sched_class)
  1303			return false;
  1304	
  1305		if (!task_on_rq_queued(p))
  1306			return false;
  1307	
  1308		return true;
  1309	}
  1310	
  1311	bool sched_can_stop_tick(struct rq *rq)
  1312	{
  1313		int fifo_nr_running;
  1314	
  1315		/* Deadline tasks, even if single, need the tick */
  1316		if (rq->dl.dl_nr_running)
  1317			return false;
  1318	
  1319		/*
  1320		 * If there are more than one RR tasks, we need the tick to affect the
  1321		 * actual RR behaviour.
  1322		 */
  1323		if (rq->rt.rr_nr_running) {
  1324			if (rq->rt.rr_nr_running == 1)
  1325				return true;
  1326			else
  1327				return false;
  1328		}
  1329	
  1330		/*
  1331		 * If there's no RR tasks, but FIFO tasks, we can skip the tick, no
  1332		 * forced preemption between FIFO tasks.
  1333		 */
  1334		fifo_nr_running = rq->rt.rt_nr_running - rq->rt.rr_nr_running;
  1335		if (fifo_nr_running)
  1336			return true;
  1337	
  1338		/*
  1339		 * If there are no DL,RR/FIFO tasks, there must only be CFS or SCX tasks
  1340		 * left. For CFS, if there's more than one we need the tick for
  1341		 * involuntary preemption. For SCX, ask.
  1342		 */
  1343		if (scx_enabled() && !scx_can_stop_tick(rq))
  1344			return false;
  1345	
  1346		if (rq->cfs.nr_running > 1)
  1347			return false;
  1348	
  1349		/*
  1350		 * If there is one task and it has CFS runtime bandwidth constraints
  1351		 * and it's on the cpu now we don't want to stop the tick.
  1352		 * This check prevents clearing the bit if a newly enqueued task here is
  1353		 * dequeued by migrating while the constrained task continues to run.
  1354		 * E.g. going from 2->1 without going through pick_next_task().
  1355		 */
  1356		if (__need_bw_check(rq, rq->curr)) {
  1357			if (cfs_task_bw_constrained(rq->curr))
  1358				return false;
  1359		}
  1360	
  1361		return true;
  1362	}
  1363	#endif /* CONFIG_NO_HZ_FULL */
  1364	#endif /* CONFIG_SMP */
  1365	
  1366	#if defined(CONFIG_RT_GROUP_SCHED) || (defined(CONFIG_FAIR_GROUP_SCHED) && \
  1367				(defined(CONFIG_SMP) || defined(CONFIG_CFS_BANDWIDTH)))
  1368	/*
  1369	 * Iterate task_group tree rooted at *from, calling @down when first entering a
  1370	 * node and @up when leaving it for the final time.
  1371	 *
  1372	 * Caller must hold rcu_lock or sufficient equivalent.
  1373	 */
  1374	int walk_tg_tree_from(struct task_group *from,
  1375				     tg_visitor down, tg_visitor up, void *data)
  1376	{
  1377		struct task_group *parent, *child;
  1378		int ret;
  1379	
  1380		parent = from;
  1381	
  1382	down:
  1383		ret = (*down)(parent, data);
  1384		if (ret)
  1385			goto out;
  1386		list_for_each_entry_rcu(child, &parent->children, siblings) {
  1387			parent = child;
  1388			goto down;
  1389	
  1390	up:
  1391			continue;
  1392		}
  1393		ret = (*up)(parent, data);
  1394		if (ret || parent == from)
  1395			goto out;
  1396	
  1397		child = parent;
  1398		parent = parent->parent;
  1399		if (parent)
  1400			goto up;
  1401	out:
  1402		return ret;
  1403	}
  1404	
  1405	int tg_nop(struct task_group *tg, void *data)
  1406	{
  1407		return 0;
  1408	}
  1409	#endif
  1410	
  1411	void set_load_weight(struct task_struct *p, bool update_load)
  1412	{
  1413		int prio = p->static_prio - MAX_RT_PRIO;
  1414		struct load_weight lw;
  1415	
  1416		if (task_has_idle_policy(p)) {
  1417			lw.weight = scale_load(WEIGHT_IDLEPRIO);
  1418			lw.inv_weight = WMULT_IDLEPRIO;
  1419		} else {
  1420			lw.weight = scale_load(sched_prio_to_weight[prio]);
  1421			lw.inv_weight = sched_prio_to_wmult[prio];
  1422		}
  1423	
  1424		/*
  1425		 * SCHED_OTHER tasks have to update their load when changing their
  1426		 * weight
  1427		 */
  1428		if (update_load && p->sched_class->reweight_task)
  1429			p->sched_class->reweight_task(task_rq(p), p, &lw);
  1430		else
  1431			p->se.load = lw;
  1432	}
  1433	
  1434	#ifdef CONFIG_UCLAMP_TASK
  1435	/*
  1436	 * Serializes updates of utilization clamp values
  1437	 *
  1438	 * The (slow-path) user-space triggers utilization clamp value updates which
  1439	 * can require updates on (fast-path) scheduler's data structures used to
  1440	 * support enqueue/dequeue operations.
  1441	 * While the per-CPU rq lock protects fast-path update operations, user-space
  1442	 * requests are serialized using a mutex to reduce the risk of conflicting
  1443	 * updates or API abuses.
  1444	 */
  1445	static __maybe_unused DEFINE_MUTEX(uclamp_mutex);
  1446	
  1447	/* Max allowed minimum utilization */
  1448	static unsigned int __maybe_unused sysctl_sched_uclamp_util_min = SCHED_CAPACITY_SCALE;
  1449	
  1450	/* Max allowed maximum utilization */
  1451	static unsigned int __maybe_unused sysctl_sched_uclamp_util_max = SCHED_CAPACITY_SCALE;
  1452	
  1453	/*
  1454	 * By default RT tasks run at the maximum performance point/capacity of the
  1455	 * system. Uclamp enforces this by always setting UCLAMP_MIN of RT tasks to
  1456	 * SCHED_CAPACITY_SCALE.
  1457	 *
  1458	 * This knob allows admins to change the default behavior when uclamp is being
  1459	 * used. In battery powered devices, particularly, running at the maximum
  1460	 * capacity and frequency will increase energy consumption and shorten the
  1461	 * battery life.
  1462	 *
  1463	 * This knob only affects RT tasks that their uclamp_se->user_defined == false.
  1464	 *
  1465	 * This knob will not override the system default sched_util_clamp_min defined
  1466	 * above.
  1467	 */
  1468	unsigned int sysctl_sched_uclamp_util_min_rt_default = SCHED_CAPACITY_SCALE;
  1469	
  1470	/* All clamps are required to be less or equal than these values */
  1471	static struct uclamp_se uclamp_default[UCLAMP_CNT];
  1472	
  1473	/*
  1474	 * This static key is used to reduce the uclamp overhead in the fast path. It
  1475	 * primarily disables the call to uclamp_rq_{inc, dec}() in
  1476	 * enqueue/dequeue_task().
  1477	 *
  1478	 * This allows users to continue to enable uclamp in their kernel config with
  1479	 * minimum uclamp overhead in the fast path.
  1480	 *
  1481	 * As soon as userspace modifies any of the uclamp knobs, the static key is
  1482	 * enabled, since we have an actual users that make use of uclamp
  1483	 * functionality.
  1484	 *
  1485	 * The knobs that would enable this static key are:
  1486	 *
  1487	 *   * A task modifying its uclamp value with sched_setattr().
  1488	 *   * An admin modifying the sysctl_sched_uclamp_{min, max} via procfs.
  1489	 *   * An admin modifying the cgroup cpu.uclamp.{min, max}
  1490	 */
  1491	DEFINE_STATIC_KEY_FALSE(sched_uclamp_used);
  1492	
  1493	static inline unsigned int
  1494	uclamp_idle_value(struct rq *rq, enum uclamp_id clamp_id,
  1495			  unsigned int clamp_value)
  1496	{
  1497		/*
  1498		 * Avoid blocked utilization pushing up the frequency when we go
  1499		 * idle (which drops the max-clamp) by retaining the last known
  1500		 * max-clamp.
  1501		 */
  1502		if (clamp_id == UCLAMP_MAX) {
  1503			rq->uclamp_flags |= UCLAMP_FLAG_IDLE;
  1504			return clamp_value;
  1505		}
  1506	
  1507		return uclamp_none(UCLAMP_MIN);
  1508	}
  1509	
  1510	static inline void uclamp_idle_reset(struct rq *rq, enum uclamp_id clamp_id,
  1511					     unsigned int clamp_value)
  1512	{
  1513		/* Reset max-clamp retention only on idle exit */
  1514		if (!(rq->uclamp_flags & UCLAMP_FLAG_IDLE))
  1515			return;
  1516	
  1517		uclamp_rq_set(rq, clamp_id, clamp_value);
  1518	}
  1519	
  1520	static inline
  1521	unsigned int uclamp_rq_max_value(struct rq *rq, enum uclamp_id clamp_id,
  1522					   unsigned int clamp_value)
  1523	{
  1524		struct uclamp_bucket *bucket = rq->uclamp[clamp_id].bucket;
  1525		int bucket_id = UCLAMP_BUCKETS - 1;
  1526	
  1527		/*
  1528		 * Since both min and max clamps are max aggregated, find the
  1529		 * top most bucket with tasks in.
  1530		 */
  1531		for ( ; bucket_id >= 0; bucket_id--) {
  1532			if (!bucket[bucket_id].tasks)
  1533				continue;
  1534			return bucket[bucket_id].value;
  1535		}
  1536	
  1537		/* No tasks -- default clamp values */
  1538		return uclamp_idle_value(rq, clamp_id, clamp_value);
  1539	}
  1540	
  1541	static void __uclamp_update_util_min_rt_default(struct task_struct *p)
  1542	{
  1543		unsigned int default_util_min;
  1544		struct uclamp_se *uc_se;
  1545	
  1546		lockdep_assert_held(&p->pi_lock);
  1547	
  1548		uc_se = &p->uclamp_req[UCLAMP_MIN];
  1549	
  1550		/* Only sync if user didn't override the default */
  1551		if (uc_se->user_defined)
  1552			return;
  1553	
  1554		default_util_min = sysctl_sched_uclamp_util_min_rt_default;
  1555		uclamp_se_set(uc_se, default_util_min, false);
  1556	}
  1557	
  1558	static void uclamp_update_util_min_rt_default(struct task_struct *p)
  1559	{
  1560		if (!rt_task(p))
  1561			return;
  1562	
  1563		/* Protect updates to p->uclamp_* */
  1564		guard(task_rq_lock)(p);
  1565		__uclamp_update_util_min_rt_default(p);
  1566	}
  1567	
  1568	static inline struct uclamp_se
  1569	uclamp_tg_restrict(struct task_struct *p, enum uclamp_id clamp_id)
  1570	{
  1571		/* Copy by value as we could modify it */
  1572		struct uclamp_se uc_req = p->uclamp_req[clamp_id];
  1573	#ifdef CONFIG_UCLAMP_TASK_GROUP
  1574		unsigned int tg_min, tg_max, value;
  1575	
  1576		/*
  1577		 * Tasks in autogroups or root task group will be
  1578		 * restricted by system defaults.
  1579		 */
  1580		if (task_group_is_autogroup(task_group(p)))
  1581			return uc_req;
  1582		if (task_group(p) == &root_task_group)
  1583			return uc_req;
  1584	
  1585		tg_min = task_group(p)->uclamp[UCLAMP_MIN].value;
  1586		tg_max = task_group(p)->uclamp[UCLAMP_MAX].value;
  1587		value = uc_req.value;
  1588		value = clamp(value, tg_min, tg_max);
  1589		uclamp_se_set(&uc_req, value, false);
  1590	#endif
  1591	
  1592		return uc_req;
  1593	}
  1594	
  1595	/*
  1596	 * The effective clamp bucket index of a task depends on, by increasing
  1597	 * priority:
  1598	 * - the task specific clamp value, when explicitly requested from userspace
  1599	 * - the task group effective clamp value, for tasks not either in the root
  1600	 *   group or in an autogroup
  1601	 * - the system default clamp value, defined by the sysadmin
  1602	 */
  1603	static inline struct uclamp_se
  1604	uclamp_eff_get(struct task_struct *p, enum uclamp_id clamp_id)
  1605	{
  1606		struct uclamp_se uc_req = uclamp_tg_restrict(p, clamp_id);
  1607		struct uclamp_se uc_max = uclamp_default[clamp_id];
  1608	
  1609		/* System default restrictions always apply */
  1610		if (unlikely(uc_req.value > uc_max.value))
  1611			return uc_max;
  1612	
  1613		return uc_req;
  1614	}
  1615	
  1616	unsigned long uclamp_eff_value(struct task_struct *p, enum uclamp_id clamp_id)
  1617	{
  1618		struct uclamp_se uc_eff;
  1619	
  1620		/* Task currently refcounted: use back-annotated (effective) value */
  1621		if (p->uclamp[clamp_id].active)
  1622			return (unsigned long)p->uclamp[clamp_id].value;
  1623	
  1624		uc_eff = uclamp_eff_get(p, clamp_id);
  1625	
  1626		return (unsigned long)uc_eff.value;
  1627	}
  1628	
  1629	/*
  1630	 * When a task is enqueued on a rq, the clamp bucket currently defined by the
  1631	 * task's uclamp::bucket_id is refcounted on that rq. This also immediately
  1632	 * updates the rq's clamp value if required.
  1633	 *
  1634	 * Tasks can have a task-specific value requested from user-space, track
  1635	 * within each bucket the maximum value for tasks refcounted in it.
  1636	 * This "local max aggregation" allows to track the exact "requested" value
  1637	 * for each bucket when all its RUNNABLE tasks require the same clamp.
  1638	 */
  1639	static inline void uclamp_rq_inc_id(struct rq *rq, struct task_struct *p,
  1640					    enum uclamp_id clamp_id)
  1641	{
  1642		struct uclamp_rq *uc_rq = &rq->uclamp[clamp_id];
  1643		struct uclamp_se *uc_se = &p->uclamp[clamp_id];
  1644		struct uclamp_bucket *bucket;
  1645	
  1646		lockdep_assert_rq_held(rq);
  1647	
  1648		/* Update task effective clamp */
  1649		p->uclamp[clamp_id] = uclamp_eff_get(p, clamp_id);
  1650	
  1651		bucket = &uc_rq->bucket[uc_se->bucket_id];
  1652		bucket->tasks++;
  1653		uc_se->active = true;
  1654	
  1655		uclamp_idle_reset(rq, clamp_id, uc_se->value);
  1656	
  1657		/*
  1658		 * Local max aggregation: rq buckets always track the max
  1659		 * "requested" clamp value of its RUNNABLE tasks.
  1660		 */
  1661		if (bucket->tasks == 1 || uc_se->value > bucket->value)
  1662			bucket->value = uc_se->value;
  1663	
  1664		if (uc_se->value > uclamp_rq_get(rq, clamp_id))
  1665			uclamp_rq_set(rq, clamp_id, uc_se->value);
  1666	}
  1667	
  1668	/*
  1669	 * When a task is dequeued from a rq, the clamp bucket refcounted by the task
  1670	 * is released. If this is the last task reference counting the rq's max
  1671	 * active clamp value, then the rq's clamp value is updated.
  1672	 *
  1673	 * Both refcounted tasks and rq's cached clamp values are expected to be
  1674	 * always valid. If it's detected they are not, as defensive programming,
  1675	 * enforce the expected state and warn.
  1676	 */
  1677	static inline void uclamp_rq_dec_id(struct rq *rq, struct task_struct *p,
  1678					    enum uclamp_id clamp_id)
  1679	{
  1680		struct uclamp_rq *uc_rq = &rq->uclamp[clamp_id];
  1681		struct uclamp_se *uc_se = &p->uclamp[clamp_id];
  1682		struct uclamp_bucket *bucket;
  1683		unsigned int bkt_clamp;
  1684		unsigned int rq_clamp;
  1685	
  1686		lockdep_assert_rq_held(rq);
  1687	
  1688		/*
  1689		 * If sched_uclamp_used was enabled after task @p was enqueued,
  1690		 * we could end up with unbalanced call to uclamp_rq_dec_id().
  1691		 *
  1692		 * In this case the uc_se->active flag should be false since no uclamp
  1693		 * accounting was performed at enqueue time and we can just return
  1694		 * here.
  1695		 *
  1696		 * Need to be careful of the following enqueue/dequeue ordering
  1697		 * problem too
  1698		 *
  1699		 *	enqueue(taskA)
  1700		 *	// sched_uclamp_used gets enabled
  1701		 *	enqueue(taskB)
  1702		 *	dequeue(taskA)
  1703		 *	// Must not decrement bucket->tasks here
  1704		 *	dequeue(taskB)
  1705		 *
  1706		 * where we could end up with stale data in uc_se and
  1707		 * bucket[uc_se->bucket_id].
  1708		 *
  1709		 * The following check here eliminates the possibility of such race.
  1710		 */
  1711		if (unlikely(!uc_se->active))
  1712			return;
  1713	
  1714		bucket = &uc_rq->bucket[uc_se->bucket_id];
  1715	
  1716		SCHED_WARN_ON(!bucket->tasks);
  1717		if (likely(bucket->tasks))
  1718			bucket->tasks--;
  1719	
  1720		uc_se->active = false;
  1721	
  1722		/*
  1723		 * Keep "local max aggregation" simple and accept to (possibly)
  1724		 * overboost some RUNNABLE tasks in the same bucket.
  1725		 * The rq clamp bucket value is reset to its base value whenever
  1726		 * there are no more RUNNABLE tasks refcounting it.
  1727		 */
  1728		if (likely(bucket->tasks))
  1729			return;
  1730	
  1731		rq_clamp = uclamp_rq_get(rq, clamp_id);
  1732		/*
  1733		 * Defensive programming: this should never happen. If it happens,
  1734		 * e.g. due to future modification, warn and fix up the expected value.
  1735		 */
  1736		SCHED_WARN_ON(bucket->value > rq_clamp);
  1737		if (bucket->value >= rq_clamp) {
  1738			bkt_clamp = uclamp_rq_max_value(rq, clamp_id, uc_se->value);
  1739			uclamp_rq_set(rq, clamp_id, bkt_clamp);
  1740		}
  1741	}
  1742	
  1743	static inline void uclamp_rq_inc(struct rq *rq, struct task_struct *p)
  1744	{
  1745		enum uclamp_id clamp_id;
  1746	
  1747		/*
  1748		 * Avoid any overhead until uclamp is actually used by the userspace.
  1749		 *
  1750		 * The condition is constructed such that a NOP is generated when
  1751		 * sched_uclamp_used is disabled.
  1752		 */
  1753		if (!static_branch_unlikely(&sched_uclamp_used))
  1754			return;
  1755	
  1756		if (unlikely(!p->sched_class->uclamp_enabled))
  1757			return;
  1758	
  1759		if (p->se.sched_delayed)
  1760			return;
  1761	
  1762		for_each_clamp_id(clamp_id)
  1763			uclamp_rq_inc_id(rq, p, clamp_id);
  1764	
  1765		/* Reset clamp idle holding when there is one RUNNABLE task */
  1766		if (rq->uclamp_flags & UCLAMP_FLAG_IDLE)
  1767			rq->uclamp_flags &= ~UCLAMP_FLAG_IDLE;
  1768	}
  1769	
  1770	static inline void uclamp_rq_dec(struct rq *rq, struct task_struct *p)
  1771	{
  1772		enum uclamp_id clamp_id;
  1773	
  1774		/*
  1775		 * Avoid any overhead until uclamp is actually used by the userspace.
  1776		 *
  1777		 * The condition is constructed such that a NOP is generated when
  1778		 * sched_uclamp_used is disabled.
  1779		 */
  1780		if (!static_branch_unlikely(&sched_uclamp_used))
  1781			return;
  1782	
  1783		if (unlikely(!p->sched_class->uclamp_enabled))
  1784			return;
  1785	
  1786		if (p->se.sched_delayed)
  1787			return;
  1788	
  1789		for_each_clamp_id(clamp_id)
  1790			uclamp_rq_dec_id(rq, p, clamp_id);
  1791	}
  1792	
  1793	static inline void uclamp_rq_reinc_id(struct rq *rq, struct task_struct *p,
  1794					      enum uclamp_id clamp_id)
  1795	{
  1796		if (!p->uclamp[clamp_id].active)
  1797			return;
  1798	
  1799		uclamp_rq_dec_id(rq, p, clamp_id);
  1800		uclamp_rq_inc_id(rq, p, clamp_id);
  1801	
  1802		/*
  1803		 * Make sure to clear the idle flag if we've transiently reached 0
  1804		 * active tasks on rq.
  1805		 */
  1806		if (clamp_id == UCLAMP_MAX && (rq->uclamp_flags & UCLAMP_FLAG_IDLE))
  1807			rq->uclamp_flags &= ~UCLAMP_FLAG_IDLE;
  1808	}
  1809	
  1810	static inline void
  1811	uclamp_update_active(struct task_struct *p)
  1812	{
  1813		enum uclamp_id clamp_id;
  1814		struct rq_flags rf;
  1815		struct rq *rq;
  1816	
  1817		/*
  1818		 * Lock the task and the rq where the task is (or was) queued.
  1819		 *
  1820		 * We might lock the (previous) rq of a !RUNNABLE task, but that's the
  1821		 * price to pay to safely serialize util_{min,max} updates with
  1822		 * enqueues, dequeues and migration operations.
  1823		 * This is the same locking schema used by __set_cpus_allowed_ptr().
  1824		 */
  1825		rq = task_rq_lock(p, &rf);
  1826	
  1827		/*
  1828		 * Setting the clamp bucket is serialized by task_rq_lock().
  1829		 * If the task is not yet RUNNABLE and its task_struct is not
  1830		 * affecting a valid clamp bucket, the next time it's enqueued,
  1831		 * it will already see the updated clamp bucket value.
  1832		 */
  1833		for_each_clamp_id(clamp_id)
  1834			uclamp_rq_reinc_id(rq, p, clamp_id);
  1835	
  1836		task_rq_unlock(rq, p, &rf);
  1837	}
  1838	
  1839	#ifdef CONFIG_UCLAMP_TASK_GROUP
  1840	static inline void
  1841	uclamp_update_active_tasks(struct cgroup_subsys_state *css)
  1842	{
  1843		struct css_task_iter it;
  1844		struct task_struct *p;
  1845	
  1846		css_task_iter_start(css, 0, &it);
  1847		while ((p = css_task_iter_next(&it)))
  1848			uclamp_update_active(p);
  1849		css_task_iter_end(&it);
  1850	}
  1851	
  1852	static void cpu_util_update_eff(struct cgroup_subsys_state *css);
  1853	#endif
  1854	
  1855	#ifdef CONFIG_SYSCTL
  1856	#ifdef CONFIG_UCLAMP_TASK_GROUP
  1857	static void uclamp_update_root_tg(void)
  1858	{
  1859		struct task_group *tg = &root_task_group;
  1860	
  1861		uclamp_se_set(&tg->uclamp_req[UCLAMP_MIN],
  1862			      sysctl_sched_uclamp_util_min, false);
  1863		uclamp_se_set(&tg->uclamp_req[UCLAMP_MAX],
  1864			      sysctl_sched_uclamp_util_max, false);
  1865	
  1866		guard(rcu)();
  1867		cpu_util_update_eff(&root_task_group.css);
  1868	}
  1869	#else
  1870	static void uclamp_update_root_tg(void) { }
  1871	#endif
  1872	
  1873	static void uclamp_sync_util_min_rt_default(void)
  1874	{
  1875		struct task_struct *g, *p;
  1876	
  1877		/*
  1878		 * copy_process()			sysctl_uclamp
  1879		 *					  uclamp_min_rt = X;
  1880		 *   write_lock(&tasklist_lock)		  read_lock(&tasklist_lock)
  1881		 *   // link thread			  smp_mb__after_spinlock()
  1882		 *   write_unlock(&tasklist_lock)	  read_unlock(&tasklist_lock);
  1883		 *   sched_post_fork()			  for_each_process_thread()
  1884		 *     __uclamp_sync_rt()		    __uclamp_sync_rt()
  1885		 *
  1886		 * Ensures that either sched_post_fork() will observe the new
  1887		 * uclamp_min_rt or for_each_process_thread() will observe the new
  1888		 * task.
  1889		 */
  1890		read_lock(&tasklist_lock);
  1891		smp_mb__after_spinlock();
  1892		read_unlock(&tasklist_lock);
  1893	
  1894		guard(rcu)();
  1895		for_each_process_thread(g, p)
  1896			uclamp_update_util_min_rt_default(p);
  1897	}
  1898	
  1899	static int sysctl_sched_uclamp_handler(const struct ctl_table *table, int write,
  1900					void *buffer, size_t *lenp, loff_t *ppos)
  1901	{
  1902		bool update_root_tg = false;
  1903		int old_min, old_max, old_min_rt;
  1904		int result;
  1905	
  1906		guard(mutex)(&uclamp_mutex);
  1907	
  1908		old_min = sysctl_sched_uclamp_util_min;
  1909		old_max = sysctl_sched_uclamp_util_max;
  1910		old_min_rt = sysctl_sched_uclamp_util_min_rt_default;
  1911	
  1912		result = proc_dointvec(table, write, buffer, lenp, ppos);
  1913		if (result)
  1914			goto undo;
  1915		if (!write)
  1916			return 0;
  1917	
  1918		if (sysctl_sched_uclamp_util_min > sysctl_sched_uclamp_util_max ||
  1919		    sysctl_sched_uclamp_util_max > SCHED_CAPACITY_SCALE	||
  1920		    sysctl_sched_uclamp_util_min_rt_default > SCHED_CAPACITY_SCALE) {
  1921	
  1922			result = -EINVAL;
  1923			goto undo;
  1924		}
  1925	
  1926		if (old_min != sysctl_sched_uclamp_util_min) {
  1927			uclamp_se_set(&uclamp_default[UCLAMP_MIN],
  1928				      sysctl_sched_uclamp_util_min, false);
  1929			update_root_tg = true;
  1930		}
  1931		if (old_max != sysctl_sched_uclamp_util_max) {
  1932			uclamp_se_set(&uclamp_default[UCLAMP_MAX],
  1933				      sysctl_sched_uclamp_util_max, false);
  1934			update_root_tg = true;
  1935		}
  1936	
  1937		if (update_root_tg) {
  1938			static_branch_enable(&sched_uclamp_used);
  1939			uclamp_update_root_tg();
  1940		}
  1941	
  1942		if (old_min_rt != sysctl_sched_uclamp_util_min_rt_default) {
  1943			static_branch_enable(&sched_uclamp_used);
  1944			uclamp_sync_util_min_rt_default();
  1945		}
  1946	
  1947		/*
  1948		 * We update all RUNNABLE tasks only when task groups are in use.
  1949		 * Otherwise, keep it simple and do just a lazy update at each next
  1950		 * task enqueue time.
  1951		 */
  1952		return 0;
  1953	
  1954	undo:
  1955		sysctl_sched_uclamp_util_min = old_min;
  1956		sysctl_sched_uclamp_util_max = old_max;
  1957		sysctl_sched_uclamp_util_min_rt_default = old_min_rt;
  1958		return result;
  1959	}
  1960	#endif
  1961	
  1962	static void uclamp_fork(struct task_struct *p)
  1963	{
  1964		enum uclamp_id clamp_id;
  1965	
  1966		/*
  1967		 * We don't need to hold task_rq_lock() when updating p->uclamp_* here
  1968		 * as the task is still at its early fork stages.
  1969		 */
  1970		for_each_clamp_id(clamp_id)
  1971			p->uclamp[clamp_id].active = false;
  1972	
  1973		if (likely(!p->sched_reset_on_fork))
  1974			return;
  1975	
  1976		for_each_clamp_id(clamp_id) {
  1977			uclamp_se_set(&p->uclamp_req[clamp_id],
  1978				      uclamp_none(clamp_id), false);
  1979		}
  1980	}
  1981	
  1982	static void uclamp_post_fork(struct task_struct *p)
  1983	{
  1984		uclamp_update_util_min_rt_default(p);
  1985	}
  1986	
  1987	static void __init init_uclamp_rq(struct rq *rq)
  1988	{
  1989		enum uclamp_id clamp_id;
  1990		struct uclamp_rq *uc_rq = rq->uclamp;
  1991	
  1992		for_each_clamp_id(clamp_id) {
  1993			uc_rq[clamp_id] = (struct uclamp_rq) {
  1994				.value = uclamp_none(clamp_id)
  1995			};
  1996		}
  1997	
  1998		rq->uclamp_flags = UCLAMP_FLAG_IDLE;
  1999	}
  2000	
  2001	static void __init init_uclamp(void)
  2002	{
  2003		struct uclamp_se uc_max = {};
  2004		enum uclamp_id clamp_id;
  2005		int cpu;
  2006	
  2007		for_each_possible_cpu(cpu)
  2008			init_uclamp_rq(cpu_rq(cpu));
  2009	
  2010		for_each_clamp_id(clamp_id) {
  2011			uclamp_se_set(&init_task.uclamp_req[clamp_id],
  2012				      uclamp_none(clamp_id), false);
  2013		}
  2014	
  2015		/* System defaults allow max clamp values for both indexes */
  2016		uclamp_se_set(&uc_max, uclamp_none(UCLAMP_MAX), false);
  2017		for_each_clamp_id(clamp_id) {
  2018			uclamp_default[clamp_id] = uc_max;
  2019	#ifdef CONFIG_UCLAMP_TASK_GROUP
  2020			root_task_group.uclamp_req[clamp_id] = uc_max;
  2021			root_task_group.uclamp[clamp_id] = uc_max;
  2022	#endif
  2023		}
  2024	}
  2025	
  2026	#else /* !CONFIG_UCLAMP_TASK */
  2027	static inline void uclamp_rq_inc(struct rq *rq, struct task_struct *p) { }
  2028	static inline void uclamp_rq_dec(struct rq *rq, struct task_struct *p) { }
  2029	static inline void uclamp_fork(struct task_struct *p) { }
  2030	static inline void uclamp_post_fork(struct task_struct *p) { }
  2031	static inline void init_uclamp(void) { }
  2032	#endif /* CONFIG_UCLAMP_TASK */
  2033	
  2034	bool sched_task_on_rq(struct task_struct *p)
  2035	{
  2036		return task_on_rq_queued(p);
  2037	}
  2038	
  2039	unsigned long get_wchan(struct task_struct *p)
  2040	{
  2041		unsigned long ip = 0;
  2042		unsigned int state;
  2043	
  2044		if (!p || p == current)
  2045			return 0;
  2046	
  2047		/* Only get wchan if task is blocked and we can keep it that way. */
  2048		raw_spin_lock_irq(&p->pi_lock);
  2049		state = READ_ONCE(p->__state);
  2050		smp_rmb(); /* see try_to_wake_up() */
  2051		if (state != TASK_RUNNING && state != TASK_WAKING && !p->on_rq)
  2052			ip = __get_wchan(p);
  2053		raw_spin_unlock_irq(&p->pi_lock);
  2054	
  2055		return ip;
  2056	}
  2057	
  2058	void enqueue_task(struct rq *rq, struct task_struct *p, int flags)
  2059	{
  2060		if (!(flags & ENQUEUE_NOCLOCK))
  2061			update_rq_clock(rq);
  2062	
  2063		p->sched_class->enqueue_task(rq, p, flags);
  2064		/*
  2065		 * Must be after ->enqueue_task() because ENQUEUE_DELAYED can clear
  2066		 * ->sched_delayed.
  2067		 */
  2068		uclamp_rq_inc(rq, p);
  2069	
  2070		psi_enqueue(p, flags);
  2071	
  2072		if (!(flags & ENQUEUE_RESTORE))
  2073			sched_info_enqueue(rq, p);
  2074	
  2075		if (sched_core_enabled(rq))
  2076			sched_core_enqueue(rq, p);
  2077	}
  2078	
  2079	/*
  2080	 * Must only return false when DEQUEUE_SLEEP.
  2081	 */
  2082	inline bool dequeue_task(struct rq *rq, struct task_struct *p, int flags)
  2083	{
  2084		if (sched_core_enabled(rq))
  2085			sched_core_dequeue(rq, p, flags);
  2086	
  2087		if (!(flags & DEQUEUE_NOCLOCK))
  2088			update_rq_clock(rq);
  2089	
  2090		if (!(flags & DEQUEUE_SAVE))
  2091			sched_info_dequeue(rq, p);
  2092	
  2093		psi_dequeue(p, flags);
  2094	
  2095		/*
  2096		 * Must be before ->dequeue_task() because ->dequeue_task() can 'fail'
  2097		 * and mark the task ->sched_delayed.
  2098		 */
  2099		uclamp_rq_dec(rq, p);
  2100		return p->sched_class->dequeue_task(rq, p, flags);
  2101	}
  2102	
  2103	void activate_task(struct rq *rq, struct task_struct *p, int flags)
  2104	{
  2105		if (task_on_rq_migrating(p))
  2106			flags |= ENQUEUE_MIGRATED;
  2107		if (flags & ENQUEUE_MIGRATED)
  2108			sched_mm_cid_migrate_to(rq, p);
  2109	
  2110		enqueue_task(rq, p, flags);
  2111	
  2112		WRITE_ONCE(p->on_rq, TASK_ON_RQ_QUEUED);
  2113		ASSERT_EXCLUSIVE_WRITER(p->on_rq);
  2114	}
  2115	
  2116	void deactivate_task(struct rq *rq, struct task_struct *p, int flags)
  2117	{
  2118		SCHED_WARN_ON(flags & DEQUEUE_SLEEP);
  2119	
  2120		WRITE_ONCE(p->on_rq, TASK_ON_RQ_MIGRATING);
  2121		ASSERT_EXCLUSIVE_WRITER(p->on_rq);
  2122	
  2123		/*
  2124		 * Code explicitly relies on TASK_ON_RQ_MIGRATING begin set *before*
  2125		 * dequeue_task() and cleared *after* enqueue_task().
  2126		 */
  2127	
  2128		dequeue_task(rq, p, flags);
  2129	}
  2130	
  2131	static void block_task(struct rq *rq, struct task_struct *p, int flags)
  2132	{
  2133		if (dequeue_task(rq, p, DEQUEUE_SLEEP | flags))
  2134			__block_task(rq, p);
  2135	}
  2136	
  2137	/**
  2138	 * task_curr - is this task currently executing on a CPU?
  2139	 * @p: the task in question.
  2140	 *
  2141	 * Return: 1 if the task is currently executing. 0 otherwise.
  2142	 */
  2143	inline int task_curr(const struct task_struct *p)
  2144	{
  2145		return cpu_curr(task_cpu(p)) == p;
  2146	}
  2147	
  2148	/*
  2149	 * ->switching_to() is called with the pi_lock and rq_lock held and must not
  2150	 * mess with locking.
  2151	 */
  2152	void check_class_changing(struct rq *rq, struct task_struct *p,
  2153				  const struct sched_class *prev_class)
  2154	{
  2155		if (prev_class != p->sched_class && p->sched_class->switching_to)
  2156			p->sched_class->switching_to(rq, p);
  2157	}
  2158	
  2159	/*
  2160	 * switched_from, switched_to and prio_changed must _NOT_ drop rq->lock,
  2161	 * use the balance_callback list if you want balancing.
  2162	 *
  2163	 * this means any call to check_class_changed() must be followed by a call to
  2164	 * balance_callback().
  2165	 */
  2166	void check_class_changed(struct rq *rq, struct task_struct *p,
  2167				 const struct sched_class *prev_class,
  2168				 int oldprio)
  2169	{
  2170		if (prev_class != p->sched_class) {
  2171			if (prev_class->switched_from)
  2172				prev_class->switched_from(rq, p);
  2173	
  2174			p->sched_class->switched_to(rq, p);
  2175		} else if (oldprio != p->prio || dl_task(p))
  2176			p->sched_class->prio_changed(rq, p, oldprio);
  2177	}
  2178	
  2179	void wakeup_preempt(struct rq *rq, struct task_struct *p, int flags)
  2180	{
  2181		struct task_struct *donor = rq->donor;
  2182	
  2183		if (p->sched_class == donor->sched_class)
  2184			donor->sched_class->wakeup_preempt(rq, p, flags);
  2185		else if (sched_class_above(p->sched_class, donor->sched_class))
  2186			resched_curr(rq);
  2187	
  2188		/*
  2189		 * A queue event has occurred, and we're going to schedule.  In
  2190		 * this case, we can save a useless back to back clock update.
  2191		 */
  2192		if (task_on_rq_queued(donor) && test_tsk_need_resched(rq->curr))
  2193			rq_clock_skip_update(rq);
  2194	}
  2195	
  2196	static __always_inline
  2197	int __task_state_match(struct task_struct *p, unsigned int state)
  2198	{
  2199		if (READ_ONCE(p->__state) & state)
  2200			return 1;
  2201	
  2202		if (READ_ONCE(p->saved_state) & state)
  2203			return -1;
  2204	
  2205		return 0;
  2206	}
  2207	
  2208	static __always_inline
  2209	int task_state_match(struct task_struct *p, unsigned int state)
  2210	{
  2211		/*
  2212		 * Serialize against current_save_and_set_rtlock_wait_state(),
  2213		 * current_restore_rtlock_saved_state(), and __refrigerator().
  2214		 */
  2215		guard(raw_spinlock_irq)(&p->pi_lock);
  2216		return __task_state_match(p, state);
  2217	}
  2218	
  2219	/*
  2220	 * wait_task_inactive - wait for a thread to unschedule.
  2221	 *
  2222	 * Wait for the thread to block in any of the states set in @match_state.
  2223	 * If it changes, i.e. @p might have woken up, then return zero.  When we
  2224	 * succeed in waiting for @p to be off its CPU, we return a positive number
  2225	 * (its total switch count).  If a second call a short while later returns the
  2226	 * same number, the caller can be sure that @p has remained unscheduled the
  2227	 * whole time.
  2228	 *
  2229	 * The caller must ensure that the task *will* unschedule sometime soon,
  2230	 * else this function might spin for a *long* time. This function can't
  2231	 * be called with interrupts off, or it may introduce deadlock with
  2232	 * smp_call_function() if an IPI is sent by the same process we are
  2233	 * waiting to become inactive.
  2234	 */
  2235	unsigned long wait_task_inactive(struct task_struct *p, unsigned int match_state)
  2236	{
  2237		int running, queued, match;
  2238		struct rq_flags rf;
  2239		unsigned long ncsw;
  2240		struct rq *rq;
  2241	
  2242		for (;;) {
  2243			/*
  2244			 * We do the initial early heuristics without holding
  2245			 * any task-queue locks at all. We'll only try to get
  2246			 * the runqueue lock when things look like they will
  2247			 * work out!
  2248			 */
  2249			rq = task_rq(p);
  2250	
  2251			/*
  2252			 * If the task is actively running on another CPU
  2253			 * still, just relax and busy-wait without holding
  2254			 * any locks.
  2255			 *
  2256			 * NOTE! Since we don't hold any locks, it's not
  2257			 * even sure that "rq" stays as the right runqueue!
  2258			 * But we don't care, since "task_on_cpu()" will
  2259			 * return false if the runqueue has changed and p
  2260			 * is actually now running somewhere else!
  2261			 */
  2262			while (task_on_cpu(rq, p)) {
  2263				if (!task_state_match(p, match_state))
  2264					return 0;
  2265				cpu_relax();
  2266			}
  2267	
  2268			/*
  2269			 * Ok, time to look more closely! We need the rq
  2270			 * lock now, to be *sure*. If we're wrong, we'll
  2271			 * just go back and repeat.
  2272			 */
  2273			rq = task_rq_lock(p, &rf);
  2274			trace_sched_wait_task(p);
  2275			running = task_on_cpu(rq, p);
  2276			queued = task_on_rq_queued(p);
  2277			ncsw = 0;
  2278			if ((match = __task_state_match(p, match_state))) {
  2279				/*
  2280				 * When matching on p->saved_state, consider this task
  2281				 * still queued so it will wait.
  2282				 */
  2283				if (match < 0)
  2284					queued = 1;
  2285				ncsw = p->nvcsw | LONG_MIN; /* sets MSB */
  2286			}
  2287			task_rq_unlock(rq, p, &rf);
  2288	
  2289			/*
  2290			 * If it changed from the expected state, bail out now.
  2291			 */
  2292			if (unlikely(!ncsw))
  2293				break;
  2294	
  2295			/*
  2296			 * Was it really running after all now that we
  2297			 * checked with the proper locks actually held?
  2298			 *
  2299			 * Oops. Go back and try again..
  2300			 */
  2301			if (unlikely(running)) {
  2302				cpu_relax();
  2303				continue;
  2304			}
  2305	
  2306			/*
  2307			 * It's not enough that it's not actively running,
  2308			 * it must be off the runqueue _entirely_, and not
  2309			 * preempted!
  2310			 *
  2311			 * So if it was still runnable (but just not actively
  2312			 * running right now), it's preempted, and we should
  2313			 * yield - it could be a while.
  2314			 */
  2315			if (unlikely(queued)) {
  2316				ktime_t to = NSEC_PER_SEC / HZ;
  2317	
  2318				set_current_state(TASK_UNINTERRUPTIBLE);
  2319				schedule_hrtimeout(&to, HRTIMER_MODE_REL_HARD);
  2320				continue;
  2321			}
  2322	
  2323			/*
  2324			 * Ahh, all good. It wasn't running, and it wasn't
  2325			 * runnable, which means that it will never become
  2326			 * running in the future either. We're all done!
  2327			 */
  2328			break;
  2329		}
  2330	
  2331		return ncsw;
  2332	}
  2333	
  2334	#ifdef CONFIG_SMP
  2335	
  2336	static void
  2337	__do_set_cpus_allowed(struct task_struct *p, struct affinity_context *ctx);
  2338	
  2339	static void migrate_disable_switch(struct rq *rq, struct task_struct *p)
  2340	{
  2341		struct affinity_context ac = {
  2342			.new_mask  = cpumask_of(rq->cpu),
  2343			.flags     = SCA_MIGRATE_DISABLE,
  2344		};
  2345	
  2346		if (likely(!p->migration_disabled))
  2347			return;
  2348	
  2349		if (p->cpus_ptr != &p->cpus_mask)
  2350			return;
  2351	
  2352		/*
  2353		 * Violates locking rules! See comment in __do_set_cpus_allowed().
  2354		 */
  2355		__do_set_cpus_allowed(p, &ac);
  2356	}
  2357	
  2358	void migrate_disable(void)
  2359	{
  2360		struct task_struct *p = current;
  2361	
  2362		if (p->migration_disabled) {
  2363	#ifdef CONFIG_DEBUG_PREEMPT
  2364			/*
  2365			 *Warn about overflow half-way through the range.
  2366			 */
  2367			WARN_ON_ONCE((s16)p->migration_disabled < 0);
  2368	#endif
  2369			p->migration_disabled++;
  2370			return;
  2371		}
  2372	
  2373		guard(preempt)();
  2374		this_rq()->nr_pinned++;
  2375		p->migration_disabled = 1;
  2376	}
  2377	EXPORT_SYMBOL_GPL(migrate_disable);
  2378	
  2379	void migrate_enable(void)
  2380	{
  2381		struct task_struct *p = current;
  2382		struct affinity_context ac = {
  2383			.new_mask  = &p->cpus_mask,
  2384			.flags     = SCA_MIGRATE_ENABLE,
  2385		};
  2386	
  2387	#ifdef CONFIG_DEBUG_PREEMPT
  2388		/*
  2389		 * Check both overflow from migrate_disable() and superfluous
  2390		 * migrate_enable().
  2391		 */
  2392		if (WARN_ON_ONCE((s16)p->migration_disabled <= 0))
  2393			return;
  2394	#endif
  2395	
  2396		if (p->migration_disabled > 1) {
  2397			p->migration_disabled--;
  2398			return;
  2399		}
  2400	
  2401		/*
  2402		 * Ensure stop_task runs either before or after this, and that
  2403		 * __set_cpus_allowed_ptr(SCA_MIGRATE_ENABLE) doesn't schedule().
  2404		 */
  2405		guard(preempt)();
  2406		if (p->cpus_ptr != &p->cpus_mask)
  2407			__set_cpus_allowed_ptr(p, &ac);
  2408		/*
  2409		 * Mustn't clear migration_disabled() until cpus_ptr points back at the
  2410		 * regular cpus_mask, otherwise things that race (eg.
  2411		 * select_fallback_rq) get confused.
  2412		 */
  2413		barrier();
  2414		p->migration_disabled = 0;
  2415		this_rq()->nr_pinned--;
  2416	}
  2417	EXPORT_SYMBOL_GPL(migrate_enable);
  2418	
  2419	static inline bool rq_has_pinned_tasks(struct rq *rq)
  2420	{
  2421		return rq->nr_pinned;
  2422	}
  2423	
  2424	/*
  2425	 * Per-CPU kthreads are allowed to run on !active && online CPUs, see
  2426	 * __set_cpus_allowed_ptr() and select_fallback_rq().
  2427	 */
  2428	static inline bool is_cpu_allowed(struct task_struct *p, int cpu)
  2429	{
  2430		/* When not in the task's cpumask, no point in looking further. */
  2431		if (!task_allowed_on_cpu(p, cpu))
  2432			return false;
  2433	
  2434		/* migrate_disabled() must be allowed to finish. */
  2435		if (is_migration_disabled(p))
  2436			return cpu_online(cpu);
  2437	
  2438		/* Non kernel threads are not allowed during either online or offline. */
  2439		if (!(p->flags & PF_KTHREAD))
  2440			return cpu_active(cpu);
  2441	
  2442		/* KTHREAD_IS_PER_CPU is always allowed. */
  2443		if (kthread_is_per_cpu(p))
  2444			return cpu_online(cpu);
  2445	
  2446		/* Regular kernel threads don't get to stay during offline. */
  2447		if (cpu_dying(cpu))
  2448			return false;
  2449	
  2450		/* But are allowed during online. */
  2451		return cpu_online(cpu);
  2452	}
  2453	
  2454	/*
  2455	 * This is how migration works:
  2456	 *
  2457	 * 1) we invoke migration_cpu_stop() on the target CPU using
  2458	 *    stop_one_cpu().
  2459	 * 2) stopper starts to run (implicitly forcing the migrated thread
  2460	 *    off the CPU)
  2461	 * 3) it checks whether the migrated task is still in the wrong runqueue.
  2462	 * 4) if it's in the wrong runqueue then the migration thread removes
  2463	 *    it and puts it into the right queue.
  2464	 * 5) stopper completes and stop_one_cpu() returns and the migration
  2465	 *    is done.
  2466	 */
  2467	
  2468	/*
  2469	 * move_queued_task - move a queued task to new rq.
  2470	 *
  2471	 * Returns (locked) new rq. Old rq's lock is released.
  2472	 */
  2473	static struct rq *move_queued_task(struct rq *rq, struct rq_flags *rf,
  2474					   struct task_struct *p, int new_cpu)
  2475	{
  2476		lockdep_assert_rq_held(rq);
  2477	
  2478		deactivate_task(rq, p, DEQUEUE_NOCLOCK);
  2479		set_task_cpu(p, new_cpu);
  2480		rq_unlock(rq, rf);
  2481	
  2482		rq = cpu_rq(new_cpu);
  2483	
  2484		rq_lock(rq, rf);
  2485		WARN_ON_ONCE(task_cpu(p) != new_cpu);
  2486		activate_task(rq, p, 0);
  2487		wakeup_preempt(rq, p, 0);
  2488	
  2489		return rq;
  2490	}
  2491	
  2492	struct migration_arg {
  2493		struct task_struct		*task;
  2494		int				dest_cpu;
  2495		struct set_affinity_pending	*pending;
  2496	};
  2497	
  2498	/*
  2499	 * @refs: number of wait_for_completion()
  2500	 * @stop_pending: is @stop_work in use
  2501	 */
  2502	struct set_affinity_pending {
  2503		refcount_t		refs;
  2504		unsigned int		stop_pending;
  2505		struct completion	done;
  2506		struct cpu_stop_work	stop_work;
  2507		struct migration_arg	arg;
  2508	};
  2509	
  2510	/*
  2511	 * Move (not current) task off this CPU, onto the destination CPU. We're doing
  2512	 * this because either it can't run here any more (set_cpus_allowed()
  2513	 * away from this CPU, or CPU going down), or because we're
  2514	 * attempting to rebalance this task on exec (sched_exec).
  2515	 *
  2516	 * So we race with normal scheduler movements, but that's OK, as long
  2517	 * as the task is no longer on this CPU.
  2518	 */
  2519	static struct rq *__migrate_task(struct rq *rq, struct rq_flags *rf,
  2520					 struct task_struct *p, int dest_cpu)
  2521	{
  2522		/* Affinity changed (again). */
  2523		if (!is_cpu_allowed(p, dest_cpu))
  2524			return rq;
  2525	
  2526		rq = move_queued_task(rq, rf, p, dest_cpu);
  2527	
  2528		return rq;
  2529	}
  2530	
  2531	/*
  2532	 * migration_cpu_stop - this will be executed by a high-prio stopper thread
  2533	 * and performs thread migration by bumping thread off CPU then
  2534	 * 'pushing' onto another runqueue.
  2535	 */
  2536	static int migration_cpu_stop(void *data)
  2537	{
  2538		struct migration_arg *arg = data;
  2539		struct set_affinity_pending *pending = arg->pending;
  2540		struct task_struct *p = arg->task;
  2541		struct rq *rq = this_rq();
  2542		bool complete = false;
  2543		struct rq_flags rf;
  2544	
  2545		/*
  2546		 * The original target CPU might have gone down and we might
  2547		 * be on another CPU but it doesn't matter.
  2548		 */
  2549		local_irq_save(rf.flags);
  2550		/*
  2551		 * We need to explicitly wake pending tasks before running
  2552		 * __migrate_task() such that we will not miss enforcing cpus_ptr
  2553		 * during wakeups, see set_cpus_allowed_ptr()'s TASK_WAKING test.
  2554		 */
  2555		flush_smp_call_function_queue();
  2556	
  2557		raw_spin_lock(&p->pi_lock);
  2558		rq_lock(rq, &rf);
  2559	
  2560		/*
  2561		 * If we were passed a pending, then ->stop_pending was set, thus
  2562		 * p->migration_pending must have remained stable.
  2563		 */
  2564		WARN_ON_ONCE(pending && pending != p->migration_pending);
  2565	
  2566		/*
  2567		 * If task_rq(p) != rq, it cannot be migrated here, because we're
  2568		 * holding rq->lock, if p->on_rq == 0 it cannot get enqueued because
  2569		 * we're holding p->pi_lock.
  2570		 */
  2571		if (task_rq(p) == rq) {
  2572			if (is_migration_disabled(p))
  2573				goto out;
  2574	
  2575			if (pending) {
  2576				p->migration_pending = NULL;
  2577				complete = true;
  2578	
  2579				if (cpumask_test_cpu(task_cpu(p), &p->cpus_mask))
  2580					goto out;
  2581			}
  2582	
  2583			if (task_on_rq_queued(p)) {
  2584				update_rq_clock(rq);
  2585				rq = __migrate_task(rq, &rf, p, arg->dest_cpu);
  2586			} else {
  2587				p->wake_cpu = arg->dest_cpu;
  2588			}
  2589	
  2590			/*
  2591			 * XXX __migrate_task() can fail, at which point we might end
  2592			 * up running on a dodgy CPU, AFAICT this can only happen
  2593			 * during CPU hotplug, at which point we'll get pushed out
  2594			 * anyway, so it's probably not a big deal.
  2595			 */
  2596	
  2597		} else if (pending) {
  2598			/*
  2599			 * This happens when we get migrated between migrate_enable()'s
  2600			 * preempt_enable() and scheduling the stopper task. At that
  2601			 * point we're a regular task again and not current anymore.
  2602			 *
  2603			 * A !PREEMPT kernel has a giant hole here, which makes it far
  2604			 * more likely.
  2605			 */
  2606	
  2607			/*
  2608			 * The task moved before the stopper got to run. We're holding
  2609			 * ->pi_lock, so the allowed mask is stable - if it got
  2610			 * somewhere allowed, we're done.
  2611			 */
  2612			if (cpumask_test_cpu(task_cpu(p), p->cpus_ptr)) {
  2613				p->migration_pending = NULL;
  2614				complete = true;
  2615				goto out;
  2616			}
  2617	
  2618			/*
  2619			 * When migrate_enable() hits a rq mis-match we can't reliably
  2620			 * determine is_migration_disabled() and so have to chase after
  2621			 * it.
  2622			 */
  2623			WARN_ON_ONCE(!pending->stop_pending);
  2624			preempt_disable();
  2625			task_rq_unlock(rq, p, &rf);
  2626			stop_one_cpu_nowait(task_cpu(p), migration_cpu_stop,
  2627					    &pending->arg, &pending->stop_work);
  2628			preempt_enable();
  2629			return 0;
  2630		}
  2631	out:
  2632		if (pending)
  2633			pending->stop_pending = false;
  2634		task_rq_unlock(rq, p, &rf);
  2635	
  2636		if (complete)
  2637			complete_all(&pending->done);
  2638	
  2639		return 0;
  2640	}
  2641	
  2642	int push_cpu_stop(void *arg)
  2643	{
  2644		struct rq *lowest_rq = NULL, *rq = this_rq();
  2645		struct task_struct *p = arg;
  2646	
  2647		raw_spin_lock_irq(&p->pi_lock);
  2648		raw_spin_rq_lock(rq);
  2649	
  2650		if (task_rq(p) != rq)
  2651			goto out_unlock;
  2652	
  2653		if (is_migration_disabled(p)) {
  2654			p->migration_flags |= MDF_PUSH;
  2655			goto out_unlock;
  2656		}
  2657	
  2658		p->migration_flags &= ~MDF_PUSH;
  2659	
  2660		if (p->sched_class->find_lock_rq)
  2661			lowest_rq = p->sched_class->find_lock_rq(p, rq);
  2662	
  2663		if (!lowest_rq)
  2664			goto out_unlock;
  2665	
  2666		// XXX validate p is still the highest prio task
  2667		if (task_rq(p) == rq) {
  2668			move_queued_task_locked(rq, lowest_rq, p);
  2669			resched_curr(lowest_rq);
  2670		}
  2671	
  2672		double_unlock_balance(rq, lowest_rq);
  2673	
  2674	out_unlock:
  2675		rq->push_busy = false;
  2676		raw_spin_rq_unlock(rq);
  2677		raw_spin_unlock_irq(&p->pi_lock);
  2678	
  2679		put_task_struct(p);
  2680		return 0;
  2681	}
  2682	
  2683	/*
  2684	 * sched_class::set_cpus_allowed must do the below, but is not required to
  2685	 * actually call this function.
  2686	 */
  2687	void set_cpus_allowed_common(struct task_struct *p, struct affinity_context *ctx)
  2688	{
  2689		if (ctx->flags & (SCA_MIGRATE_ENABLE | SCA_MIGRATE_DISABLE)) {
  2690			p->cpus_ptr = ctx->new_mask;
  2691			return;
  2692		}
  2693	
  2694		cpumask_copy(&p->cpus_mask, ctx->new_mask);
  2695		p->nr_cpus_allowed = cpumask_weight(ctx->new_mask);
  2696	
  2697		/*
  2698		 * Swap in a new user_cpus_ptr if SCA_USER flag set
  2699		 */
  2700		if (ctx->flags & SCA_USER)
  2701			swap(p->user_cpus_ptr, ctx->user_mask);
  2702	}
  2703	
  2704	static void
  2705	__do_set_cpus_allowed(struct task_struct *p, struct affinity_context *ctx)
  2706	{
  2707		struct rq *rq = task_rq(p);
  2708		bool queued, running;
  2709	
  2710		/*
  2711		 * This here violates the locking rules for affinity, since we're only
  2712		 * supposed to change these variables while holding both rq->lock and
  2713		 * p->pi_lock.
  2714		 *
  2715		 * HOWEVER, it magically works, because ttwu() is the only code that
  2716		 * accesses these variables under p->pi_lock and only does so after
  2717		 * smp_cond_load_acquire(&p->on_cpu, !VAL), and we're in __schedule()
  2718		 * before finish_task().
  2719		 *
  2720		 * XXX do further audits, this smells like something putrid.
  2721		 */
  2722		if (ctx->flags & SCA_MIGRATE_DISABLE)
  2723			SCHED_WARN_ON(!p->on_cpu);
  2724		else
  2725			lockdep_assert_held(&p->pi_lock);
  2726	
  2727		queued = task_on_rq_queued(p);
  2728		running = task_current_donor(rq, p);
  2729	
  2730		if (queued) {
  2731			/*
  2732			 * Because __kthread_bind() calls this on blocked tasks without
  2733			 * holding rq->lock.
  2734			 */
  2735			lockdep_assert_rq_held(rq);
  2736			dequeue_task(rq, p, DEQUEUE_SAVE | DEQUEUE_NOCLOCK);
  2737		}
  2738		if (running)
  2739			put_prev_task(rq, p);
  2740	
  2741		p->sched_class->set_cpus_allowed(p, ctx);
  2742		mm_set_cpus_allowed(p->mm, ctx->new_mask);
  2743	
  2744		if (queued)
  2745			enqueue_task(rq, p, ENQUEUE_RESTORE | ENQUEUE_NOCLOCK);
  2746		if (running)
  2747			set_next_task(rq, p);
  2748	}
  2749	
  2750	/*
  2751	 * Used for kthread_bind() and select_fallback_rq(), in both cases the user
  2752	 * affinity (if any) should be destroyed too.
  2753	 */
  2754	void do_set_cpus_allowed(struct task_struct *p, const struct cpumask *new_mask)
  2755	{
  2756		struct affinity_context ac = {
  2757			.new_mask  = new_mask,
  2758			.user_mask = NULL,
  2759			.flags     = SCA_USER,	/* clear the user requested mask */
  2760		};
  2761		union cpumask_rcuhead {
  2762			cpumask_t cpumask;
  2763			struct rcu_head rcu;
  2764		};
  2765	
  2766		__do_set_cpus_allowed(p, &ac);
  2767	
  2768		/*
  2769		 * Because this is called with p->pi_lock held, it is not possible
  2770		 * to use kfree() here (when PREEMPT_RT=y), therefore punt to using
  2771		 * kfree_rcu().
  2772		 */
  2773		kfree_rcu((union cpumask_rcuhead *)ac.user_mask, rcu);
  2774	}
  2775	
  2776	int dup_user_cpus_ptr(struct task_struct *dst, struct task_struct *src,
  2777			      int node)
  2778	{
  2779		cpumask_t *user_mask;
  2780		unsigned long flags;
  2781	
  2782		/*
  2783		 * Always clear dst->user_cpus_ptr first as their user_cpus_ptr's
  2784		 * may differ by now due to racing.
  2785		 */
  2786		dst->user_cpus_ptr = NULL;
  2787	
  2788		/*
  2789		 * This check is racy and losing the race is a valid situation.
  2790		 * It is not worth the extra overhead of taking the pi_lock on
  2791		 * every fork/clone.
  2792		 */
  2793		if (data_race(!src->user_cpus_ptr))
  2794			return 0;
  2795	
  2796		user_mask = alloc_user_cpus_ptr(node);
  2797		if (!user_mask)
  2798			return -ENOMEM;
  2799	
  2800		/*
  2801		 * Use pi_lock to protect content of user_cpus_ptr
  2802		 *
  2803		 * Though unlikely, user_cpus_ptr can be reset to NULL by a concurrent
  2804		 * do_set_cpus_allowed().
  2805		 */
  2806		raw_spin_lock_irqsave(&src->pi_lock, flags);
  2807		if (src->user_cpus_ptr) {
  2808			swap(dst->user_cpus_ptr, user_mask);
  2809			cpumask_copy(dst->user_cpus_ptr, src->user_cpus_ptr);
  2810		}
  2811		raw_spin_unlock_irqrestore(&src->pi_lock, flags);
  2812	
  2813		if (unlikely(user_mask))
  2814			kfree(user_mask);
  2815	
  2816		return 0;
  2817	}
  2818	
  2819	static inline struct cpumask *clear_user_cpus_ptr(struct task_struct *p)
  2820	{
  2821		struct cpumask *user_mask = NULL;
  2822	
  2823		swap(p->user_cpus_ptr, user_mask);
  2824	
  2825		return user_mask;
  2826	}
  2827	
  2828	void release_user_cpus_ptr(struct task_struct *p)
  2829	{
  2830		kfree(clear_user_cpus_ptr(p));
  2831	}
  2832	
  2833	/*
  2834	 * This function is wildly self concurrent; here be dragons.
  2835	 *
  2836	 *
  2837	 * When given a valid mask, __set_cpus_allowed_ptr() must block until the
  2838	 * designated task is enqueued on an allowed CPU. If that task is currently
  2839	 * running, we have to kick it out using the CPU stopper.
  2840	 *
  2841	 * Migrate-Disable comes along and tramples all over our nice sandcastle.
  2842	 * Consider:
  2843	 *
  2844	 *     Initial conditions: P0->cpus_mask = [0, 1]
  2845	 *
  2846	 *     P0@CPU0                  P1
  2847	 *
  2848	 *     migrate_disable();
  2849	 *     <preempted>
  2850	 *                              set_cpus_allowed_ptr(P0, [1]);
  2851	 *
  2852	 * P1 *cannot* return from this set_cpus_allowed_ptr() call until P0 executes
  2853	 * its outermost migrate_enable() (i.e. it exits its Migrate-Disable region).
  2854	 * This means we need the following scheme:
  2855	 *
  2856	 *     P0@CPU0                  P1
  2857	 *
  2858	 *     migrate_disable();
  2859	 *     <preempted>
  2860	 *                              set_cpus_allowed_ptr(P0, [1]);
  2861	 *                                <blocks>
  2862	 *     <resumes>
  2863	 *     migrate_enable();
  2864	 *       __set_cpus_allowed_ptr();
  2865	 *       <wakes local stopper>
  2866	 *                         `--> <woken on migration completion>
  2867	 *
  2868	 * Now the fun stuff: there may be several P1-like tasks, i.e. multiple
  2869	 * concurrent set_cpus_allowed_ptr(P0, [*]) calls. CPU affinity changes of any
  2870	 * task p are serialized by p->pi_lock, which we can leverage: the one that
  2871	 * should come into effect at the end of the Migrate-Disable region is the last
  2872	 * one. This means we only need to track a single cpumask (i.e. p->cpus_mask),
  2873	 * but we still need to properly signal those waiting tasks at the appropriate
  2874	 * moment.
  2875	 *
  2876	 * This is implemented using struct set_affinity_pending. The first
  2877	 * __set_cpus_allowed_ptr() caller within a given Migrate-Disable region will
  2878	 * setup an instance of that struct and install it on the targeted task_struct.
  2879	 * Any and all further callers will reuse that instance. Those then wait for
  2880	 * a completion signaled at the tail of the CPU stopper callback (1), triggered
  2881	 * on the end of the Migrate-Disable region (i.e. outermost migrate_enable()).
  2882	 *
  2883	 *
  2884	 * (1) In the cases covered above. There is one more where the completion is
  2885	 * signaled within affine_move_task() itself: when a subsequent affinity request
  2886	 * occurs after the stopper bailed out due to the targeted task still being
  2887	 * Migrate-Disable. Consider:
  2888	 *
  2889	 *     Initial conditions: P0->cpus_mask = [0, 1]
  2890	 *
  2891	 *     CPU0		  P1				P2
  2892	 *     <P0>
  2893	 *       migrate_disable();
  2894	 *       <preempted>
  2895	 *                        set_cpus_allowed_ptr(P0, [1]);
  2896	 *                          <blocks>
  2897	 *     <migration/0>
  2898	 *       migration_cpu_stop()
  2899	 *         is_migration_disabled()
  2900	 *           <bails>
  2901	 *                                                       set_cpus_allowed_ptr(P0, [0, 1]);
  2902	 *                                                         <signal completion>
  2903	 *                          <awakes>
  2904	 *
  2905	 * Note that the above is safe vs a concurrent migrate_enable(), as any
  2906	 * pending affinity completion is preceded by an uninstallation of
  2907	 * p->migration_pending done with p->pi_lock held.
  2908	 */
  2909	static int affine_move_task(struct rq *rq, struct task_struct *p, struct rq_flags *rf,
  2910				    int dest_cpu, unsigned int flags)
  2911		__releases(rq->lock)
  2912		__releases(p->pi_lock)
  2913	{
  2914		struct set_affinity_pending my_pending = { }, *pending = NULL;
  2915		bool stop_pending, complete = false;
  2916	
  2917		/* Can the task run on the task's current CPU? If so, we're done */
  2918		if (cpumask_test_cpu(task_cpu(p), &p->cpus_mask)) {
  2919			struct task_struct *push_task = NULL;
  2920	
  2921			if ((flags & SCA_MIGRATE_ENABLE) &&
  2922			    (p->migration_flags & MDF_PUSH) && !rq->push_busy) {
  2923				rq->push_busy = true;
  2924				push_task = get_task_struct(p);
  2925			}
  2926	
  2927			/*
  2928			 * If there are pending waiters, but no pending stop_work,
  2929			 * then complete now.
  2930			 */
  2931			pending = p->migration_pending;
  2932			if (pending && !pending->stop_pending) {
  2933				p->migration_pending = NULL;
  2934				complete = true;
  2935			}
  2936	
  2937			preempt_disable();
  2938			task_rq_unlock(rq, p, rf);
  2939			if (push_task) {
  2940				stop_one_cpu_nowait(rq->cpu, push_cpu_stop,
  2941						    p, &rq->push_work);
  2942			}
  2943			preempt_enable();
  2944	
  2945			if (complete)
  2946				complete_all(&pending->done);
  2947	
  2948			return 0;
  2949		}
  2950	
  2951		if (!(flags & SCA_MIGRATE_ENABLE)) {
  2952			/* serialized by p->pi_lock */
  2953			if (!p->migration_pending) {
  2954				/* Install the request */
  2955				refcount_set(&my_pending.refs, 1);
  2956				init_completion(&my_pending.done);
  2957				my_pending.arg = (struct migration_arg) {
  2958					.task = p,
  2959					.dest_cpu = dest_cpu,
  2960					.pending = &my_pending,
  2961				};
  2962	
  2963				p->migration_pending = &my_pending;
  2964			} else {
  2965				pending = p->migration_pending;
  2966				refcount_inc(&pending->refs);
  2967				/*
  2968				 * Affinity has changed, but we've already installed a
  2969				 * pending. migration_cpu_stop() *must* see this, else
  2970				 * we risk a completion of the pending despite having a
  2971				 * task on a disallowed CPU.
  2972				 *
  2973				 * Serialized by p->pi_lock, so this is safe.
  2974				 */
  2975				pending->arg.dest_cpu = dest_cpu;
  2976			}
  2977		}
  2978		pending = p->migration_pending;
  2979		/*
  2980		 * - !MIGRATE_ENABLE:
  2981		 *   we'll have installed a pending if there wasn't one already.
  2982		 *
  2983		 * - MIGRATE_ENABLE:
  2984		 *   we're here because the current CPU isn't matching anymore,
  2985		 *   the only way that can happen is because of a concurrent
  2986		 *   set_cpus_allowed_ptr() call, which should then still be
  2987		 *   pending completion.
  2988		 *
  2989		 * Either way, we really should have a @pending here.
  2990		 */
  2991		if (WARN_ON_ONCE(!pending)) {
  2992			task_rq_unlock(rq, p, rf);
  2993			return -EINVAL;
  2994		}
  2995	
  2996		if (task_on_cpu(rq, p) || READ_ONCE(p->__state) == TASK_WAKING) {
  2997			/*
  2998			 * MIGRATE_ENABLE gets here because 'p == current', but for
  2999			 * anything else we cannot do is_migration_disabled(), punt
  3000			 * and have the stopper function handle it all race-free.
  3001			 */
  3002			stop_pending = pending->stop_pending;
  3003			if (!stop_pending)
  3004				pending->stop_pending = true;
  3005	
  3006			if (flags & SCA_MIGRATE_ENABLE)
  3007				p->migration_flags &= ~MDF_PUSH;
  3008	
  3009			preempt_disable();
  3010			task_rq_unlock(rq, p, rf);
  3011			if (!stop_pending) {
  3012				stop_one_cpu_nowait(cpu_of(rq), migration_cpu_stop,
  3013						    &pending->arg, &pending->stop_work);
  3014			}
  3015			preempt_enable();
  3016	
  3017			if (flags & SCA_MIGRATE_ENABLE)
  3018				return 0;
  3019		} else {
  3020	
  3021			if (!is_migration_disabled(p)) {
  3022				if (task_on_rq_queued(p))
  3023					rq = move_queued_task(rq, rf, p, dest_cpu);
  3024	
  3025				if (!pending->stop_pending) {
  3026					p->migration_pending = NULL;
  3027					complete = true;
  3028				}
  3029			}
  3030			task_rq_unlock(rq, p, rf);
  3031	
  3032			if (complete)
  3033				complete_all(&pending->done);
  3034		}
  3035	
  3036		wait_for_completion(&pending->done);
  3037	
  3038		if (refcount_dec_and_test(&pending->refs))
  3039			wake_up_var(&pending->refs); /* No UaF, just an address */
  3040	
  3041		/*
  3042		 * Block the original owner of &pending until all subsequent callers
  3043		 * have seen the completion and decremented the refcount
  3044		 */
  3045		wait_var_event(&my_pending.refs, !refcount_read(&my_pending.refs));
  3046	
  3047		/* ARGH */
  3048		WARN_ON_ONCE(my_pending.stop_pending);
  3049	
  3050		return 0;
  3051	}
  3052	
  3053	/*
  3054	 * Called with both p->pi_lock and rq->lock held; drops both before returning.
  3055	 */
  3056	static int __set_cpus_allowed_ptr_locked(struct task_struct *p,
  3057						 struct affinity_context *ctx,
  3058						 struct rq *rq,
  3059						 struct rq_flags *rf)
  3060		__releases(rq->lock)
  3061		__releases(p->pi_lock)
  3062	{
  3063		const struct cpumask *cpu_allowed_mask = task_cpu_possible_mask(p);
  3064		const struct cpumask *cpu_valid_mask = cpu_active_mask;
  3065		bool kthread = p->flags & PF_KTHREAD;
  3066		unsigned int dest_cpu;
  3067		int ret = 0;
  3068	
  3069		update_rq_clock(rq);
  3070	
  3071		if (kthread || is_migration_disabled(p)) {
  3072			/*
  3073			 * Kernel threads are allowed on online && !active CPUs,
  3074			 * however, during cpu-hot-unplug, even these might get pushed
  3075			 * away if not KTHREAD_IS_PER_CPU.
  3076			 *
  3077			 * Specifically, migration_disabled() tasks must not fail the
  3078			 * cpumask_any_and_distribute() pick below, esp. so on
  3079			 * SCA_MIGRATE_ENABLE, otherwise we'll not call
  3080			 * set_cpus_allowed_common() and actually reset p->cpus_ptr.
  3081			 */
  3082			cpu_valid_mask = cpu_online_mask;
  3083		}
  3084	
  3085		if (!kthread && !cpumask_subset(ctx->new_mask, cpu_allowed_mask)) {
  3086			ret = -EINVAL;
  3087			goto out;
  3088		}
  3089	
  3090		/*
  3091		 * Must re-check here, to close a race against __kthread_bind(),
  3092		 * sched_setaffinity() is not guaranteed to observe the flag.
  3093		 */
  3094		if ((ctx->flags & SCA_CHECK) && (p->flags & PF_NO_SETAFFINITY)) {
  3095			ret = -EINVAL;
  3096			goto out;
  3097		}
  3098	
  3099		if (!(ctx->flags & SCA_MIGRATE_ENABLE)) {
  3100			if (cpumask_equal(&p->cpus_mask, ctx->new_mask)) {
  3101				if (ctx->flags & SCA_USER)
  3102					swap(p->user_cpus_ptr, ctx->user_mask);
  3103				goto out;
  3104			}
  3105	
  3106			if (WARN_ON_ONCE(p == current &&
  3107					 is_migration_disabled(p) &&
  3108					 !cpumask_test_cpu(task_cpu(p), ctx->new_mask))) {
  3109				ret = -EBUSY;
  3110				goto out;
  3111			}
  3112		}
  3113	
  3114		/*
  3115		 * Picking a ~random cpu helps in cases where we are changing affinity
  3116		 * for groups of tasks (ie. cpuset), so that load balancing is not
  3117		 * immediately required to distribute the tasks within their new mask.
  3118		 */
  3119		dest_cpu = cpumask_any_and_distribute(cpu_valid_mask, ctx->new_mask);
  3120		if (dest_cpu >= nr_cpu_ids) {
  3121			ret = -EINVAL;
  3122			goto out;
  3123		}
  3124	
  3125		__do_set_cpus_allowed(p, ctx);
  3126	
  3127		return affine_move_task(rq, p, rf, dest_cpu, ctx->flags);
  3128	
  3129	out:
  3130		task_rq_unlock(rq, p, rf);
  3131	
  3132		return ret;
  3133	}
  3134	
  3135	/*
  3136	 * Change a given task's CPU affinity. Migrate the thread to a
  3137	 * proper CPU and schedule it away if the CPU it's executing on
  3138	 * is removed from the allowed bitmask.
  3139	 *
  3140	 * NOTE: the caller must have a valid reference to the task, the
  3141	 * task must not exit() & deallocate itself prematurely. The
  3142	 * call is not atomic; no spinlocks may be held.
  3143	 */
  3144	int __set_cpus_allowed_ptr(struct task_struct *p, struct affinity_context *ctx)
  3145	{
  3146		struct rq_flags rf;
  3147		struct rq *rq;
  3148	
  3149		rq = task_rq_lock(p, &rf);
  3150		/*
  3151		 * Masking should be skipped if SCA_USER or any of the SCA_MIGRATE_*
  3152		 * flags are set.
  3153		 */
  3154		if (p->user_cpus_ptr &&
  3155		    !(ctx->flags & (SCA_USER | SCA_MIGRATE_ENABLE | SCA_MIGRATE_DISABLE)) &&
  3156		    cpumask_and(rq->scratch_mask, ctx->new_mask, p->user_cpus_ptr))
  3157			ctx->new_mask = rq->scratch_mask;
  3158	
  3159		return __set_cpus_allowed_ptr_locked(p, ctx, rq, &rf);
  3160	}
  3161	
  3162	int set_cpus_allowed_ptr(struct task_struct *p, const struct cpumask *new_mask)
  3163	{
  3164		struct affinity_context ac = {
  3165			.new_mask  = new_mask,
  3166			.flags     = 0,
  3167		};
  3168	
  3169		return __set_cpus_allowed_ptr(p, &ac);
  3170	}
  3171	EXPORT_SYMBOL_GPL(set_cpus_allowed_ptr);
  3172	
  3173	/*
  3174	 * Change a given task's CPU affinity to the intersection of its current
  3175	 * affinity mask and @subset_mask, writing the resulting mask to @new_mask.
  3176	 * If user_cpus_ptr is defined, use it as the basis for restricting CPU
  3177	 * affinity or use cpu_online_mask instead.
  3178	 *
  3179	 * If the resulting mask is empty, leave the affinity unchanged and return
  3180	 * -EINVAL.
  3181	 */
  3182	static int restrict_cpus_allowed_ptr(struct task_struct *p,
  3183					     struct cpumask *new_mask,
  3184					     const struct cpumask *subset_mask)
  3185	{
  3186		struct affinity_context ac = {
  3187			.new_mask  = new_mask,
  3188			.flags     = 0,
  3189		};
  3190		struct rq_flags rf;
  3191		struct rq *rq;
  3192		int err;
  3193	
  3194		rq = task_rq_lock(p, &rf);
  3195	
  3196		/*
  3197		 * Forcefully restricting the affinity of a deadline task is
  3198		 * likely to cause problems, so fail and noisily override the
  3199		 * mask entirely.
  3200		 */
  3201		if (task_has_dl_policy(p) && dl_bandwidth_enabled()) {
  3202			err = -EPERM;
  3203			goto err_unlock;
  3204		}
  3205	
  3206		if (!cpumask_and(new_mask, task_user_cpus(p), subset_mask)) {
  3207			err = -EINVAL;
  3208			goto err_unlock;
  3209		}
  3210	
  3211		return __set_cpus_allowed_ptr_locked(p, &ac, rq, &rf);
  3212	
  3213	err_unlock:
  3214		task_rq_unlock(rq, p, &rf);
  3215		return err;
  3216	}
  3217	
  3218	/*
  3219	 * Restrict the CPU affinity of task @p so that it is a subset of
  3220	 * task_cpu_possible_mask() and point @p->user_cpus_ptr to a copy of the
  3221	 * old affinity mask. If the resulting mask is empty, we warn and walk
  3222	 * up the cpuset hierarchy until we find a suitable mask.
  3223	 */
  3224	void force_compatible_cpus_allowed_ptr(struct task_struct *p)
  3225	{
  3226		cpumask_var_t new_mask;
  3227		const struct cpumask *override_mask = task_cpu_possible_mask(p);
  3228	
  3229		alloc_cpumask_var(&new_mask, GFP_KERNEL);
  3230	
  3231		/*
  3232		 * __migrate_task() can fail silently in the face of concurrent
  3233		 * offlining of the chosen destination CPU, so take the hotplug
  3234		 * lock to ensure that the migration succeeds.
  3235		 */
  3236		cpus_read_lock();
  3237		if (!cpumask_available(new_mask))
  3238			goto out_set_mask;
  3239	
  3240		if (!restrict_cpus_allowed_ptr(p, new_mask, override_mask))
  3241			goto out_free_mask;
  3242	
  3243		/*
  3244		 * We failed to find a valid subset of the affinity mask for the
  3245		 * task, so override it based on its cpuset hierarchy.
  3246		 */
  3247		cpuset_cpus_allowed(p, new_mask);
  3248		override_mask = new_mask;
  3249	
  3250	out_set_mask:
  3251		if (printk_ratelimit()) {
  3252			printk_deferred("Overriding affinity for process %d (%s) to CPUs %*pbl\n",
  3253					task_pid_nr(p), p->comm,
  3254					cpumask_pr_args(override_mask));
  3255		}
  3256	
  3257		WARN_ON(set_cpus_allowed_ptr(p, override_mask));
  3258	out_free_mask:
  3259		cpus_read_unlock();
  3260		free_cpumask_var(new_mask);
  3261	}
  3262	
  3263	/*
  3264	 * Restore the affinity of a task @p which was previously restricted by a
  3265	 * call to force_compatible_cpus_allowed_ptr().
  3266	 *
  3267	 * It is the caller's responsibility to serialise this with any calls to
  3268	 * force_compatible_cpus_allowed_ptr(@p).
  3269	 */
  3270	void relax_compatible_cpus_allowed_ptr(struct task_struct *p)
  3271	{
  3272		struct affinity_context ac = {
  3273			.new_mask  = task_user_cpus(p),
  3274			.flags     = 0,
  3275		};
  3276		int ret;
  3277	
  3278		/*
  3279		 * Try to restore the old affinity mask with __sched_setaffinity().
  3280		 * Cpuset masking will be done there too.
  3281		 */
  3282		ret = __sched_setaffinity(p, &ac);
  3283		WARN_ON_ONCE(ret);
  3284	}
  3285	
  3286	void set_task_cpu(struct task_struct *p, unsigned int new_cpu)
  3287	{
  3288	#ifdef CONFIG_SCHED_DEBUG
  3289		unsigned int state = READ_ONCE(p->__state);
  3290	
  3291		/*
  3292		 * We should never call set_task_cpu() on a blocked task,
  3293		 * ttwu() will sort out the placement.
  3294		 */
  3295		WARN_ON_ONCE(state != TASK_RUNNING && state != TASK_WAKING && !p->on_rq);
  3296	
  3297		/*
  3298		 * Migrating fair class task must have p->on_rq = TASK_ON_RQ_MIGRATING,
  3299		 * because schedstat_wait_{start,end} rebase migrating task's wait_start
  3300		 * time relying on p->on_rq.
  3301		 */
  3302		WARN_ON_ONCE(state == TASK_RUNNING &&
  3303			     p->sched_class == &fair_sched_class &&
  3304			     (p->on_rq && !task_on_rq_migrating(p)));
  3305	
  3306	#ifdef CONFIG_LOCKDEP
  3307		/*
  3308		 * The caller should hold either p->pi_lock or rq->lock, when changing
  3309		 * a task's CPU. ->pi_lock for waking tasks, rq->lock for runnable tasks.
  3310		 *
  3311		 * sched_move_task() holds both and thus holding either pins the cgroup,
  3312		 * see task_group().
  3313		 *
  3314		 * Furthermore, all task_rq users should acquire both locks, see
  3315		 * task_rq_lock().
  3316		 */
  3317		WARN_ON_ONCE(debug_locks && !(lockdep_is_held(&p->pi_lock) ||
  3318					      lockdep_is_held(__rq_lockp(task_rq(p)))));
  3319	#endif
  3320		/*
  3321		 * Clearly, migrating tasks to offline CPUs is a fairly daft thing.
  3322		 */
  3323		WARN_ON_ONCE(!cpu_online(new_cpu));
  3324	
  3325		WARN_ON_ONCE(is_migration_disabled(p));
  3326	#endif
  3327	
  3328		trace_sched_migrate_task(p, new_cpu);
  3329	
  3330		if (task_cpu(p) != new_cpu) {
  3331			if (p->sched_class->migrate_task_rq)
  3332				p->sched_class->migrate_task_rq(p, new_cpu);
  3333			p->se.nr_migrations++;
  3334			rseq_migrate(p);
  3335			sched_mm_cid_migrate_from(p);
  3336			perf_event_task_migrate(p);
  3337		}
  3338	
  3339		__set_task_cpu(p, new_cpu);
  3340	}
  3341	
  3342	#ifdef CONFIG_NUMA_BALANCING
  3343	static void __migrate_swap_task(struct task_struct *p, int cpu)
  3344	{
  3345		if (task_on_rq_queued(p)) {
  3346			struct rq *src_rq, *dst_rq;
  3347			struct rq_flags srf, drf;
  3348	
  3349			src_rq = task_rq(p);
  3350			dst_rq = cpu_rq(cpu);
  3351	
  3352			rq_pin_lock(src_rq, &srf);
  3353			rq_pin_lock(dst_rq, &drf);
  3354	
  3355			move_queued_task_locked(src_rq, dst_rq, p);
  3356			wakeup_preempt(dst_rq, p, 0);
  3357	
  3358			rq_unpin_lock(dst_rq, &drf);
  3359			rq_unpin_lock(src_rq, &srf);
  3360	
  3361		} else {
  3362			/*
  3363			 * Task isn't running anymore; make it appear like we migrated
  3364			 * it before it went to sleep. This means on wakeup we make the
  3365			 * previous CPU our target instead of where it really is.
  3366			 */
  3367			p->wake_cpu = cpu;
  3368		}
  3369	}
  3370	
  3371	struct migration_swap_arg {
  3372		struct task_struct *src_task, *dst_task;
  3373		int src_cpu, dst_cpu;
  3374	};
  3375	
  3376	static int migrate_swap_stop(void *data)
  3377	{
  3378		struct migration_swap_arg *arg = data;
  3379		struct rq *src_rq, *dst_rq;
  3380	
  3381		if (!cpu_active(arg->src_cpu) || !cpu_active(arg->dst_cpu))
  3382			return -EAGAIN;
  3383	
  3384		src_rq = cpu_rq(arg->src_cpu);
  3385		dst_rq = cpu_rq(arg->dst_cpu);
  3386	
  3387		guard(double_raw_spinlock)(&arg->src_task->pi_lock, &arg->dst_task->pi_lock);
  3388		guard(double_rq_lock)(src_rq, dst_rq);
  3389	
  3390		if (task_cpu(arg->dst_task) != arg->dst_cpu)
  3391			return -EAGAIN;
  3392	
  3393		if (task_cpu(arg->src_task) != arg->src_cpu)
  3394			return -EAGAIN;
  3395	
  3396		if (!cpumask_test_cpu(arg->dst_cpu, arg->src_task->cpus_ptr))
  3397			return -EAGAIN;
  3398	
  3399		if (!cpumask_test_cpu(arg->src_cpu, arg->dst_task->cpus_ptr))
  3400			return -EAGAIN;
  3401	
  3402		__migrate_swap_task(arg->src_task, arg->dst_cpu);
  3403		__migrate_swap_task(arg->dst_task, arg->src_cpu);
  3404	
  3405		return 0;
  3406	}
  3407	
  3408	/*
  3409	 * Cross migrate two tasks
  3410	 */
  3411	int migrate_swap(struct task_struct *cur, struct task_struct *p,
  3412			int target_cpu, int curr_cpu)
  3413	{
  3414		struct migration_swap_arg arg;
  3415		int ret = -EINVAL;
  3416	
  3417		arg = (struct migration_swap_arg){
  3418			.src_task = cur,
  3419			.src_cpu = curr_cpu,
  3420			.dst_task = p,
  3421			.dst_cpu = target_cpu,
  3422		};
  3423	
  3424		if (arg.src_cpu == arg.dst_cpu)
  3425			goto out;
  3426	
  3427		/*
  3428		 * These three tests are all lockless; this is OK since all of them
  3429		 * will be re-checked with proper locks held further down the line.
  3430		 */
  3431		if (!cpu_active(arg.src_cpu) || !cpu_active(arg.dst_cpu))
  3432			goto out;
  3433	
  3434		if (!cpumask_test_cpu(arg.dst_cpu, arg.src_task->cpus_ptr))
  3435			goto out;
  3436	
  3437		if (!cpumask_test_cpu(arg.src_cpu, arg.dst_task->cpus_ptr))
  3438			goto out;
  3439	
  3440		trace_sched_swap_numa(cur, arg.src_cpu, p, arg.dst_cpu);
  3441		ret = stop_two_cpus(arg.dst_cpu, arg.src_cpu, migrate_swap_stop, &arg);
  3442	
  3443	out:
  3444		return ret;
  3445	}
  3446	#endif /* CONFIG_NUMA_BALANCING */
  3447	
  3448	/***
  3449	 * kick_process - kick a running thread to enter/exit the kernel
  3450	 * @p: the to-be-kicked thread
  3451	 *
  3452	 * Cause a process which is running on another CPU to enter
  3453	 * kernel-mode, without any delay. (to get signals handled.)
  3454	 *
  3455	 * NOTE: this function doesn't have to take the runqueue lock,
  3456	 * because all it wants to ensure is that the remote task enters
  3457	 * the kernel. If the IPI races and the task has been migrated
  3458	 * to another CPU then no harm is done and the purpose has been
  3459	 * achieved as well.
  3460	 */
  3461	void kick_process(struct task_struct *p)
  3462	{
  3463		guard(preempt)();
  3464		int cpu = task_cpu(p);
  3465	
  3466		if ((cpu != smp_processor_id()) && task_curr(p))
  3467			smp_send_reschedule(cpu);
  3468	}
  3469	EXPORT_SYMBOL_GPL(kick_process);
  3470	
  3471	/*
  3472	 * ->cpus_ptr is protected by both rq->lock and p->pi_lock
  3473	 *
  3474	 * A few notes on cpu_active vs cpu_online:
  3475	 *
  3476	 *  - cpu_active must be a subset of cpu_online
  3477	 *
  3478	 *  - on CPU-up we allow per-CPU kthreads on the online && !active CPU,
  3479	 *    see __set_cpus_allowed_ptr(). At this point the newly online
  3480	 *    CPU isn't yet part of the sched domains, and balancing will not
  3481	 *    see it.
  3482	 *
  3483	 *  - on CPU-down we clear cpu_active() to mask the sched domains and
  3484	 *    avoid the load balancer to place new tasks on the to be removed
  3485	 *    CPU. Existing tasks will remain running there and will be taken
  3486	 *    off.
  3487	 *
  3488	 * This means that fallback selection must not select !active CPUs.
  3489	 * And can assume that any active CPU must be online. Conversely
  3490	 * select_task_rq() below may allow selection of !active CPUs in order
  3491	 * to satisfy the above rules.
  3492	 */
  3493	static int select_fallback_rq(int cpu, struct task_struct *p)
  3494	{
  3495		int nid = cpu_to_node(cpu);
  3496		const struct cpumask *nodemask = NULL;
  3497		enum { cpuset, possible, fail } state = cpuset;
  3498		int dest_cpu;
  3499	
  3500		/*
  3501		 * If the node that the CPU is on has been offlined, cpu_to_node()
  3502		 * will return -1. There is no CPU on the node, and we should
  3503		 * select the CPU on the other node.
  3504		 */
  3505		if (nid != -1) {
  3506			nodemask = cpumask_of_node(nid);
  3507	
  3508			/* Look for allowed, online CPU in same node. */
  3509			for_each_cpu(dest_cpu, nodemask) {
  3510				if (is_cpu_allowed(p, dest_cpu))
  3511					return dest_cpu;
  3512			}
  3513		}
  3514	
  3515		for (;;) {
  3516			/* Any allowed, online CPU? */
  3517			for_each_cpu(dest_cpu, p->cpus_ptr) {
  3518				if (!is_cpu_allowed(p, dest_cpu))
  3519					continue;
  3520	
  3521				goto out;
  3522			}
  3523	
  3524			/* No more Mr. Nice Guy. */
  3525			switch (state) {
  3526			case cpuset:
  3527				if (cpuset_cpus_allowed_fallback(p)) {
  3528					state = possible;
  3529					break;
  3530				}
  3531				fallthrough;
  3532			case possible:
  3533				/*
  3534				 * XXX When called from select_task_rq() we only
  3535				 * hold p->pi_lock and again violate locking order.
  3536				 *
  3537				 * More yuck to audit.
  3538				 */
  3539				do_set_cpus_allowed(p, task_cpu_possible_mask(p));
  3540				state = fail;
  3541				break;
  3542			case fail:
  3543				BUG();
  3544				break;
  3545			}
  3546		}
  3547	
  3548	out:
  3549		if (state != cpuset) {
  3550			/*
  3551			 * Don't tell them about moving exiting tasks or
  3552			 * kernel threads (both mm NULL), since they never
  3553			 * leave kernel.
  3554			 */
  3555			if (p->mm && printk_ratelimit()) {
  3556				printk_deferred("process %d (%s) no longer affine to cpu%d\n",
  3557						task_pid_nr(p), p->comm, cpu);
  3558			}
  3559		}
  3560	
  3561		return dest_cpu;
  3562	}
  3563	
  3564	/*
  3565	 * The caller (fork, wakeup) owns p->pi_lock, ->cpus_ptr is stable.
  3566	 */
  3567	static inline
  3568	int select_task_rq(struct task_struct *p, int cpu, int *wake_flags)
  3569	{
  3570		lockdep_assert_held(&p->pi_lock);
  3571	
  3572		if (p->nr_cpus_allowed > 1 && !is_migration_disabled(p)) {
  3573			cpu = p->sched_class->select_task_rq(p, cpu, *wake_flags);
  3574			*wake_flags |= WF_RQ_SELECTED;
  3575		} else {
  3576			cpu = cpumask_any(p->cpus_ptr);
  3577		}
  3578	
  3579		/*
  3580		 * In order not to call set_task_cpu() on a blocking task we need
  3581		 * to rely on ttwu() to place the task on a valid ->cpus_ptr
  3582		 * CPU.
  3583		 *
  3584		 * Since this is common to all placement strategies, this lives here.
  3585		 *
  3586		 * [ this allows ->select_task() to simply return task_cpu(p) and
  3587		 *   not worry about this generic constraint ]
  3588		 */
  3589		if (unlikely(!is_cpu_allowed(p, cpu)))
  3590			cpu = select_fallback_rq(task_cpu(p), p);
  3591	
  3592		return cpu;
  3593	}
  3594	
  3595	void sched_set_stop_task(int cpu, struct task_struct *stop)
  3596	{
  3597		static struct lock_class_key stop_pi_lock;
  3598		struct sched_param param = { .sched_priority = MAX_RT_PRIO - 1 };
  3599		struct task_struct *old_stop = cpu_rq(cpu)->stop;
  3600	
  3601		if (stop) {
  3602			/*
  3603			 * Make it appear like a SCHED_FIFO task, its something
  3604			 * userspace knows about and won't get confused about.
  3605			 *
  3606			 * Also, it will make PI more or less work without too
  3607			 * much confusion -- but then, stop work should not
  3608			 * rely on PI working anyway.
  3609			 */
  3610			sched_setscheduler_nocheck(stop, SCHED_FIFO, &param);
  3611	
  3612			stop->sched_class = &stop_sched_class;
  3613	
  3614			/*
  3615			 * The PI code calls rt_mutex_setprio() with ->pi_lock held to
  3616			 * adjust the effective priority of a task. As a result,
  3617			 * rt_mutex_setprio() can trigger (RT) balancing operations,
  3618			 * which can then trigger wakeups of the stop thread to push
  3619			 * around the current task.
  3620			 *
  3621			 * The stop task itself will never be part of the PI-chain, it
  3622			 * never blocks, therefore that ->pi_lock recursion is safe.
  3623			 * Tell lockdep about this by placing the stop->pi_lock in its
  3624			 * own class.
  3625			 */
  3626			lockdep_set_class(&stop->pi_lock, &stop_pi_lock);
  3627		}
  3628	
  3629		cpu_rq(cpu)->stop = stop;
  3630	
  3631		if (old_stop) {
  3632			/*
  3633			 * Reset it back to a normal scheduling class so that
  3634			 * it can die in pieces.
  3635			 */
  3636			old_stop->sched_class = &rt_sched_class;
  3637		}
  3638	}
  3639	
  3640	#else /* CONFIG_SMP */
  3641	
  3642	static inline void migrate_disable_switch(struct rq *rq, struct task_struct *p) { }
  3643	
  3644	static inline bool rq_has_pinned_tasks(struct rq *rq)
  3645	{
  3646		return false;
  3647	}
  3648	
  3649	#endif /* !CONFIG_SMP */
  3650	
  3651	static void
  3652	ttwu_stat(struct task_struct *p, int cpu, int wake_flags)
  3653	{
  3654		struct rq *rq;
  3655	
  3656		if (!schedstat_enabled())
  3657			return;
  3658	
  3659		rq = this_rq();
  3660	
  3661	#ifdef CONFIG_SMP
  3662		if (cpu == rq->cpu) {
  3663			__schedstat_inc(rq->ttwu_local);
  3664			__schedstat_inc(p->stats.nr_wakeups_local);
  3665		} else {
  3666			struct sched_domain *sd;
  3667	
  3668			__schedstat_inc(p->stats.nr_wakeups_remote);
  3669	
  3670			guard(rcu)();
  3671			for_each_domain(rq->cpu, sd) {
  3672				if (cpumask_test_cpu(cpu, sched_domain_span(sd))) {
  3673					__schedstat_inc(sd->ttwu_wake_remote);
  3674					break;
  3675				}
  3676			}
  3677		}
  3678	
  3679		if (wake_flags & WF_MIGRATED)
  3680			__schedstat_inc(p->stats.nr_wakeups_migrate);
  3681	#endif /* CONFIG_SMP */
  3682	
  3683		__schedstat_inc(rq->ttwu_count);
  3684		__schedstat_inc(p->stats.nr_wakeups);
  3685	
  3686		if (wake_flags & WF_SYNC)
  3687			__schedstat_inc(p->stats.nr_wakeups_sync);
  3688	}
  3689	
  3690	/*
  3691	 * Mark the task runnable.
  3692	 */
  3693	static inline void ttwu_do_wakeup(struct task_struct *p)
  3694	{
  3695		WRITE_ONCE(p->__state, TASK_RUNNING);
  3696		trace_sched_wakeup(p);
  3697	}
  3698	
  3699	static void
  3700	ttwu_do_activate(struct rq *rq, struct task_struct *p, int wake_flags,
  3701			 struct rq_flags *rf)
  3702	{
  3703		int en_flags = ENQUEUE_WAKEUP | ENQUEUE_NOCLOCK;
  3704	
  3705		lockdep_assert_rq_held(rq);
  3706	
  3707		if (p->sched_contributes_to_load)
  3708			rq->nr_uninterruptible--;
  3709	
  3710	#ifdef CONFIG_SMP
  3711		if (wake_flags & WF_RQ_SELECTED)
  3712			en_flags |= ENQUEUE_RQ_SELECTED;
  3713		if (wake_flags & WF_MIGRATED)
  3714			en_flags |= ENQUEUE_MIGRATED;
  3715		else
  3716	#endif
  3717		if (p->in_iowait) {
  3718			delayacct_blkio_end(p);
  3719			atomic_dec(&task_rq(p)->nr_iowait);
  3720		}
  3721	
  3722		activate_task(rq, p, en_flags);
  3723		wakeup_preempt(rq, p, wake_flags);
  3724	
  3725		ttwu_do_wakeup(p);
  3726	
  3727	#ifdef CONFIG_SMP
  3728		if (p->sched_class->task_woken) {
  3729			/*
  3730			 * Our task @p is fully woken up and running; so it's safe to
  3731			 * drop the rq->lock, hereafter rq is only used for statistics.
  3732			 */
  3733			rq_unpin_lock(rq, rf);
  3734			p->sched_class->task_woken(rq, p);
  3735			rq_repin_lock(rq, rf);
  3736		}
  3737	
  3738		if (rq->idle_stamp) {
  3739			u64 delta = rq_clock(rq) - rq->idle_stamp;
  3740			u64 max = 2*rq->max_idle_balance_cost;
  3741	
  3742			update_avg(&rq->avg_idle, delta);
  3743	
  3744			if (rq->avg_idle > max)
  3745				rq->avg_idle = max;
  3746	
  3747			rq->idle_stamp = 0;
  3748		}
  3749	#endif
  3750	}
  3751	
  3752	/*
  3753	 * Consider @p being inside a wait loop:
  3754	 *
  3755	 *   for (;;) {
  3756	 *      set_current_state(TASK_UNINTERRUPTIBLE);
  3757	 *
  3758	 *      if (CONDITION)
  3759	 *         break;
  3760	 *
  3761	 *      schedule();
  3762	 *   }
  3763	 *   __set_current_state(TASK_RUNNING);
  3764	 *
  3765	 * between set_current_state() and schedule(). In this case @p is still
  3766	 * runnable, so all that needs doing is change p->state back to TASK_RUNNING in
  3767	 * an atomic manner.
  3768	 *
  3769	 * By taking task_rq(p)->lock we serialize against schedule(), if @p->on_rq
  3770	 * then schedule() must still happen and p->state can be changed to
  3771	 * TASK_RUNNING. Otherwise we lost the race, schedule() has happened, and we
  3772	 * need to do a full wakeup with enqueue.
  3773	 *
  3774	 * Returns: %true when the wakeup is done,
  3775	 *          %false otherwise.
  3776	 */
  3777	static int ttwu_runnable(struct task_struct *p, int wake_flags)
  3778	{
  3779		struct rq_flags rf;
  3780		struct rq *rq;
  3781		int ret = 0;
  3782	
  3783		rq = __task_rq_lock(p, &rf);
  3784		if (task_on_rq_queued(p)) {
  3785			update_rq_clock(rq);
  3786			if (p->se.sched_delayed)
  3787				enqueue_task(rq, p, ENQUEUE_NOCLOCK | ENQUEUE_DELAYED);
  3788			if (!task_on_cpu(rq, p)) {
  3789				/*
  3790				 * When on_rq && !on_cpu the task is preempted, see if
  3791				 * it should preempt the task that is current now.
  3792				 */
  3793				wakeup_preempt(rq, p, wake_flags);
  3794			}
  3795			ttwu_do_wakeup(p);
  3796			ret = 1;
  3797		}
  3798		__task_rq_unlock(rq, &rf);
  3799	
  3800		return ret;
  3801	}
  3802	
  3803	#ifdef CONFIG_SMP
  3804	void sched_ttwu_pending(void *arg)
  3805	{
  3806		struct llist_node *llist = arg;
  3807		struct rq *rq = this_rq();
  3808		struct task_struct *p, *t;
  3809		struct rq_flags rf;
  3810	
  3811		if (!llist)
  3812			return;
  3813	
  3814		rq_lock_irqsave(rq, &rf);
  3815		update_rq_clock(rq);
  3816	
  3817		llist_for_each_entry_safe(p, t, llist, wake_entry.llist) {
  3818			if (WARN_ON_ONCE(p->on_cpu))
  3819				smp_cond_load_acquire(&p->on_cpu, !VAL);
  3820	
  3821			if (WARN_ON_ONCE(task_cpu(p) != cpu_of(rq)))
  3822				set_task_cpu(p, cpu_of(rq));
  3823	
  3824			ttwu_do_activate(rq, p, p->sched_remote_wakeup ? WF_MIGRATED : 0, &rf);
  3825		}
  3826	
  3827		/*
  3828		 * Must be after enqueueing at least once task such that
  3829		 * idle_cpu() does not observe a false-negative -- if it does,
  3830		 * it is possible for select_idle_siblings() to stack a number
  3831		 * of tasks on this CPU during that window.
  3832		 *
  3833		 * It is OK to clear ttwu_pending when another task pending.
  3834		 * We will receive IPI after local IRQ enabled and then enqueue it.
  3835		 * Since now nr_running > 0, idle_cpu() will always get correct result.
  3836		 */
  3837		WRITE_ONCE(rq->ttwu_pending, 0);
  3838		rq_unlock_irqrestore(rq, &rf);
  3839	}
  3840	
  3841	/*
  3842	 * Prepare the scene for sending an IPI for a remote smp_call
  3843	 *
  3844	 * Returns true if the caller can proceed with sending the IPI.
  3845	 * Returns false otherwise.
  3846	 */
  3847	bool call_function_single_prep_ipi(int cpu)
  3848	{
  3849		if (set_nr_if_polling(cpu_rq(cpu)->idle)) {
  3850			trace_sched_wake_idle_without_ipi(cpu);
  3851			return false;
  3852		}
  3853	
  3854		return true;
  3855	}
  3856	
  3857	/*
  3858	 * Queue a task on the target CPUs wake_list and wake the CPU via IPI if
  3859	 * necessary. The wakee CPU on receipt of the IPI will queue the task
  3860	 * via sched_ttwu_wakeup() for activation so the wakee incurs the cost
  3861	 * of the wakeup instead of the waker.
  3862	 */
  3863	static void __ttwu_queue_wakelist(struct task_struct *p, int cpu, int wake_flags)
  3864	{
  3865		struct rq *rq = cpu_rq(cpu);
  3866	
  3867		p->sched_remote_wakeup = !!(wake_flags & WF_MIGRATED);
  3868	
  3869		WRITE_ONCE(rq->ttwu_pending, 1);
  3870		__smp_call_single_queue(cpu, &p->wake_entry.llist);
  3871	}
  3872	
  3873	void wake_up_if_idle(int cpu)
  3874	{
  3875		struct rq *rq = cpu_rq(cpu);
  3876	
  3877		guard(rcu)();
  3878		if (is_idle_task(rcu_dereference(rq->curr))) {
  3879			guard(rq_lock_irqsave)(rq);
  3880			if (is_idle_task(rq->curr))
  3881				resched_curr(rq);
  3882		}
  3883	}
  3884	
  3885	bool cpus_equal_capacity(int this_cpu, int that_cpu)
  3886	{
  3887		if (!sched_asym_cpucap_active())
  3888			return true;
  3889	
  3890		if (this_cpu == that_cpu)
  3891			return true;
  3892	
  3893		return arch_scale_cpu_capacity(this_cpu) == arch_scale_cpu_capacity(that_cpu);
  3894	}
  3895	
  3896	bool cpus_share_cache(int this_cpu, int that_cpu)
  3897	{
  3898		if (this_cpu == that_cpu)
  3899			return true;
  3900	
  3901		return per_cpu(sd_llc_id, this_cpu) == per_cpu(sd_llc_id, that_cpu);
  3902	}
  3903	
  3904	/*
  3905	 * Whether CPUs are share cache resources, which means LLC on non-cluster
  3906	 * machines and LLC tag or L2 on machines with clusters.
  3907	 */
  3908	bool cpus_share_resources(int this_cpu, int that_cpu)
  3909	{
  3910		if (this_cpu == that_cpu)
  3911			return true;
  3912	
  3913		return per_cpu(sd_share_id, this_cpu) == per_cpu(sd_share_id, that_cpu);
  3914	}
  3915	
  3916	static inline bool ttwu_queue_cond(struct task_struct *p, int cpu)
  3917	{
  3918		/*
  3919		 * The BPF scheduler may depend on select_task_rq() being invoked during
  3920		 * wakeups. In addition, @p may end up executing on a different CPU
  3921		 * regardless of what happens in the wakeup path making the ttwu_queue
  3922		 * optimization less meaningful. Skip if on SCX.
  3923		 */
  3924		if (task_on_scx(p))
  3925			return false;
  3926	
  3927		/*
  3928		 * Do not complicate things with the async wake_list while the CPU is
  3929		 * in hotplug state.
  3930		 */
  3931		if (!cpu_active(cpu))
  3932			return false;
  3933	
  3934		/* Ensure the task will still be allowed to run on the CPU. */
  3935		if (!cpumask_test_cpu(cpu, p->cpus_ptr))
  3936			return false;
  3937	
  3938		/*
  3939		 * If the CPU does not share cache, then queue the task on the
  3940		 * remote rqs wakelist to avoid accessing remote data.
  3941		 */
  3942		if (!cpus_share_cache(smp_processor_id(), cpu))
  3943			return true;
  3944	
  3945		if (cpu == smp_processor_id())
  3946			return false;
  3947	
  3948		/*
  3949		 * If the wakee cpu is idle, or the task is descheduling and the
  3950		 * only running task on the CPU, then use the wakelist to offload
  3951		 * the task activation to the idle (or soon-to-be-idle) CPU as
  3952		 * the current CPU is likely busy. nr_running is checked to
  3953		 * avoid unnecessary task stacking.
  3954		 *
  3955		 * Note that we can only get here with (wakee) p->on_rq=0,
  3956		 * p->on_cpu can be whatever, we've done the dequeue, so
  3957		 * the wakee has been accounted out of ->nr_running.
  3958		 */
  3959		if (!cpu_rq(cpu)->nr_running)
  3960			return true;
  3961	
  3962		return false;
  3963	}
  3964	
  3965	static bool ttwu_queue_wakelist(struct task_struct *p, int cpu, int wake_flags)
  3966	{
  3967		if (sched_feat(TTWU_QUEUE) && ttwu_queue_cond(p, cpu)) {
  3968			sched_clock_cpu(cpu); /* Sync clocks across CPUs */
  3969			__ttwu_queue_wakelist(p, cpu, wake_flags);
  3970			return true;
  3971		}
  3972	
  3973		return false;
  3974	}
  3975	
  3976	#else /* !CONFIG_SMP */
  3977	
  3978	static inline bool ttwu_queue_wakelist(struct task_struct *p, int cpu, int wake_flags)
  3979	{
  3980		return false;
  3981	}
  3982	
  3983	#endif /* CONFIG_SMP */
  3984	
  3985	static void ttwu_queue(struct task_struct *p, int cpu, int wake_flags)
  3986	{
  3987		struct rq *rq = cpu_rq(cpu);
  3988		struct rq_flags rf;
  3989	
  3990		if (ttwu_queue_wakelist(p, cpu, wake_flags))
  3991			return;
  3992	
  3993		rq_lock(rq, &rf);
  3994		update_rq_clock(rq);
  3995		ttwu_do_activate(rq, p, wake_flags, &rf);
  3996		rq_unlock(rq, &rf);
  3997	}
  3998	
  3999	/*
  4000	 * Invoked from try_to_wake_up() to check whether the task can be woken up.
  4001	 *
  4002	 * The caller holds p::pi_lock if p != current or has preemption
  4003	 * disabled when p == current.
  4004	 *
  4005	 * The rules of saved_state:
  4006	 *
  4007	 *   The related locking code always holds p::pi_lock when updating
  4008	 *   p::saved_state, which means the code is fully serialized in both cases.
  4009	 *
  4010	 *   For PREEMPT_RT, the lock wait and lock wakeups happen via TASK_RTLOCK_WAIT.
  4011	 *   No other bits set. This allows to distinguish all wakeup scenarios.
  4012	 *
  4013	 *   For FREEZER, the wakeup happens via TASK_FROZEN. No other bits set. This
  4014	 *   allows us to prevent early wakeup of tasks before they can be run on
  4015	 *   asymmetric ISA architectures (eg ARMv9).
  4016	 */
  4017	static __always_inline
  4018	bool ttwu_state_match(struct task_struct *p, unsigned int state, int *success)
  4019	{
  4020		int match;
  4021	
  4022		if (IS_ENABLED(CONFIG_DEBUG_PREEMPT)) {
  4023			WARN_ON_ONCE((state & TASK_RTLOCK_WAIT) &&
  4024				     state != TASK_RTLOCK_WAIT);
  4025		}
  4026	
  4027		*success = !!(match = __task_state_match(p, state));
  4028	
  4029		/*
  4030		 * Saved state preserves the task state across blocking on
  4031		 * an RT lock or TASK_FREEZABLE tasks.  If the state matches,
  4032		 * set p::saved_state to TASK_RUNNING, but do not wake the task
  4033		 * because it waits for a lock wakeup or __thaw_task(). Also
  4034		 * indicate success because from the regular waker's point of
  4035		 * view this has succeeded.
  4036		 *
  4037		 * After acquiring the lock the task will restore p::__state
  4038		 * from p::saved_state which ensures that the regular
  4039		 * wakeup is not lost. The restore will also set
  4040		 * p::saved_state to TASK_RUNNING so any further tests will
  4041		 * not result in false positives vs. @success
  4042		 */
  4043		if (match < 0)
  4044			p->saved_state = TASK_RUNNING;
  4045	
  4046		return match > 0;
  4047	}
  4048	
  4049	/*
  4050	 * Notes on Program-Order guarantees on SMP systems.
  4051	 *
  4052	 *  MIGRATION
  4053	 *
  4054	 * The basic program-order guarantee on SMP systems is that when a task [t]
  4055	 * migrates, all its activity on its old CPU [c0] happens-before any subsequent
  4056	 * execution on its new CPU [c1].
  4057	 *
  4058	 * For migration (of runnable tasks) this is provided by the following means:
  4059	 *
  4060	 *  A) UNLOCK of the rq(c0)->lock scheduling out task t
  4061	 *  B) migration for t is required to synchronize *both* rq(c0)->lock and
  4062	 *     rq(c1)->lock (if not at the same time, then in that order).
  4063	 *  C) LOCK of the rq(c1)->lock scheduling in task
  4064	 *
  4065	 * Release/acquire chaining guarantees that B happens after A and C after B.
  4066	 * Note: the CPU doing B need not be c0 or c1
  4067	 *
  4068	 * Example:
  4069	 *
  4070	 *   CPU0            CPU1            CPU2
  4071	 *
  4072	 *   LOCK rq(0)->lock
  4073	 *   sched-out X
  4074	 *   sched-in Y
  4075	 *   UNLOCK rq(0)->lock
  4076	 *
  4077	 *                                   LOCK rq(0)->lock // orders against CPU0
  4078	 *                                   dequeue X
  4079	 *                                   UNLOCK rq(0)->lock
  4080	 *
  4081	 *                                   LOCK rq(1)->lock
  4082	 *                                   enqueue X
  4083	 *                                   UNLOCK rq(1)->lock
  4084	 *
  4085	 *                   LOCK rq(1)->lock // orders against CPU2
  4086	 *                   sched-out Z
  4087	 *                   sched-in X
  4088	 *                   UNLOCK rq(1)->lock
  4089	 *
  4090	 *
  4091	 *  BLOCKING -- aka. SLEEP + WAKEUP
  4092	 *
  4093	 * For blocking we (obviously) need to provide the same guarantee as for
  4094	 * migration. However the means are completely different as there is no lock
  4095	 * chain to provide order. Instead we do:
  4096	 *
  4097	 *   1) smp_store_release(X->on_cpu, 0)   -- finish_task()
  4098	 *   2) smp_cond_load_acquire(!X->on_cpu) -- try_to_wake_up()
  4099	 *
  4100	 * Example:
  4101	 *
  4102	 *   CPU0 (schedule)  CPU1 (try_to_wake_up) CPU2 (schedule)
  4103	 *
  4104	 *   LOCK rq(0)->lock LOCK X->pi_lock
  4105	 *   dequeue X
  4106	 *   sched-out X
  4107	 *   smp_store_release(X->on_cpu, 0);
  4108	 *
  4109	 *                    smp_cond_load_acquire(&X->on_cpu, !VAL);
  4110	 *                    X->state = WAKING
  4111	 *                    set_task_cpu(X,2)
  4112	 *
  4113	 *                    LOCK rq(2)->lock
  4114	 *                    enqueue X
  4115	 *                    X->state = RUNNING
  4116	 *                    UNLOCK rq(2)->lock
  4117	 *
  4118	 *                                          LOCK rq(2)->lock // orders against CPU1
  4119	 *                                          sched-out Z
  4120	 *                                          sched-in X
  4121	 *                                          UNLOCK rq(2)->lock
  4122	 *
  4123	 *                    UNLOCK X->pi_lock
  4124	 *   UNLOCK rq(0)->lock
  4125	 *
  4126	 *
  4127	 * However, for wakeups there is a second guarantee we must provide, namely we
  4128	 * must ensure that CONDITION=1 done by the caller can not be reordered with
  4129	 * accesses to the task state; see try_to_wake_up() and set_current_state().
  4130	 */
  4131	
  4132	/**
  4133	 * try_to_wake_up - wake up a thread
  4134	 * @p: the thread to be awakened
  4135	 * @state: the mask of task states that can be woken
  4136	 * @wake_flags: wake modifier flags (WF_*)
  4137	 *
  4138	 * Conceptually does:
  4139	 *
  4140	 *   If (@state & @p->state) @p->state = TASK_RUNNING.
  4141	 *
  4142	 * If the task was not queued/runnable, also place it back on a runqueue.
  4143	 *
  4144	 * This function is atomic against schedule() which would dequeue the task.
  4145	 *
  4146	 * It issues a full memory barrier before accessing @p->state, see the comment
  4147	 * with set_current_state().
  4148	 *
  4149	 * Uses p->pi_lock to serialize against concurrent wake-ups.
  4150	 *
  4151	 * Relies on p->pi_lock stabilizing:
  4152	 *  - p->sched_class
  4153	 *  - p->cpus_ptr
  4154	 *  - p->sched_task_group
  4155	 * in order to do migration, see its use of select_task_rq()/set_task_cpu().
  4156	 *
  4157	 * Tries really hard to only take one task_rq(p)->lock for performance.
  4158	 * Takes rq->lock in:
  4159	 *  - ttwu_runnable()    -- old rq, unavoidable, see comment there;
  4160	 *  - ttwu_queue()       -- new rq, for enqueue of the task;
  4161	 *  - psi_ttwu_dequeue() -- much sadness :-( accounting will kill us.
  4162	 *
  4163	 * As a consequence we race really badly with just about everything. See the
  4164	 * many memory barriers and their comments for details.
  4165	 *
  4166	 * Return: %true if @p->state changes (an actual wakeup was done),
  4167	 *	   %false otherwise.
  4168	 */
  4169	int try_to_wake_up(struct task_struct *p, unsigned int state, int wake_flags)
  4170	{
  4171		guard(preempt)();
  4172		int cpu, success = 0;
  4173	
  4174		wake_flags |= WF_TTWU;
  4175	
  4176		if (p == current) {
  4177			/*
  4178			 * We're waking current, this means 'p->on_rq' and 'task_cpu(p)
  4179			 * == smp_processor_id()'. Together this means we can special
  4180			 * case the whole 'p->on_rq && ttwu_runnable()' case below
  4181			 * without taking any locks.
  4182			 *
  4183			 * Specifically, given current runs ttwu() we must be before
  4184			 * schedule()'s block_task(), as such this must not observe
  4185			 * sched_delayed.
  4186			 *
  4187			 * In particular:
  4188			 *  - we rely on Program-Order guarantees for all the ordering,
  4189			 *  - we're serialized against set_special_state() by virtue of
  4190			 *    it disabling IRQs (this allows not taking ->pi_lock).
  4191			 */
  4192			SCHED_WARN_ON(p->se.sched_delayed);
  4193			if (!ttwu_state_match(p, state, &success))
  4194				goto out;
  4195	
  4196			trace_sched_waking(p);
  4197			ttwu_do_wakeup(p);
  4198			goto out;
  4199		}
  4200	
  4201		/*
  4202		 * If we are going to wake up a thread waiting for CONDITION we
  4203		 * need to ensure that CONDITION=1 done by the caller can not be
  4204		 * reordered with p->state check below. This pairs with smp_store_mb()
  4205		 * in set_current_state() that the waiting thread does.
  4206		 */
  4207		scoped_guard (raw_spinlock_irqsave, &p->pi_lock) {
  4208			smp_mb__after_spinlock();
  4209			if (!ttwu_state_match(p, state, &success))
  4210				break;
  4211	
  4212			trace_sched_waking(p);
  4213	
  4214			/*
  4215			 * Ensure we load p->on_rq _after_ p->state, otherwise it would
  4216			 * be possible to, falsely, observe p->on_rq == 0 and get stuck
  4217			 * in smp_cond_load_acquire() below.
  4218			 *
  4219			 * sched_ttwu_pending()			try_to_wake_up()
  4220			 *   STORE p->on_rq = 1			  LOAD p->state
  4221			 *   UNLOCK rq->lock
  4222			 *
  4223			 * __schedule() (switch to task 'p')
  4224			 *   LOCK rq->lock			  smp_rmb();
  4225			 *   smp_mb__after_spinlock();
  4226			 *   UNLOCK rq->lock
  4227			 *
  4228			 * [task p]
  4229			 *   STORE p->state = UNINTERRUPTIBLE	  LOAD p->on_rq
  4230			 *
  4231			 * Pairs with the LOCK+smp_mb__after_spinlock() on rq->lock in
  4232			 * __schedule().  See the comment for smp_mb__after_spinlock().
  4233			 *
  4234			 * A similar smp_rmb() lives in __task_needs_rq_lock().
  4235			 */
  4236			smp_rmb();
  4237			if (READ_ONCE(p->on_rq) && ttwu_runnable(p, wake_flags))
  4238				break;
  4239	
  4240	#ifdef CONFIG_SMP
  4241			/*
  4242			 * Ensure we load p->on_cpu _after_ p->on_rq, otherwise it would be
  4243			 * possible to, falsely, observe p->on_cpu == 0.
  4244			 *
  4245			 * One must be running (->on_cpu == 1) in order to remove oneself
  4246			 * from the runqueue.
  4247			 *
  4248			 * __schedule() (switch to task 'p')	try_to_wake_up()
  4249			 *   STORE p->on_cpu = 1		  LOAD p->on_rq
  4250			 *   UNLOCK rq->lock
  4251			 *
  4252			 * __schedule() (put 'p' to sleep)
  4253			 *   LOCK rq->lock			  smp_rmb();
  4254			 *   smp_mb__after_spinlock();
  4255			 *   STORE p->on_rq = 0			  LOAD p->on_cpu
  4256			 *
  4257			 * Pairs with the LOCK+smp_mb__after_spinlock() on rq->lock in
  4258			 * __schedule().  See the comment for smp_mb__after_spinlock().
  4259			 *
  4260			 * Form a control-dep-acquire with p->on_rq == 0 above, to ensure
  4261			 * schedule()'s deactivate_task() has 'happened' and p will no longer
  4262			 * care about it's own p->state. See the comment in __schedule().
  4263			 */
  4264			smp_acquire__after_ctrl_dep();
  4265	
  4266			/*
  4267			 * We're doing the wakeup (@success == 1), they did a dequeue (p->on_rq
  4268			 * == 0), which means we need to do an enqueue, change p->state to
  4269			 * TASK_WAKING such that we can unlock p->pi_lock before doing the
  4270			 * enqueue, such as ttwu_queue_wakelist().
  4271			 */
  4272			WRITE_ONCE(p->__state, TASK_WAKING);
  4273	
  4274			/*
  4275			 * If the owning (remote) CPU is still in the middle of schedule() with
  4276			 * this task as prev, considering queueing p on the remote CPUs wake_list
  4277			 * which potentially sends an IPI instead of spinning on p->on_cpu to
  4278			 * let the waker make forward progress. This is safe because IRQs are
  4279			 * disabled and the IPI will deliver after on_cpu is cleared.
  4280			 *
  4281			 * Ensure we load task_cpu(p) after p->on_cpu:
  4282			 *
  4283			 * set_task_cpu(p, cpu);
  4284			 *   STORE p->cpu = @cpu
  4285			 * __schedule() (switch to task 'p')
  4286			 *   LOCK rq->lock
  4287			 *   smp_mb__after_spin_lock()		smp_cond_load_acquire(&p->on_cpu)
  4288			 *   STORE p->on_cpu = 1		LOAD p->cpu
  4289			 *
  4290			 * to ensure we observe the correct CPU on which the task is currently
  4291			 * scheduling.
  4292			 */
  4293			if (smp_load_acquire(&p->on_cpu) &&
  4294			    ttwu_queue_wakelist(p, task_cpu(p), wake_flags))
  4295				break;
  4296	
  4297			/*
  4298			 * If the owning (remote) CPU is still in the middle of schedule() with
  4299			 * this task as prev, wait until it's done referencing the task.
  4300			 *
  4301			 * Pairs with the smp_store_release() in finish_task().
  4302			 *
  4303			 * This ensures that tasks getting woken will be fully ordered against
  4304			 * their previous state and preserve Program Order.
  4305			 */
  4306			smp_cond_load_acquire(&p->on_cpu, !VAL);
  4307	
  4308			cpu = select_task_rq(p, p->wake_cpu, &wake_flags);
  4309			if (task_cpu(p) != cpu) {
  4310				if (p->in_iowait) {
  4311					delayacct_blkio_end(p);
  4312					atomic_dec(&task_rq(p)->nr_iowait);
  4313				}
  4314	
  4315				wake_flags |= WF_MIGRATED;
  4316				psi_ttwu_dequeue(p);
  4317				set_task_cpu(p, cpu);
  4318			}
  4319	#else
  4320			cpu = task_cpu(p);
  4321	#endif /* CONFIG_SMP */
  4322	
  4323			ttwu_queue(p, cpu, wake_flags);
  4324		}
  4325	out:
  4326		if (success)
  4327			ttwu_stat(p, task_cpu(p), wake_flags);
  4328	
  4329		return success;
  4330	}
  4331	
  4332	static bool __task_needs_rq_lock(struct task_struct *p)
  4333	{
  4334		unsigned int state = READ_ONCE(p->__state);
  4335	
  4336		/*
  4337		 * Since pi->lock blocks try_to_wake_up(), we don't need rq->lock when
  4338		 * the task is blocked. Make sure to check @state since ttwu() can drop
  4339		 * locks at the end, see ttwu_queue_wakelist().
  4340		 */
  4341		if (state == TASK_RUNNING || state == TASK_WAKING)
  4342			return true;
  4343	
  4344		/*
  4345		 * Ensure we load p->on_rq after p->__state, otherwise it would be
  4346		 * possible to, falsely, observe p->on_rq == 0.
  4347		 *
  4348		 * See try_to_wake_up() for a longer comment.
  4349		 */
  4350		smp_rmb();
  4351		if (p->on_rq)
  4352			return true;
  4353	
  4354	#ifdef CONFIG_SMP
  4355		/*
  4356		 * Ensure the task has finished __schedule() and will not be referenced
  4357		 * anymore. Again, see try_to_wake_up() for a longer comment.
  4358		 */
  4359		smp_rmb();
  4360		smp_cond_load_acquire(&p->on_cpu, !VAL);
  4361	#endif
  4362	
  4363		return false;
  4364	}
  4365	
  4366	/**
  4367	 * task_call_func - Invoke a function on task in fixed state
  4368	 * @p: Process for which the function is to be invoked, can be @current.
  4369	 * @func: Function to invoke.
  4370	 * @arg: Argument to function.
  4371	 *
  4372	 * Fix the task in it's current state by avoiding wakeups and or rq operations
  4373	 * and call @func(@arg) on it.  This function can use task_is_runnable() and
  4374	 * task_curr() to work out what the state is, if required.  Given that @func
  4375	 * can be invoked with a runqueue lock held, it had better be quite
  4376	 * lightweight.
  4377	 *
  4378	 * Returns:
  4379	 *   Whatever @func returns
  4380	 */
  4381	int task_call_func(struct task_struct *p, task_call_f func, void *arg)
  4382	{
  4383		struct rq *rq = NULL;
  4384		struct rq_flags rf;
  4385		int ret;
  4386	
  4387		raw_spin_lock_irqsave(&p->pi_lock, rf.flags);
  4388	
  4389		if (__task_needs_rq_lock(p))
  4390			rq = __task_rq_lock(p, &rf);
  4391	
  4392		/*
  4393		 * At this point the task is pinned; either:
  4394		 *  - blocked and we're holding off wakeups	 (pi->lock)
  4395		 *  - woken, and we're holding off enqueue	 (rq->lock)
  4396		 *  - queued, and we're holding off schedule	 (rq->lock)
  4397		 *  - running, and we're holding off de-schedule (rq->lock)
  4398		 *
  4399		 * The called function (@func) can use: task_curr(), p->on_rq and
  4400		 * p->__state to differentiate between these states.
  4401		 */
  4402		ret = func(p, arg);
  4403	
  4404		if (rq)
  4405			rq_unlock(rq, &rf);
  4406	
  4407		raw_spin_unlock_irqrestore(&p->pi_lock, rf.flags);
  4408		return ret;
  4409	}
  4410	
  4411	/**
  4412	 * cpu_curr_snapshot - Return a snapshot of the currently running task
  4413	 * @cpu: The CPU on which to snapshot the task.
  4414	 *
  4415	 * Returns the task_struct pointer of the task "currently" running on
  4416	 * the specified CPU.
  4417	 *
  4418	 * If the specified CPU was offline, the return value is whatever it
  4419	 * is, perhaps a pointer to the task_struct structure of that CPU's idle
  4420	 * task, but there is no guarantee.  Callers wishing a useful return
  4421	 * value must take some action to ensure that the specified CPU remains
  4422	 * online throughout.
  4423	 *
  4424	 * This function executes full memory barriers before and after fetching
  4425	 * the pointer, which permits the caller to confine this function's fetch
  4426	 * with respect to the caller's accesses to other shared variables.
  4427	 */
  4428	struct task_struct *cpu_curr_snapshot(int cpu)
  4429	{
  4430		struct rq *rq = cpu_rq(cpu);
  4431		struct task_struct *t;
  4432		struct rq_flags rf;
  4433	
  4434		rq_lock_irqsave(rq, &rf);
  4435		smp_mb__after_spinlock(); /* Pairing determined by caller's synchronization design. */
  4436		t = rcu_dereference(cpu_curr(cpu));
  4437		rq_unlock_irqrestore(rq, &rf);
  4438		smp_mb(); /* Pairing determined by caller's synchronization design. */
  4439	
  4440		return t;
  4441	}
  4442	
  4443	/**
  4444	 * wake_up_process - Wake up a specific process
  4445	 * @p: The process to be woken up.
  4446	 *
  4447	 * Attempt to wake up the nominated process and move it to the set of runnable
  4448	 * processes.
  4449	 *
  4450	 * Return: 1 if the process was woken up, 0 if it was already running.
  4451	 *
  4452	 * This function executes a full memory barrier before accessing the task state.
  4453	 */
  4454	int wake_up_process(struct task_struct *p)
  4455	{
  4456		return try_to_wake_up(p, TASK_NORMAL, 0);
  4457	}
  4458	EXPORT_SYMBOL(wake_up_process);
  4459	
  4460	int wake_up_state(struct task_struct *p, unsigned int state)
  4461	{
  4462		return try_to_wake_up(p, state, 0);
  4463	}
  4464	
  4465	/*
  4466	 * Perform scheduler related setup for a newly forked process p.
  4467	 * p is forked by current.
  4468	 *
  4469	 * __sched_fork() is basic setup which is also used by sched_init() to
  4470	 * initialize the boot CPU's idle task.
  4471	 */
  4472	static void __sched_fork(unsigned long clone_flags, struct task_struct *p)
  4473	{
  4474		p->on_rq			= 0;
  4475	
  4476		p->se.on_rq			= 0;
  4477		p->se.exec_start		= 0;
  4478		p->se.sum_exec_runtime		= 0;
  4479		p->se.prev_sum_exec_runtime	= 0;
  4480		p->se.nr_migrations		= 0;
  4481		p->se.vruntime			= 0;
  4482		p->se.vlag			= 0;
  4483		INIT_LIST_HEAD(&p->se.group_node);
  4484	
  4485		/* A delayed task cannot be in clone(). */
  4486		SCHED_WARN_ON(p->se.sched_delayed);
  4487	
  4488	#ifdef CONFIG_FAIR_GROUP_SCHED
  4489		p->se.cfs_rq			= NULL;
  4490	#endif
  4491	
  4492	#ifdef CONFIG_SCHEDSTATS
  4493		/* Even if schedstat is disabled, there should not be garbage */
  4494		memset(&p->stats, 0, sizeof(p->stats));
  4495	#endif
  4496	
  4497		init_dl_entity(&p->dl);
  4498	
  4499		INIT_LIST_HEAD(&p->rt.run_list);
  4500		p->rt.timeout		= 0;
  4501		p->rt.time_slice	= sched_rr_timeslice;
  4502		p->rt.on_rq		= 0;
  4503		p->rt.on_list		= 0;
  4504	
  4505	#ifdef CONFIG_SCHED_CLASS_EXT
  4506		init_scx_entity(&p->scx);
  4507	#endif
  4508	
  4509	#ifdef CONFIG_PREEMPT_NOTIFIERS
  4510		INIT_HLIST_HEAD(&p->preempt_notifiers);
  4511	#endif
  4512	
  4513	#ifdef CONFIG_COMPACTION
  4514		p->capture_control = NULL;
  4515	#endif
  4516		init_numa_balancing(clone_flags, p);
  4517	#ifdef CONFIG_SMP
  4518		p->wake_entry.u_flags = CSD_TYPE_TTWU;
  4519		p->migration_pending = NULL;
  4520	#endif
  4521		init_sched_mm_cid(p);
  4522	}
  4523	
  4524	DEFINE_STATIC_KEY_FALSE(sched_numa_balancing);
  4525	
  4526	#ifdef CONFIG_NUMA_BALANCING
  4527	
  4528	int sysctl_numa_balancing_mode;
  4529	
  4530	static void __set_numabalancing_state(bool enabled)
  4531	{
  4532		if (enabled)
  4533			static_branch_enable(&sched_numa_balancing);
  4534		else
  4535			static_branch_disable(&sched_numa_balancing);
  4536	}
  4537	
  4538	void set_numabalancing_state(bool enabled)
  4539	{
  4540		if (enabled)
  4541			sysctl_numa_balancing_mode = NUMA_BALANCING_NORMAL;
  4542		else
  4543			sysctl_numa_balancing_mode = NUMA_BALANCING_DISABLED;
  4544		__set_numabalancing_state(enabled);
  4545	}
  4546	
  4547	#ifdef CONFIG_PROC_SYSCTL
  4548	static void reset_memory_tiering(void)
  4549	{
  4550		struct pglist_data *pgdat;
  4551	
  4552		for_each_online_pgdat(pgdat) {
  4553			pgdat->nbp_threshold = 0;
  4554			pgdat->nbp_th_nr_cand = node_page_state(pgdat, PGPROMOTE_CANDIDATE);
  4555			pgdat->nbp_th_start = jiffies_to_msecs(jiffies);
  4556		}
  4557	}
  4558	
  4559	static int sysctl_numa_balancing(const struct ctl_table *table, int write,
  4560				  void *buffer, size_t *lenp, loff_t *ppos)
  4561	{
  4562		struct ctl_table t;
  4563		int err;
  4564		int state = sysctl_numa_balancing_mode;
  4565	
  4566		if (write && !capable(CAP_SYS_ADMIN))
  4567			return -EPERM;
  4568	
  4569		t = *table;
  4570		t.data = &state;
  4571		err = proc_dointvec_minmax(&t, write, buffer, lenp, ppos);
  4572		if (err < 0)
  4573			return err;
  4574		if (write) {
  4575			if (!(sysctl_numa_balancing_mode & NUMA_BALANCING_MEMORY_TIERING) &&
  4576			    (state & NUMA_BALANCING_MEMORY_TIERING))
  4577				reset_memory_tiering();
  4578			sysctl_numa_balancing_mode = state;
  4579			__set_numabalancing_state(state);
  4580		}
  4581		return err;
  4582	}
  4583	#endif
  4584	#endif
  4585	
  4586	#ifdef CONFIG_SCHEDSTATS
  4587	
  4588	DEFINE_STATIC_KEY_FALSE(sched_schedstats);
  4589	
  4590	static void set_schedstats(bool enabled)
  4591	{
  4592		if (enabled)
  4593			static_branch_enable(&sched_schedstats);
  4594		else
  4595			static_branch_disable(&sched_schedstats);
  4596	}
  4597	
  4598	void force_schedstat_enabled(void)
  4599	{
  4600		if (!schedstat_enabled()) {
  4601			pr_info("kernel profiling enabled schedstats, disable via kernel.sched_schedstats.\n");
  4602			static_branch_enable(&sched_schedstats);
  4603		}
  4604	}
  4605	
  4606	static int __init setup_schedstats(char *str)
  4607	{
  4608		int ret = 0;
  4609		if (!str)
  4610			goto out;
  4611	
  4612		if (!strcmp(str, "enable")) {
  4613			set_schedstats(true);
  4614			ret = 1;
  4615		} else if (!strcmp(str, "disable")) {
  4616			set_schedstats(false);
  4617			ret = 1;
  4618		}
  4619	out:
  4620		if (!ret)
  4621			pr_warn("Unable to parse schedstats=\n");
  4622	
  4623		return ret;
  4624	}
  4625	__setup("schedstats=", setup_schedstats);
  4626	
  4627	#ifdef CONFIG_PROC_SYSCTL
  4628	static int sysctl_schedstats(const struct ctl_table *table, int write, void *buffer,
  4629			size_t *lenp, loff_t *ppos)
  4630	{
  4631		struct ctl_table t;
  4632		int err;
  4633		int state = static_branch_likely(&sched_schedstats);
  4634	
  4635		if (write && !capable(CAP_SYS_ADMIN))
  4636			return -EPERM;
  4637	
  4638		t = *table;
  4639		t.data = &state;
  4640		err = proc_dointvec_minmax(&t, write, buffer, lenp, ppos);
  4641		if (err < 0)
  4642			return err;
  4643		if (write)
  4644			set_schedstats(state);
  4645		return err;
  4646	}
  4647	#endif /* CONFIG_PROC_SYSCTL */
  4648	#endif /* CONFIG_SCHEDSTATS */
  4649	
  4650	#ifdef CONFIG_SYSCTL
  4651	static struct ctl_table sched_core_sysctls[] = {
  4652	#ifdef CONFIG_SCHEDSTATS
  4653		{
  4654			.procname       = "sched_schedstats",
  4655			.data           = NULL,
  4656			.maxlen         = sizeof(unsigned int),
  4657			.mode           = 0644,
  4658			.proc_handler   = sysctl_schedstats,
  4659			.extra1         = SYSCTL_ZERO,
  4660			.extra2         = SYSCTL_ONE,
  4661		},
  4662	#endif /* CONFIG_SCHEDSTATS */
  4663	#ifdef CONFIG_UCLAMP_TASK
  4664		{
  4665			.procname       = "sched_util_clamp_min",
  4666			.data           = &sysctl_sched_uclamp_util_min,
  4667			.maxlen         = sizeof(unsigned int),
  4668			.mode           = 0644,
  4669			.proc_handler   = sysctl_sched_uclamp_handler,
  4670		},
  4671		{
  4672			.procname       = "sched_util_clamp_max",
  4673			.data           = &sysctl_sched_uclamp_util_max,
  4674			.maxlen         = sizeof(unsigned int),
  4675			.mode           = 0644,
  4676			.proc_handler   = sysctl_sched_uclamp_handler,
  4677		},
  4678		{
  4679			.procname       = "sched_util_clamp_min_rt_default",
  4680			.data           = &sysctl_sched_uclamp_util_min_rt_default,
  4681			.maxlen         = sizeof(unsigned int),
  4682			.mode           = 0644,
  4683			.proc_handler   = sysctl_sched_uclamp_handler,
  4684		},
  4685	#endif /* CONFIG_UCLAMP_TASK */
  4686	#ifdef CONFIG_NUMA_BALANCING
  4687		{
  4688			.procname	= "numa_balancing",
  4689			.data		= NULL, /* filled in by handler */
  4690			.maxlen		= sizeof(unsigned int),
  4691			.mode		= 0644,
  4692			.proc_handler	= sysctl_numa_balancing,
  4693			.extra1		= SYSCTL_ZERO,
  4694			.extra2		= SYSCTL_FOUR,
  4695		},
  4696	#endif /* CONFIG_NUMA_BALANCING */
  4697	};
  4698	static int __init sched_core_sysctl_init(void)
  4699	{
  4700		register_sysctl_init("kernel", sched_core_sysctls);
  4701		return 0;
  4702	}
  4703	late_initcall(sched_core_sysctl_init);
  4704	#endif /* CONFIG_SYSCTL */
  4705	
  4706	/*
  4707	 * fork()/clone()-time setup:
  4708	 */
  4709	int sched_fork(unsigned long clone_flags, struct task_struct *p)
  4710	{
  4711		__sched_fork(clone_flags, p);
  4712		/*
  4713		 * We mark the process as NEW here. This guarantees that
  4714		 * nobody will actually run it, and a signal or other external
  4715		 * event cannot wake it up and insert it on the runqueue either.
  4716		 */
  4717		p->__state = TASK_NEW;
  4718	
  4719		/*
  4720		 * Make sure we do not leak PI boosting priority to the child.
  4721		 */
  4722		p->prio = current->normal_prio;
  4723	
  4724		uclamp_fork(p);
  4725	
  4726		/*
  4727		 * Revert to default priority/policy on fork if requested.
  4728		 */
  4729		if (unlikely(p->sched_reset_on_fork)) {
  4730			if (task_has_dl_policy(p) || task_has_rt_policy(p)) {
  4731				p->policy = SCHED_NORMAL;
  4732				p->static_prio = NICE_TO_PRIO(0);
  4733				p->rt_priority = 0;
  4734			} else if (PRIO_TO_NICE(p->static_prio) < 0)
  4735				p->static_prio = NICE_TO_PRIO(0);
  4736	
  4737			p->prio = p->normal_prio = p->static_prio;
  4738			set_load_weight(p, false);
  4739			p->se.custom_slice = 0;
  4740			p->se.slice = sysctl_sched_base_slice;
  4741	
  4742			/*
  4743			 * We don't need the reset flag anymore after the fork. It has
  4744			 * fulfilled its duty:
  4745			 */
  4746			p->sched_reset_on_fork = 0;
  4747		}
  4748	
  4749		if (dl_prio(p->prio))
  4750			return -EAGAIN;
  4751	
  4752		scx_pre_fork(p);
  4753	
  4754		if (rt_prio(p->prio)) {
  4755			p->sched_class = &rt_sched_class;
  4756	#ifdef CONFIG_SCHED_CLASS_EXT
  4757		} else if (task_should_scx(p->policy)) {
  4758			p->sched_class = &ext_sched_class;
  4759	#endif
  4760		} else {
  4761			p->sched_class = &fair_sched_class;
  4762		}
  4763	
  4764		init_entity_runnable_average(&p->se);
  4765	
  4766	
  4767	#ifdef CONFIG_SCHED_INFO
  4768		if (likely(sched_info_on()))
  4769			memset(&p->sched_info, 0, sizeof(p->sched_info));
  4770	#endif
  4771	#if defined(CONFIG_SMP)
  4772		p->on_cpu = 0;
  4773	#endif
  4774		init_task_preempt_count(p);
  4775	#ifdef CONFIG_SMP
  4776		plist_node_init(&p->pushable_tasks, MAX_PRIO);
  4777		RB_CLEAR_NODE(&p->pushable_dl_tasks);
  4778	#endif
  4779		return 0;
  4780	}
  4781	
  4782	int sched_cgroup_fork(struct task_struct *p, struct kernel_clone_args *kargs)
  4783	{
  4784		unsigned long flags;
  4785	
  4786		/*
  4787		 * Because we're not yet on the pid-hash, p->pi_lock isn't strictly
  4788		 * required yet, but lockdep gets upset if rules are violated.
  4789		 */
  4790		raw_spin_lock_irqsave(&p->pi_lock, flags);
  4791	#ifdef CONFIG_CGROUP_SCHED
  4792		if (1) {
  4793			struct task_group *tg;
  4794			tg = container_of(kargs->cset->subsys[cpu_cgrp_id],
  4795					  struct task_group, css);
  4796			tg = autogroup_task_group(p, tg);
  4797			p->sched_task_group = tg;
  4798		}
  4799	#endif
  4800		rseq_migrate(p);
  4801		/*
  4802		 * We're setting the CPU for the first time, we don't migrate,
  4803		 * so use __set_task_cpu().
  4804		 */
  4805		__set_task_cpu(p, smp_processor_id());
  4806		if (p->sched_class->task_fork)
  4807			p->sched_class->task_fork(p);
  4808		raw_spin_unlock_irqrestore(&p->pi_lock, flags);
  4809	
  4810		return scx_fork(p);
  4811	}
  4812	
  4813	void sched_cancel_fork(struct task_struct *p)
  4814	{
  4815		scx_cancel_fork(p);
  4816	}
  4817	
  4818	void sched_post_fork(struct task_struct *p)
  4819	{
  4820		uclamp_post_fork(p);
  4821		scx_post_fork(p);
  4822	}
  4823	
  4824	unsigned long to_ratio(u64 period, u64 runtime)
  4825	{
  4826		if (runtime == RUNTIME_INF)
  4827			return BW_UNIT;
  4828	
  4829		/*
  4830		 * Doing this here saves a lot of checks in all
  4831		 * the calling paths, and returning zero seems
  4832		 * safe for them anyway.
  4833		 */
  4834		if (period == 0)
  4835			return 0;
  4836	
  4837		return div64_u64(runtime << BW_SHIFT, period);
  4838	}
  4839	
  4840	/*
  4841	 * wake_up_new_task - wake up a newly created task for the first time.
  4842	 *
  4843	 * This function will do some initial scheduler statistics housekeeping
  4844	 * that must be done for every newly created context, then puts the task
  4845	 * on the runqueue and wakes it.
  4846	 */
  4847	void wake_up_new_task(struct task_struct *p)
  4848	{
  4849		struct rq_flags rf;
  4850		struct rq *rq;
  4851		int wake_flags = WF_FORK;
  4852	
  4853		raw_spin_lock_irqsave(&p->pi_lock, rf.flags);
  4854		WRITE_ONCE(p->__state, TASK_RUNNING);
  4855	#ifdef CONFIG_SMP
  4856		/*
  4857		 * Fork balancing, do it here and not earlier because:
  4858		 *  - cpus_ptr can change in the fork path
  4859		 *  - any previously selected CPU might disappear through hotplug
  4860		 *
  4861		 * Use __set_task_cpu() to avoid calling sched_class::migrate_task_rq,
  4862		 * as we're not fully set-up yet.
  4863		 */
  4864		p->recent_used_cpu = task_cpu(p);
  4865		rseq_migrate(p);
  4866		__set_task_cpu(p, select_task_rq(p, task_cpu(p), &wake_flags));
  4867	#endif
  4868		rq = __task_rq_lock(p, &rf);
  4869		update_rq_clock(rq);
  4870		post_init_entity_util_avg(p);
  4871	
  4872		activate_task(rq, p, ENQUEUE_NOCLOCK | ENQUEUE_INITIAL);
  4873		trace_sched_wakeup_new(p);
  4874		wakeup_preempt(rq, p, wake_flags);
  4875	#ifdef CONFIG_SMP
  4876		if (p->sched_class->task_woken) {
  4877			/*
  4878			 * Nothing relies on rq->lock after this, so it's fine to
  4879			 * drop it.
  4880			 */
  4881			rq_unpin_lock(rq, &rf);
  4882			p->sched_class->task_woken(rq, p);
  4883			rq_repin_lock(rq, &rf);
  4884		}
  4885	#endif
  4886		task_rq_unlock(rq, p, &rf);
  4887	}
  4888	
  4889	#ifdef CONFIG_PREEMPT_NOTIFIERS
  4890	
  4891	static DEFINE_STATIC_KEY_FALSE(preempt_notifier_key);
  4892	
  4893	void preempt_notifier_inc(void)
  4894	{
  4895		static_branch_inc(&preempt_notifier_key);
  4896	}
  4897	EXPORT_SYMBOL_GPL(preempt_notifier_inc);
  4898	
  4899	void preempt_notifier_dec(void)
  4900	{
  4901		static_branch_dec(&preempt_notifier_key);
  4902	}
  4903	EXPORT_SYMBOL_GPL(preempt_notifier_dec);
  4904	
  4905	/**
  4906	 * preempt_notifier_register - tell me when current is being preempted & rescheduled
  4907	 * @notifier: notifier struct to register
  4908	 */
  4909	void preempt_notifier_register(struct preempt_notifier *notifier)
  4910	{
  4911		if (!static_branch_unlikely(&preempt_notifier_key))
  4912			WARN(1, "registering preempt_notifier while notifiers disabled\n");
  4913	
  4914		hlist_add_head(&notifier->link, &current->preempt_notifiers);
  4915	}
  4916	EXPORT_SYMBOL_GPL(preempt_notifier_register);
  4917	
  4918	/**
  4919	 * preempt_notifier_unregister - no longer interested in preemption notifications
  4920	 * @notifier: notifier struct to unregister
  4921	 *
  4922	 * This is *not* safe to call from within a preemption notifier.
  4923	 */
  4924	void preempt_notifier_unregister(struct preempt_notifier *notifier)
  4925	{
  4926		hlist_del(&notifier->link);
  4927	}
  4928	EXPORT_SYMBOL_GPL(preempt_notifier_unregister);
  4929	
  4930	static void __fire_sched_in_preempt_notifiers(struct task_struct *curr)
  4931	{
  4932		struct preempt_notifier *notifier;
  4933	
  4934		hlist_for_each_entry(notifier, &curr->preempt_notifiers, link)
  4935			notifier->ops->sched_in(notifier, raw_smp_processor_id());
  4936	}
  4937	
  4938	static __always_inline void fire_sched_in_preempt_notifiers(struct task_struct *curr)
  4939	{
  4940		if (static_branch_unlikely(&preempt_notifier_key))
  4941			__fire_sched_in_preempt_notifiers(curr);
  4942	}
  4943	
  4944	static void
  4945	__fire_sched_out_preempt_notifiers(struct task_struct *curr,
  4946					   struct task_struct *next)
  4947	{
  4948		struct preempt_notifier *notifier;
  4949	
  4950		hlist_for_each_entry(notifier, &curr->preempt_notifiers, link)
  4951			notifier->ops->sched_out(notifier, next);
  4952	}
  4953	
  4954	static __always_inline void
  4955	fire_sched_out_preempt_notifiers(struct task_struct *curr,
  4956					 struct task_struct *next)
  4957	{
  4958		if (static_branch_unlikely(&preempt_notifier_key))
  4959			__fire_sched_out_preempt_notifiers(curr, next);
  4960	}
  4961	
  4962	#else /* !CONFIG_PREEMPT_NOTIFIERS */
  4963	
  4964	static inline void fire_sched_in_preempt_notifiers(struct task_struct *curr)
  4965	{
  4966	}
  4967	
  4968	static inline void
  4969	fire_sched_out_preempt_notifiers(struct task_struct *curr,
  4970					 struct task_struct *next)
  4971	{
  4972	}
  4973	
  4974	#endif /* CONFIG_PREEMPT_NOTIFIERS */
  4975	
  4976	static inline void prepare_task(struct task_struct *next)
  4977	{
  4978	#ifdef CONFIG_SMP
  4979		/*
  4980		 * Claim the task as running, we do this before switching to it
  4981		 * such that any running task will have this set.
  4982		 *
  4983		 * See the smp_load_acquire(&p->on_cpu) case in ttwu() and
  4984		 * its ordering comment.
  4985		 */
  4986		WRITE_ONCE(next->on_cpu, 1);
  4987	#endif
  4988	}
  4989	
  4990	static inline void finish_task(struct task_struct *prev)
  4991	{
  4992	#ifdef CONFIG_SMP
  4993		/*
  4994		 * This must be the very last reference to @prev from this CPU. After
  4995		 * p->on_cpu is cleared, the task can be moved to a different CPU. We
  4996		 * must ensure this doesn't happen until the switch is completely
  4997		 * finished.
  4998		 *
  4999		 * In particular, the load of prev->state in finish_task_switch() must
  5000		 * happen before this.
  5001		 *
  5002		 * Pairs with the smp_cond_load_acquire() in try_to_wake_up().
  5003		 */
  5004		smp_store_release(&prev->on_cpu, 0);
  5005	#endif
  5006	}
  5007	
  5008	#ifdef CONFIG_SMP
  5009	
  5010	static void do_balance_callbacks(struct rq *rq, struct balance_callback *head)
  5011	{
  5012		void (*func)(struct rq *rq);
  5013		struct balance_callback *next;
  5014	
  5015		lockdep_assert_rq_held(rq);
  5016	
  5017		while (head) {
  5018			func = (void (*)(struct rq *))head->func;
  5019			next = head->next;
  5020			head->next = NULL;
  5021			head = next;
  5022	
  5023			func(rq);
  5024		}
  5025	}
  5026	
  5027	static void balance_push(struct rq *rq);
  5028	
  5029	/*
  5030	 * balance_push_callback is a right abuse of the callback interface and plays
  5031	 * by significantly different rules.
  5032	 *
  5033	 * Where the normal balance_callback's purpose is to be ran in the same context
  5034	 * that queued it (only later, when it's safe to drop rq->lock again),
  5035	 * balance_push_callback is specifically targeted at __schedule().
  5036	 *
  5037	 * This abuse is tolerated because it places all the unlikely/odd cases behind
  5038	 * a single test, namely: rq->balance_callback == NULL.
  5039	 */
  5040	struct balance_callback balance_push_callback = {
  5041		.next = NULL,
  5042		.func = balance_push,
  5043	};
  5044	
  5045	static inline struct balance_callback *
  5046	__splice_balance_callbacks(struct rq *rq, bool split)
  5047	{
  5048		struct balance_callback *head = rq->balance_callback;
  5049	
  5050		if (likely(!head))
  5051			return NULL;
  5052	
  5053		lockdep_assert_rq_held(rq);
  5054		/*
  5055		 * Must not take balance_push_callback off the list when
  5056		 * splice_balance_callbacks() and balance_callbacks() are not
  5057		 * in the same rq->lock section.
  5058		 *
  5059		 * In that case it would be possible for __schedule() to interleave
  5060		 * and observe the list empty.
  5061		 */
  5062		if (split && head == &balance_push_callback)
  5063			head = NULL;
  5064		else
  5065			rq->balance_callback = NULL;
  5066	
  5067		return head;
  5068	}
  5069	
  5070	struct balance_callback *splice_balance_callbacks(struct rq *rq)
  5071	{
  5072		return __splice_balance_callbacks(rq, true);
  5073	}
  5074	
  5075	static void __balance_callbacks(struct rq *rq)
  5076	{
  5077		do_balance_callbacks(rq, __splice_balance_callbacks(rq, false));
  5078	}
  5079	
  5080	void balance_callbacks(struct rq *rq, struct balance_callback *head)
  5081	{
  5082		unsigned long flags;
  5083	
  5084		if (unlikely(head)) {
  5085			raw_spin_rq_lock_irqsave(rq, flags);
  5086			do_balance_callbacks(rq, head);
  5087			raw_spin_rq_unlock_irqrestore(rq, flags);
  5088		}
  5089	}
  5090	
  5091	#else
  5092	
  5093	static inline void __balance_callbacks(struct rq *rq)
  5094	{
  5095	}
  5096	
  5097	#endif
  5098	
  5099	static inline void
  5100	prepare_lock_switch(struct rq *rq, struct task_struct *next, struct rq_flags *rf)
  5101	{
  5102		/*
  5103		 * Since the runqueue lock will be released by the next
  5104		 * task (which is an invalid locking op but in the case
  5105		 * of the scheduler it's an obvious special-case), so we
  5106		 * do an early lockdep release here:
  5107		 */
  5108		rq_unpin_lock(rq, rf);
  5109		spin_release(&__rq_lockp(rq)->dep_map, _THIS_IP_);
  5110	#ifdef CONFIG_DEBUG_SPINLOCK
  5111		/* this is a valid case when another task releases the spinlock */
  5112		rq_lockp(rq)->owner = next;
  5113	#endif
  5114	}
  5115	
  5116	static inline void finish_lock_switch(struct rq *rq)
  5117	{
  5118		/*
  5119		 * If we are tracking spinlock dependencies then we have to
  5120		 * fix up the runqueue lock - which gets 'carried over' from
  5121		 * prev into current:
  5122		 */
  5123		spin_acquire(&__rq_lockp(rq)->dep_map, 0, 0, _THIS_IP_);
  5124		__balance_callbacks(rq);
  5125		raw_spin_rq_unlock_irq(rq);
  5126	}
  5127	
  5128	/*
  5129	 * NOP if the arch has not defined these:
  5130	 */
  5131	
  5132	#ifndef prepare_arch_switch
  5133	# define prepare_arch_switch(next)	do { } while (0)
  5134	#endif
  5135	
  5136	#ifndef finish_arch_post_lock_switch
  5137	# define finish_arch_post_lock_switch()	do { } while (0)
  5138	#endif
  5139	
  5140	static inline void kmap_local_sched_out(void)
  5141	{
  5142	#ifdef CONFIG_KMAP_LOCAL
  5143		if (unlikely(current->kmap_ctrl.idx))
  5144			__kmap_local_sched_out();
  5145	#endif
  5146	}
  5147	
  5148	static inline void kmap_local_sched_in(void)
  5149	{
  5150	#ifdef CONFIG_KMAP_LOCAL
  5151		if (unlikely(current->kmap_ctrl.idx))
  5152			__kmap_local_sched_in();
  5153	#endif
  5154	}
  5155	
  5156	/**
  5157	 * prepare_task_switch - prepare to switch tasks
  5158	 * @rq: the runqueue preparing to switch
  5159	 * @prev: the current task that is being switched out
  5160	 * @next: the task we are going to switch to.
  5161	 *
  5162	 * This is called with the rq lock held and interrupts off. It must
  5163	 * be paired with a subsequent finish_task_switch after the context
  5164	 * switch.
  5165	 *
  5166	 * prepare_task_switch sets up locking and calls architecture specific
  5167	 * hooks.
  5168	 */
  5169	static inline void
  5170	prepare_task_switch(struct rq *rq, struct task_struct *prev,
  5171			    struct task_struct *next)
  5172	{
  5173		kcov_prepare_switch(prev);
  5174		sched_info_switch(rq, prev, next);
  5175		perf_event_task_sched_out(prev, next);
  5176		rseq_preempt(prev);
  5177		fire_sched_out_preempt_notifiers(prev, next);
  5178		kmap_local_sched_out();
  5179		prepare_task(next);
  5180		prepare_arch_switch(next);
  5181	}
  5182	
  5183	/**
  5184	 * finish_task_switch - clean up after a task-switch
  5185	 * @prev: the thread we just switched away from.
  5186	 *
  5187	 * finish_task_switch must be called after the context switch, paired
  5188	 * with a prepare_task_switch call before the context switch.
  5189	 * finish_task_switch will reconcile locking set up by prepare_task_switch,
  5190	 * and do any other architecture-specific cleanup actions.
  5191	 *
  5192	 * Note that we may have delayed dropping an mm in context_switch(). If
  5193	 * so, we finish that here outside of the runqueue lock. (Doing it
  5194	 * with the lock held can cause deadlocks; see schedule() for
  5195	 * details.)
  5196	 *
  5197	 * The context switch have flipped the stack from under us and restored the
  5198	 * local variables which were saved when this task called schedule() in the
  5199	 * past. 'prev == current' is still correct but we need to recalculate this_rq
  5200	 * because prev may have moved to another CPU.
  5201	 */
  5202	static struct rq *finish_task_switch(struct task_struct *prev)
  5203		__releases(rq->lock)
  5204	{
  5205		struct rq *rq = this_rq();
  5206		struct mm_struct *mm = rq->prev_mm;
  5207		unsigned int prev_state;
  5208	
  5209		/*
  5210		 * The previous task will have left us with a preempt_count of 2
  5211		 * because it left us after:
  5212		 *
  5213		 *	schedule()
  5214		 *	  preempt_disable();			// 1
  5215		 *	  __schedule()
  5216		 *	    raw_spin_lock_irq(&rq->lock)	// 2
  5217		 *
  5218		 * Also, see FORK_PREEMPT_COUNT.
  5219		 */
  5220		if (WARN_ONCE(preempt_count() != 2*PREEMPT_DISABLE_OFFSET,
  5221			      "corrupted preempt_count: %s/%d/0x%x\n",
  5222			      current->comm, current->pid, preempt_count()))
  5223			preempt_count_set(FORK_PREEMPT_COUNT);
  5224	
  5225		rq->prev_mm = NULL;
  5226	
  5227		/*
  5228		 * A task struct has one reference for the use as "current".
  5229		 * If a task dies, then it sets TASK_DEAD in tsk->state and calls
  5230		 * schedule one last time. The schedule call will never return, and
  5231		 * the scheduled task must drop that reference.
  5232		 *
  5233		 * We must observe prev->state before clearing prev->on_cpu (in
  5234		 * finish_task), otherwise a concurrent wakeup can get prev
  5235		 * running on another CPU and we could rave with its RUNNING -> DEAD
  5236		 * transition, resulting in a double drop.
  5237		 */
  5238		prev_state = READ_ONCE(prev->__state);
  5239		vtime_task_switch(prev);
  5240		perf_event_task_sched_in(prev, current);
  5241		finish_task(prev);
  5242		tick_nohz_task_switch();
  5243		finish_lock_switch(rq);
  5244		finish_arch_post_lock_switch();
  5245		kcov_finish_switch(current);
  5246		/*
  5247		 * kmap_local_sched_out() is invoked with rq::lock held and
  5248		 * interrupts disabled. There is no requirement for that, but the
  5249		 * sched out code does not have an interrupt enabled section.
  5250		 * Restoring the maps on sched in does not require interrupts being
  5251		 * disabled either.
  5252		 */
  5253		kmap_local_sched_in();
  5254	
  5255		fire_sched_in_preempt_notifiers(current);
  5256		/*
  5257		 * When switching through a kernel thread, the loop in
  5258		 * membarrier_{private,global}_expedited() may have observed that
  5259		 * kernel thread and not issued an IPI. It is therefore possible to
  5260		 * schedule between user->kernel->user threads without passing though
  5261		 * switch_mm(). Membarrier requires a barrier after storing to
  5262		 * rq->curr, before returning to userspace, so provide them here:
  5263		 *
  5264		 * - a full memory barrier for {PRIVATE,GLOBAL}_EXPEDITED, implicitly
  5265		 *   provided by mmdrop_lazy_tlb(),
  5266		 * - a sync_core for SYNC_CORE.
  5267		 */
  5268		if (mm) {
  5269			membarrier_mm_sync_core_before_usermode(mm);
  5270			mmdrop_lazy_tlb_sched(mm);
  5271		}
  5272	
  5273		if (unlikely(prev_state == TASK_DEAD)) {
  5274			if (prev->sched_class->task_dead)
  5275				prev->sched_class->task_dead(prev);
  5276	
  5277			/* Task is done with its stack. */
  5278			put_task_stack(prev);
  5279	
  5280			put_task_struct_rcu_user(prev);
  5281		}
  5282	
  5283		return rq;
  5284	}
  5285	
  5286	/**
  5287	 * schedule_tail - first thing a freshly forked thread must call.
  5288	 * @prev: the thread we just switched away from.
  5289	 */
  5290	asmlinkage __visible void schedule_tail(struct task_struct *prev)
  5291		__releases(rq->lock)
  5292	{
  5293		/*
  5294		 * New tasks start with FORK_PREEMPT_COUNT, see there and
  5295		 * finish_task_switch() for details.
  5296		 *
  5297		 * finish_task_switch() will drop rq->lock() and lower preempt_count
  5298		 * and the preempt_enable() will end up enabling preemption (on
  5299		 * PREEMPT_COUNT kernels).
  5300		 */
  5301	
  5302		finish_task_switch(prev);
  5303		preempt_enable();
  5304	
  5305		if (current->set_child_tid)
  5306			put_user(task_pid_vnr(current), current->set_child_tid);
  5307	
  5308		calculate_sigpending();
  5309	}
  5310	
  5311	/*
  5312	 * context_switch - switch to the new MM and the new thread's register state.
  5313	 */
  5314	static __always_inline struct rq *
  5315	context_switch(struct rq *rq, struct task_struct *prev,
  5316		       struct task_struct *next, struct rq_flags *rf)
  5317	{
  5318		prepare_task_switch(rq, prev, next);
  5319	
  5320		/*
  5321		 * For paravirt, this is coupled with an exit in switch_to to
  5322		 * combine the page table reload and the switch backend into
  5323		 * one hypercall.
  5324		 */
  5325		arch_start_context_switch(prev);
  5326	
  5327		/*
  5328		 * kernel -> kernel   lazy + transfer active
  5329		 *   user -> kernel   lazy + mmgrab_lazy_tlb() active
  5330		 *
  5331		 * kernel ->   user   switch + mmdrop_lazy_tlb() active
  5332		 *   user ->   user   switch
  5333		 *
  5334		 * switch_mm_cid() needs to be updated if the barriers provided
  5335		 * by context_switch() are modified.
  5336		 */
  5337		if (!next->mm) {                                // to kernel
  5338			enter_lazy_tlb(prev->active_mm, next);
  5339	
  5340			next->active_mm = prev->active_mm;
  5341			if (prev->mm)                           // from user
  5342				mmgrab_lazy_tlb(prev->active_mm);
  5343			else
  5344				prev->active_mm = NULL;
  5345		} else {                                        // to user
  5346			membarrier_switch_mm(rq, prev->active_mm, next->mm);
  5347			/*
  5348			 * sys_membarrier() requires an smp_mb() between setting
  5349			 * rq->curr / membarrier_switch_mm() and returning to userspace.
  5350			 *
  5351			 * The below provides this either through switch_mm(), or in
  5352			 * case 'prev->active_mm == next->mm' through
  5353			 * finish_task_switch()'s mmdrop().
  5354			 */
  5355			switch_mm_irqs_off(prev->active_mm, next->mm, next);
  5356			lru_gen_use_mm(next->mm);
  5357	
  5358			if (!prev->mm) {                        // from kernel
  5359				/* will mmdrop_lazy_tlb() in finish_task_switch(). */
  5360				rq->prev_mm = prev->active_mm;
  5361				prev->active_mm = NULL;
  5362			}
  5363		}
  5364	
  5365		/* switch_mm_cid() requires the memory barriers above. */
  5366		switch_mm_cid(rq, prev, next);
  5367	
  5368		prepare_lock_switch(rq, next, rf);
  5369	
  5370		/* Here we just switch the register state and the stack. */
  5371		switch_to(prev, next, prev);
  5372		barrier();
  5373	
  5374		return finish_task_switch(prev);
  5375	}
  5376	
  5377	/*
  5378	 * nr_running and nr_context_switches:
  5379	 *
  5380	 * externally visible scheduler statistics: current number of runnable
  5381	 * threads, total number of context switches performed since bootup.
  5382	 */
  5383	unsigned int nr_running(void)
  5384	{
  5385		unsigned int i, sum = 0;
  5386	
  5387		for_each_online_cpu(i)
  5388			sum += cpu_rq(i)->nr_running;
  5389	
  5390		return sum;
  5391	}
  5392	
  5393	/*
  5394	 * Check if only the current task is running on the CPU.
  5395	 *
  5396	 * Caution: this function does not check that the caller has disabled
  5397	 * preemption, thus the result might have a time-of-check-to-time-of-use
  5398	 * race.  The caller is responsible to use it correctly, for example:
  5399	 *
  5400	 * - from a non-preemptible section (of course)
  5401	 *
  5402	 * - from a thread that is bound to a single CPU
  5403	 *
  5404	 * - in a loop with very short iterations (e.g. a polling loop)
  5405	 */
  5406	bool single_task_running(void)
  5407	{
  5408		return raw_rq()->nr_running == 1;
  5409	}
  5410	EXPORT_SYMBOL(single_task_running);
  5411	
  5412	unsigned long long nr_context_switches_cpu(int cpu)
  5413	{
  5414		return cpu_rq(cpu)->nr_switches;
  5415	}
  5416	
  5417	unsigned long long nr_context_switches(void)
  5418	{
  5419		int i;
  5420		unsigned long long sum = 0;
  5421	
  5422		for_each_possible_cpu(i)
  5423			sum += cpu_rq(i)->nr_switches;
  5424	
  5425		return sum;
  5426	}
  5427	
  5428	/*
  5429	 * Consumers of these two interfaces, like for example the cpuidle menu
  5430	 * governor, are using nonsensical data. Preferring shallow idle state selection
  5431	 * for a CPU that has IO-wait which might not even end up running the task when
  5432	 * it does become runnable.
  5433	 */
  5434	
  5435	unsigned int nr_iowait_cpu(int cpu)
  5436	{
  5437		return atomic_read(&cpu_rq(cpu)->nr_iowait);
  5438	}
  5439	
  5440	/*
  5441	 * IO-wait accounting, and how it's mostly bollocks (on SMP).
  5442	 *
  5443	 * The idea behind IO-wait account is to account the idle time that we could
  5444	 * have spend running if it were not for IO. That is, if we were to improve the
  5445	 * storage performance, we'd have a proportional reduction in IO-wait time.
  5446	 *
  5447	 * This all works nicely on UP, where, when a task blocks on IO, we account
  5448	 * idle time as IO-wait, because if the storage were faster, it could've been
  5449	 * running and we'd not be idle.
  5450	 *
  5451	 * This has been extended to SMP, by doing the same for each CPU. This however
  5452	 * is broken.
  5453	 *
  5454	 * Imagine for instance the case where two tasks block on one CPU, only the one
  5455	 * CPU will have IO-wait accounted, while the other has regular idle. Even
  5456	 * though, if the storage were faster, both could've ran at the same time,
  5457	 * utilising both CPUs.
  5458	 *
  5459	 * This means, that when looking globally, the current IO-wait accounting on
  5460	 * SMP is a lower bound, by reason of under accounting.
  5461	 *
  5462	 * Worse, since the numbers are provided per CPU, they are sometimes
  5463	 * interpreted per CPU, and that is nonsensical. A blocked task isn't strictly
  5464	 * associated with any one particular CPU, it can wake to another CPU than it
  5465	 * blocked on. This means the per CPU IO-wait number is meaningless.
  5466	 *
  5467	 * Task CPU affinities can make all that even more 'interesting'.
  5468	 */
  5469	
  5470	unsigned int nr_iowait(void)
  5471	{
  5472		unsigned int i, sum = 0;
  5473	
  5474		for_each_possible_cpu(i)
  5475			sum += nr_iowait_cpu(i);
  5476	
  5477		return sum;
  5478	}
  5479	
  5480	#ifdef CONFIG_SMP
  5481	
  5482	/*
  5483	 * sched_exec - execve() is a valuable balancing opportunity, because at
  5484	 * this point the task has the smallest effective memory and cache footprint.
  5485	 */
  5486	void sched_exec(void)
  5487	{
  5488		struct task_struct *p = current;
  5489		struct migration_arg arg;
  5490		int dest_cpu;
  5491	
  5492		scoped_guard (raw_spinlock_irqsave, &p->pi_lock) {
  5493			dest_cpu = p->sched_class->select_task_rq(p, task_cpu(p), WF_EXEC);
  5494			if (dest_cpu == smp_processor_id())
  5495				return;
  5496	
  5497			if (unlikely(!cpu_active(dest_cpu)))
  5498				return;
  5499	
  5500			arg = (struct migration_arg){ p, dest_cpu };
  5501		}
  5502		stop_one_cpu(task_cpu(p), migration_cpu_stop, &arg);
  5503	}
  5504	
  5505	#endif
  5506	
  5507	DEFINE_PER_CPU(struct kernel_stat, kstat);
  5508	DEFINE_PER_CPU(struct kernel_cpustat, kernel_cpustat);
  5509	
  5510	EXPORT_PER_CPU_SYMBOL(kstat);
  5511	EXPORT_PER_CPU_SYMBOL(kernel_cpustat);
  5512	
  5513	/*
  5514	 * The function fair_sched_class.update_curr accesses the struct curr
  5515	 * and its field curr->exec_start; when called from task_sched_runtime(),
  5516	 * we observe a high rate of cache misses in practice.
  5517	 * Prefetching this data results in improved performance.
  5518	 */
  5519	static inline void prefetch_curr_exec_start(struct task_struct *p)
  5520	{
  5521	#ifdef CONFIG_FAIR_GROUP_SCHED
  5522		struct sched_entity *curr = p->se.cfs_rq->curr;
  5523	#else
  5524		struct sched_entity *curr = task_rq(p)->cfs.curr;
  5525	#endif
  5526		prefetch(curr);
  5527		prefetch(&curr->exec_start);
  5528	}
  5529	
  5530	/*
  5531	 * Return accounted runtime for the task.
  5532	 * In case the task is currently running, return the runtime plus current's
  5533	 * pending runtime that have not been accounted yet.
  5534	 */
  5535	unsigned long long task_sched_runtime(struct task_struct *p)
  5536	{
  5537		struct rq_flags rf;
  5538		struct rq *rq;
  5539		u64 ns;
  5540	
  5541	#if defined(CONFIG_64BIT) && defined(CONFIG_SMP)
  5542		/*
  5543		 * 64-bit doesn't need locks to atomically read a 64-bit value.
  5544		 * So we have a optimization chance when the task's delta_exec is 0.
  5545		 * Reading ->on_cpu is racy, but this is OK.
  5546		 *
  5547		 * If we race with it leaving CPU, we'll take a lock. So we're correct.
  5548		 * If we race with it entering CPU, unaccounted time is 0. This is
  5549		 * indistinguishable from the read occurring a few cycles earlier.
  5550		 * If we see ->on_cpu without ->on_rq, the task is leaving, and has
  5551		 * been accounted, so we're correct here as well.
  5552		 */
  5553		if (!p->on_cpu || !task_on_rq_queued(p))
  5554			return p->se.sum_exec_runtime;
  5555	#endif
  5556	
  5557		rq = task_rq_lock(p, &rf);
  5558		/*
  5559		 * Must be ->curr _and_ ->on_rq.  If dequeued, we would
  5560		 * project cycles that may never be accounted to this
  5561		 * thread, breaking clock_gettime().
  5562		 */
  5563		if (task_current_donor(rq, p) && task_on_rq_queued(p)) {
  5564			prefetch_curr_exec_start(p);
  5565			update_rq_clock(rq);
  5566			p->sched_class->update_curr(rq);
  5567		}
  5568		ns = p->se.sum_exec_runtime;
  5569		task_rq_unlock(rq, p, &rf);
  5570	
  5571		return ns;
  5572	}
  5573	
  5574	#ifdef CONFIG_SCHED_DEBUG
  5575	static u64 cpu_resched_latency(struct rq *rq)
  5576	{
  5577		int latency_warn_ms = READ_ONCE(sysctl_resched_latency_warn_ms);
  5578		u64 resched_latency, now = rq_clock(rq);
  5579		static bool warned_once;
  5580	
  5581		if (sysctl_resched_latency_warn_once && warned_once)
  5582			return 0;
  5583	
  5584		if (!need_resched() || !latency_warn_ms)
  5585			return 0;
  5586	
  5587		if (system_state == SYSTEM_BOOTING)
  5588			return 0;
  5589	
  5590		if (!rq->last_seen_need_resched_ns) {
  5591			rq->last_seen_need_resched_ns = now;
  5592			rq->ticks_without_resched = 0;
  5593			return 0;
  5594		}
  5595	
  5596		rq->ticks_without_resched++;
  5597		resched_latency = now - rq->last_seen_need_resched_ns;
  5598		if (resched_latency <= latency_warn_ms * NSEC_PER_MSEC)
  5599			return 0;
  5600	
  5601		warned_once = true;
  5602	
  5603		return resched_latency;
  5604	}
  5605	
  5606	static int __init setup_resched_latency_warn_ms(char *str)
  5607	{
  5608		long val;
  5609	
  5610		if ((kstrtol(str, 0, &val))) {
  5611			pr_warn("Unable to set resched_latency_warn_ms\n");
  5612			return 1;
  5613		}
  5614	
  5615		sysctl_resched_latency_warn_ms = val;
  5616		return 1;
  5617	}
  5618	__setup("resched_latency_warn_ms=", setup_resched_latency_warn_ms);
  5619	#else
  5620	static inline u64 cpu_resched_latency(struct rq *rq) { return 0; }
  5621	#endif /* CONFIG_SCHED_DEBUG */
  5622	
  5623	/*
  5624	 * This function gets called by the timer code, with HZ frequency.
  5625	 * We call it with interrupts disabled.
  5626	 */
  5627	void sched_tick(void)
  5628	{
  5629		int cpu = smp_processor_id();
  5630		struct rq *rq = cpu_rq(cpu);
  5631		/* accounting goes to the donor task */
  5632		struct task_struct *donor;
  5633		struct rq_flags rf;
  5634		unsigned long hw_pressure;
  5635		u64 resched_latency;
  5636	
  5637		if (housekeeping_cpu(cpu, HK_TYPE_KERNEL_NOISE))
  5638			arch_scale_freq_tick();
  5639	
  5640		sched_clock_tick();
  5641	
  5642		rq_lock(rq, &rf);
  5643		donor = rq->donor;
  5644	
  5645		psi_account_irqtime(rq, donor, NULL);
  5646	
  5647		update_rq_clock(rq);
  5648		hw_pressure = arch_scale_hw_pressure(cpu_of(rq));
  5649		update_hw_load_avg(rq_clock_task(rq), rq, hw_pressure);
  5650	
  5651		if (dynamic_preempt_lazy() && tif_test_bit(TIF_NEED_RESCHED_LAZY))
  5652			resched_curr(rq);
  5653	
  5654		donor->sched_class->task_tick(rq, donor, 0);
  5655		if (sched_feat(LATENCY_WARN))
  5656			resched_latency = cpu_resched_latency(rq);
  5657		calc_global_load_tick(rq);
  5658		sched_core_tick(rq);
  5659		task_tick_mm_cid(rq, donor);
  5660		scx_tick(rq);
  5661	
  5662		rq_unlock(rq, &rf);
  5663	
  5664		if (sched_feat(LATENCY_WARN) && resched_latency)
  5665			resched_latency_warn(cpu, resched_latency);
  5666	
  5667		perf_event_task_tick();
  5668	
  5669		if (donor->flags & PF_WQ_WORKER)
  5670			wq_worker_tick(donor);
  5671	
  5672	#ifdef CONFIG_SMP
  5673		if (!scx_switched_all()) {
  5674			rq->idle_balance = idle_cpu(cpu);
  5675			sched_balance_trigger(rq);
  5676		}
  5677	#endif
  5678	}
  5679	
  5680	#ifdef CONFIG_NO_HZ_FULL
  5681	
  5682	struct tick_work {
  5683		int			cpu;
  5684		atomic_t		state;
  5685		struct delayed_work	work;
  5686	};
  5687	/* Values for ->state, see diagram below. */
  5688	#define TICK_SCHED_REMOTE_OFFLINE	0
  5689	#define TICK_SCHED_REMOTE_OFFLINING	1
  5690	#define TICK_SCHED_REMOTE_RUNNING	2
  5691	
  5692	/*
  5693	 * State diagram for ->state:
  5694	 *
  5695	 *
  5696	 *          TICK_SCHED_REMOTE_OFFLINE
  5697	 *                    |   ^
  5698	 *                    |   |
  5699	 *                    |   | sched_tick_remote()
  5700	 *                    |   |
  5701	 *                    |   |
  5702	 *                    +--TICK_SCHED_REMOTE_OFFLINING
  5703	 *                    |   ^
  5704	 *                    |   |
  5705	 * sched_tick_start() |   | sched_tick_stop()
  5706	 *                    |   |
  5707	 *                    V   |
  5708	 *          TICK_SCHED_REMOTE_RUNNING
  5709	 *
  5710	 *
  5711	 * Other transitions get WARN_ON_ONCE(), except that sched_tick_remote()
  5712	 * and sched_tick_start() are happy to leave the state in RUNNING.
  5713	 */
  5714	
  5715	static struct tick_work __percpu *tick_work_cpu;
  5716	
  5717	static void sched_tick_remote(struct work_struct *work)
  5718	{
  5719		struct delayed_work *dwork = to_delayed_work(work);
  5720		struct tick_work *twork = container_of(dwork, struct tick_work, work);
  5721		int cpu = twork->cpu;
  5722		struct rq *rq = cpu_rq(cpu);
  5723		int os;
  5724	
  5725		/*
  5726		 * Handle the tick only if it appears the remote CPU is running in full
  5727		 * dynticks mode. The check is racy by nature, but missing a tick or
  5728		 * having one too much is no big deal because the scheduler tick updates
  5729		 * statistics and checks timeslices in a time-independent way, regardless
  5730		 * of when exactly it is running.
  5731		 */
  5732		if (tick_nohz_tick_stopped_cpu(cpu)) {
  5733			guard(rq_lock_irq)(rq);
  5734			struct task_struct *curr = rq->curr;
  5735	
  5736			if (cpu_online(cpu)) {
  5737				/*
  5738				 * Since this is a remote tick for full dynticks mode,
  5739				 * we are always sure that there is no proxy (only a
  5740				 * single task is running).
  5741				 */
  5742				SCHED_WARN_ON(rq->curr != rq->donor);
  5743				update_rq_clock(rq);
  5744	
  5745				if (!is_idle_task(curr)) {
  5746					/*
  5747					 * Make sure the next tick runs within a
  5748					 * reasonable amount of time.
  5749					 */
  5750					u64 delta = rq_clock_task(rq) - curr->se.exec_start;
  5751					WARN_ON_ONCE(delta > (u64)NSEC_PER_SEC * 3);
  5752				}
  5753				curr->sched_class->task_tick(rq, curr, 0);
  5754	
  5755				calc_load_nohz_remote(rq);
  5756			}
  5757		}
  5758	
  5759		/*
  5760		 * Run the remote tick once per second (1Hz). This arbitrary
  5761		 * frequency is large enough to avoid overload but short enough
  5762		 * to keep scheduler internal stats reasonably up to date.  But
  5763		 * first update state to reflect hotplug activity if required.
  5764		 */
  5765		os = atomic_fetch_add_unless(&twork->state, -1, TICK_SCHED_REMOTE_RUNNING);
  5766		WARN_ON_ONCE(os == TICK_SCHED_REMOTE_OFFLINE);
  5767		if (os == TICK_SCHED_REMOTE_RUNNING)
  5768			queue_delayed_work(system_unbound_wq, dwork, HZ);
  5769	}
  5770	
  5771	static void sched_tick_start(int cpu)
  5772	{
  5773		int os;
  5774		struct tick_work *twork;
  5775	
  5776		if (housekeeping_cpu(cpu, HK_TYPE_KERNEL_NOISE))
  5777			return;
  5778	
  5779		WARN_ON_ONCE(!tick_work_cpu);
  5780	
  5781		twork = per_cpu_ptr(tick_work_cpu, cpu);
  5782		os = atomic_xchg(&twork->state, TICK_SCHED_REMOTE_RUNNING);
  5783		WARN_ON_ONCE(os == TICK_SCHED_REMOTE_RUNNING);
  5784		if (os == TICK_SCHED_REMOTE_OFFLINE) {
  5785			twork->cpu = cpu;
  5786			INIT_DELAYED_WORK(&twork->work, sched_tick_remote);
  5787			queue_delayed_work(system_unbound_wq, &twork->work, HZ);
  5788		}
  5789	}
  5790	
  5791	#ifdef CONFIG_HOTPLUG_CPU
  5792	static void sched_tick_stop(int cpu)
  5793	{
  5794		struct tick_work *twork;
  5795		int os;
  5796	
  5797		if (housekeeping_cpu(cpu, HK_TYPE_KERNEL_NOISE))
  5798			return;
  5799	
  5800		WARN_ON_ONCE(!tick_work_cpu);
  5801	
  5802		twork = per_cpu_ptr(tick_work_cpu, cpu);
  5803		/* There cannot be competing actions, but don't rely on stop-machine. */
  5804		os = atomic_xchg(&twork->state, TICK_SCHED_REMOTE_OFFLINING);
  5805		WARN_ON_ONCE(os != TICK_SCHED_REMOTE_RUNNING);
  5806		/* Don't cancel, as this would mess up the state machine. */
  5807	}
  5808	#endif /* CONFIG_HOTPLUG_CPU */
  5809	
  5810	int __init sched_tick_offload_init(void)
  5811	{
  5812		tick_work_cpu = alloc_percpu(struct tick_work);
  5813		BUG_ON(!tick_work_cpu);
  5814		return 0;
  5815	}
  5816	
  5817	#else /* !CONFIG_NO_HZ_FULL */
  5818	static inline void sched_tick_start(int cpu) { }
  5819	static inline void sched_tick_stop(int cpu) { }
  5820	#endif
  5821	
  5822	#if defined(CONFIG_PREEMPTION) && (defined(CONFIG_DEBUG_PREEMPT) || \
  5823					defined(CONFIG_TRACE_PREEMPT_TOGGLE))
  5824	/*
  5825	 * If the value passed in is equal to the current preempt count
  5826	 * then we just disabled preemption. Start timing the latency.
  5827	 */
  5828	static inline void preempt_latency_start(int val)
  5829	{
  5830		if (preempt_count() == val) {
  5831			unsigned long ip = get_lock_parent_ip();
  5832	#ifdef CONFIG_DEBUG_PREEMPT
  5833			current->preempt_disable_ip = ip;
  5834	#endif
  5835			trace_preempt_off(CALLER_ADDR0, ip);
  5836		}
  5837	}
  5838	
  5839	void preempt_count_add(int val)
  5840	{
  5841	#ifdef CONFIG_DEBUG_PREEMPT
  5842		/*
  5843		 * Underflow?
  5844		 */
  5845		if (DEBUG_LOCKS_WARN_ON((preempt_count() < 0)))
  5846			return;
  5847	#endif
  5848		__preempt_count_add(val);
  5849	#ifdef CONFIG_DEBUG_PREEMPT
  5850		/*
  5851		 * Spinlock count overflowing soon?
  5852		 */
  5853		DEBUG_LOCKS_WARN_ON((preempt_count() & PREEMPT_MASK) >=
  5854					PREEMPT_MASK - 10);
  5855	#endif
  5856		preempt_latency_start(val);
  5857	}
  5858	EXPORT_SYMBOL(preempt_count_add);
  5859	NOKPROBE_SYMBOL(preempt_count_add);
  5860	
  5861	/*
  5862	 * If the value passed in equals to the current preempt count
  5863	 * then we just enabled preemption. Stop timing the latency.
  5864	 */
  5865	static inline void preempt_latency_stop(int val)
  5866	{
  5867		if (preempt_count() == val)
  5868			trace_preempt_on(CALLER_ADDR0, get_lock_parent_ip());
  5869	}
  5870	
  5871	void preempt_count_sub(int val)
  5872	{
  5873	#ifdef CONFIG_DEBUG_PREEMPT
  5874		/*
  5875		 * Underflow?
  5876		 */
  5877		if (DEBUG_LOCKS_WARN_ON(val > preempt_count()))
  5878			return;
  5879		/*
  5880		 * Is the spinlock portion underflowing?
  5881		 */
  5882		if (DEBUG_LOCKS_WARN_ON((val < PREEMPT_MASK) &&
  5883				!(preempt_count() & PREEMPT_MASK)))
  5884			return;
  5885	#endif
  5886	
  5887		preempt_latency_stop(val);
  5888		__preempt_count_sub(val);
  5889	}
  5890	EXPORT_SYMBOL(preempt_count_sub);
  5891	NOKPROBE_SYMBOL(preempt_count_sub);
  5892	
  5893	#else
  5894	static inline void preempt_latency_start(int val) { }
  5895	static inline void preempt_latency_stop(int val) { }
  5896	#endif
  5897	
  5898	static inline unsigned long get_preempt_disable_ip(struct task_struct *p)
  5899	{
  5900	#ifdef CONFIG_DEBUG_PREEMPT
  5901		return p->preempt_disable_ip;
  5902	#else
  5903		return 0;
  5904	#endif
  5905	}
  5906	
  5907	/*
  5908	 * Print scheduling while atomic bug:
  5909	 */
  5910	static noinline void __schedule_bug(struct task_struct *prev)
  5911	{
  5912		/* Save this before calling printk(), since that will clobber it */
  5913		unsigned long preempt_disable_ip = get_preempt_disable_ip(current);
  5914	
  5915		if (oops_in_progress)
  5916			return;
  5917	
  5918		printk(KERN_ERR "BUG: scheduling while atomic: %s/%d/0x%08x\n",
  5919			prev->comm, prev->pid, preempt_count());
  5920	
  5921		debug_show_held_locks(prev);
  5922		print_modules();
  5923		if (irqs_disabled())
  5924			print_irqtrace_events(prev);
  5925		if (IS_ENABLED(CONFIG_DEBUG_PREEMPT)) {
  5926			pr_err("Preemption disabled at:");
  5927			print_ip_sym(KERN_ERR, preempt_disable_ip);
  5928		}
  5929		check_panic_on_warn("scheduling while atomic");
  5930	
  5931		dump_stack();
  5932		add_taint(TAINT_WARN, LOCKDEP_STILL_OK);
  5933	}
  5934	
  5935	/*
  5936	 * Various schedule()-time debugging checks and statistics:
  5937	 */
  5938	static inline void schedule_debug(struct task_struct *prev, bool preempt)
  5939	{
  5940	#ifdef CONFIG_SCHED_STACK_END_CHECK
  5941		if (task_stack_end_corrupted(prev))
  5942			panic("corrupted stack end detected inside scheduler\n");
  5943	
  5944		if (task_scs_end_corrupted(prev))
  5945			panic("corrupted shadow stack detected inside scheduler\n");
  5946	#endif
  5947	
  5948	#ifdef CONFIG_DEBUG_ATOMIC_SLEEP
  5949		if (!preempt && READ_ONCE(prev->__state) && prev->non_block_count) {
  5950			printk(KERN_ERR "BUG: scheduling in a non-blocking section: %s/%d/%i\n",
  5951				prev->comm, prev->pid, prev->non_block_count);
  5952			dump_stack();
  5953			add_taint(TAINT_WARN, LOCKDEP_STILL_OK);
  5954		}
  5955	#endif
  5956	
  5957		if (unlikely(in_atomic_preempt_off())) {
  5958			__schedule_bug(prev);
  5959			preempt_count_set(PREEMPT_DISABLED);
  5960		}
  5961		rcu_sleep_check();
  5962		SCHED_WARN_ON(ct_state() == CT_STATE_USER);
  5963	
  5964		profile_hit(SCHED_PROFILING, __builtin_return_address(0));
  5965	
  5966		schedstat_inc(this_rq()->sched_count);
  5967	}
  5968	
  5969	static void prev_balance(struct rq *rq, struct task_struct *prev,
  5970				 struct rq_flags *rf)
  5971	{
  5972		const struct sched_class *start_class = prev->sched_class;
  5973		const struct sched_class *class;
  5974	
  5975	#ifdef CONFIG_SCHED_CLASS_EXT
  5976		/*
  5977		 * SCX requires a balance() call before every pick_task() including when
  5978		 * waking up from SCHED_IDLE. If @start_class is below SCX, start from
  5979		 * SCX instead. Also, set a flag to detect missing balance() call.
  5980		 */
  5981		if (scx_enabled()) {
  5982			rq->scx.flags |= SCX_RQ_BAL_PENDING;
  5983			if (sched_class_above(&ext_sched_class, start_class))
  5984				start_class = &ext_sched_class;
  5985		}
  5986	#endif
  5987	
  5988		/*
  5989		 * We must do the balancing pass before put_prev_task(), such
  5990		 * that when we release the rq->lock the task is in the same
  5991		 * state as before we took rq->lock.
  5992		 *
  5993		 * We can terminate the balance pass as soon as we know there is
  5994		 * a runnable task of @class priority or higher.
  5995		 */
  5996		for_active_class_range(class, start_class, &idle_sched_class) {
  5997			if (class->balance && class->balance(rq, prev, rf))
  5998				break;
  5999		}
  6000	}
  6001	
  6002	/*
  6003	 * Pick up the highest-prio task:
  6004	 */
  6005	static inline struct task_struct *
  6006	__pick_next_task(struct rq *rq, struct task_struct *prev, struct rq_flags *rf)
  6007	{
  6008		const struct sched_class *class;
  6009		struct task_struct *p;
  6010	
  6011		rq->dl_server = NULL;
  6012	
  6013		if (scx_enabled())
  6014			goto restart;
  6015	
  6016		/*
  6017		 * Optimization: we know that if all tasks are in the fair class we can
  6018		 * call that function directly, but only if the @prev task wasn't of a
  6019		 * higher scheduling class, because otherwise those lose the
  6020		 * opportunity to pull in more work from other CPUs.
  6021		 */
  6022		if (likely(!sched_class_above(prev->sched_class, &fair_sched_class) &&
  6023			   rq->nr_running == rq->cfs.h_nr_running)) {
  6024	
  6025			p = pick_next_task_fair(rq, prev, rf);
  6026			if (unlikely(p == RETRY_TASK))
  6027				goto restart;
  6028	
  6029			/* Assume the next prioritized class is idle_sched_class */
  6030			if (!p) {
  6031				p = pick_task_idle(rq);
  6032				put_prev_set_next_task(rq, prev, p);
  6033			}
  6034	
  6035			return p;
  6036		}
  6037	
  6038	restart:
  6039		prev_balance(rq, prev, rf);
  6040	
  6041		for_each_active_class(class) {
  6042			if (class->pick_next_task) {
  6043				p = class->pick_next_task(rq, prev);
  6044				if (p)
  6045					return p;
  6046			} else {
  6047				p = class->pick_task(rq);
  6048				if (p) {
  6049					put_prev_set_next_task(rq, prev, p);
  6050					return p;
  6051				}
  6052			}
  6053		}
  6054	
  6055		BUG(); /* The idle class should always have a runnable task. */
  6056	}
  6057	
  6058	#ifdef CONFIG_SCHED_CORE
  6059	static inline bool is_task_rq_idle(struct task_struct *t)
  6060	{
  6061		return (task_rq(t)->idle == t);
  6062	}
  6063	
  6064	static inline bool cookie_equals(struct task_struct *a, unsigned long cookie)
  6065	{
  6066		return is_task_rq_idle(a) || (a->core_cookie == cookie);
  6067	}
  6068	
  6069	static inline bool cookie_match(struct task_struct *a, struct task_struct *b)
  6070	{
  6071		if (is_task_rq_idle(a) || is_task_rq_idle(b))
  6072			return true;
  6073	
  6074		return a->core_cookie == b->core_cookie;
  6075	}
  6076	
  6077	static inline struct task_struct *pick_task(struct rq *rq)
  6078	{
  6079		const struct sched_class *class;
  6080		struct task_struct *p;
  6081	
  6082		rq->dl_server = NULL;
  6083	
  6084		for_each_active_class(class) {
  6085			p = class->pick_task(rq);
  6086			if (p)
  6087				return p;
  6088		}
  6089	
  6090		BUG(); /* The idle class should always have a runnable task. */
  6091	}
  6092	
  6093	extern void task_vruntime_update(struct rq *rq, struct task_struct *p, bool in_fi);
  6094	
  6095	static void queue_core_balance(struct rq *rq);
  6096	
  6097	static struct task_struct *
  6098	pick_next_task(struct rq *rq, struct task_struct *prev, struct rq_flags *rf)
  6099	{
  6100		struct task_struct *next, *p, *max = NULL;
  6101		const struct cpumask *smt_mask;
  6102		bool fi_before = false;
  6103		bool core_clock_updated = (rq == rq->core);
  6104		unsigned long cookie;
  6105		int i, cpu, occ = 0;
  6106		struct rq *rq_i;
  6107		bool need_sync;
  6108	
  6109		if (!sched_core_enabled(rq))
  6110			return __pick_next_task(rq, prev, rf);
  6111	
  6112		cpu = cpu_of(rq);
  6113	
  6114		/* Stopper task is switching into idle, no need core-wide selection. */
  6115		if (cpu_is_offline(cpu)) {
  6116			/*
  6117			 * Reset core_pick so that we don't enter the fastpath when
  6118			 * coming online. core_pick would already be migrated to
  6119			 * another cpu during offline.
  6120			 */
  6121			rq->core_pick = NULL;
  6122			rq->core_dl_server = NULL;
  6123			return __pick_next_task(rq, prev, rf);
  6124		}
  6125	
  6126		/*
  6127		 * If there were no {en,de}queues since we picked (IOW, the task
  6128		 * pointers are all still valid), and we haven't scheduled the last
  6129		 * pick yet, do so now.
  6130		 *
  6131		 * rq->core_pick can be NULL if no selection was made for a CPU because
  6132		 * it was either offline or went offline during a sibling's core-wide
  6133		 * selection. In this case, do a core-wide selection.
  6134		 */
  6135		if (rq->core->core_pick_seq == rq->core->core_task_seq &&
  6136		    rq->core->core_pick_seq != rq->core_sched_seq &&
  6137		    rq->core_pick) {
  6138			WRITE_ONCE(rq->core_sched_seq, rq->core->core_pick_seq);
  6139	
  6140			next = rq->core_pick;
  6141			rq->dl_server = rq->core_dl_server;
  6142			rq->core_pick = NULL;
  6143			rq->core_dl_server = NULL;
  6144			goto out_set_next;
  6145		}
  6146	
  6147		prev_balance(rq, prev, rf);
  6148	
  6149		smt_mask = cpu_smt_mask(cpu);
  6150		need_sync = !!rq->core->core_cookie;
  6151	
  6152		/* reset state */
  6153		rq->core->core_cookie = 0UL;
  6154		if (rq->core->core_forceidle_count) {
  6155			if (!core_clock_updated) {
  6156				update_rq_clock(rq->core);
  6157				core_clock_updated = true;
  6158			}
  6159			sched_core_account_forceidle(rq);
  6160			/* reset after accounting force idle */
  6161			rq->core->core_forceidle_start = 0;
  6162			rq->core->core_forceidle_count = 0;
  6163			rq->core->core_forceidle_occupation = 0;
  6164			need_sync = true;
  6165			fi_before = true;
  6166		}
  6167	
  6168		/*
  6169		 * core->core_task_seq, core->core_pick_seq, rq->core_sched_seq
  6170		 *
  6171		 * @task_seq guards the task state ({en,de}queues)
  6172		 * @pick_seq is the @task_seq we did a selection on
  6173		 * @sched_seq is the @pick_seq we scheduled
  6174		 *
  6175		 * However, preemptions can cause multiple picks on the same task set.
  6176		 * 'Fix' this by also increasing @task_seq for every pick.
  6177		 */
  6178		rq->core->core_task_seq++;
  6179	
  6180		/*
  6181		 * Optimize for common case where this CPU has no cookies
  6182		 * and there are no cookied tasks running on siblings.
  6183		 */
  6184		if (!need_sync) {
  6185			next = pick_task(rq);
  6186			if (!next->core_cookie) {
  6187				rq->core_pick = NULL;
  6188				rq->core_dl_server = NULL;
  6189				/*
  6190				 * For robustness, update the min_vruntime_fi for
  6191				 * unconstrained picks as well.
  6192				 */
  6193				WARN_ON_ONCE(fi_before);
  6194				task_vruntime_update(rq, next, false);
  6195				goto out_set_next;
  6196			}
  6197		}
  6198	
  6199		/*
  6200		 * For each thread: do the regular task pick and find the max prio task
  6201		 * amongst them.
  6202		 *
  6203		 * Tie-break prio towards the current CPU
  6204		 */
  6205		for_each_cpu_wrap(i, smt_mask, cpu) {
  6206			rq_i = cpu_rq(i);
  6207	
  6208			/*
  6209			 * Current cpu always has its clock updated on entrance to
  6210			 * pick_next_task(). If the current cpu is not the core,
  6211			 * the core may also have been updated above.
  6212			 */
  6213			if (i != cpu && (rq_i != rq->core || !core_clock_updated))
  6214				update_rq_clock(rq_i);
  6215	
  6216			rq_i->core_pick = p = pick_task(rq_i);
  6217			rq_i->core_dl_server = rq_i->dl_server;
  6218	
  6219			if (!max || prio_less(max, p, fi_before))
  6220				max = p;
  6221		}
  6222	
  6223		cookie = rq->core->core_cookie = max->core_cookie;
  6224	
  6225		/*
  6226		 * For each thread: try and find a runnable task that matches @max or
  6227		 * force idle.
  6228		 */
  6229		for_each_cpu(i, smt_mask) {
  6230			rq_i = cpu_rq(i);
  6231			p = rq_i->core_pick;
  6232	
  6233			if (!cookie_equals(p, cookie)) {
  6234				p = NULL;
  6235				if (cookie)
  6236					p = sched_core_find(rq_i, cookie);
  6237				if (!p)
  6238					p = idle_sched_class.pick_task(rq_i);
  6239			}
  6240	
  6241			rq_i->core_pick = p;
  6242			rq_i->core_dl_server = NULL;
  6243	
  6244			if (p == rq_i->idle) {
  6245				if (rq_i->nr_running) {
  6246					rq->core->core_forceidle_count++;
  6247					if (!fi_before)
  6248						rq->core->core_forceidle_seq++;
  6249				}
  6250			} else {
  6251				occ++;
  6252			}
  6253		}
  6254	
  6255		if (schedstat_enabled() && rq->core->core_forceidle_count) {
  6256			rq->core->core_forceidle_start = rq_clock(rq->core);
  6257			rq->core->core_forceidle_occupation = occ;
  6258		}
  6259	
  6260		rq->core->core_pick_seq = rq->core->core_task_seq;
  6261		next = rq->core_pick;
  6262		rq->core_sched_seq = rq->core->core_pick_seq;
  6263	
  6264		/* Something should have been selected for current CPU */
  6265		WARN_ON_ONCE(!next);
  6266	
  6267		/*
  6268		 * Reschedule siblings
  6269		 *
  6270		 * NOTE: L1TF -- at this point we're no longer running the old task and
  6271		 * sending an IPI (below) ensures the sibling will no longer be running
  6272		 * their task. This ensures there is no inter-sibling overlap between
  6273		 * non-matching user state.
  6274		 */
  6275		for_each_cpu(i, smt_mask) {
  6276			rq_i = cpu_rq(i);
  6277	
  6278			/*
  6279			 * An online sibling might have gone offline before a task
  6280			 * could be picked for it, or it might be offline but later
  6281			 * happen to come online, but its too late and nothing was
  6282			 * picked for it.  That's Ok - it will pick tasks for itself,
  6283			 * so ignore it.
  6284			 */
  6285			if (!rq_i->core_pick)
  6286				continue;
  6287	
  6288			/*
  6289			 * Update for new !FI->FI transitions, or if continuing to be in !FI:
  6290			 * fi_before     fi      update?
  6291			 *  0            0       1
  6292			 *  0            1       1
  6293			 *  1            0       1
  6294			 *  1            1       0
  6295			 */
  6296			if (!(fi_before && rq->core->core_forceidle_count))
  6297				task_vruntime_update(rq_i, rq_i->core_pick, !!rq->core->core_forceidle_count);
  6298	
  6299			rq_i->core_pick->core_occupation = occ;
  6300	
  6301			if (i == cpu) {
  6302				rq_i->core_pick = NULL;
  6303				rq_i->core_dl_server = NULL;
  6304				continue;
  6305			}
  6306	
  6307			/* Did we break L1TF mitigation requirements? */
  6308			WARN_ON_ONCE(!cookie_match(next, rq_i->core_pick));
  6309	
  6310			if (rq_i->curr == rq_i->core_pick) {
  6311				rq_i->core_pick = NULL;
  6312				rq_i->core_dl_server = NULL;
  6313				continue;
  6314			}
  6315	
  6316			resched_curr(rq_i);
  6317		}
  6318	
  6319	out_set_next:
  6320		put_prev_set_next_task(rq, prev, next);
  6321		if (rq->core->core_forceidle_count && next == rq->idle)
  6322			queue_core_balance(rq);
  6323	
  6324		return next;
  6325	}
  6326	
  6327	static bool try_steal_cookie(int this, int that)
  6328	{
  6329		struct rq *dst = cpu_rq(this), *src = cpu_rq(that);
  6330		struct task_struct *p;
  6331		unsigned long cookie;
  6332		bool success = false;
  6333	
  6334		guard(irq)();
  6335		guard(double_rq_lock)(dst, src);
  6336	
  6337		cookie = dst->core->core_cookie;
  6338		if (!cookie)
  6339			return false;
  6340	
  6341		if (dst->curr != dst->idle)
  6342			return false;
  6343	
  6344		p = sched_core_find(src, cookie);
  6345		if (!p)
  6346			return false;
  6347	
  6348		do {
  6349			if (p == src->core_pick || p == src->curr)
  6350				goto next;
  6351	
  6352			if (!is_cpu_allowed(p, this))
  6353				goto next;
  6354	
  6355			if (p->core_occupation > dst->idle->core_occupation)
  6356				goto next;
  6357			/*
  6358			 * sched_core_find() and sched_core_next() will ensure
  6359			 * that task @p is not throttled now, we also need to
  6360			 * check whether the runqueue of the destination CPU is
  6361			 * being throttled.
  6362			 */
  6363			if (sched_task_is_throttled(p, this))
  6364				goto next;
  6365	
  6366			move_queued_task_locked(src, dst, p);
  6367			resched_curr(dst);
  6368	
  6369			success = true;
  6370			break;
  6371	
  6372	next:
  6373			p = sched_core_next(p, cookie);
  6374		} while (p);
  6375	
  6376		return success;
  6377	}
  6378	
  6379	static bool steal_cookie_task(int cpu, struct sched_domain *sd)
  6380	{
  6381		int i;
  6382	
  6383		for_each_cpu_wrap(i, sched_domain_span(sd), cpu + 1) {
  6384			if (i == cpu)
  6385				continue;
  6386	
  6387			if (need_resched())
  6388				break;
  6389	
  6390			if (try_steal_cookie(cpu, i))
  6391				return true;
  6392		}
  6393	
  6394		return false;
  6395	}
  6396	
  6397	static void sched_core_balance(struct rq *rq)
  6398	{
  6399		struct sched_domain *sd;
  6400		int cpu = cpu_of(rq);
  6401	
  6402		guard(preempt)();
  6403		guard(rcu)();
  6404	
  6405		raw_spin_rq_unlock_irq(rq);
  6406		for_each_domain(cpu, sd) {
  6407			if (need_resched())
  6408				break;
  6409	
  6410			if (steal_cookie_task(cpu, sd))
  6411				break;
  6412		}
  6413		raw_spin_rq_lock_irq(rq);
  6414	}
  6415	
  6416	static DEFINE_PER_CPU(struct balance_callback, core_balance_head);
  6417	
  6418	static void queue_core_balance(struct rq *rq)
  6419	{
  6420		if (!sched_core_enabled(rq))
  6421			return;
  6422	
  6423		if (!rq->core->core_cookie)
  6424			return;
  6425	
  6426		if (!rq->nr_running) /* not forced idle */
  6427			return;
  6428	
  6429		queue_balance_callback(rq, &per_cpu(core_balance_head, rq->cpu), sched_core_balance);
  6430	}
  6431	
  6432	DEFINE_LOCK_GUARD_1(core_lock, int,
  6433			    sched_core_lock(*_T->lock, &_T->flags),
  6434			    sched_core_unlock(*_T->lock, &_T->flags),
  6435			    unsigned long flags)
  6436	
  6437	static void sched_core_cpu_starting(unsigned int cpu)
  6438	{
  6439		const struct cpumask *smt_mask = cpu_smt_mask(cpu);
  6440		struct rq *rq = cpu_rq(cpu), *core_rq = NULL;
  6441		int t;
  6442	
  6443		guard(core_lock)(&cpu);
  6444	
  6445		WARN_ON_ONCE(rq->core != rq);
  6446	
  6447		/* if we're the first, we'll be our own leader */
  6448		if (cpumask_weight(smt_mask) == 1)
  6449			return;
  6450	
  6451		/* find the leader */
  6452		for_each_cpu(t, smt_mask) {
  6453			if (t == cpu)
  6454				continue;
  6455			rq = cpu_rq(t);
  6456			if (rq->core == rq) {
  6457				core_rq = rq;
  6458				break;
  6459			}
  6460		}
  6461	
  6462		if (WARN_ON_ONCE(!core_rq)) /* whoopsie */
  6463			return;
  6464	
  6465		/* install and validate core_rq */
  6466		for_each_cpu(t, smt_mask) {
  6467			rq = cpu_rq(t);
  6468	
  6469			if (t == cpu)
  6470				rq->core = core_rq;
  6471	
  6472			WARN_ON_ONCE(rq->core != core_rq);
  6473		}
  6474	}
  6475	
  6476	static void sched_core_cpu_deactivate(unsigned int cpu)
  6477	{
  6478		const struct cpumask *smt_mask = cpu_smt_mask(cpu);
  6479		struct rq *rq = cpu_rq(cpu), *core_rq = NULL;
  6480		int t;
  6481	
  6482		guard(core_lock)(&cpu);
  6483	
  6484		/* if we're the last man standing, nothing to do */
  6485		if (cpumask_weight(smt_mask) == 1) {
  6486			WARN_ON_ONCE(rq->core != rq);
  6487			return;
  6488		}
  6489	
  6490		/* if we're not the leader, nothing to do */
  6491		if (rq->core != rq)
  6492			return;
  6493	
  6494		/* find a new leader */
  6495		for_each_cpu(t, smt_mask) {
  6496			if (t == cpu)
  6497				continue;
  6498			core_rq = cpu_rq(t);
  6499			break;
  6500		}
  6501	
  6502		if (WARN_ON_ONCE(!core_rq)) /* impossible */
  6503			return;
  6504	
  6505		/* copy the shared state to the new leader */
  6506		core_rq->core_task_seq             = rq->core_task_seq;
  6507		core_rq->core_pick_seq             = rq->core_pick_seq;
  6508		core_rq->core_cookie               = rq->core_cookie;
  6509		core_rq->core_forceidle_count      = rq->core_forceidle_count;
  6510		core_rq->core_forceidle_seq        = rq->core_forceidle_seq;
  6511		core_rq->core_forceidle_occupation = rq->core_forceidle_occupation;
  6512	
  6513		/*
  6514		 * Accounting edge for forced idle is handled in pick_next_task().
  6515		 * Don't need another one here, since the hotplug thread shouldn't
  6516		 * have a cookie.
  6517		 */
  6518		core_rq->core_forceidle_start = 0;
  6519	
  6520		/* install new leader */
  6521		for_each_cpu(t, smt_mask) {
  6522			rq = cpu_rq(t);
  6523			rq->core = core_rq;
  6524		}
  6525	}
  6526	
  6527	static inline void sched_core_cpu_dying(unsigned int cpu)
  6528	{
  6529		struct rq *rq = cpu_rq(cpu);
  6530	
  6531		if (rq->core != rq)
  6532			rq->core = rq;
  6533	}
  6534	
  6535	#else /* !CONFIG_SCHED_CORE */
  6536	
  6537	static inline void sched_core_cpu_starting(unsigned int cpu) {}
  6538	static inline void sched_core_cpu_deactivate(unsigned int cpu) {}
  6539	static inline void sched_core_cpu_dying(unsigned int cpu) {}
  6540	
  6541	static struct task_struct *
  6542	pick_next_task(struct rq *rq, struct task_struct *prev, struct rq_flags *rf)
  6543	{
  6544		return __pick_next_task(rq, prev, rf);
  6545	}
  6546	
  6547	#endif /* CONFIG_SCHED_CORE */
  6548	
  6549	/*
  6550	 * Constants for the sched_mode argument of __schedule().
  6551	 *
  6552	 * The mode argument allows RT enabled kernels to differentiate a
  6553	 * preemption from blocking on an 'sleeping' spin/rwlock.
  6554	 */
  6555	#define SM_IDLE			(-1)
  6556	#define SM_NONE			0
  6557	#define SM_PREEMPT		1
  6558	#define SM_RTLOCK_WAIT		2
  6559	
  6560	/*
  6561	 * Helper function for __schedule()
  6562	 *
  6563	 * If a task does not have signals pending, deactivate it
  6564	 * Otherwise marks the task's __state as RUNNING
  6565	 */
  6566	static bool try_to_block_task(struct rq *rq, struct task_struct *p,
  6567				      unsigned long task_state)
  6568	{
  6569		int flags = DEQUEUE_NOCLOCK;
  6570	
  6571		if (signal_pending_state(task_state, p)) {
  6572			WRITE_ONCE(p->__state, TASK_RUNNING);
  6573			return false;
  6574		}
  6575	
  6576		p->sched_contributes_to_load =
  6577			(task_state & TASK_UNINTERRUPTIBLE) &&
  6578			!(task_state & TASK_NOLOAD) &&
  6579			!(task_state & TASK_FROZEN);
  6580	
  6581		if (unlikely(is_special_task_state(task_state)))
  6582			flags |= DEQUEUE_SPECIAL;
  6583	
  6584		/*
  6585		 * __schedule()			ttwu()
  6586		 *   prev_state = prev->state;    if (p->on_rq && ...)
  6587		 *   if (prev_state)		    goto out;
  6588		 *     p->on_rq = 0;		  smp_acquire__after_ctrl_dep();
  6589		 *				  p->state = TASK_WAKING
  6590		 *
  6591		 * Where __schedule() and ttwu() have matching control dependencies.
  6592		 *
  6593		 * After this, schedule() must not care about p->state any more.
  6594		 */
  6595		block_task(rq, p, flags);
  6596		return true;
  6597	}
  6598	
  6599	/*
  6600	 * __schedule() is the main scheduler function.
  6601	 *
  6602	 * The main means of driving the scheduler and thus entering this function are:
  6603	 *
  6604	 *   1. Explicit blocking: mutex, semaphore, waitqueue, etc.
  6605	 *
  6606	 *   2. TIF_NEED_RESCHED flag is checked on interrupt and userspace return
  6607	 *      paths. For example, see arch/x86/entry_64.S.
  6608	 *
  6609	 *      To drive preemption between tasks, the scheduler sets the flag in timer
  6610	 *      interrupt handler sched_tick().
  6611	 *
  6612	 *   3. Wakeups don't really cause entry into schedule(). They add a
  6613	 *      task to the run-queue and that's it.
  6614	 *
  6615	 *      Now, if the new task added to the run-queue preempts the current
  6616	 *      task, then the wakeup sets TIF_NEED_RESCHED and schedule() gets
  6617	 *      called on the nearest possible occasion:
  6618	 *
  6619	 *       - If the kernel is preemptible (CONFIG_PREEMPTION=y):
  6620	 *
  6621	 *         - in syscall or exception context, at the next outmost
  6622	 *           preempt_enable(). (this might be as soon as the wake_up()'s
  6623	 *           spin_unlock()!)
  6624	 *
  6625	 *         - in IRQ context, return from interrupt-handler to
  6626	 *           preemptible context
  6627	 *
  6628	 *       - If the kernel is not preemptible (CONFIG_PREEMPTION is not set)
  6629	 *         then at the next:
  6630	 *
  6631	 *          - cond_resched() call
  6632	 *          - explicit schedule() call
  6633	 *          - return from syscall or exception to user-space
  6634	 *          - return from interrupt-handler to user-space
  6635	 *
  6636	 * WARNING: must be called with preemption disabled!
  6637	 */
  6638	static void __sched notrace __schedule(int sched_mode)
  6639	{
  6640		struct task_struct *prev, *next;
  6641		/*
  6642		 * On PREEMPT_RT kernel, SM_RTLOCK_WAIT is noted
  6643		 * as a preemption by schedule_debug() and RCU.
  6644		 */
  6645		bool preempt = sched_mode > SM_NONE;
  6646		bool block = false;
  6647		unsigned long *switch_count;
  6648		unsigned long prev_state;
  6649		struct rq_flags rf;
  6650		struct rq *rq;
  6651		int cpu;
  6652	
  6653		cpu = smp_processor_id();
  6654		rq = cpu_rq(cpu);
  6655		prev = rq->curr;
  6656	
  6657		schedule_debug(prev, preempt);
  6658	
  6659		if (sched_feat(HRTICK) || sched_feat(HRTICK_DL))
  6660			hrtick_clear(rq);
  6661	
  6662		local_irq_disable();
  6663		rcu_note_context_switch(preempt);
  6664	
  6665		/*
  6666		 * Make sure that signal_pending_state()->signal_pending() below
  6667		 * can't be reordered with __set_current_state(TASK_INTERRUPTIBLE)
  6668		 * done by the caller to avoid the race with signal_wake_up():
  6669		 *
  6670		 * __set_current_state(@state)		signal_wake_up()
  6671		 * schedule()				  set_tsk_thread_flag(p, TIF_SIGPENDING)
  6672		 *					  wake_up_state(p, state)
  6673		 *   LOCK rq->lock			    LOCK p->pi_state
  6674		 *   smp_mb__after_spinlock()		    smp_mb__after_spinlock()
  6675		 *     if (signal_pending_state())	    if (p->state & @state)
  6676		 *
  6677		 * Also, the membarrier system call requires a full memory barrier
  6678		 * after coming from user-space, before storing to rq->curr; this
  6679		 * barrier matches a full barrier in the proximity of the membarrier
  6680		 * system call exit.
  6681		 */
  6682		rq_lock(rq, &rf);
  6683		smp_mb__after_spinlock();
  6684	
  6685		/* Promote REQ to ACT */
  6686		rq->clock_update_flags <<= 1;
  6687		update_rq_clock(rq);
  6688		rq->clock_update_flags = RQCF_UPDATED;
  6689	
  6690		switch_count = &prev->nivcsw;
  6691	
  6692		/* Task state changes only considers SM_PREEMPT as preemption */
  6693		preempt = sched_mode == SM_PREEMPT;
  6694	
  6695		/*
  6696		 * We must load prev->state once (task_struct::state is volatile), such
  6697		 * that we form a control dependency vs deactivate_task() below.
  6698		 */
  6699		prev_state = READ_ONCE(prev->__state);
  6700		if (sched_mode == SM_IDLE) {
  6701			/* SCX must consult the BPF scheduler to tell if rq is empty */
  6702			if (!rq->nr_running && !scx_enabled()) {
  6703				next = prev;
  6704				goto picked;
  6705			}
  6706		} else if (!preempt && prev_state) {
  6707			block = try_to_block_task(rq, prev, prev_state);
  6708			switch_count = &prev->nvcsw;
  6709		}
  6710	
  6711		next = pick_next_task(rq, prev, &rf);
  6712		rq_set_donor(rq, next);
  6713	picked:
  6714		clear_tsk_need_resched(prev);
  6715		clear_preempt_need_resched();
  6716	#ifdef CONFIG_SCHED_DEBUG
  6717		rq->last_seen_need_resched_ns = 0;
  6718	#endif
  6719	
  6720		if (likely(prev != next)) {
  6721			rq->nr_switches++;
  6722			/*
  6723			 * RCU users of rcu_dereference(rq->curr) may not see
  6724			 * changes to task_struct made by pick_next_task().
  6725			 */
  6726			RCU_INIT_POINTER(rq->curr, next);
  6727			/*
  6728			 * The membarrier system call requires each architecture
  6729			 * to have a full memory barrier after updating
  6730			 * rq->curr, before returning to user-space.
  6731			 *
  6732			 * Here are the schemes providing that barrier on the
  6733			 * various architectures:
  6734			 * - mm ? switch_mm() : mmdrop() for x86, s390, sparc, PowerPC,
  6735			 *   RISC-V.  switch_mm() relies on membarrier_arch_switch_mm()
  6736			 *   on PowerPC and on RISC-V.
  6737			 * - finish_lock_switch() for weakly-ordered
  6738			 *   architectures where spin_unlock is a full barrier,
  6739			 * - switch_to() for arm64 (weakly-ordered, spin_unlock
  6740			 *   is a RELEASE barrier),
  6741			 *
  6742			 * The barrier matches a full barrier in the proximity of
  6743			 * the membarrier system call entry.
  6744			 *
  6745			 * On RISC-V, this barrier pairing is also needed for the
  6746			 * SYNC_CORE command when switching between processes, cf.
  6747			 * the inline comments in membarrier_arch_switch_mm().
  6748			 */
  6749			++*switch_count;
  6750	
  6751			migrate_disable_switch(rq, prev);
  6752			psi_account_irqtime(rq, prev, next);
  6753			psi_sched_switch(prev, next, block);
  6754	
  6755			trace_sched_switch(preempt, prev, next, prev_state);
  6756	
  6757			/* Also unlocks the rq: */
  6758			rq = context_switch(rq, prev, next, &rf);
  6759		} else {
  6760			rq_unpin_lock(rq, &rf);
  6761			__balance_callbacks(rq);
  6762			raw_spin_rq_unlock_irq(rq);
  6763		}
  6764	}
  6765	
  6766	void __noreturn do_task_dead(void)
  6767	{
  6768		/* Causes final put_task_struct in finish_task_switch(): */
  6769		set_special_state(TASK_DEAD);
  6770	
  6771		/* Tell freezer to ignore us: */
  6772		current->flags |= PF_NOFREEZE;
  6773	
  6774		__schedule(SM_NONE);
  6775		BUG();
  6776	
  6777		/* Avoid "noreturn function does return" - but don't continue if BUG() is a NOP: */
  6778		for (;;)
  6779			cpu_relax();
  6780	}
  6781	
  6782	static inline void sched_submit_work(struct task_struct *tsk)
  6783	{
  6784		static DEFINE_WAIT_OVERRIDE_MAP(sched_map, LD_WAIT_CONFIG);
  6785		unsigned int task_flags;
  6786	
  6787		/*
  6788		 * Establish LD_WAIT_CONFIG context to ensure none of the code called
  6789		 * will use a blocking primitive -- which would lead to recursion.
  6790		 */
  6791		lock_map_acquire_try(&sched_map);
  6792	
  6793		task_flags = tsk->flags;
  6794		/*
  6795		 * If a worker goes to sleep, notify and ask workqueue whether it
  6796		 * wants to wake up a task to maintain concurrency.
  6797		 */
  6798		if (task_flags & PF_WQ_WORKER)
  6799			wq_worker_sleeping(tsk);
  6800		else if (task_flags & PF_IO_WORKER)
  6801			io_wq_worker_sleeping(tsk);
  6802	
  6803		/*
  6804		 * spinlock and rwlock must not flush block requests.  This will
  6805		 * deadlock if the callback attempts to acquire a lock which is
  6806		 * already acquired.
  6807		 */
  6808		SCHED_WARN_ON(current->__state & TASK_RTLOCK_WAIT);
  6809	
  6810		/*
  6811		 * If we are going to sleep and we have plugged IO queued,
  6812		 * make sure to submit it to avoid deadlocks.
  6813		 */
  6814		blk_flush_plug(tsk->plug, true);
  6815	
  6816		lock_map_release(&sched_map);
  6817	}
  6818	
  6819	static void sched_update_worker(struct task_struct *tsk)
  6820	{
  6821		if (tsk->flags & (PF_WQ_WORKER | PF_IO_WORKER | PF_BLOCK_TS)) {
  6822			if (tsk->flags & PF_BLOCK_TS)
  6823				blk_plug_invalidate_ts(tsk);
  6824			if (tsk->flags & PF_WQ_WORKER)
  6825				wq_worker_running(tsk);
  6826			else if (tsk->flags & PF_IO_WORKER)
  6827				io_wq_worker_running(tsk);
  6828		}
  6829	}
  6830	
  6831	static __always_inline void __schedule_loop(int sched_mode)
  6832	{
  6833		do {
  6834			preempt_disable();
  6835			__schedule(sched_mode);
  6836			sched_preempt_enable_no_resched();
  6837		} while (need_resched());
  6838	}
  6839	
  6840	asmlinkage __visible void __sched schedule(void)
  6841	{
  6842		struct task_struct *tsk = current;
  6843	
  6844	#ifdef CONFIG_RT_MUTEXES
  6845		lockdep_assert(!tsk->sched_rt_mutex);
  6846	#endif
  6847	
  6848		if (!task_is_running(tsk))
  6849			sched_submit_work(tsk);
  6850		__schedule_loop(SM_NONE);
  6851		sched_update_worker(tsk);
  6852	}
  6853	EXPORT_SYMBOL(schedule);
  6854	
  6855	/*
  6856	 * synchronize_rcu_tasks() makes sure that no task is stuck in preempted
  6857	 * state (have scheduled out non-voluntarily) by making sure that all
  6858	 * tasks have either left the run queue or have gone into user space.
  6859	 * As idle tasks do not do either, they must not ever be preempted
  6860	 * (schedule out non-voluntarily).
  6861	 *
  6862	 * schedule_idle() is similar to schedule_preempt_disable() except that it
  6863	 * never enables preemption because it does not call sched_submit_work().
  6864	 */
  6865	void __sched schedule_idle(void)
  6866	{
  6867		/*
  6868		 * As this skips calling sched_submit_work(), which the idle task does
  6869		 * regardless because that function is a NOP when the task is in a
  6870		 * TASK_RUNNING state, make sure this isn't used someplace that the
  6871		 * current task can be in any other state. Note, idle is always in the
  6872		 * TASK_RUNNING state.
  6873		 */
  6874		WARN_ON_ONCE(current->__state);
  6875		do {
  6876			__schedule(SM_IDLE);
  6877		} while (need_resched());
  6878	}
  6879	
  6880	#if defined(CONFIG_CONTEXT_TRACKING_USER) && !defined(CONFIG_HAVE_CONTEXT_TRACKING_USER_OFFSTACK)
  6881	asmlinkage __visible void __sched schedule_user(void)
  6882	{
  6883		/*
  6884		 * If we come here after a random call to set_need_resched(),
  6885		 * or we have been woken up remotely but the IPI has not yet arrived,
  6886		 * we haven't yet exited the RCU idle mode. Do it here manually until
  6887		 * we find a better solution.
  6888		 *
  6889		 * NB: There are buggy callers of this function.  Ideally we
  6890		 * should warn if prev_state != CT_STATE_USER, but that will trigger
  6891		 * too frequently to make sense yet.
  6892		 */
  6893		enum ctx_state prev_state = exception_enter();
  6894		schedule();
  6895		exception_exit(prev_state);
  6896	}
  6897	#endif
  6898	
  6899	/**
  6900	 * schedule_preempt_disabled - called with preemption disabled
  6901	 *
  6902	 * Returns with preemption disabled. Note: preempt_count must be 1
  6903	 */
  6904	void __sched schedule_preempt_disabled(void)
  6905	{
  6906		sched_preempt_enable_no_resched();
  6907		schedule();
  6908		preempt_disable();
  6909	}
  6910	
  6911	#ifdef CONFIG_PREEMPT_RT
  6912	void __sched notrace schedule_rtlock(void)
  6913	{
  6914		__schedule_loop(SM_RTLOCK_WAIT);
  6915	}
  6916	NOKPROBE_SYMBOL(schedule_rtlock);
  6917	#endif
  6918	
  6919	static void __sched notrace preempt_schedule_common(void)
  6920	{
  6921		do {
  6922			/*
  6923			 * Because the function tracer can trace preempt_count_sub()
  6924			 * and it also uses preempt_enable/disable_notrace(), if
  6925			 * NEED_RESCHED is set, the preempt_enable_notrace() called
  6926			 * by the function tracer will call this function again and
  6927			 * cause infinite recursion.
  6928			 *
  6929			 * Preemption must be disabled here before the function
  6930			 * tracer can trace. Break up preempt_disable() into two
  6931			 * calls. One to disable preemption without fear of being
  6932			 * traced. The other to still record the preemption latency,
  6933			 * which can also be traced by the function tracer.
  6934			 */
  6935			preempt_disable_notrace();
  6936			preempt_latency_start(1);
  6937			__schedule(SM_PREEMPT);
  6938			preempt_latency_stop(1);
  6939			preempt_enable_no_resched_notrace();
  6940	
  6941			/*
  6942			 * Check again in case we missed a preemption opportunity
  6943			 * between schedule and now.
  6944			 */
  6945		} while (need_resched());
  6946	}
  6947	
  6948	#ifdef CONFIG_PREEMPTION
  6949	/*
  6950	 * This is the entry point to schedule() from in-kernel preemption
  6951	 * off of preempt_enable.
  6952	 */
  6953	asmlinkage __visible void __sched notrace preempt_schedule(void)
  6954	{
  6955		/*
  6956		 * If there is a non-zero preempt_count or interrupts are disabled,
  6957		 * we do not want to preempt the current task. Just return..
  6958		 */
  6959		if (likely(!preemptible()))
  6960			return;
  6961		preempt_schedule_common();
  6962	}
  6963	NOKPROBE_SYMBOL(preempt_schedule);
  6964	EXPORT_SYMBOL(preempt_schedule);
  6965	
  6966	#ifdef CONFIG_PREEMPT_DYNAMIC
  6967	#if defined(CONFIG_HAVE_PREEMPT_DYNAMIC_CALL)
  6968	#ifndef preempt_schedule_dynamic_enabled
  6969	#define preempt_schedule_dynamic_enabled	preempt_schedule
  6970	#define preempt_schedule_dynamic_disabled	NULL
  6971	#endif
  6972	DEFINE_STATIC_CALL(preempt_schedule, preempt_schedule_dynamic_enabled);
  6973	EXPORT_STATIC_CALL_TRAMP(preempt_schedule);
  6974	#elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY)
  6975	static DEFINE_STATIC_KEY_TRUE(sk_dynamic_preempt_schedule);
  6976	void __sched notrace dynamic_preempt_schedule(void)
  6977	{
  6978		if (!static_branch_unlikely(&sk_dynamic_preempt_schedule))
  6979			return;
  6980		preempt_schedule();
  6981	}
  6982	NOKPROBE_SYMBOL(dynamic_preempt_schedule);
  6983	EXPORT_SYMBOL(dynamic_preempt_schedule);
  6984	#endif
  6985	#endif
  6986	
  6987	/**
  6988	 * preempt_schedule_notrace - preempt_schedule called by tracing
  6989	 *
  6990	 * The tracing infrastructure uses preempt_enable_notrace to prevent
  6991	 * recursion and tracing preempt enabling caused by the tracing
  6992	 * infrastructure itself. But as tracing can happen in areas coming
  6993	 * from userspace or just about to enter userspace, a preempt enable
  6994	 * can occur before user_exit() is called. This will cause the scheduler
  6995	 * to be called when the system is still in usermode.
  6996	 *
  6997	 * To prevent this, the preempt_enable_notrace will use this function
  6998	 * instead of preempt_schedule() to exit user context if needed before
  6999	 * calling the scheduler.
  7000	 */
  7001	asmlinkage __visible void __sched notrace preempt_schedule_notrace(void)
  7002	{
  7003		enum ctx_state prev_ctx;
  7004	
  7005		if (likely(!preemptible()))
  7006			return;
  7007	
  7008		do {
  7009			/*
  7010			 * Because the function tracer can trace preempt_count_sub()
  7011			 * and it also uses preempt_enable/disable_notrace(), if
  7012			 * NEED_RESCHED is set, the preempt_enable_notrace() called
  7013			 * by the function tracer will call this function again and
  7014			 * cause infinite recursion.
  7015			 *
  7016			 * Preemption must be disabled here before the function
  7017			 * tracer can trace. Break up preempt_disable() into two
  7018			 * calls. One to disable preemption without fear of being
  7019			 * traced. The other to still record the preemption latency,
  7020			 * which can also be traced by the function tracer.
  7021			 */
  7022			preempt_disable_notrace();
  7023			preempt_latency_start(1);
  7024			/*
  7025			 * Needs preempt disabled in case user_exit() is traced
  7026			 * and the tracer calls preempt_enable_notrace() causing
  7027			 * an infinite recursion.
  7028			 */
  7029			prev_ctx = exception_enter();
  7030			__schedule(SM_PREEMPT);
  7031			exception_exit(prev_ctx);
  7032	
  7033			preempt_latency_stop(1);
  7034			preempt_enable_no_resched_notrace();
  7035		} while (need_resched());
  7036	}
  7037	EXPORT_SYMBOL_GPL(preempt_schedule_notrace);
  7038	
  7039	#ifdef CONFIG_PREEMPT_DYNAMIC
  7040	#if defined(CONFIG_HAVE_PREEMPT_DYNAMIC_CALL)
  7041	#ifndef preempt_schedule_notrace_dynamic_enabled
  7042	#define preempt_schedule_notrace_dynamic_enabled	preempt_schedule_notrace
  7043	#define preempt_schedule_notrace_dynamic_disabled	NULL
  7044	#endif
  7045	DEFINE_STATIC_CALL(preempt_schedule_notrace, preempt_schedule_notrace_dynamic_enabled);
  7046	EXPORT_STATIC_CALL_TRAMP(preempt_schedule_notrace);
  7047	#elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY)
  7048	static DEFINE_STATIC_KEY_TRUE(sk_dynamic_preempt_schedule_notrace);
  7049	void __sched notrace dynamic_preempt_schedule_notrace(void)
  7050	{
  7051		if (!static_branch_unlikely(&sk_dynamic_preempt_schedule_notrace))
  7052			return;
  7053		preempt_schedule_notrace();
  7054	}
  7055	NOKPROBE_SYMBOL(dynamic_preempt_schedule_notrace);
  7056	EXPORT_SYMBOL(dynamic_preempt_schedule_notrace);
  7057	#endif
  7058	#endif
  7059	
  7060	#endif /* CONFIG_PREEMPTION */
  7061	
  7062	/*
  7063	 * This is the entry point to schedule() from kernel preemption
  7064	 * off of IRQ context.
  7065	 * Note, that this is called and return with IRQs disabled. This will
  7066	 * protect us against recursive calling from IRQ contexts.
  7067	 */
  7068	asmlinkage __visible void __sched preempt_schedule_irq(void)
  7069	{
  7070		enum ctx_state prev_state;
  7071	
  7072		/* Catch callers which need to be fixed */
  7073		BUG_ON(preempt_count() || !irqs_disabled());
  7074	
  7075		prev_state = exception_enter();
  7076	
  7077		do {
  7078			preempt_disable();
  7079			local_irq_enable();
  7080			__schedule(SM_PREEMPT);
  7081			local_irq_disable();
  7082			sched_preempt_enable_no_resched();
  7083		} while (need_resched());
  7084	
  7085		exception_exit(prev_state);
  7086	}
  7087	
  7088	int default_wake_function(wait_queue_entry_t *curr, unsigned mode, int wake_flags,
  7089				  void *key)
  7090	{
  7091		WARN_ON_ONCE(IS_ENABLED(CONFIG_SCHED_DEBUG) && wake_flags & ~(WF_SYNC|WF_CURRENT_CPU));
  7092		return try_to_wake_up(curr->private, mode, wake_flags);
  7093	}
  7094	EXPORT_SYMBOL(default_wake_function);
  7095	
  7096	const struct sched_class *__setscheduler_class(int policy, int prio)
  7097	{
  7098		if (dl_prio(prio))
  7099			return &dl_sched_class;
  7100	
  7101		if (rt_prio(prio))
  7102			return &rt_sched_class;
  7103	
  7104	#ifdef CONFIG_SCHED_CLASS_EXT
  7105		if (task_should_scx(policy))
  7106			return &ext_sched_class;
  7107	#endif
  7108	
  7109		return &fair_sched_class;
  7110	}
  7111	
  7112	#ifdef CONFIG_RT_MUTEXES
  7113	
  7114	/*
  7115	 * Would be more useful with typeof()/auto_type but they don't mix with
  7116	 * bit-fields. Since it's a local thing, use int. Keep the generic sounding
  7117	 * name such that if someone were to implement this function we get to compare
  7118	 * notes.
  7119	 */
  7120	#define fetch_and_set(x, v) ({ int _x = (x); (x) = (v); _x; })
  7121	
  7122	void rt_mutex_pre_schedule(void)
  7123	{
  7124		lockdep_assert(!fetch_and_set(current->sched_rt_mutex, 1));
  7125		sched_submit_work(current);
  7126	}
  7127	
  7128	void rt_mutex_schedule(void)
  7129	{
  7130		lockdep_assert(current->sched_rt_mutex);
  7131		__schedule_loop(SM_NONE);
  7132	}
  7133	
  7134	void rt_mutex_post_schedule(void)
  7135	{
  7136		sched_update_worker(current);
  7137		lockdep_assert(fetch_and_set(current->sched_rt_mutex, 0));
  7138	}
  7139	
  7140	/*
  7141	 * rt_mutex_setprio - set the current priority of a task
  7142	 * @p: task to boost
  7143	 * @pi_task: donor task
  7144	 *
  7145	 * This function changes the 'effective' priority of a task. It does
  7146	 * not touch ->normal_prio like __setscheduler().
  7147	 *
  7148	 * Used by the rt_mutex code to implement priority inheritance
  7149	 * logic. Call site only calls if the priority of the task changed.
  7150	 */
  7151	void rt_mutex_setprio(struct task_struct *p, struct task_struct *pi_task)
  7152	{
  7153		int prio, oldprio, queued, running, queue_flag =
  7154			DEQUEUE_SAVE | DEQUEUE_MOVE | DEQUEUE_NOCLOCK;
  7155		const struct sched_class *prev_class, *next_class;
  7156		struct rq_flags rf;
  7157		struct rq *rq;
  7158	
  7159		/* XXX used to be waiter->prio, not waiter->task->prio */
  7160		prio = __rt_effective_prio(pi_task, p->normal_prio);
  7161	
  7162		/*
  7163		 * If nothing changed; bail early.
  7164		 */
  7165		if (p->pi_top_task == pi_task && prio == p->prio && !dl_prio(prio))
  7166			return;
  7167	
  7168		rq = __task_rq_lock(p, &rf);
  7169		update_rq_clock(rq);
  7170		/*
  7171		 * Set under pi_lock && rq->lock, such that the value can be used under
  7172		 * either lock.
  7173		 *
  7174		 * Note that there is loads of tricky to make this pointer cache work
  7175		 * right. rt_mutex_slowunlock()+rt_mutex_postunlock() work together to
  7176		 * ensure a task is de-boosted (pi_task is set to NULL) before the
  7177		 * task is allowed to run again (and can exit). This ensures the pointer
  7178		 * points to a blocked task -- which guarantees the task is present.
  7179		 */
  7180		p->pi_top_task = pi_task;
  7181	
  7182		/*
  7183		 * For FIFO/RR we only need to set prio, if that matches we're done.
  7184		 */
  7185		if (prio == p->prio && !dl_prio(prio))
  7186			goto out_unlock;
  7187	
  7188		/*
  7189		 * Idle task boosting is a no-no in general. There is one
  7190		 * exception, when PREEMPT_RT and NOHZ is active:
  7191		 *
  7192		 * The idle task calls get_next_timer_interrupt() and holds
  7193		 * the timer wheel base->lock on the CPU and another CPU wants
  7194		 * to access the timer (probably to cancel it). We can safely
  7195		 * ignore the boosting request, as the idle CPU runs this code
  7196		 * with interrupts disabled and will complete the lock
  7197		 * protected section without being interrupted. So there is no
  7198		 * real need to boost.
  7199		 */
  7200		if (unlikely(p == rq->idle)) {
  7201			WARN_ON(p != rq->curr);
  7202			WARN_ON(p->pi_blocked_on);
  7203			goto out_unlock;
  7204		}
  7205	
  7206		trace_sched_pi_setprio(p, pi_task);
  7207		oldprio = p->prio;
  7208	
  7209		if (oldprio == prio)
  7210			queue_flag &= ~DEQUEUE_MOVE;
  7211	
  7212		prev_class = p->sched_class;
  7213		next_class = __setscheduler_class(p->policy, prio);
  7214	
  7215		if (prev_class != next_class && p->se.sched_delayed)
  7216			dequeue_task(rq, p, DEQUEUE_SLEEP | DEQUEUE_DELAYED | DEQUEUE_NOCLOCK);
  7217	
  7218		queued = task_on_rq_queued(p);
  7219		running = task_current_donor(rq, p);
  7220		if (queued)
  7221			dequeue_task(rq, p, queue_flag);
  7222		if (running)
  7223			put_prev_task(rq, p);
  7224	
  7225		/*
  7226		 * Boosting condition are:
  7227		 * 1. -rt task is running and holds mutex A
  7228		 *      --> -dl task blocks on mutex A
  7229		 *
  7230		 * 2. -dl task is running and holds mutex A
  7231		 *      --> -dl task blocks on mutex A and could preempt the
  7232		 *          running task
  7233		 */
  7234		if (dl_prio(prio)) {
  7235			if (!dl_prio(p->normal_prio) ||
  7236			    (pi_task && dl_prio(pi_task->prio) &&
  7237			     dl_entity_preempt(&pi_task->dl, &p->dl))) {
  7238				p->dl.pi_se = pi_task->dl.pi_se;
  7239				queue_flag |= ENQUEUE_REPLENISH;
  7240			} else {
  7241				p->dl.pi_se = &p->dl;
  7242			}
  7243		} else if (rt_prio(prio)) {
  7244			if (dl_prio(oldprio))
  7245				p->dl.pi_se = &p->dl;
  7246			if (oldprio < prio)
  7247				queue_flag |= ENQUEUE_HEAD;
  7248		} else {
  7249			if (dl_prio(oldprio))
  7250				p->dl.pi_se = &p->dl;
  7251			if (rt_prio(oldprio))
  7252				p->rt.timeout = 0;
  7253		}
  7254	
  7255		p->sched_class = next_class;
  7256		p->prio = prio;
  7257	
  7258		check_class_changing(rq, p, prev_class);
  7259	
  7260		if (queued)
  7261			enqueue_task(rq, p, queue_flag);
  7262		if (running)
  7263			set_next_task(rq, p);
  7264	
  7265		check_class_changed(rq, p, prev_class, oldprio);
  7266	out_unlock:
  7267		/* Avoid rq from going away on us: */
  7268		preempt_disable();
  7269	
  7270		rq_unpin_lock(rq, &rf);
  7271		__balance_callbacks(rq);
  7272		raw_spin_rq_unlock(rq);
  7273	
  7274		preempt_enable();
  7275	}
  7276	#endif
  7277	
  7278	#if !defined(CONFIG_PREEMPTION) || defined(CONFIG_PREEMPT_DYNAMIC)
  7279	int __sched __cond_resched(void)
  7280	{
  7281		if (should_resched(0)) {
  7282			preempt_schedule_common();
  7283			return 1;
  7284		}
  7285		/*
  7286		 * In preemptible kernels, ->rcu_read_lock_nesting tells the tick
  7287		 * whether the current CPU is in an RCU read-side critical section,
  7288		 * so the tick can report quiescent states even for CPUs looping
  7289		 * in kernel context.  In contrast, in non-preemptible kernels,
  7290		 * RCU readers leave no in-memory hints, which means that CPU-bound
  7291		 * processes executing in kernel context might never report an
  7292		 * RCU quiescent state.  Therefore, the following code causes
  7293		 * cond_resched() to report a quiescent state, but only when RCU
  7294		 * is in urgent need of one.
  7295		 */
  7296	#ifndef CONFIG_PREEMPT_RCU
  7297		rcu_all_qs();
  7298	#endif
  7299		return 0;
  7300	}
  7301	EXPORT_SYMBOL(__cond_resched);
  7302	#endif
  7303	
  7304	#ifdef CONFIG_PREEMPT_DYNAMIC
  7305	#if defined(CONFIG_HAVE_PREEMPT_DYNAMIC_CALL)
  7306	#define cond_resched_dynamic_enabled	__cond_resched
  7307	#define cond_resched_dynamic_disabled	((void *)&__static_call_return0)
  7308	DEFINE_STATIC_CALL_RET0(cond_resched, __cond_resched);
  7309	EXPORT_STATIC_CALL_TRAMP(cond_resched);
  7310	
  7311	#define might_resched_dynamic_enabled	__cond_resched
  7312	#define might_resched_dynamic_disabled	((void *)&__static_call_return0)
  7313	DEFINE_STATIC_CALL_RET0(might_resched, __cond_resched);
  7314	EXPORT_STATIC_CALL_TRAMP(might_resched);
  7315	#elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY)
  7316	static DEFINE_STATIC_KEY_FALSE(sk_dynamic_cond_resched);
  7317	int __sched dynamic_cond_resched(void)
  7318	{
  7319		klp_sched_try_switch();
  7320		if (!static_branch_unlikely(&sk_dynamic_cond_resched))
  7321			return 0;
  7322		return __cond_resched();
  7323	}
  7324	EXPORT_SYMBOL(dynamic_cond_resched);
  7325	
  7326	static DEFINE_STATIC_KEY_FALSE(sk_dynamic_might_resched);
  7327	int __sched dynamic_might_resched(void)
  7328	{
  7329		if (!static_branch_unlikely(&sk_dynamic_might_resched))
  7330			return 0;
  7331		return __cond_resched();
  7332	}
  7333	EXPORT_SYMBOL(dynamic_might_resched);
  7334	#endif
  7335	#endif
  7336	
  7337	/*
  7338	 * __cond_resched_lock() - if a reschedule is pending, drop the given lock,
  7339	 * call schedule, and on return reacquire the lock.
  7340	 *
  7341	 * This works OK both with and without CONFIG_PREEMPTION. We do strange low-level
  7342	 * operations here to prevent schedule() from being called twice (once via
  7343	 * spin_unlock(), once by hand).
  7344	 */
  7345	int __cond_resched_lock(spinlock_t *lock)
  7346	{
  7347		int resched = should_resched(PREEMPT_LOCK_OFFSET);
  7348		int ret = 0;
  7349	
  7350		lockdep_assert_held(lock);
  7351	
  7352		if (spin_needbreak(lock) || resched) {
  7353			spin_unlock(lock);
  7354			if (!_cond_resched())
  7355				cpu_relax();
  7356			ret = 1;
  7357			spin_lock(lock);
  7358		}
  7359		return ret;
  7360	}
  7361	EXPORT_SYMBOL(__cond_resched_lock);
  7362	
  7363	int __cond_resched_rwlock_read(rwlock_t *lock)
  7364	{
  7365		int resched = should_resched(PREEMPT_LOCK_OFFSET);
  7366		int ret = 0;
  7367	
  7368		lockdep_assert_held_read(lock);
  7369	
  7370		if (rwlock_needbreak(lock) || resched) {
  7371			read_unlock(lock);
  7372			if (!_cond_resched())
  7373				cpu_relax();
  7374			ret = 1;
  7375			read_lock(lock);
  7376		}
  7377		return ret;
  7378	}
  7379	EXPORT_SYMBOL(__cond_resched_rwlock_read);
  7380	
  7381	int __cond_resched_rwlock_write(rwlock_t *lock)
  7382	{
  7383		int resched = should_resched(PREEMPT_LOCK_OFFSET);
  7384		int ret = 0;
  7385	
  7386		lockdep_assert_held_write(lock);
  7387	
  7388		if (rwlock_needbreak(lock) || resched) {
  7389			write_unlock(lock);
  7390			if (!_cond_resched())
  7391				cpu_relax();
  7392			ret = 1;
  7393			write_lock(lock);
  7394		}
  7395		return ret;
  7396	}
  7397	EXPORT_SYMBOL(__cond_resched_rwlock_write);
  7398	
  7399	#ifdef CONFIG_PREEMPT_DYNAMIC
  7400	
  7401	#ifdef CONFIG_GENERIC_IRQ_ENTRY
> 7402	#include <linux/irq-entry-common.h>
  7403	#endif
  7404	

-- 
0-DAY CI Kernel Test Service
https://github.com/intel/lkp-tests/wiki

^ permalink raw reply	[flat|nested] 3+ messages in thread
* [PATCH -next v5 00/22] arm64: entry: Convert to generic entry
@ 2024-12-06 10:17 Jinjie Ruan
  2024-12-06 10:17 ` [PATCH -next v5 09/22] entry: Split generic entry into irq and syscall Jinjie Ruan
  0 siblings, 1 reply; 3+ messages in thread
From: Jinjie Ruan @ 2024-12-06 10:17 UTC (permalink / raw)
  To: catalin.marinas, will, oleg, sstabellini, tglx, peterz, luto,
	mingo, juri.lelli, vincent.guittot, dietmar.eggemann, rostedt,
	bsegall, mgorman, vschneid, kees, wad, akpm, samitolvanen,
	masahiroy, hca, aliceryhl, rppt, xur, paulmck, arnd, mbenes,
	puranjay, mark.rutland, ruanjinjie, pcc, ardb, sudeep.holla,
	guohanjun, rafael, liuwei09, dwmw, Jonathan.Cameron, liaochang1,
	kristina.martsenko, ptosi, broonie, thiago.bauermann,
	kevin.brodsky, joey.gouly, liuyuntao12, leobras, linux-kernel,
	linux-arm-kernel, xen-devel

Currently, x86, Riscv, Loongarch use the generic entry. Convert arm64
to use the generic entry infrastructure from kernel/entry/*. The generic
entry makes maintainers' work easier and codes more elegant, which aslo
removed a lot of duplicate code.

The main steps are as follows:
- Make arm64 easier to use irqentry_enter/exit().
- Make arm64 closer to the PREEMPT_DYNAMIC code of generic entry.
- Split generic entry into generic irq entry and generic syscall to
  make the single patch more concentrated in switching to one thing.
- Switch to generic irq entry.
- Make arm64 closer to the generic syscall code.
- Switch to generic entry completely.

Changes in v5:
- Not change arm32 and keep inerrupts_enabled() macro for gicv3 driver.
- Move irqentry_state definition into arch/arm64/kernel/entry-common.c.
- Avoid removing the __enter_from_*() and __exit_to_*() wrappers.
- Update "irqentry_state_t ret/irq_state" to "state"
  to keep it consistently.
- Use generic irq entry header for PREEMPT_DYNAMIC after split
  the generic entry.
- Also refactor the ARM64 syscall code.
- Introduce arch_ptrace_report_syscall_entry/exit(), instead of
  arch_pre/post_report_syscall_entry/exit() to simplify code.
- Make the syscall patches clear separation.
- Update the commit message.

Changes in v4:
- Rework/cleanup split into a few patches as Mark suggested.
- Replace interrupts_enabled() macro with regs_irqs_disabled(), instead
  of left it here.
- Remove rcu and lockdep state in pt_regs by using temporary
  irqentry_state_t as Mark suggested.
- Remove some unnecessary intermediate functions to make it clear.
- Rework preempt irq and PREEMPT_DYNAMIC code
  to make the switch more clear.
- arch_prepare_*_entry/exit() -> arch_pre_*_entry/exit().
- Expand the arch functions comment.
- Make arch functions closer to its caller.
- Declare saved_reg in for block.
- Remove arch_exit_to_kernel_mode_prepare(), arch_enter_from_kernel_mode().
- Adjust "Add few arch functions to use generic entry" patch to be
  the penultimate.
- Update the commit message.
- Add suggested-by.

Changes in v3:
- Test the MTE test cases.
- Handle forget_syscall() in arch_post_report_syscall_entry()
- Make the arch funcs not use __weak as Thomas suggested, so move
  the arch funcs to entry-common.h, and make arch_forget_syscall() folded
  in arch_post_report_syscall_entry() as suggested.
- Move report_single_step() to thread_info.h for arm64
- Change __always_inline() to inline, add inline for the other arch funcs.
- Remove unused signal.h for entry-common.h.
- Add Suggested-by.
- Update the commit message.

Changes in v2:
- Add tested-by.
- Fix a bug that not call arch_post_report_syscall_entry() in
  syscall_trace_enter() if ptrace_report_syscall_entry() return not zero.
- Refactor report_syscall().
- Add comment for arch_prepare_report_syscall_exit().
- Adjust entry-common.h header file inclusion to alphabetical order.
- Update the commit message.

Jinjie Ruan (22):
  arm64: ptrace: Replace interrupts_enabled() with regs_irqs_disabled()
  arm64: entry: Refactor the entry and exit for exceptions from EL1
  arm64: entry: Move arm64_preempt_schedule_irq() into
    __exit_to_kernel_mode()
  arm64: entry: Rework arm64_preempt_schedule_irq()
  arm64: entry: Use preempt_count() and need_resched() helper
  arm64: entry: Expand the need_irq_preemption() macro ahead
  arm64: entry: preempt_schedule_irq() only if PREEMPTION enabled
  arm64: entry: Use different helpers to check resched for
    PREEMPT_DYNAMIC
  entry: Split generic entry into irq and syscall
  entry: Add arch_irqentry_exit_need_resched() for arm64
  arm64: entry: Switch to generic IRQ entry
  arm64/ptrace: Split report_syscall() function
  arm64/ptrace: Refactor syscall_trace_enter()
  arm64/ptrace: Refactor syscall_trace_exit()
  arm64/ptrace: Refator el0_svc_common()
  entry: Make syscall_exit_to_user_mode_prepare() not static
  arm64/ptrace: Return early for ptrace_report_syscall_entry() error
  arm64/ptrace: Expand secure_computing() in place
  arm64/ptrace: Use syscall_get_arguments() heleper
  entry: Add arch_ptrace_report_syscall_entry/exit()
  entry: Add has_syscall_work() helepr
  arm64: entry: Convert to generic entry

 MAINTAINERS                           |   1 +
 arch/Kconfig                          |   8 +
 arch/arm64/Kconfig                    |   1 +
 arch/arm64/include/asm/daifflags.h    |   2 +-
 arch/arm64/include/asm/entry-common.h | 134 +++++++++
 arch/arm64/include/asm/preempt.h      |   2 -
 arch/arm64/include/asm/ptrace.h       |  11 +-
 arch/arm64/include/asm/syscall.h      |   6 +-
 arch/arm64/include/asm/thread_info.h  |  23 +-
 arch/arm64/include/asm/xen/events.h   |   2 +-
 arch/arm64/kernel/acpi.c              |   2 +-
 arch/arm64/kernel/debug-monitors.c    |   9 +-
 arch/arm64/kernel/entry-common.c      | 377 ++++++++-----------------
 arch/arm64/kernel/ptrace.c            |  90 ------
 arch/arm64/kernel/sdei.c              |   2 +-
 arch/arm64/kernel/signal.c            |   3 +-
 arch/arm64/kernel/syscall.c           |  31 +-
 include/linux/entry-common.h          | 384 +------------------------
 include/linux/irq-entry-common.h      | 389 ++++++++++++++++++++++++++
 kernel/entry/Makefile                 |   3 +-
 kernel/entry/common.c                 | 176 ++----------
 kernel/entry/syscall-common.c         | 198 +++++++++++++
 kernel/sched/core.c                   |   8 +-
 23 files changed, 909 insertions(+), 953 deletions(-)
 create mode 100644 arch/arm64/include/asm/entry-common.h
 create mode 100644 include/linux/irq-entry-common.h
 create mode 100644 kernel/entry/syscall-common.c

-- 
2.34.1



^ permalink raw reply	[flat|nested] 3+ messages in thread

end of thread, other threads:[~2025-02-10 12:17 UTC | newest]

Thread overview: 3+ messages (download: mbox.gz follow: Atom feed
-- links below jump to the message on this page --
2024-12-09  4:04 [PATCH -next v5 09/22] entry: Split generic entry into irq and syscall kernel test robot
  -- strict thread matches above, loose matches on Subject: below --
2024-12-06 10:17 [PATCH -next v5 00/22] arm64: entry: Convert to generic entry Jinjie Ruan
2024-12-06 10:17 ` [PATCH -next v5 09/22] entry: Split generic entry into irq and syscall Jinjie Ruan
2025-02-10 12:04   ` Mark Rutland

This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.