Linux virtualization list
 help / color / mirror / Atom feed
From: Yury Norov <ynorov@nvidia.com>
To: Shrikanth Hegde <sshegde@linux.ibm.com>
Cc: linux-kernel@vger.kernel.org, mingo@kernel.org,
	peterz@infradead.org, juri.lelli@redhat.com,
	vincent.guittot@linaro.org, yury.norov@gmail.com,
	kprateek.nayak@amd.com, iii@linux.ibm.com, corbet@lwn.net,
	tglx@kernel.org, gregkh@linuxfoundation.org, pbonzini@redhat.com,
	seanjc@google.com, vschneid@redhat.com, huschle@linux.ibm.com,
	rostedt@goodmis.org, dietmar.eggemann@arm.com,
	maddy@linux.ibm.com, srikar@linux.ibm.com, hdanton@sina.com,
	chleroy@kernel.org, vineeth@bitbyteword.org, frederic@kernel.org,
	arighi@nvidia.com, pauld@redhat.com, christian.loehle@arm.com,
	tj@kernel.org, tommaso.cucinotta@gmail.com, maz@kernel.org,
	rafael@kernel.org, rdunlap@infradead.org, kernellwp@gmail.com,
	linux-doc@vger.kernel.org, jgross@suse.com,
	virtualization@lists.linux.dev
Subject: Re: [PATCH v9 02/11] cpumask: Introduce cpu_preferred_mask
Date: Fri, 24 Jul 2026 16:00:11 -0400	[thread overview]
Message-ID: <amPES7GsB5sVCDjY@yury> (raw)
In-Reply-To: <20260724140732.2683314-3-sshegde@linux.ibm.com>

On Fri, Jul 24, 2026 at 07:37:23PM +0530, Shrikanth Hegde wrote:
> Provide cpu_preferred_mask infrastructure. Define get/set macros
> which could be used to get/set CPU state as preferred.
> 
> PREFERRED_CPU config will be selected by the driver which handles
> steal time values. It is going to set/clear preferred CPU state.
> This driver will be called steal_governor and it is introduced in
> subsequent patches. It periodically samples the steal time and
> decides on preferred CPU state.
> 
> A CPU is set to preferred when it becomes active. Later it may be
> marked as non-preferred depending on steal time values with
> steal_governor being enabled.
> 
> Always maintain design construct of preferred is subset of active.
> i.e. preferred ⊆ active ⊆ online ⊆ present ⊆ possible
> 
> With PREFERRED_CPU=n, ensure set_cpu_preferred is a nop and get
> method returns the active state in that case.
> 
> Signed-off-by: Shrikanth Hegde <sshegde@linux.ibm.com>
> ---
>  include/linux/cpumask.h | 24 ++++++++++++++++++++++++
>  kernel/Kconfig.preempt  |  4 ++++
>  kernel/cpu.c            |  6 ++++++
>  kernel/sched/core.c     |  5 +++++
>  4 files changed, 39 insertions(+)
> 
> diff --git a/include/linux/cpumask.h b/include/linux/cpumask.h
> index d3cda0544954..34d08a3d80e1 100644
> --- a/include/linux/cpumask.h
> +++ b/include/linux/cpumask.h
> @@ -122,12 +122,20 @@ extern struct cpumask __cpu_enabled_mask;
>  extern struct cpumask __cpu_present_mask;
>  extern struct cpumask __cpu_active_mask;
>  extern struct cpumask __cpu_dying_mask;
> +
> +#ifdef CONFIG_PREFERRED_CPU
> +extern struct cpumask __cpu_preferred_mask;
> +#else
> +#define __cpu_preferred_mask __cpu_active_mask
> +#endif
> +
>  #define cpu_possible_mask ((const struct cpumask *)&__cpu_possible_mask)
>  #define cpu_online_mask   ((const struct cpumask *)&__cpu_online_mask)
>  #define cpu_enabled_mask   ((const struct cpumask *)&__cpu_enabled_mask)
>  #define cpu_present_mask  ((const struct cpumask *)&__cpu_present_mask)
>  #define cpu_active_mask   ((const struct cpumask *)&__cpu_active_mask)
>  #define cpu_dying_mask    ((const struct cpumask *)&__cpu_dying_mask)
> +#define cpu_preferred_mask ((const struct cpumask *)&__cpu_preferred_mask)
>  
>  extern atomic_t __num_online_cpus;
>  extern unsigned int __num_possible_cpus;
> @@ -1164,6 +1172,12 @@ void init_cpu_possible(const struct cpumask *src);
>  #define set_cpu_active(cpu, active)	assign_cpu((cpu), &__cpu_active_mask, (active))
>  #define set_cpu_dying(cpu, dying)	assign_cpu((cpu), &__cpu_dying_mask, (dying))
>  
> +#ifdef CONFIG_PREFERRED_CPU
> +#define set_cpu_preferred(cpu, preferred) assign_cpu((cpu), &__cpu_preferred_mask, (preferred))
> +#else
> +#define set_cpu_preferred(cpu, preferred) do { } while (0)
> +#endif
> +
>  void set_cpu_online(unsigned int cpu, bool online);
>  void set_cpu_possible(unsigned int cpu, bool possible);
>  
> @@ -1258,6 +1272,11 @@ static __always_inline bool cpu_dying(unsigned int cpu)
>  	return cpumask_test_cpu(cpu, cpu_dying_mask);
>  }
>  
> +static __always_inline bool cpu_preferred(unsigned int cpu)
> +{
> +	return cpumask_test_cpu(cpu, cpu_preferred_mask);
> +}
> +
>  #else
>  
>  #define num_online_cpus()	1U
> @@ -1296,6 +1315,11 @@ static __always_inline bool cpu_dying(unsigned int cpu)
>  	return false;
>  }
>  
> +static __always_inline bool cpu_preferred(unsigned int cpu)
> +{
> +	return cpu == 0;
> +}
> +
>  #endif /* NR_CPUS > 1 */
>  
>  #define cpu_is_offline(cpu)	unlikely(!cpu_online(cpu))
> diff --git a/kernel/Kconfig.preempt b/kernel/Kconfig.preempt
> index 88c594c6d7fc..de789b274ba3 100644
> --- a/kernel/Kconfig.preempt
> +++ b/kernel/Kconfig.preempt
> @@ -192,3 +192,7 @@ config SCHED_CLASS_EXT
>  	  For more information:
>  	    Documentation/scheduler/sched-ext.rst
>  	    https://github.com/sched-ext/scx
> +
> +config PREFERRED_CPU
> +	bool
> +	depends on SMP && PARAVIRT
> diff --git a/kernel/cpu.c b/kernel/cpu.c
> index b3c8553d7bd6..376d297a6292 100644
> --- a/kernel/cpu.c
> +++ b/kernel/cpu.c
> @@ -3103,6 +3103,11 @@ EXPORT_SYMBOL(__cpu_dying_mask);
>  atomic_t __num_online_cpus __read_mostly;
>  EXPORT_SYMBOL(__num_online_cpus);
>  
> +#ifdef CONFIG_PREFERRED_CPU
> +struct cpumask __cpu_preferred_mask __read_mostly;
> +EXPORT_SYMBOL_GPL(__cpu_preferred_mask);
> +#endif
> +
>  void init_cpu_present(const struct cpumask *src)
>  {
>  	cpumask_copy(&__cpu_present_mask, src);
> @@ -3160,6 +3165,7 @@ void __init boot_cpu_init(void)
>  	/* Mark the boot cpu "present", "online" etc for SMP and UP case */
>  	set_cpu_online(cpu, true);
>  	set_cpu_active(cpu, true);
> +	set_cpu_preferred(cpu, true);
>  	set_cpu_present(cpu, true);
>  	set_cpu_possible(cpu, true);
>  
> diff --git a/kernel/sched/core.c b/kernel/sched/core.c
> index 2e7cde033a31..a45f7c308329 100644
> --- a/kernel/sched/core.c
> +++ b/kernel/sched/core.c
> @@ -8690,6 +8690,9 @@ int sched_cpu_activate(unsigned int cpu)
>  	 */
>  	sched_set_rq_online(rq, cpu);
>  
> +	/* preferred is subset of active and follows its state */
> +	set_cpu_preferred(cpu, true);
> +
>  	return 0;
>  }
>  
> @@ -8703,6 +8706,8 @@ int sched_cpu_deactivate(unsigned int cpu)
>  	if (ret)
>  		return ret;
>  
> +	set_cpu_preferred(cpu, false);
> +

Is it possible that this CPU would be the last preferred CPU in the
system? If so, you'll make the preferred mask empty.

In v9 you disabled integrity check while the steal time is withing the
threshold, so this condition may stay undetected quite a long.

Can you add another integrity check here? If you're going to remove
the last preferred CPU, you need to force-enable some alternative.
Something like:

        if (cpumask_nth(1, cpu_preferred_mask) >= nr_cpu_ids) {
                new_cpu = cpumask_any_andnot_but(cpu_active_mask, cpu_preferred_mask, cpu);
                if (!WARN_ON(new_cpu >= nr_cpu_ids))
                        set_cpu_preferred(new_cpu);
        }

        set_cpu_preferred(cpu, false);
                
>  	/*
>  	 * Remove CPU from nohz.idle_cpus_mask to prevent participating in
>  	 * load balancing when not active
> -- 
> 2.47.3

  reply	other threads:[~2026-07-24 20:00 UTC|newest]

Thread overview: 19+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-07-24 14:07 [PATCH v9 00/11] sched, steal_governor: Introduce preferred CPUs and steal-driven vCPU backoff Shrikanth Hegde
2026-07-24 14:07 ` [PATCH v9 01/11] sched/docs: Document cpu_preferred_mask and Preferred CPU concept Shrikanth Hegde
2026-07-24 21:45   ` Yury Norov
2026-07-24 14:07 ` [PATCH v9 02/11] cpumask: Introduce cpu_preferred_mask Shrikanth Hegde
2026-07-24 20:00   ` Yury Norov [this message]
2026-07-24 14:07 ` [PATCH v9 03/11] sysfs: Add preferred CPU file Shrikanth Hegde
2026-07-24 14:07 ` [PATCH v9 04/11] sched/core: Try to use a preferred CPU in is_cpu_allowed Shrikanth Hegde
2026-07-24 14:07 ` [PATCH v9 05/11] sched/fair: Load balance only among preferred CPUs Shrikanth Hegde
2026-07-24 21:40   ` Yury Norov
2026-07-24 14:07 ` [PATCH v9 06/11] sched/core: Push current task from non preferred CPU Shrikanth Hegde
2026-07-24 22:04   ` Yury Norov
2026-07-24 14:07 ` [PATCH v9 07/11] sched/debug: Add migration stats due to non preferred CPUs Shrikanth Hegde
2026-07-24 14:07 ` [PATCH v9 08/11] virt: Introduce steal governor driver Shrikanth Hegde
2026-07-24 14:07 ` [PATCH v9 09/11] virt/steal_governor: Add control knobs for handling steal values Shrikanth Hegde
2026-07-24 14:07 ` [PATCH v9 10/11] virt/steal_governor: Implement steal_governor policy loop Shrikanth Hegde
2026-07-24 21:05   ` Yury Norov
2026-07-24 14:07 ` [PATCH v9 11/11] virt/steal_governor: Enable the driver Shrikanth Hegde
2026-07-24 22:07 ` [PATCH v9 00/11] sched, steal_governor: Introduce preferred CPUs and steal-driven vCPU backoff Yury Norov
2026-07-25  1:53   ` Shrikanth Hegde

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=amPES7GsB5sVCDjY@yury \
    --to=ynorov@nvidia.com \
    --cc=arighi@nvidia.com \
    --cc=chleroy@kernel.org \
    --cc=christian.loehle@arm.com \
    --cc=corbet@lwn.net \
    --cc=dietmar.eggemann@arm.com \
    --cc=frederic@kernel.org \
    --cc=gregkh@linuxfoundation.org \
    --cc=hdanton@sina.com \
    --cc=huschle@linux.ibm.com \
    --cc=iii@linux.ibm.com \
    --cc=jgross@suse.com \
    --cc=juri.lelli@redhat.com \
    --cc=kernellwp@gmail.com \
    --cc=kprateek.nayak@amd.com \
    --cc=linux-doc@vger.kernel.org \
    --cc=linux-kernel@vger.kernel.org \
    --cc=maddy@linux.ibm.com \
    --cc=maz@kernel.org \
    --cc=mingo@kernel.org \
    --cc=pauld@redhat.com \
    --cc=pbonzini@redhat.com \
    --cc=peterz@infradead.org \
    --cc=rafael@kernel.org \
    --cc=rdunlap@infradead.org \
    --cc=rostedt@goodmis.org \
    --cc=seanjc@google.com \
    --cc=srikar@linux.ibm.com \
    --cc=sshegde@linux.ibm.com \
    --cc=tglx@kernel.org \
    --cc=tj@kernel.org \
    --cc=tommaso.cucinotta@gmail.com \
    --cc=vincent.guittot@linaro.org \
    --cc=vineeth@bitbyteword.org \
    --cc=virtualization@lists.linux.dev \
    --cc=vschneid@redhat.com \
    --cc=yury.norov@gmail.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox