Linux Power Management development
 help / color / mirror / Atom feed
From: Anthony Harivel <aharivel@redhat.com>
To: linux-pm@vger.kernel.org
Cc: rafael@kernel.org, daniel.lezcano@linaro.org, seanjc@google.com,
	pbonzini@redhat.com, kvm@vger.kernel.org,
	Anthony Harivel <aharivel@redhat.com>
Subject: [PATCH 1/1] cpuidle: add per-CPU latency_limit_ns sysfs attribute
Date: Tue, 15 Sep 2026 15:13:07 +0200	[thread overview]
Message-ID: <20260915131310.1053834-2-aharivel@redhat.com> (raw)
In-Reply-To: <20260915131310.1053834-1-aharivel@redhat.com>

Add a per-CPU sysfs attribute at:
  /sys/devices/system/cpu/cpuN/cpuidle/latency_limit_ns

This allows privileged userspace to set a governor-respected upper
bound on the exit latency for idle state selection on a given CPU.
When set (non-zero), cpuidle_governor_latency_req() returns the
minimum of the existing PM QoS constraints and latency_limit_ns,
so all governors (menu, TEO, haltpoll) automatically respect it
without per-governor modifications.

Use case: cloud operators running mixed workloads can cap idle
depth on CPUs pinned to latency-sensitive VMs while allowing other
CPUs to enter deep C-states for energy savings. The interface is
latency-based (nanoseconds) rather than C-state-index-based, making
it portable across Intel/AMD/ARM without uarch-specific tuning.

Signed-off-by: Anthony Harivel <aharivel@redhat.com>
---
 drivers/cpuidle/governor.c | 10 +++++++++-
 drivers/cpuidle/sysfs.c    | 36 ++++++++++++++++++++++++++++++++++++
 include/linux/cpuidle.h    |  1 +
 3 files changed, 46 insertions(+), 1 deletion(-)

diff --git a/drivers/cpuidle/governor.c b/drivers/cpuidle/governor.c
index 5d0e7f78c6c5..4c6f77ce2028 100644
--- a/drivers/cpuidle/governor.c
+++ b/drivers/cpuidle/governor.c
@@ -112,6 +112,8 @@ s64 cpuidle_governor_latency_req(unsigned int cpu)
 	int device_req = dev_pm_qos_raw_resume_latency(device);
 	int global_req = cpu_latency_qos_limit();
 	int global_wake_req = cpu_wakeup_latency_qos_limit();
+	struct cpuidle_device *dev;
+	s64 result;
 
 	if (global_req > global_wake_req)
 		global_req = global_wake_req;
@@ -119,5 +121,11 @@ s64 cpuidle_governor_latency_req(unsigned int cpu)
 	if (device_req > global_req)
 		device_req = global_req;
 
-	return (s64)device_req * NSEC_PER_USEC;
+	result = (s64)device_req * NSEC_PER_USEC;
+
+	dev = per_cpu(cpuidle_devices, cpu);
+	if (dev && dev->latency_limit_ns && dev->latency_limit_ns < result)
+		result = dev->latency_limit_ns;
+
+	return result;
 }
diff --git a/drivers/cpuidle/sysfs.c b/drivers/cpuidle/sysfs.c
index b81d22479234..d060a4b7facc 100644
--- a/drivers/cpuidle/sysfs.c
+++ b/drivers/cpuidle/sysfs.c
@@ -207,8 +207,44 @@ static void cpuidle_sysfs_release(struct kobject *kobj)
 	complete(&kdev->kobj_unregister);
 }
 
+static ssize_t show_latency_limit_ns(struct cpuidle_device *dev, char *buf)
+{
+	return sysfs_emit(buf, "%llu\n", dev->latency_limit_ns);
+}
+
+static ssize_t store_latency_limit_ns(struct cpuidle_device *dev,
+				       const char *buf, size_t count)
+{
+	u64 value;
+	int err;
+
+	if (!capable(CAP_SYS_ADMIN))
+		return -EPERM;
+
+	err = kstrtou64(buf, 0, &value);
+	if (err)
+		return err;
+
+	dev->latency_limit_ns = value;
+
+	return count;
+}
+
+static struct cpuidle_attr attr_latency_limit_ns = {
+	.attr = { .name = "latency_limit_ns", .mode = 0644 },
+	.show = show_latency_limit_ns,
+	.store = store_latency_limit_ns,
+};
+
+static struct attribute *cpuidle_device_default_attrs[] = {
+	&attr_latency_limit_ns.attr,
+	NULL,
+};
+ATTRIBUTE_GROUPS(cpuidle_device_default);
+
 static const struct kobj_type ktype_cpuidle = {
 	.sysfs_ops = &cpuidle_sysfs_ops,
+	.default_groups = cpuidle_device_default_groups,
 	.release = cpuidle_sysfs_release,
 };
 
diff --git a/include/linux/cpuidle.h b/include/linux/cpuidle.h
index a2485348def3..f70d6288e13d 100644
--- a/include/linux/cpuidle.h
+++ b/include/linux/cpuidle.h
@@ -101,6 +101,7 @@ struct cpuidle_device {
 	u64			last_residency_ns;
 	u64			poll_limit_ns;
 	u64			forced_idle_latency_limit_ns;
+	u64			latency_limit_ns;
 	struct cpuidle_state_usage	states_usage[CPUIDLE_STATE_MAX];
 	struct cpuidle_state_kobj *kobjs[CPUIDLE_STATE_MAX];
 	struct cpuidle_driver_kobj *kobj_driver;
-- 
2.55.0


  reply	other threads:[~2026-09-15 13:13 UTC|newest]

Thread overview: 3+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-15 13:13 [PATCH RFC 0/1] cpuidle: add per-CPU latency_limit_ns sysfs attribute Anthony Harivel
2026-09-15 13:13 ` Anthony Harivel [this message]
2026-09-25 17:36   ` [PATCH 1/1] " Rafael J. Wysocki (Intel)

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260915131310.1053834-2-aharivel@redhat.com \
    --to=aharivel@redhat.com \
    --cc=daniel.lezcano@linaro.org \
    --cc=kvm@vger.kernel.org \
    --cc=linux-pm@vger.kernel.org \
    --cc=pbonzini@redhat.com \
    --cc=rafael@kernel.org \
    --cc=seanjc@google.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox