Linux Power Management development
 help / color / mirror / Atom feed
From: Baorui.Liu <baorliu@amd.com>
To: <linux-pm@vger.kernel.org>
Cc: <rafael@kernel.org>, <viresh.kumar@linaro.org>,
	<saravanak@kernel.org>, <mst@redhat.com>, <jasowangio@gmail.com>,
	<eperezma@redhat.com>, <xuanzhuo@linux.alibaba.com>,
	<virtualization@lists.linux.dev>, Baorui.Liu <baorliu@amd.com>
Subject: [RFC PATCH 1/1] cpufreq: virtio: add driver to report host frequency to guests
Date: Wed, 16 Sep 2026 15:37:35 +0800	[thread overview]
Message-ID: <20260916073735.541-2-baorliu@amd.com> (raw)
In-Reply-To: <20260916073735.541-1-baorliu@amd.com>

Some virtualization stacks pin each vCPU to a pCPU. Guest software
still needs the current host frequency for energy accounting, but
the guest cannot access host sysfs.

Add a virtio frontend that queries the host for the frequency of
the mapped pCPU and exposes it through the cpufreq .get() callback.
Frequency tables are placeholder OPPs; the driver does not change
the physical frequency.

RFC: virtio device id 42 is not allocated by the virtio spec yet.
Signed-off-by: Baorui.Liu <baorliu@amd.com>
---
 drivers/cpufreq/Kconfig          |  14 ++
 drivers/cpufreq/Makefile         |   1 +
 drivers/cpufreq/virtio-cpufreq.c | 235 +++++++++++++++++++++++++++++++
 include/uapi/linux/virtio_ids.h  |   1 +
 4 files changed, 251 insertions(+)
 create mode 100644 drivers/cpufreq/virtio-cpufreq.c

diff --git a/drivers/cpufreq/Kconfig b/drivers/cpufreq/Kconfig
index db83f3365698..ba0e38f5dce5 100644
--- a/drivers/cpufreq/Kconfig
+++ b/drivers/cpufreq/Kconfig
@@ -242,6 +242,20 @@ config CPUFREQ_VIRT
 
 	  If in doubt, say N.
 
+config CPUFREQ_VIRTIO
+	tristate "Virtio CPU frequency driver"
+	depends on VIRTIO
+	help
+	  This option adds a virtio cpufreq frontend. The guest sends a
+	  vCPU id to the host backend and receives the mapped physical
+	  CPU frequency in kHz through the cpufreq .get() callback.
+
+	  This is intended for virtualization stacks (for example Xen)
+	  that pin each vCPU to a pCPU and expose host frequency to the
+	  guest. It does not change the host frequency.
+
+	  If in doubt, say N.
+
 config CPUFREQ_DT_PLATDEV
 	bool "Generic DT based cpufreq platdev driver"
 	depends on OF
diff --git a/drivers/cpufreq/Makefile b/drivers/cpufreq/Makefile
index 6c7a39b7f8d2..9d9bb7ea1300 100644
--- a/drivers/cpufreq/Makefile
+++ b/drivers/cpufreq/Makefile
@@ -18,6 +18,7 @@ obj-$(CONFIG_CPUFREQ_DT)		+= cpufreq-dt.o
 obj-$(CONFIG_CPUFREQ_DT_RUST)		+= rcpufreq_dt.o
 obj-$(CONFIG_CPUFREQ_DT_PLATDEV)	+= cpufreq-dt-platdev.o
 obj-$(CONFIG_CPUFREQ_VIRT)		+= virtual-cpufreq.o
+obj-$(CONFIG_CPUFREQ_VIRTIO)		+= virtio-cpufreq.o
 
 # Traces
 CFLAGS_amd-pstate-trace.o               := -I$(src)
diff --git a/drivers/cpufreq/virtio-cpufreq.c b/drivers/cpufreq/virtio-cpufreq.c
new file mode 100644
index 000000000000..23fde862a313
--- /dev/null
+++ b/drivers/cpufreq/virtio-cpufreq.c
@@ -0,0 +1,235 @@
+// SPDX-License-Identifier: GPL-2.0-only
+/*
+ * Virtio cpufreq frontend
+ *
+ * Query the host for the frequency of the pCPU that currently runs a
+ * given vCPU. Used when the guest cannot read host cpufreq sysfs, for
+ * example a Xen domain that needs the value for energy accounting.
+ *
+ * The on-wire protocol is a packed { cpu_id, freq_khz } request that
+ * must match the QEMU virtio-cpufreq backend.
+ */
+
+#include <linux/completion.h>
+#include <linux/cpufreq.h>
+#include <linux/cpumask.h>
+#include <linux/module.h>
+#include <linux/mutex.h>
+#include <linux/scatterlist.h>
+#include <linux/slab.h>
+#include <linux/virtio.h>
+#include <linux/virtio_config.h>
+#include <linux/virtio_ids.h>
+
+#define DRV_NAME "virtio-cpufreq"
+
+struct virtio_cpufreq_req {
+	__le32 cpu_id;
+	__le32 freq_khz;
+} __packed;
+
+struct virtio_cpufreq {
+	struct virtio_device *vdev;
+	struct virtqueue *vq;
+	struct mutex lock; /* serializes virtqueue requests */
+};
+
+static struct virtio_cpufreq *virtio_cpufreq_dev;
+
+/*
+ * Placeholder OPPs so cpufreq has a table. Real platforms should
+ * populate this from the host or from firmware. Values match the
+ * existing QEMU/Android backend used for bring-up.
+ */
+static struct cpufreq_frequency_table virtio_freq_table[] = {
+	{ .frequency = 1400000 },
+	{ .frequency = 1700000 },
+	{ .frequency = 3000000 },
+	{ .frequency = CPUFREQ_TABLE_END },
+};
+
+static int virtio_cpufreq_init_policy(struct cpufreq_policy *policy)
+{
+	cpumask_clear(policy->cpus);
+	cpumask_set_cpu(policy->cpu, policy->cpus);
+
+	policy->freq_table = virtio_freq_table;
+	policy->cpuinfo.min_freq = 1400000;
+	policy->cpuinfo.max_freq = 3000000;
+	policy->cpuinfo.transition_latency = 0;
+
+	return 0;
+}
+
+static int virtio_cpufreq_target_index(struct cpufreq_policy *policy,
+				       unsigned int index)
+{
+	struct cpufreq_freqs freqs;
+
+	freqs.old = policy->cur;
+	freqs.new = policy->freq_table[index].frequency;
+
+	/*
+	 * This frontend does not change host frequency. It only keeps
+	 * the cpufreq core in sync so userspace can observe values.
+	 */
+	cpufreq_freq_transition_begin(policy, &freqs);
+	cpufreq_freq_transition_end(policy, &freqs, 0);
+
+	return 0;
+}
+
+static unsigned int virtio_cpufreq_fallback(unsigned int cpu)
+{
+	struct cpufreq_policy *policy;
+	unsigned int freq = 0;
+
+	policy = cpufreq_cpu_get(cpu);
+	if (policy) {
+		freq = policy->cur;
+		cpufreq_cpu_put(policy);
+	}
+
+	return freq;
+}
+
+static void virtio_cpufreq_vq_cb(struct virtqueue *vq)
+{
+	struct completion *done;
+	unsigned int len;
+
+	while ((done = virtqueue_get_buf(vq, &len)) != NULL)
+		complete(done);
+}
+
+static unsigned int virtio_cpufreq_get(unsigned int cpu)
+{
+	struct virtio_cpufreq *vc = virtio_cpufreq_dev;
+	struct virtio_cpufreq_req *req;
+	struct scatterlist out_sg, in_sg, *sgs[2];
+	struct completion done;
+	unsigned int freq_khz;
+	int ret;
+
+	if (!vc || !vc->vq)
+		return virtio_cpufreq_fallback(cpu);
+
+	req = kzalloc_obj(*req, GFP_KERNEL);
+	if (!req)
+		return virtio_cpufreq_fallback(cpu);
+
+	req->cpu_id = cpu_to_le32(cpu);
+
+	init_completion(&done);
+	sg_init_one(&out_sg, req, sizeof(*req));
+	sg_init_one(&in_sg, req, sizeof(*req));
+	sgs[0] = &out_sg;
+	sgs[1] = &in_sg;
+
+	mutex_lock(&vc->lock);
+	ret = virtqueue_add_sgs(vc->vq, sgs, 1, 1, &done, GFP_KERNEL);
+	if (ret) {
+		mutex_unlock(&vc->lock);
+		kfree(req);
+		return virtio_cpufreq_fallback(cpu);
+	}
+
+	virtqueue_kick(vc->vq);
+	ret = wait_for_completion_timeout(&done, msecs_to_jiffies(1000));
+	mutex_unlock(&vc->lock);
+
+	if (!ret) {
+		/*
+		 * The buffer may still be on the virtqueue. Leak it
+		 * rather than freeing while the host may still write.
+		 */
+		return virtio_cpufreq_fallback(cpu);
+	}
+
+	freq_khz = le32_to_cpu(req->freq_khz);
+	kfree(req);
+
+	if (!freq_khz)
+		return virtio_cpufreq_fallback(cpu);
+
+	return freq_khz;
+}
+
+static struct cpufreq_driver virtio_cpufreq_driver = {
+	.name		= DRV_NAME,
+	.flags		= CPUFREQ_CONST_LOOPS,
+	.init		= virtio_cpufreq_init_policy,
+	.verify		= cpufreq_generic_frequency_table_verify,
+	.target_index	= virtio_cpufreq_target_index,
+	.get		= virtio_cpufreq_get,
+	.attr		= cpufreq_generic_attr,
+};
+
+static int virtio_cpufreq_probe(struct virtio_device *vdev)
+{
+	struct virtio_cpufreq *vc;
+	struct virtqueue *vq;
+	int ret;
+
+	vc = devm_kzalloc(&vdev->dev, sizeof(*vc), GFP_KERNEL);
+	if (!vc)
+		return -ENOMEM;
+
+	mutex_init(&vc->lock);
+	vc->vdev = vdev;
+
+	vq = virtio_find_single_vq(vdev, virtio_cpufreq_vq_cb, "requests");
+	if (IS_ERR(vq))
+		return PTR_ERR(vq);
+
+	vc->vq = vq;
+	vdev->priv = vc;
+	virtio_cpufreq_dev = vc;
+	virtio_device_ready(vdev);
+
+	ret = cpufreq_register_driver(&virtio_cpufreq_driver);
+	if (ret) {
+		vdev->config->del_vqs(vdev);
+		virtio_cpufreq_dev = NULL;
+		return ret;
+	}
+
+	return 0;
+}
+
+static void virtio_cpufreq_remove(struct virtio_device *vdev)
+{
+	cpufreq_unregister_driver(&virtio_cpufreq_driver);
+	virtio_cpufreq_dev = NULL;
+	vdev->config->del_vqs(vdev);
+}
+
+static const struct virtio_device_id id_table[] = {
+	{ VIRTIO_ID_CPUFREQ, VIRTIO_DEV_ANY_ID },
+	{ 0 },
+};
+
+static struct virtio_driver virtio_cpufreq_virtio_driver = {
+	.driver.name	= DRV_NAME,
+	.driver.owner	= THIS_MODULE,
+	.id_table	= id_table,
+	.probe		= virtio_cpufreq_probe,
+	.remove		= virtio_cpufreq_remove,
+};
+
+static int __init virtio_cpufreq_mod_init(void)
+{
+	return register_virtio_driver(&virtio_cpufreq_virtio_driver);
+}
+
+static void __exit virtio_cpufreq_mod_exit(void)
+{
+	unregister_virtio_driver(&virtio_cpufreq_virtio_driver);
+}
+
+module_init(virtio_cpufreq_mod_init);
+module_exit(virtio_cpufreq_mod_exit);
+
+MODULE_DEVICE_TABLE(virtio, id_table);
+MODULE_DESCRIPTION("Virtio cpufreq frontend");
+MODULE_LICENSE("GPL");
diff --git a/include/uapi/linux/virtio_ids.h b/include/uapi/linux/virtio_ids.h
index f9056af0c622..f7e357f3bf94 100644
--- a/include/uapi/linux/virtio_ids.h
+++ b/include/uapi/linux/virtio_ids.h
@@ -68,6 +68,7 @@
 #define VIRTIO_ID_AUDIO_POLICY		39 /* virtio audio policy */
 #define VIRTIO_ID_BT			40 /* virtio bluetooth */
 #define VIRTIO_ID_GPIO			41 /* virtio gpio */
+#define VIRTIO_ID_CPUFREQ		42 /* virtio cpufreq */
 #define VIRTIO_ID_SPI			45 /* virtio spi */
 #define VIRTIO_ID_MEDIA			48 /* virtio media */
 
-- 
2.34.1


  reply	other threads:[~2026-09-16  7:38 UTC|newest]

Thread overview: 6+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-16  7:37 [RFC PATCH 0/1] cpufreq: virtio frontend to report host CPU frequency Baorui.Liu
2026-09-16  7:37 ` Baorui.Liu [this message]
2026-09-22  9:02 ` [RFC PATCH v2 " Baorui.Liu
2026-09-22  9:22   ` Michael S. Tsirkin
2026-09-22  9:02 ` [RFC PATCH v2 1/1] cpufreq: virtio: add driver to report host frequency to guests Baorui.Liu
2026-09-24  7:18   ` Saravana Kannan

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260916073735.541-2-baorliu@amd.com \
    --to=baorliu@amd.com \
    --cc=eperezma@redhat.com \
    --cc=jasowangio@gmail.com \
    --cc=linux-pm@vger.kernel.org \
    --cc=mst@redhat.com \
    --cc=rafael@kernel.org \
    --cc=saravanak@kernel.org \
    --cc=viresh.kumar@linaro.org \
    --cc=virtualization@lists.linux.dev \
    --cc=xuanzhuo@linux.alibaba.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox