From: Joshua Yeong <joshua.yeong@starfivetech.com>
To: robh@kernel.org, krzk+dt@kernel.org, conor+dt@kernel.org,
pjw@kernel.org, palmer@dabbelt.com, aou@eecs.berkeley.edu,
rafael@kernel.org, viresh.kumar@linaro.org, ulfh@kernel.org,
rahul@summations.net, anup@brainfault.org, lftan.linux@gmail.com
Cc: alex@ghiti.fr, joshua.yeong@starfivetech.com,
linux-riscv@lists.infradead.org, devicetree@vger.kernel.org,
linux-pm@vger.kernel.org, linux-kernel@vger.kernel.org
Subject: [PATCH v2 5/7] cpufreq: Add RISC-V RPMI cpufreq driver
Date: Thu, 8 Oct 2026 17:10:29 +0800 [thread overview]
Message-ID: <20261008091032.2832333-6-joshua.yeong@starfivetech.com> (raw)
In-Reply-To: <20261008091032.2832333-1-joshua.yeong@starfivetech.com>
Add a cpufreq driver for the RISC-V RPMI performance domains that CPUs
name through "performance-domains". The CPUs that name the same domain
share a policy, the levels the domain advertises become its frequency
table, and the level is set through the domain's fast channel when it
has one, so that the governor can switch frequency from the scheduler.
An energy model is registered from the power cost of each level.
The driver is a front-end over the RPMI performance service group core,
which creates the device it binds to when a CPU names the provider.
Signed-off-by: Joshua Yeong <joshua.yeong@starfivetech.com>
---
drivers/cpufreq/Kconfig | 16 +
drivers/cpufreq/Makefile | 4 +
drivers/cpufreq/riscv-rpmi-cpufreq.c | 294 ++++++++++++++++++
.../firmware/riscv/riscv-rpmi-performance.c | 55 ++++
4 files changed, 369 insertions(+)
create mode 100644 drivers/cpufreq/riscv-rpmi-cpufreq.c
diff --git a/drivers/cpufreq/Kconfig b/drivers/cpufreq/Kconfig
index db83f3365698..f012b8268b5e 100644
--- a/drivers/cpufreq/Kconfig
+++ b/drivers/cpufreq/Kconfig
@@ -364,6 +364,22 @@ config ACPI_CPPC_CPUFREQ_FIE
If in doubt, say N.
+config RISCV_RPMI_CPUFREQ
+ tristate "RISC-V RPMI Based CPUFreq driver"
+ depends on RISCV_RPMI_PERFORMANCE
+ default RISCV
+ select PM_OPP
+ help
+ This adds the CPUfreq driver support for RISC-V platforms whose CPU
+ performance is managed through the RPMI performance service group.
+ CPUs reference their domain through the "performance-domains"
+ property. The performance domains it drives are enumerated by the
+ RPMI performance service group core, which is what owns the mailbox
+ channel.
+
+ To compile this driver as a module, choose M here: the
+ module will be called riscv-rpmi-cpufreq.
+
endif # CPU_FREQ
endmenu
diff --git a/drivers/cpufreq/Makefile b/drivers/cpufreq/Makefile
index 6c7a39b7f8d2..265a6bccef4b 100644
--- a/drivers/cpufreq/Makefile
+++ b/drivers/cpufreq/Makefile
@@ -96,6 +96,10 @@ obj-$(CONFIG_CPU_FREQ_PMAC64) += pmac64-cpufreq.o
obj-$(CONFIG_PPC_PASEMI_CPUFREQ) += pasemi-cpufreq.o
obj-$(CONFIG_POWERNV_CPUFREQ) += powernv-cpufreq.o
+##################################################################################
+# RISC-V platform drivers
+obj-$(CONFIG_RISCV_RPMI_CPUFREQ) += riscv-rpmi-cpufreq.o
+
##################################################################################
# Other platform drivers
obj-$(CONFIG_BMIPS_CPUFREQ) += bmips-cpufreq.o
diff --git a/drivers/cpufreq/riscv-rpmi-cpufreq.c b/drivers/cpufreq/riscv-rpmi-cpufreq.c
new file mode 100644
index 000000000000..1a5d0248f879
--- /dev/null
+++ b/drivers/cpufreq/riscv-rpmi-cpufreq.c
@@ -0,0 +1,294 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * RISC-V RPMI Based CPUFreq Driver
+ *
+ * Copyright (C) 2026 Shanghai StarFive Technology Co., Ltd.
+ *
+ * Drives the performance domains that CPUs share. The RPMI protocol and the
+ * domain enumeration live in the performance service group core, which owns
+ * the mailbox channel and creates the device this driver binds to.
+ */
+
+#define pr_fmt(fmt) "riscv-rpmi-cpufreq: " fmt
+
+#include <linux/cpufreq.h>
+#include <linux/energy_model.h>
+#include <linux/firmware/riscv/riscv-rpmi-performance.h>
+#include <linux/of.h>
+#include <linux/platform_device.h>
+#include <linux/pm_opp.h>
+#include <linux/slab.h>
+
+struct rpmi_perf_cpufreq_data {
+ int nr_opp;
+ struct device *cpu_dev;
+ struct rpmi_perf_domain *domain;
+};
+
+static int rpmi_perf_set_target_index(struct cpufreq_policy *policy, unsigned int index)
+{
+ struct rpmi_perf_cpufreq_data *data = policy->driver_data;
+ u32 level;
+ int ret;
+
+ /*
+ * cpufreq indexes its own frequency table, which is not the same thing
+ * as an RPMI level index. Go through the frequency so that the two
+ * only have to agree on what they mean, not on how they are numbered.
+ */
+ ret = rpmi_perf_domain_freq_to_level(data->domain,
+ policy->freq_table[index].frequency,
+ &level);
+ if (ret)
+ return ret;
+
+ if (rpmi_perf_domain_has_fast_channel(data->domain))
+ return rpmi_perf_domain_set_level_fast(data->domain, level);
+
+ return rpmi_perf_domain_set_level(data->domain, level);
+}
+
+static unsigned int rpmi_perf_fast_switch(struct cpufreq_policy *policy,
+ unsigned int target_freq)
+{
+ struct rpmi_perf_cpufreq_data *data = policy->driver_data;
+ u32 level;
+
+ if (rpmi_perf_domain_freq_to_level(data->domain, target_freq, &level))
+ return 0;
+
+ if (rpmi_perf_domain_set_level_fast(data->domain, level))
+ return 0;
+
+ return target_freq;
+}
+
+static unsigned int rpmi_perf_get_rate(unsigned int cpu)
+{
+ struct cpufreq_policy *policy = cpufreq_cpu_get_raw(cpu);
+ struct rpmi_perf_cpufreq_data *data;
+ u32 cpufreq, level;
+
+ if (!policy)
+ return 0;
+
+ data = policy->driver_data;
+
+ if (rpmi_perf_domain_get_level(data->domain, &level))
+ return 0;
+
+ if (rpmi_perf_domain_level_to_freq(data->domain, level, &cpufreq))
+ return 0;
+
+ return cpufreq;
+}
+
+static int rpmi_perf_init(struct cpufreq_policy *policy)
+{
+ struct cpufreq_frequency_table *freq_table;
+ struct rpmi_perf_cpufreq_data *data;
+ struct rpmi_perf_domain *domain;
+ struct platform_device *pdev = cpufreq_get_driver_data();
+ struct rpmi_perf **mpxy_perf = dev_get_platdata(&pdev->dev);
+ struct of_phandle_args args;
+ int ret, nr_opp;
+ struct device *cpu_dev;
+
+ cpu_dev = get_cpu_device(policy->cpu);
+ if (!cpu_dev) {
+ pr_err("failed to get cpu%d device\n", policy->cpu);
+ return -ENODEV;
+ }
+
+ data = kzalloc(sizeof(*data), GFP_KERNEL);
+ if (!data)
+ return -ENOMEM;
+
+ ret = of_perf_domain_get_sharing_cpumask(policy->cpu,
+ "performance-domains",
+ "#performance-domain-cells",
+ policy->cpus, &args);
+ if (ret) {
+ dev_err(cpu_dev, "%s: failed to get performance domain info: %d\n",
+ __func__, ret);
+ goto out_free_priv;
+ }
+
+ /* A domain ID only means something to the provider the CPU names. */
+ if (args.np != dev_of_node(pdev->dev.parent)) {
+ dev_err(cpu_dev, "performance domain of %pOF, not of %pOF\n",
+ args.np, dev_of_node(pdev->dev.parent));
+ of_node_put(args.np);
+ ret = -ENODEV;
+ goto out_free_priv;
+ }
+
+ domain = rpmi_perf_domain_by_id(*mpxy_perf, args.args[0]);
+ of_node_put(args.np);
+ if (!domain) {
+ dev_err(cpu_dev, "performance domain %u is not usable\n",
+ args.args[0]);
+ ret = -EINVAL;
+ goto out_free_priv;
+ }
+
+ ret = rpmi_perf_domain_opps_add(domain, cpu_dev);
+ if (ret) {
+ dev_warn(cpu_dev, "failed to add opps to the device\n");
+ goto out_free_priv;
+ }
+
+ nr_opp = dev_pm_opp_get_opp_count(cpu_dev);
+ if (nr_opp <= 0) {
+ dev_err(cpu_dev, "performance domain has no operating points\n");
+ ret = -ENODEV;
+ goto out_free_opp;
+ }
+
+ ret = dev_pm_opp_init_cpufreq_table(cpu_dev, &freq_table);
+ if (ret) {
+ dev_err(cpu_dev, "failed to init cpufreq table: %d\n", ret);
+ goto out_free_opp;
+ }
+
+ data->cpu_dev = cpu_dev;
+ data->nr_opp = nr_opp;
+ data->domain = domain;
+
+ /* Allow DVFS request for any domain from any CPU */
+ policy->dvfs_possible_from_any_cpu = true;
+ policy->driver_data = data;
+ policy->freq_table = freq_table;
+
+ policy->cpuinfo.transition_latency =
+ rpmi_perf_domain_trans_latency_us(domain) * 1000;
+ policy->fast_switch_possible = rpmi_perf_domain_has_fast_channel(domain);
+
+ return 0;
+
+out_free_opp:
+ dev_pm_opp_remove_all_dynamic(cpu_dev);
+
+out_free_priv:
+ kfree(data);
+
+ return ret;
+}
+
+static void rpmi_perf_exit(struct cpufreq_policy *policy)
+{
+ struct rpmi_perf_cpufreq_data *data = policy->driver_data;
+
+ dev_pm_opp_free_cpufreq_table(data->cpu_dev, &policy->freq_table);
+ dev_pm_opp_remove_all_dynamic(data->cpu_dev);
+ kfree(data);
+}
+
+static int __maybe_unused
+rpmi_perf_get_cpu_power(struct device *cpu_dev, unsigned long *uW,
+ unsigned long *kHz)
+{
+ struct rpmi_perf_cpufreq_data *data;
+ struct rpmi_perf_level level;
+ struct cpufreq_policy *policy;
+ u32 idx, count;
+
+ policy = cpufreq_cpu_get_raw(cpu_dev->id);
+ if (!policy)
+ return -EINVAL;
+
+ data = policy->driver_data;
+ count = rpmi_perf_domain_level_count(data->domain);
+
+ /* The levels are in ascending order of frequency. */
+ for (idx = 0; idx < count; idx++) {
+ if (rpmi_perf_domain_level_info(data->domain, idx, &level))
+ return -EINVAL;
+
+ if (level.clock_freq < *kHz)
+ continue;
+
+ *uW = level.power_cost;
+ *kHz = level.clock_freq;
+ return 0;
+ }
+
+ /* No level at or above *kHz: there is no state to describe. */
+ return -EINVAL;
+}
+
+static void rpmi_perf_register_em(struct cpufreq_policy *policy)
+{
+ struct em_data_callback em_cb = EM_DATA_CB(rpmi_perf_get_cpu_power);
+ struct rpmi_perf_cpufreq_data *data = policy->driver_data;
+
+ em_dev_register_perf_domain(get_cpu_device(policy->cpu), data->nr_opp,
+ &em_cb, policy->cpus, true);
+}
+
+static struct cpufreq_driver rpmi_perf_cpufreq_driver = {
+ .name = "rpmi-cpufreq",
+ .flags = CPUFREQ_HAVE_GOVERNOR_PER_POLICY |
+ CPUFREQ_NEED_INITIAL_FREQ_CHECK |
+ CPUFREQ_IS_COOLING_DEV,
+ .verify = cpufreq_generic_frequency_table_verify,
+ .target_index = rpmi_perf_set_target_index,
+ .fast_switch = rpmi_perf_fast_switch,
+ .get = rpmi_perf_get_rate,
+ .init = rpmi_perf_init,
+ .exit = rpmi_perf_exit,
+ .register_em = rpmi_perf_register_em,
+};
+
+static int rpmi_cpufreq_probe(struct platform_device *pdev)
+{
+ struct rpmi_perf **mpxy_perf = dev_get_platdata(&pdev->dev);
+ struct device *dev = &pdev->dev;
+ int ret;
+
+ if (!mpxy_perf || !*mpxy_perf)
+ return -EINVAL;
+
+ /*
+ * There is one cpufreq driver for the whole system, so only one
+ * provider can drive the CPUs. Refuse another one rather than point
+ * the registered driver at its domains.
+ */
+ if (rpmi_perf_cpufreq_driver.driver_data)
+ return dev_err_probe(dev, -EBUSY,
+ "CPUs are already driven by another provider\n");
+
+ rpmi_perf_cpufreq_driver.driver_data = pdev;
+
+ ret = cpufreq_register_driver(&rpmi_perf_cpufreq_driver);
+ if (ret) {
+ rpmi_perf_cpufreq_driver.driver_data = NULL;
+ return dev_err_probe(dev, ret, "registering cpufreq failed\n");
+ }
+
+ dev_info(dev, "%d MPXY cpufreq domains registered\n",
+ rpmi_perf_num_domains(*mpxy_perf));
+
+ return 0;
+}
+
+static void rpmi_cpufreq_remove(struct platform_device *pdev)
+{
+ cpufreq_unregister_driver(&rpmi_perf_cpufreq_driver);
+ rpmi_perf_cpufreq_driver.driver_data = NULL;
+}
+
+static struct platform_driver rpmi_cpufreq_platdrv = {
+ .driver = {
+ .name = "riscv-rpmi-cpufreq",
+ },
+ .probe = rpmi_cpufreq_probe,
+ .remove = rpmi_cpufreq_remove,
+};
+
+module_platform_driver(rpmi_cpufreq_platdrv);
+
+MODULE_ALIAS("platform:riscv-rpmi-cpufreq");
+MODULE_AUTHOR("Joshua Yeong <joshua.yeong@starfivetech.com>");
+MODULE_DESCRIPTION("RISC-V RPMI performance cpufreq driver");
+MODULE_LICENSE("GPL");
diff --git a/drivers/firmware/riscv/riscv-rpmi-performance.c b/drivers/firmware/riscv/riscv-rpmi-performance.c
index d44802f07ab0..f91f2d861dc0 100644
--- a/drivers/firmware/riscv/riscv-rpmi-performance.c
+++ b/drivers/firmware/riscv/riscv-rpmi-performance.c
@@ -19,6 +19,7 @@
#include <linux/mailbox/riscv-rpmi-message.h>
#include <linux/module.h>
#include <linux/mutex.h>
+#include <linux/of.h>
#include <linux/platform_device.h>
#include <linux/pm_opp.h>
#include <linux/slab.h>
@@ -995,6 +996,56 @@ static void rpmi_perf_mbox_chan_release(void *data)
mbox_free_channel((struct mbox_chan *)data);
}
+/*
+ * cpufreq has no device tree node of its own: a CPU names its domain through
+ * "performance-domains" on the CPU node. Create the device it binds to, but
+ * only when some CPU actually points back here.
+ */
+static bool rpmi_perf_cpus_present(struct device *dev)
+{
+ struct device_node *cpu_np;
+ struct of_phandle_args args;
+ int ret;
+
+ for_each_of_cpu_node(cpu_np) {
+ ret = of_parse_phandle_with_args(cpu_np, "performance-domains",
+ "#performance-domain-cells", 0,
+ &args);
+ if (ret)
+ continue;
+
+ if (args.np == dev_of_node(dev)) {
+ of_node_put(args.np);
+ of_node_put(cpu_np);
+ return true;
+ }
+ of_node_put(args.np);
+ }
+
+ return false;
+}
+
+static void rpmi_perf_frontend_unregister(void *data)
+{
+ platform_device_unregister((struct platform_device *)data);
+}
+
+static int rpmi_perf_cpufreq_register(struct device *dev, struct rpmi_perf *perf)
+{
+ struct platform_device *pdev;
+
+ if (!IS_ENABLED(CONFIG_RISCV_RPMI_CPUFREQ) || !rpmi_perf_cpus_present(dev))
+ return 0;
+
+ pdev = platform_device_register_data(dev, "riscv-rpmi-cpufreq",
+ PLATFORM_DEVID_AUTO, &perf,
+ sizeof(perf));
+ if (IS_ERR(pdev))
+ return PTR_ERR(pdev);
+
+ return devm_add_action_or_reset(dev, rpmi_perf_frontend_unregister, pdev);
+}
+
static int rpmi_perf_probe(struct platform_device *pdev)
{
struct device *dev = &pdev->dev;
@@ -1078,6 +1129,10 @@ static int rpmi_perf_probe(struct platform_device *pdev)
dev_set_drvdata(dev, mpxy_perf);
+ ret = rpmi_perf_cpufreq_register(dev, mpxy_perf);
+ if (ret)
+ return dev_err_probe(dev, ret, "failed to register cpufreq device\n");
+
dev_info(dev, "%d MPXY performance domains registered\n", num_domains);
return 0;
--
2.43.0
next prev parent reply other threads:[~2026-10-08 9:12 UTC|newest]
Thread overview: 12+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-10-08 9:10 [PATCH v2 0/7] Add RISC-V RPMI performance service support Joshua Yeong
2026-10-08 9:10 ` [PATCH v2 1/7] dt-bindings: dvfs: Add RPMI performance service message proxy bindings Joshua Yeong
2026-10-08 9:10 ` [PATCH v2 2/7] dt-bindings: dvfs: Add RPMI performance service bindings Joshua Yeong
2026-10-08 10:41 ` Conor Dooley
2026-10-08 9:10 ` [PATCH v2 3/7] dt-bindings: riscv: cpus: document performance-domains property Joshua Yeong
2026-10-08 9:10 ` [PATCH v2 4/7] firmware: riscv: Add RPMI performance service Joshua Yeong
2026-10-08 9:23 ` sashiko-bot
2026-10-08 9:10 ` Joshua Yeong [this message]
2026-10-08 9:23 ` [PATCH v2 5/7] cpufreq: Add RISC-V RPMI cpufreq driver sashiko-bot
2026-10-08 9:10 ` [PATCH v2 6/7] pmdomain: riscv: Add RPMI performance domains as power domains Joshua Yeong
2026-10-08 9:27 ` sashiko-bot
2026-10-08 9:10 ` [PATCH v2 7/7] MAINTAINERS: Add RISC-V RPMI performance driver Joshua Yeong
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20261008091032.2832333-6-joshua.yeong@starfivetech.com \
--to=joshua.yeong@starfivetech.com \
--cc=alex@ghiti.fr \
--cc=anup@brainfault.org \
--cc=aou@eecs.berkeley.edu \
--cc=conor+dt@kernel.org \
--cc=devicetree@vger.kernel.org \
--cc=krzk+dt@kernel.org \
--cc=lftan.linux@gmail.com \
--cc=linux-kernel@vger.kernel.org \
--cc=linux-pm@vger.kernel.org \
--cc=linux-riscv@lists.infradead.org \
--cc=palmer@dabbelt.com \
--cc=pjw@kernel.org \
--cc=rafael@kernel.org \
--cc=rahul@summations.net \
--cc=robh@kernel.org \
--cc=ulfh@kernel.org \
--cc=viresh.kumar@linaro.org \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox