* [RFC PATCH 1/5] iommu: Add per-device requestor QoS tagging and lookup helpers
2026-09-01 14:07 [RFC PATCH 0/5] resctrl: Assign devices to resource groups via IOMMU DMA QoS tagging Qinxin Xia
@ 2026-09-01 14:07 ` Qinxin Xia
2026-09-01 14:07 ` [RFC PATCH 2/5] arm_mpam: resctrl: Add arch query for device DMA QoS support Qinxin Xia
` (3 subsequent siblings)
4 siblings, 0 replies; 7+ messages in thread
From: Qinxin Xia @ 2026-09-01 14:07 UTC (permalink / raw)
To: ben.horgan, zhangzhanpeng.jasper, joro, palmer, tony.luck,
reinette.chatre, tomasz.jeznach, zengheng4, fustini, cuiyunhui,
wangzhou1, xiaqinxin
Cc: will, robin.murphy, pjw, aou, alex, Dave.Martin, james.morse,
babu.moger, corbet, shuah, jgg, kevin.tian, yuanzhu, iommu,
linuxarm, baolin.wang
Allow a device's DMA to be tagged with a QoS requestor ID so resctrl can
assign the device to a resource group.
Add helpers to program the requestor ID, track group membership changes,
and look up a device by name.
Signed-off-by: Qinxin Xia <xiaqinxin@huawei.com>
---
drivers/iommu/iommu.c | 113 ++++++++++++++++++++++++++++++++++++++++++
include/linux/iommu.h | 29 +++++++++++
2 files changed, 142 insertions(+)
diff --git a/drivers/iommu/iommu.c b/drivers/iommu/iommu.c
index cd1bca7ede9a..2f3739a755d6 100644
--- a/drivers/iommu/iommu.c
+++ b/drivers/iommu/iommu.c
@@ -43,6 +43,8 @@ static struct kset *iommu_group_kset;
static DEFINE_IDA(iommu_group_ida);
static DEFINE_IDA(iommu_global_pasid_ida);
+static const struct iommu_qos_device_ops *iommu_qos_dev_ops;
+
static unsigned int iommu_def_domain_type __read_mostly;
static bool iommu_dma_strict __read_mostly = IS_ENABLED(CONFIG_IOMMU_DEFAULT_DMA_STRICT);
static u32 iommu_cmd_line __read_mostly;
@@ -728,7 +730,10 @@ static void __iommu_group_free_device(struct iommu_group *group,
struct group_device *grp_dev)
{
struct device *dev = grp_dev->dev;
+ const struct iommu_qos_device_ops *qos_ops = READ_ONCE(iommu_qos_dev_ops);
+ if (qos_ops)
+ qos_ops->remove(dev);
sysfs_remove_link(group->devices_kobj, grp_dev->name);
sysfs_remove_link(&dev->kobj, "iommu_group");
@@ -1266,6 +1271,7 @@ static int iommu_create_device_direct_mappings(struct iommu_domain *domain,
static struct group_device *iommu_group_alloc_device(struct iommu_group *group,
struct device *dev)
{
+ const struct iommu_qos_device_ops *qos_ops;
int ret, i = 0;
struct group_device *device;
@@ -1304,6 +1310,9 @@ static struct group_device *iommu_group_alloc_device(struct iommu_group *group,
trace_add_device_to_group(group->id, dev);
+ qos_ops = READ_ONCE(iommu_qos_dev_ops);
+ if (qos_ops)
+ qos_ops->add(dev);
dev_info(dev, "Adding to iommu group %d\n", group->id);
return device;
@@ -4222,6 +4231,110 @@ void pci_dev_reset_iommu_done(struct pci_dev *pdev)
}
EXPORT_SYMBOL_GPL(pci_dev_reset_iommu_done);
+static struct iommu_group *iommu_group_kset_next_get(struct iommu_group *prev)
+{
+ struct iommu_group *group = NULL;
+ struct kobject *kobj;
+
+ if (!iommu_group_kset)
+ return NULL;
+
+ spin_lock(&iommu_group_kset->list_lock);
+ kobj = prev ? list_next_entry(&prev->kobj, entry) :
+ list_first_entry_or_null(&iommu_group_kset->list,
+ struct kobject, entry);
+ if (kobj) {
+ list_for_each_entry_from(kobj, &iommu_group_kset->list, entry) {
+ group = container_of(kobj, struct iommu_group, kobj);
+
+ /* Skip groups already on their way out (refcount 0). */
+ if (kobject_get_unless_zero(&group->kobj))
+ break;
+ group = NULL;
+ }
+ }
+ spin_unlock(&iommu_group_kset->list_lock);
+
+ return group;
+}
+
+struct device *iommu_group_find_device_by_name(const char *name)
+{
+ struct iommu_group *group, *next;
+ struct group_device *gdev;
+ struct device *dev = NULL;
+
+ if (!name)
+ return NULL;
+
+ for (group = iommu_group_kset_next_get(NULL); group; group = next) {
+ mutex_lock(&group->mutex);
+ for_each_group_device(group, gdev) {
+ if (!strcmp(gdev->name, name)) {
+ /*
+ * Take a reference while still under
+ * group->mutex: device removal also takes this
+ * mutex before freeing the device, so the
+ * pointer cannot vanish until we hold a
+ * reference. The caller must put_device().
+ */
+ dev = get_device(gdev->dev);
+ break;
+ }
+ }
+ mutex_unlock(&group->mutex);
+
+ next = dev ? NULL : iommu_group_kset_next_get(group);
+ kobject_put(&group->kobj);
+ if (dev)
+ break;
+ }
+
+ return dev;
+}
+EXPORT_SYMBOL_GPL(iommu_group_find_device_by_name);
+
+int iommu_set_dev_requestor_id(struct device *dev, u32 requestor_id, u8 pmg)
+{
+ const struct iommu_ops *ops;
+
+ if (!dev_has_iommu(dev))
+ return -EOPNOTSUPP;
+
+ ops = dev_iommu_ops(dev);
+ if (!ops->set_dev_requestor_id)
+ return -EOPNOTSUPP;
+
+ return ops->set_dev_requestor_id(dev, requestor_id, pmg);
+}
+EXPORT_SYMBOL_GPL(iommu_set_dev_requestor_id);
+
+void iommu_register_qos_device_ops(const struct iommu_qos_device_ops *ops)
+{
+ struct iommu_group *group, *next;
+ struct group_device *gdev;
+
+ /* Publish the ops first so newly probed devices see them. */
+ WRITE_ONCE(iommu_qos_dev_ops, ops);
+
+ /*
+ * Registration happens late (e.g. from resctrl's fs_initcall), after
+ * early IOMMU devices have already been added to their groups. Replay
+ * the add() callback for every device that is currently on a group so
+ * none are missed.
+ */
+ for (group = iommu_group_kset_next_get(NULL); group; group = next) {
+ mutex_lock(&group->mutex);
+ for_each_group_device(group, gdev)
+ ops->add(gdev->dev);
+ mutex_unlock(&group->mutex);
+
+ next = iommu_group_kset_next_get(group);
+ kobject_put(&group->kobj);
+ }
+}
+EXPORT_SYMBOL_GPL(iommu_register_qos_device_ops);
+
#if IS_ENABLED(CONFIG_IRQ_MSI_IOMMU)
/**
* iommu_dma_prepare_msi() - Map the MSI page in the IOMMU domain
diff --git a/include/linux/iommu.h b/include/linux/iommu.h
index ac43b8b93f14..20bb7f381064 100644
--- a/include/linux/iommu.h
+++ b/include/linux/iommu.h
@@ -340,6 +340,11 @@ struct iommu_pages_list {
#define IOMMU_PAGES_LIST_INIT(name) \
((struct iommu_pages_list){ .pages = LIST_HEAD_INIT(name.pages) })
+struct iommu_qos_device_ops {
+ void (*add)(struct device *dev);
+ void (*remove)(struct device *dev);
+};
+
#ifdef CONFIG_IOMMU_API
/**
@@ -735,6 +740,9 @@ struct iommu_ops {
struct iommu_domain *parent_domain,
const struct iommu_user_data *user_data);
+ int (*set_dev_requestor_id)(struct device *dev, u32 requestor_id,
+ u8 pmg);
+
const struct iommu_domain_ops *default_domain_ops;
struct module *owner;
struct iommu_domain *identity_domain;
@@ -1225,6 +1233,10 @@ void iommu_free_global_pasid(ioasid_t pasid);
/* PCI device reset functions */
int pci_dev_reset_iommu_prepare(struct pci_dev *pdev);
void pci_dev_reset_iommu_done(struct pci_dev *pdev);
+
+struct device *iommu_group_find_device_by_name(const char *name);
+int iommu_set_dev_requestor_id(struct device *dev, u32 requestor_id, u8 pmg);
+void iommu_register_qos_device_ops(const struct iommu_qos_device_ops *ops);
#else /* CONFIG_IOMMU_API */
struct iommu_ops {};
@@ -1557,6 +1569,23 @@ static inline int pci_dev_reset_iommu_prepare(struct pci_dev *pdev)
static inline void pci_dev_reset_iommu_done(struct pci_dev *pdev)
{
}
+
+static inline struct device *iommu_group_find_device_by_name(const char *name)
+{
+ return NULL;
+}
+
+static inline int iommu_set_dev_requestor_id(struct device *dev,
+ u32 requestor_id, u8 pmg)
+{
+ return -EOPNOTSUPP;
+}
+
+static inline void
+iommu_register_qos_device_ops(const struct iommu_qos_device_ops *ops)
+{
+}
+
#endif /* CONFIG_IOMMU_API */
#ifdef CONFIG_IRQ_MSI_IOMMU
--
2.33.0
^ permalink raw reply related [flat|nested] 7+ messages in thread* [RFC PATCH 2/5] arm_mpam: resctrl: Add arch query for device DMA QoS support
2026-09-01 14:07 [RFC PATCH 0/5] resctrl: Assign devices to resource groups via IOMMU DMA QoS tagging Qinxin Xia
2026-09-01 14:07 ` [RFC PATCH 1/5] iommu: Add per-device requestor QoS tagging and lookup helpers Qinxin Xia
@ 2026-09-01 14:07 ` Qinxin Xia
2026-09-01 14:08 ` [RFC PATCH 3/5] iommu/arm-smmu-v3: Support MPAM device DMA QoS tagging Qinxin Xia
` (2 subsequent siblings)
4 siblings, 0 replies; 7+ messages in thread
From: Qinxin Xia @ 2026-09-01 14:07 UTC (permalink / raw)
To: ben.horgan, zhangzhanpeng.jasper, joro, palmer, tony.luck,
reinette.chatre, tomasz.jeznach, zengheng4, fustini, cuiyunhui,
wangzhou1, xiaqinxin
Cc: will, robin.murphy, pjw, aou, alex, Dave.Martin, james.morse,
babu.moger, corbet, shuah, jgg, kevin.tian, yuanzhu, iommu,
linuxarm, baolin.wang
Add an architecture query so core resctrl can decide whether to expose the
"devices" file.
The file is only meaningful when an IOMMU can tag the DMA of devices behind
it with a QoS class (such as an MPAM PARTID/PMG), so the query reports
support only on platforms where that is available.
Signed-off-by: Qinxin Xia <xiaqinxin@huawei.com>
---
arch/x86/include/asm/resctrl.h | 5 +++++
drivers/resctrl/mpam_devices.c | 20 ++++++++++++++++++++
drivers/resctrl/mpam_resctrl.c | 5 +++++
include/linux/arm_mpam.h | 5 +++++
4 files changed, 35 insertions(+)
diff --git a/arch/x86/include/asm/resctrl.h b/arch/x86/include/asm/resctrl.h
index 8f6edcdcfd87..744a04f06025 100644
--- a/arch/x86/include/asm/resctrl.h
+++ b/arch/x86/include/asm/resctrl.h
@@ -71,6 +71,11 @@ static inline bool resctrl_arch_mon_capable(void)
return rdt_mon_capable;
}
+static inline bool resctrl_arch_devices_supported(void)
+{
+ return false;
+}
+
static inline void resctrl_arch_enable_mon(void)
{
static_branch_enable_cpuslocked(&rdt_mon_enable_key);
diff --git a/drivers/resctrl/mpam_devices.c b/drivers/resctrl/mpam_devices.c
index dd422c56fbb1..882184651c14 100644
--- a/drivers/resctrl/mpam_devices.c
+++ b/drivers/resctrl/mpam_devices.c
@@ -67,6 +67,26 @@ u8 mpam_pmg_max;
static bool partid_max_init, partid_max_published;
static DEFINE_SPINLOCK(partid_max_lock);
+/*
+ * Set once an IOMMU (e.g. ARM SMMU v3) has registered that it can tag
+ * device DMA with a PARTID/PMG. resctrl only exposes the "devices" file
+ * when this is true, so unsupported architectures never see it. The flag
+ * only ever flips false -> true and is read with READ_ONCE().
+ */
+static bool mpam_device_requestor_registered;
+
+void mpam_register_device_requestor(void)
+{
+ WRITE_ONCE(mpam_device_requestor_registered, true);
+}
+EXPORT_SYMBOL(mpam_register_device_requestor);
+
+bool mpam_devices_supported(void)
+{
+ return READ_ONCE(mpam_device_requestor_registered);
+}
+EXPORT_SYMBOL(mpam_devices_supported);
+
/*
* mpam is enabled once all devices have been probed from CPU online callbacks,
* scheduled via this work_struct. If access to an MSC depends on a CPU that
diff --git a/drivers/resctrl/mpam_resctrl.c b/drivers/resctrl/mpam_resctrl.c
index 9d223057953a..71ccb3db5987 100644
--- a/drivers/resctrl/mpam_resctrl.c
+++ b/drivers/resctrl/mpam_resctrl.c
@@ -97,6 +97,11 @@ bool resctrl_arch_mon_capable(void)
return l3->mon_capable;
}
+bool resctrl_arch_devices_supported(void)
+{
+ return mpam_devices_supported();
+}
+
bool resctrl_arch_is_evt_configurable(enum resctrl_event_id evt)
{
return false;
diff --git a/include/linux/arm_mpam.h b/include/linux/arm_mpam.h
index f92a36187a52..9d2deee604ce 100644
--- a/include/linux/arm_mpam.h
+++ b/include/linux/arm_mpam.h
@@ -53,6 +53,11 @@ static inline int mpam_ris_create(struct mpam_msc *msc, u8 ris_idx,
bool resctrl_arch_alloc_capable(void);
bool resctrl_arch_mon_capable(void);
+bool resctrl_arch_devices_supported(void);
+
+void mpam_register_device_requestor(void);
+bool mpam_devices_supported(void);
+
void resctrl_arch_set_cpu_default_closid(int cpu, u32 closid);
void resctrl_arch_set_closid_rmid(struct task_struct *tsk, u32 closid, u32 rmid);
void resctrl_arch_set_cpu_default_closid_rmid(int cpu, u32 closid, u32 rmid);
--
2.33.0
^ permalink raw reply related [flat|nested] 7+ messages in thread* [RFC PATCH 3/5] iommu/arm-smmu-v3: Support MPAM device DMA QoS tagging
2026-09-01 14:07 [RFC PATCH 0/5] resctrl: Assign devices to resource groups via IOMMU DMA QoS tagging Qinxin Xia
2026-09-01 14:07 ` [RFC PATCH 1/5] iommu: Add per-device requestor QoS tagging and lookup helpers Qinxin Xia
2026-09-01 14:07 ` [RFC PATCH 2/5] arm_mpam: resctrl: Add arch query for device DMA QoS support Qinxin Xia
@ 2026-09-01 14:08 ` Qinxin Xia
2026-09-01 14:08 ` [RFC PATCH 4/5] fs/resctrl: Add device-to-group QoS tracking infrastructure Qinxin Xia
2026-09-01 14:08 ` [RFC PATCH 5/5] fs/resctrl: Add a "devices" file to assign devices to groups Qinxin Xia
4 siblings, 0 replies; 7+ messages in thread
From: Qinxin Xia @ 2026-09-01 14:08 UTC (permalink / raw)
To: ben.horgan, zhangzhanpeng.jasper, joro, palmer, tony.luck,
reinette.chatre, tomasz.jeznach, zengheng4, fustini, cuiyunhui,
wangzhou1, xiaqinxin
Cc: will, robin.murphy, pjw, aou, alex, Dave.Martin, james.morse,
babu.moger, corbet, shuah, jgg, kevin.tian, yuanzhu, iommu,
linuxarm, baolin.wang
Enable tagging a master's DMA with an MPAM PARTID/PMG so resctrl can assign
a device to a resource group.
When the SMMU supports MPAM, register it with the MPAM code and let resctrl
program a device's PARTID/PMG, bounded by the hardware limits.
Signed-off-by: Qinxin Xia <xiaqinxin@huawei.com>
---
drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3.c | 96 +++++++++++++++++++++
drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3.h | 13 +++
2 files changed, 109 insertions(+)
diff --git a/drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3.c b/drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3.c
index 5732f3ba0122..7b1f65c0ee6d 100644
--- a/drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3.c
+++ b/drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3.c
@@ -11,6 +11,7 @@
#include <linux/acpi.h>
#include <linux/acpi_iort.h>
+#include <linux/arm_mpam.h>
#include <linux/bitops.h>
#include <linux/crash_dump.h>
#include <linux/delay.h>
@@ -4374,6 +4375,65 @@ static int arm_smmu_def_domain_type(struct device *dev)
return 0;
}
+static int arm_smmu_set_dev_requestor_id(struct device *dev, u32 partid,
+ u8 pmg)
+{
+ struct arm_smmu_master *master = dev_iommu_priv_get(dev);
+ struct arm_smmu_cmd cmd;
+ struct arm_smmu_cmdq_batch cmds;
+ struct arm_smmu_device *smmu;
+ u64 val;
+ int i;
+
+ /*
+ * TODO: This writes the PARTID/PMG directly into the live STEs, so the
+ * tag is lost on any later STE rewrite and can race a concurrent writer.
+ * It should be stored on arm_smmu_master and stamped in the STE
+ * generators via a group-mutex-holding path instead.
+ */
+ if (!master || !master->smmu)
+ return -ENODEV;
+ smmu = master->smmu;
+
+ if (!(smmu->features & ARM_SMMU_FEAT_MPAM))
+ return -EOPNOTSUPP;
+
+ if (partid > smmu->partid_max || pmg > smmu->pmg_max)
+ return -ERANGE;
+
+ /* Program the stream-level PARTID/PMG into every STE the master owns. */
+ arm_smmu_cmdq_batch_init_cmd(smmu, &cmds, &cmd);
+ mutex_lock(&smmu->streams_mutex);
+ for (i = 0; i < master->num_streams; i++) {
+ u32 sid = master->streams[i].id;
+ struct arm_smmu_ste *ste = arm_smmu_get_step_for_sid(smmu, sid);
+
+ if (!ste)
+ continue;
+
+ val = le64_to_cpu(ste->data[1]);
+ val &= ~STRTAB_STE_1_S1MPAM;
+ WRITE_ONCE(ste->data[1], cpu_to_le64(val));
+
+ val = le64_to_cpu(ste->data[4]);
+ val &= ~STRTAB_STE_4_PARTID;
+ val |= FIELD_PREP(STRTAB_STE_4_PARTID, partid);
+ WRITE_ONCE(ste->data[4], cpu_to_le64(val));
+
+ val = le64_to_cpu(ste->data[5]);
+ val &= ~STRTAB_STE_5_PMG;
+ val |= FIELD_PREP(STRTAB_STE_5_PMG, pmg);
+ WRITE_ONCE(ste->data[5], cpu_to_le64(val));
+
+ cmd = arm_smmu_make_cmd_cfgi_ste(sid, true);
+ arm_smmu_cmdq_batch_add_cmd_p(smmu, &cmds, &cmd);
+ }
+
+ mutex_unlock(&smmu->streams_mutex);
+ arm_smmu_cmdq_batch_submit(smmu, &cmds);
+ return 0;
+}
+
static const struct iommu_ops arm_smmu_ops = {
.identity_domain = &arm_smmu_identity_domain,
.blocked_domain = &arm_smmu_blocked_domain,
@@ -4388,6 +4448,7 @@ static const struct iommu_ops arm_smmu_ops = {
.of_xlate = arm_smmu_of_xlate,
.get_resv_regions = arm_smmu_get_resv_regions,
.page_response = arm_smmu_page_response,
+ .set_dev_requestor_id = arm_smmu_set_dev_requestor_id,
.def_domain_type = arm_smmu_def_domain_type,
.get_viommu_size = arm_smmu_get_viommu_size,
.viommu_init = arm_vsmmu_init,
@@ -5046,6 +5107,36 @@ static void arm_smmu_get_httu(struct arm_smmu_device *smmu, u32 reg)
hw_features, fw_features);
}
+static void arm_smmu_mpam_register_smmu(struct arm_smmu_device *smmu)
+{
+ u16 partid_max;
+ u8 pmg_max;
+ u32 reg;
+
+ if (!IS_ENABLED(CONFIG_ARM64_MPAM))
+ return;
+
+ if (!(smmu->features & ARM_SMMU_FEAT_MPAM))
+ return;
+
+ reg = readl_relaxed(smmu->base + ARM_SMMU_MPAMIDR);
+ if (!reg)
+ return;
+
+ partid_max = FIELD_GET(SMMU_MPAMIDR_PARTID_MAX, reg);
+ pmg_max = FIELD_GET(SMMU_MPAMIDR_PMG_MAX, reg);
+
+ smmu->partid_max = partid_max;
+ smmu->pmg_max = pmg_max;
+
+ if (mpam_register_requestor(partid_max, pmg_max)) {
+ smmu->features &= ~ARM_SMMU_FEAT_MPAM;
+ return;
+ }
+
+ mpam_register_device_requestor();
+}
+
static int arm_smmu_device_hw_probe(struct arm_smmu_device *smmu)
{
u32 reg;
@@ -5197,6 +5288,9 @@ static int arm_smmu_device_hw_probe(struct arm_smmu_device *smmu)
if (FIELD_GET(IDR3_BBM, reg) == 2)
smmu->features |= ARM_SMMU_FEAT_BBML2;
+ if (FIELD_GET(IDR3_MPAM, reg))
+ smmu->features |= ARM_SMMU_FEAT_MPAM;
+
/* IDR5 */
reg = readl_relaxed(smmu->base + ARM_SMMU_IDR5);
@@ -5261,6 +5355,8 @@ static int arm_smmu_device_hw_probe(struct arm_smmu_device *smmu)
if (arm_smmu_sva_supported(smmu))
smmu->features |= ARM_SMMU_FEAT_SVA;
+ arm_smmu_mpam_register_smmu(smmu);
+
dev_info(smmu->dev, "oas %lu-bit (features 0x%08x)\n",
smmu->oas, smmu->features);
return 0;
diff --git a/drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3.h b/drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3.h
index 50f8321e979c..f0118214c81c 100644
--- a/drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3.h
+++ b/drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3.h
@@ -59,6 +59,7 @@ struct arm_vsmmu;
#define IDR1_SIDSIZE GENMASK(5, 0)
#define ARM_SMMU_IDR3 0xc
+#define IDR3_MPAM (1 << 7)
#define IDR3_FWB (1 << 8)
#define IDR3_RIL (1 << 10)
#define IDR3_BBM GENMASK(12, 11)
@@ -171,6 +172,10 @@ struct arm_vsmmu;
#define ARM_SMMU_PRIQ_IRQ_CFG1 0xd8
#define ARM_SMMU_PRIQ_IRQ_CFG2 0xdc
+#define ARM_SMMU_MPAMIDR 0x130
+#define SMMU_MPAMIDR_PARTID_MAX GENMASK(15, 0)
+#define SMMU_MPAMIDR_PMG_MAX GENMASK(23, 16)
+
#define ARM_SMMU_REG_SZ 0xe00
/* Common MSI config fields */
@@ -300,6 +305,10 @@ static inline u32 arm_smmu_strtab_l2_idx(u32 sid)
#define STRTAB_STE_2_S2S (1UL << 57)
#define STRTAB_STE_2_S2R (1UL << 58)
+#define STRTAB_STE_1_S1MPAM (1UL << 26)
+#define STRTAB_STE_4_PARTID GENMASK_ULL(31, 16)
+#define STRTAB_STE_5_PMG GENMASK_ULL(7, 0)
+
#define STRTAB_STE_3_S2TTB_MASK GENMASK_ULL(51, 4)
/* These bits can be controlled by userspace for STRTAB_STE_0_CFG_NESTED */
@@ -927,8 +936,12 @@ struct arm_smmu_device {
#define ARM_SMMU_FEAT_BBML2 (1 << 24)
#define ARM_SMMU_FEAT_HAFT (1 << 25)
#define ARM_SMMU_FEAT_DS (1 << 26)
+#define ARM_SMMU_FEAT_MPAM (1 << 27)
u32 features;
+ u16 partid_max;
+ u8 pmg_max;
+
#define ARM_SMMU_OPT_SKIP_PREFETCH (1 << 0)
#define ARM_SMMU_OPT_PAGE0_REGS_ONLY (1 << 1)
#define ARM_SMMU_OPT_MSIPOLL (1 << 2)
--
2.33.0
^ permalink raw reply related [flat|nested] 7+ messages in thread* [RFC PATCH 4/5] fs/resctrl: Add device-to-group QoS tracking infrastructure
2026-09-01 14:07 [RFC PATCH 0/5] resctrl: Assign devices to resource groups via IOMMU DMA QoS tagging Qinxin Xia
` (2 preceding siblings ...)
2026-09-01 14:08 ` [RFC PATCH 3/5] iommu/arm-smmu-v3: Support MPAM device DMA QoS tagging Qinxin Xia
@ 2026-09-01 14:08 ` Qinxin Xia
2026-09-01 14:08 ` [RFC PATCH 5/5] fs/resctrl: Add a "devices" file to assign devices to groups Qinxin Xia
4 siblings, 0 replies; 7+ messages in thread
From: Qinxin Xia @ 2026-09-01 14:08 UTC (permalink / raw)
To: ben.horgan, zhangzhanpeng.jasper, joro, palmer, tony.luck,
reinette.chatre, tomasz.jeznach, zengheng4, fustini, cuiyunhui,
wangzhou1, xiaqinxin
Cc: will, robin.murphy, pjw, aou, alex, Dave.Martin, james.morse,
babu.moger, corbet, shuah, jgg, kevin.tian, yuanzhu, iommu,
linuxarm, baolin.wang
Add the core support for tracking which devices belong to a resource group
and tagging their DMA with the group's QoS IDs.
Devices mastering through an IOMMU are tracked automatically and default to
the root group; when a group is destroyed its devices are moved back to it.
A group with assigned devices cannot be pseudo-locked.
Signed-off-by: Qinxin Xia <xiaqinxin@huawei.com>
---
fs/resctrl/internal.h | 9 +++
fs/resctrl/pseudo_lock.c | 5 ++
fs/resctrl/rdtgroup.c | 148 +++++++++++++++++++++++++++++++++++++++
3 files changed, 162 insertions(+)
diff --git a/fs/resctrl/internal.h b/fs/resctrl/internal.h
index e62a277dee85..6457962763ff 100644
--- a/fs/resctrl/internal.h
+++ b/fs/resctrl/internal.h
@@ -202,6 +202,13 @@ struct mongroup {
u32 rmid;
};
+struct rdtdev {
+ struct list_head node;
+ struct device *dev;
+ u32 closid;
+ u32 rmid;
+};
+
/**
* struct rdtgroup - store rdtgroup's data in resctrl file system.
* @kn: kernfs node
@@ -366,6 +373,8 @@ enum rdtgrp_mode rdtgroup_mode_by_closid(int closid);
int rdtgroup_tasks_assigned(struct rdtgroup *r);
+int rdtgroup_devices_assigned(struct rdtgroup *r);
+
int closids_supported(void);
void closid_free(int closid);
diff --git a/fs/resctrl/pseudo_lock.c b/fs/resctrl/pseudo_lock.c
index dea2b4bf966f..e3b4f53aacce 100644
--- a/fs/resctrl/pseudo_lock.c
+++ b/fs/resctrl/pseudo_lock.c
@@ -531,6 +531,11 @@ int rdtgroup_locksetup_enter(struct rdtgroup *rdtgrp)
return -EINVAL;
}
+ if (rdtgroup_devices_assigned(rdtgrp)) {
+ rdt_last_cmd_puts("Devices assigned to resource group\n");
+ return -EINVAL;
+ }
+
if (!cpumask_empty(&rdtgrp->cpu_mask)) {
rdt_last_cmd_puts("CPUs assigned to resource group\n");
return -EINVAL;
diff --git a/fs/resctrl/rdtgroup.c b/fs/resctrl/rdtgroup.c
index 5dcbb0a964e8..c33891ead788 100644
--- a/fs/resctrl/rdtgroup.c
+++ b/fs/resctrl/rdtgroup.c
@@ -16,6 +16,7 @@
#include <linux/debugfs.h>
#include <linux/fs.h>
#include <linux/fs_parser.h>
+#include <linux/iommu.h>
#include <linux/sysfs.h>
#include <linux/kernfs.h>
#include <linux/once.h>
@@ -33,6 +34,9 @@
/* Mutex to protect rdtgroup access. */
DEFINE_MUTEX(rdtgroup_mutex);
+/* Mutex to protect the rdtdev_list and the rdtdev entries. */
+DEFINE_MUTEX(rdtdev_mutex);
+
static struct kernfs_root *rdt_root;
struct rdtgroup rdtgroup_default;
@@ -42,6 +46,8 @@ LIST_HEAD(rdt_all_groups);
/* list of entries for the schemata file */
LIST_HEAD(resctrl_schema_all);
+LIST_HEAD(rdtdev_list);
+
/*
* List of struct mon_data containing private data of event files for use by
* rdtgroup_mondata_show(). Protected by rdtgroup_mutex.
@@ -885,6 +891,137 @@ static int rdtgroup_rmid_show(struct kernfs_open_file *of,
return ret;
}
+static bool is_closid_match_dev(struct rdtdev *rdtdev, struct rdtgroup *r)
+{
+ return (resctrl_arch_alloc_capable() && (r->type == RDTCTRL_GROUP) &&
+ (rdtdev->closid == r->closid));
+}
+
+static bool is_rmid_match_dev(struct rdtdev *rdtdev, struct rdtgroup *r)
+{
+ return (resctrl_arch_mon_capable() && (r->type == RDTMON_GROUP) &&
+ rdtdev->rmid == r->mon.rmid && rdtdev->closid == r->mon.parent->closid);
+}
+
+static int rdtdev_set_qos(struct device *dev, u32 closid, u32 rmid)
+{
+ return iommu_set_dev_requestor_id(dev, closid, rmid);
+}
+
+static int rdtgroup_set_device(struct device *dev, struct rdtgroup *rdtgrp)
+{
+ struct rdtdev *rdtdev, *entry = NULL, *tmp;
+ u32 closid, rmid;
+ int ret;
+
+ if (!dev)
+ return -ENODEV;
+
+ if (!rdtgrp)
+ rdtgrp = &rdtgroup_default;
+
+ closid = (rdtgrp->type == RDTMON_GROUP) ?
+ rdtgrp->mon.parent->closid : rdtgrp->closid;
+ rmid = rdtgrp->mon.rmid;
+
+ guard(mutex)(&rdtdev_mutex);
+
+ list_for_each_entry(tmp, &rdtdev_list, node)
+ if (tmp->dev == dev) {
+ entry = tmp;
+ break;
+ }
+
+ if (entry) {
+ ret = rdtdev_set_qos(dev, closid, rmid);
+ if (ret)
+ return ret;
+ entry->closid = closid;
+ entry->rmid = rmid;
+ return 0;
+ }
+
+ rdtdev = kzalloc_obj(*rdtdev);
+ if (!rdtdev)
+ return -ENOMEM;
+
+ rdtdev->dev = get_device(dev);
+ rdtdev->closid = closid;
+ rdtdev->rmid = rmid;
+ list_add_tail(&rdtdev->node, &rdtdev_list);
+
+ ret = rdtdev_set_qos(dev, closid, rmid);
+ if (ret) {
+ list_del(&rdtdev->node);
+ put_device(rdtdev->dev);
+ kfree(rdtdev);
+ return ret;
+ }
+
+ return 0;
+}
+
+static void rdtgroup_remove_device(struct device *dev)
+{
+ struct rdtdev *rdtdev, *entry;
+
+ guard(mutex)(&rdtdev_mutex);
+ list_for_each_entry_safe(rdtdev, entry, &rdtdev_list, node)
+ if (rdtdev->dev == dev) {
+ list_del(&rdtdev->node);
+ put_device(rdtdev->dev);
+ kfree(rdtdev);
+ break;
+ }
+}
+
+static void rdtgroup_qos_device_add(struct device *dev)
+{
+ rdtgroup_set_device(dev, NULL);
+}
+
+static const struct iommu_qos_device_ops rdtgroup_qos_device_ops = {
+ .add = rdtgroup_qos_device_add,
+ .remove = rdtgroup_remove_device,
+};
+
+static void rdt_move_group_devices(struct rdtgroup *from, struct rdtgroup *to)
+{
+ struct rdtdev *rdtdev;
+
+ guard(mutex)(&rdtdev_mutex);
+ list_for_each_entry(rdtdev, &rdtdev_list, node)
+ if (!from || is_rmid_match_dev(rdtdev, from) ||
+ is_closid_match_dev(rdtdev, from)) {
+ /*
+ * The source group is going away and its closid/rmid
+ * will be freed and reused. Retag the device to @to,
+ * and move it in the tracking list regardless of the
+ * hardware result: leaving it on the old ids would
+ * later match a different group once they are reused.
+ * Warn if the hardware could not be updated to match.
+ */
+ if (rdtdev_set_qos(rdtdev->dev, to->closid, to->mon.rmid))
+ pr_warn("Failed to retag device %s while moving group\n",
+ dev_name(rdtdev->dev));
+ rdtdev->closid = to->closid;
+ rdtdev->rmid = to->mon.rmid;
+ }
+}
+
+int rdtgroup_devices_assigned(struct rdtgroup *r)
+{
+ struct rdtdev *rdtdev;
+
+ guard(mutex)(&rdtdev_mutex);
+ list_for_each_entry(rdtdev, &rdtdev_list, node)
+ if (is_rmid_match_dev(rdtdev, r) ||
+ is_closid_match_dev(rdtdev, r))
+ return 1;
+
+ return 0;
+}
+
#ifdef CONFIG_PROC_CPU_RESCTRL
/*
* A task can only be part of one resctrl control group and of one monitor
@@ -3024,6 +3161,9 @@ static void rmdir_all_sub(void)
/* Move all tasks to the default resource group */
rdt_move_group_tasks(NULL, &rdtgroup_default, NULL);
+ /* Move all devices to the default resource group */
+ rdt_move_group_devices(NULL, &rdtgroup_default);
+
list_for_each_entry_safe(rdtgrp, tmp, &rdt_all_groups, rdtgroup_list) {
/* Free any child rmids */
free_all_child_rdtgrp(rdtgrp);
@@ -4173,6 +4313,9 @@ static int rdtgroup_rmdir_mon(struct rdtgroup *rdtgrp, cpumask_var_t tmpmask)
/* Give any tasks back to the parent group */
rdt_move_group_tasks(rdtgrp, prdtgrp, tmpmask);
+ /* Give any devices back to the parent group */
+ rdt_move_group_devices(rdtgrp, prdtgrp);
+
/*
* Update per cpu closid/rmid of the moved CPUs first.
* Note: the closid will not change, but the arch code still needs it.
@@ -4223,6 +4366,9 @@ static int rdtgroup_rmdir_ctrl(struct rdtgroup *rdtgrp, cpumask_var_t tmpmask)
/* Give any tasks back to the default group */
rdt_move_group_tasks(rdtgrp, &rdtgroup_default, tmpmask);
+ /* Give any devices back to the default group */
+ rdt_move_group_devices(rdtgrp, &rdtgroup_default);
+
/* Give any CPUs back to the default group */
cpumask_or(&rdtgroup_default.cpu_mask,
&rdtgroup_default.cpu_mask, &rdtgrp->cpu_mask);
@@ -4826,6 +4972,8 @@ int resctrl_init(void)
if (ret)
goto cleanup_mountpoint;
+ iommu_register_qos_device_ops(&rdtgroup_qos_device_ops);
+
/*
* Adding the resctrl debugfs directory here may not be ideal since
* it would let the resctrl debugfs directory appear on the debugfs
--
2.33.0
^ permalink raw reply related [flat|nested] 7+ messages in thread* [RFC PATCH 5/5] fs/resctrl: Add a "devices" file to assign devices to groups
2026-09-01 14:07 [RFC PATCH 0/5] resctrl: Assign devices to resource groups via IOMMU DMA QoS tagging Qinxin Xia
` (3 preceding siblings ...)
2026-09-01 14:08 ` [RFC PATCH 4/5] fs/resctrl: Add device-to-group QoS tracking infrastructure Qinxin Xia
@ 2026-09-01 14:08 ` Qinxin Xia
2026-09-10 10:24 ` Ben Horgan
4 siblings, 1 reply; 7+ messages in thread
From: Qinxin Xia @ 2026-09-01 14:08 UTC (permalink / raw)
To: ben.horgan, zhangzhanpeng.jasper, joro, palmer, tony.luck,
reinette.chatre, tomasz.jeznach, zengheng4, fustini, cuiyunhui,
wangzhou1, xiaqinxin
Cc: will, robin.murphy, pjw, aou, alex, Dave.Martin, james.morse,
babu.moger, corbet, shuah, jgg, kevin.tian, yuanzhu, iommu,
linuxarm, baolin.wang
Expose the device QoS tracking through a new "devices" file. Writing a
device name assigns it to the group, tagging its DMA with the group's QoS
IDs.
The file is only shown where an IOMMU can tag device DMA.
Signed-off-by: Qinxin Xia <xiaqinxin@huawei.com>
---
Documentation/filesystems/resctrl.rst | 18 ++++-
fs/resctrl/rdtgroup.c | 101 ++++++++++++++++++++++++++
2 files changed, 117 insertions(+), 2 deletions(-)
diff --git a/Documentation/filesystems/resctrl.rst b/Documentation/filesystems/resctrl.rst
index e4b66af55ffb..aa205ccaabf6 100644
--- a/Documentation/filesystems/resctrl.rst
+++ b/Documentation/filesystems/resctrl.rst
@@ -546,8 +546,8 @@ directories can be created to monitor subsets of tasks in the CTRL_MON
group that is their ancestor. These are called "MON" groups in the rest
of this document.
-Removing a directory will move all tasks and cpus owned by the group it
-represents to the parent. Removing one of the created CTRL_MON groups
+Removing a directory will move all tasks, cpus and devices owned by the
+group it represents to the parent. Removing one of the created CTRL_MON groups
will automatically remove all MON groups below it.
Moving MON group directories to a new parent CTRL_MON group is supported
@@ -581,6 +581,20 @@ All groups contain the following files:
idle tasks. Instead, a CPU's idle task is always considered as a
member of the group owning the CPU.
+"devices":
+ Reading this file shows the list of all devices that belong to
+ this group. Writing a device name to the file will add a device to
+ the group, tagging its DMA with the group's QoS IDs. Multiple
+ devices can be added by separating the names with commas. A single
+ failure encountered while attempting to assign a device will cause
+ the operation to abort and already added devices before the failure
+ will remain in the group. Failures will be logged to
+ /sys/fs/resctrl/info/last_cmd_status.
+
+ The device name is the one listed under
+ /sys/kernel/iommu_groups/<id>/devices/. This file is only present
+ when an IOMMU can tag the DMA of devices behind it with a QoS class.
+
"cpus":
Reading this file shows a bitmask of the logical CPUs owned by
this group. Writing a mask to this file will add and remove
diff --git a/fs/resctrl/rdtgroup.c b/fs/resctrl/rdtgroup.c
index c33891ead788..3633b3521941 100644
--- a/fs/resctrl/rdtgroup.c
+++ b/fs/resctrl/rdtgroup.c
@@ -985,6 +985,92 @@ static const struct iommu_qos_device_ops rdtgroup_qos_device_ops = {
.remove = rdtgroup_remove_device,
};
+static ssize_t rdtgroup_devices_write(struct kernfs_open_file *of,
+ char *buf, size_t nbytes, loff_t off)
+{
+ struct rdtgroup *rdtgrp;
+ char *token;
+ int ret = 0;
+
+ if (!buf)
+ return -EINVAL;
+
+ rdtgrp = rdtgroup_kn_lock_live(of->kn);
+ if (!rdtgrp) {
+ rdtgroup_kn_unlock(of->kn);
+ return -ENOENT;
+ }
+ rdt_last_cmd_clear();
+
+ if (rdtgrp->mode == RDT_MODE_PSEUDO_LOCKED ||
+ rdtgrp->mode == RDT_MODE_PSEUDO_LOCKSETUP) {
+ ret = -EINVAL;
+ rdt_last_cmd_puts("Pseudo-locking in progress\n");
+ goto unlock;
+ }
+
+ while ((token = strsep(&buf, ","))) {
+ struct device *dev;
+
+ token = strim(token);
+ if (!*token) {
+ rdt_last_cmd_puts("Device list parsing error\n");
+ ret = -EINVAL;
+ break;
+ }
+
+ dev = iommu_group_find_device_by_name(token);
+ if (!dev) {
+ rdt_last_cmd_printf("No device %s\n", token);
+ ret = -ENODEV;
+ break;
+ }
+
+ ret = rdtgroup_set_device(dev, rdtgrp);
+ if (ret == -EOPNOTSUPP)
+ rdt_last_cmd_printf("Device %s does not support QoS\n",
+ token);
+ else if (ret)
+ rdt_last_cmd_printf("Error while processing device %s\n",
+ token);
+ /*
+ * rdtgroup_set_device() takes its own reference on a new
+ * rdtdev; drop the temporary reference returned by the
+ * lookup regardless of the outcome.
+ */
+ put_device(dev);
+ if (ret)
+ break;
+ }
+
+unlock:
+ rdtgroup_kn_unlock(of->kn);
+ return ret ?: nbytes;
+}
+
+static int rdtgroup_devices_show(struct kernfs_open_file *of,
+ struct seq_file *s, void *v)
+{
+ struct rdtgroup *rdtgrp;
+ struct rdtdev *rdtdev;
+
+ rdtgrp = rdtgroup_kn_lock_live(of->kn);
+ if (!rdtgrp) {
+ rdtgroup_kn_unlock(of->kn);
+ return -ENOENT;
+ }
+
+ mutex_lock(&rdtdev_mutex);
+ list_for_each_entry(rdtdev, &rdtdev_list, node)
+ if (is_rmid_match_dev(rdtdev, rdtgrp) ||
+ is_closid_match_dev(rdtdev, rdtgrp))
+ seq_printf(s, "%s\n", kobject_name(&rdtdev->dev->kobj));
+ mutex_unlock(&rdtdev_mutex);
+
+ rdtgroup_kn_unlock(of->kn);
+ return 0;
+}
+
static void rdt_move_group_devices(struct rdtgroup *from, struct rdtgroup *to)
{
struct rdtdev *rdtdev;
@@ -2295,6 +2381,14 @@ static struct rftype res_common_files[] = {
.seq_show = rdtgroup_tasks_show,
.fflags = RFTYPE_BASE,
},
+ {
+ .name = "devices",
+ .mode = 0644,
+ .kf_ops = &rdtgroup_kf_single_ops,
+ .write = rdtgroup_devices_write,
+ .seq_show = rdtgroup_devices_show,
+ .fflags = RFTYPE_BASE,
+ },
{
.name = "mon_hw_id",
.mode = 0444,
@@ -2363,6 +2457,13 @@ static int rdtgroup_add_files(struct kernfs_node *kn, unsigned long fflags)
for (rft = rfts; rft < rfts + len; rft++) {
if (rft->fflags && ((fflags & rft->fflags) == rft->fflags)) {
+ if (!strcmp(rft->name, "devices") &&
+ (!resctrl_arch_devices_supported() ||
+ ((fflags & RFTYPE_CTRL) &&
+ !resctrl_arch_alloc_capable()) ||
+ ((fflags & RFTYPE_MON) &&
+ !resctrl_arch_mon_capable())))
+ continue;
ret = rdtgroup_add_file(kn, rft);
if (ret)
goto error;
--
2.33.0
^ permalink raw reply related [flat|nested] 7+ messages in thread* Re: [RFC PATCH 5/5] fs/resctrl: Add a "devices" file to assign devices to groups
2026-09-01 14:08 ` [RFC PATCH 5/5] fs/resctrl: Add a "devices" file to assign devices to groups Qinxin Xia
@ 2026-09-10 10:24 ` Ben Horgan
0 siblings, 0 replies; 7+ messages in thread
From: Ben Horgan @ 2026-09-10 10:24 UTC (permalink / raw)
To: Qinxin Xia, zhangzhanpeng.jasper, joro, palmer, tony.luck,
reinette.chatre, tomasz.jeznach, zengheng4, fustini, cuiyunhui,
wangzhou1
Cc: will, robin.murphy, pjw, aou, alex, Dave.Martin, james.morse,
babu.moger, corbet, shuah, jgg, kevin.tian, yuanzhu, iommu,
linuxarm, baolin.wang
Hi Qinxin,
On 01/09/2026 15:08, Qinxin Xia wrote:
> Expose the device QoS tracking through a new "devices" file. Writing a
> device name assigns it to the group, tagging its DMA with the group's QoS
> IDs.
>
> The file is only shown where an IOMMU can tag device DMA.
>
> Signed-off-by: Qinxin Xia <xiaqinxin@huawei.com>
> ---
> Documentation/filesystems/resctrl.rst | 18 ++++-
> fs/resctrl/rdtgroup.c | 101 ++++++++++++++++++++++++++
> 2 files changed, 117 insertions(+), 2 deletions(-)
>
> diff --git a/Documentation/filesystems/resctrl.rst b/Documentation/filesystems/resctrl.rst
> index e4b66af55ffb..aa205ccaabf6 100644
> --- a/Documentation/filesystems/resctrl.rst
> +++ b/Documentation/filesystems/resctrl.rst
> @@ -546,8 +546,8 @@ directories can be created to monitor subsets of tasks in the CTRL_MON
> group that is their ancestor. These are called "MON" groups in the rest
> of this document.
>
> -Removing a directory will move all tasks and cpus owned by the group it
> -represents to the parent. Removing one of the created CTRL_MON groups
> +Removing a directory will move all tasks, cpus and devices owned by the
> +group it represents to the parent. Removing one of the created CTRL_MON groups
> will automatically remove all MON groups below it.
>
> Moving MON group directories to a new parent CTRL_MON group is supported
> @@ -581,6 +581,20 @@ All groups contain the following files:
> idle tasks. Instead, a CPU's idle task is always considered as a
> member of the group owning the CPU.
>
> +"devices":
> + Reading this file shows the list of all devices that belong to
> + this group. Writing a device name to the file will add a device to
> + the group, tagging its DMA with the group's QoS IDs. Multiple
> + devices can be added by separating the names with commas. A single
> + failure encountered while attempting to assign a device will cause
> + the operation to abort and already added devices before the failure
> + will remain in the group. Failures will be logged to
> + /sys/fs/resctrl/info/last_cmd_status.
> +
> + The device name is the one listed under
> + /sys/kernel/iommu_groups/<id>/devices/. This file is only present
> + when an IOMMU can tag the DMA of devices behind it with a QoS class.
What's the expectation when there is more than one device listed? It seems odd to specify the
iommu_group by device name when there may be more than one device. Would it not be better to just
use the id directly?
There are also other platform devices such as the GPU which may not be behind an SMMU/IOMMU but can
be an MPAM requester. We should also take these into account when designing the interface. One
,unworkable, idea is that under a 'devices' directory there could be two files, iommu_group and
platform_device. The iommu_group would take an iommu_group id (as per /sys/kernel/iommu_groups/<id>)
and the platform_device the name from /sys/bus/platform/devices/. I suggest separate files to avoid
any naming conflicts. However, this does leave the problem that the 'devices' directory would likely
then be considered a CTRL_MON group by tools and so not a usable interface.
One other consideration is the life cycle and scope of the resctrl domains. For instance, if the
IOMMU is upstream of a cache which supports cache allocation then the cache allocation for the
instance downstream of the IOMMU should still be configurable even if the associated CPUs are
disabled. For MPAM systems IIUC this would require extra topology information than what is currently
available in the ACPI description. I plan to bring this up with the MPAM architects. I'm not sure
what's available for other architectures. The existing 'io_alloc' and 'io_alloc_cbm' (not used on
MPAM) look to side step this by keeping them in the info directory.
Thanks,
Ben
> +
> "cpus":
> Reading this file shows a bitmask of the logical CPUs owned by
> this group. Writing a mask to this file will add and remove
> diff --git a/fs/resctrl/rdtgroup.c b/fs/resctrl/rdtgroup.c
> index c33891ead788..3633b3521941 100644
> --- a/fs/resctrl/rdtgroup.c
> +++ b/fs/resctrl/rdtgroup.c
> @@ -985,6 +985,92 @@ static const struct iommu_qos_device_ops rdtgroup_qos_device_ops = {
> .remove = rdtgroup_remove_device,
> };
>
> +static ssize_t rdtgroup_devices_write(struct kernfs_open_file *of,
> + char *buf, size_t nbytes, loff_t off)
> +{
> + struct rdtgroup *rdtgrp;
> + char *token;
> + int ret = 0;
> +
> + if (!buf)
> + return -EINVAL;
> +
> + rdtgrp = rdtgroup_kn_lock_live(of->kn);
> + if (!rdtgrp) {
> + rdtgroup_kn_unlock(of->kn);
> + return -ENOENT;
> + }
> + rdt_last_cmd_clear();
> +
> + if (rdtgrp->mode == RDT_MODE_PSEUDO_LOCKED ||
> + rdtgrp->mode == RDT_MODE_PSEUDO_LOCKSETUP) {
> + ret = -EINVAL;
> + rdt_last_cmd_puts("Pseudo-locking in progress\n");
> + goto unlock;
> + }
> +
> + while ((token = strsep(&buf, ","))) {
> + struct device *dev;
> +
> + token = strim(token);
> + if (!*token) {
> + rdt_last_cmd_puts("Device list parsing error\n");
> + ret = -EINVAL;
> + break;
> + }
> +
> + dev = iommu_group_find_device_by_name(token);
> + if (!dev) {
> + rdt_last_cmd_printf("No device %s\n", token);
> + ret = -ENODEV;
> + break;
> + }
> +
> + ret = rdtgroup_set_device(dev, rdtgrp);
> + if (ret == -EOPNOTSUPP)
> + rdt_last_cmd_printf("Device %s does not support QoS\n",
> + token);
> + else if (ret)
> + rdt_last_cmd_printf("Error while processing device %s\n",
> + token);
> + /*
> + * rdtgroup_set_device() takes its own reference on a new
> + * rdtdev; drop the temporary reference returned by the
> + * lookup regardless of the outcome.
> + */
> + put_device(dev);
> + if (ret)
> + break;
> + }
> +
> +unlock:
> + rdtgroup_kn_unlock(of->kn);
> + return ret ?: nbytes;
> +}
> +
> +static int rdtgroup_devices_show(struct kernfs_open_file *of,
> + struct seq_file *s, void *v)
> +{
> + struct rdtgroup *rdtgrp;
> + struct rdtdev *rdtdev;
> +
> + rdtgrp = rdtgroup_kn_lock_live(of->kn);
> + if (!rdtgrp) {
> + rdtgroup_kn_unlock(of->kn);
> + return -ENOENT;
> + }
> +
> + mutex_lock(&rdtdev_mutex);
> + list_for_each_entry(rdtdev, &rdtdev_list, node)
> + if (is_rmid_match_dev(rdtdev, rdtgrp) ||
> + is_closid_match_dev(rdtdev, rdtgrp))
> + seq_printf(s, "%s\n", kobject_name(&rdtdev->dev->kobj));
> + mutex_unlock(&rdtdev_mutex);
> +
> + rdtgroup_kn_unlock(of->kn);
> + return 0;
> +}
> +
> static void rdt_move_group_devices(struct rdtgroup *from, struct rdtgroup *to)
> {
> struct rdtdev *rdtdev;
> @@ -2295,6 +2381,14 @@ static struct rftype res_common_files[] = {
> .seq_show = rdtgroup_tasks_show,
> .fflags = RFTYPE_BASE,
> },
> + {
> + .name = "devices",
> + .mode = 0644,
> + .kf_ops = &rdtgroup_kf_single_ops,
> + .write = rdtgroup_devices_write,
> + .seq_show = rdtgroup_devices_show,
> + .fflags = RFTYPE_BASE,
> + },
> {
> .name = "mon_hw_id",
> .mode = 0444,
> @@ -2363,6 +2457,13 @@ static int rdtgroup_add_files(struct kernfs_node *kn, unsigned long fflags)
>
> for (rft = rfts; rft < rfts + len; rft++) {
> if (rft->fflags && ((fflags & rft->fflags) == rft->fflags)) {
> + if (!strcmp(rft->name, "devices") &&
> + (!resctrl_arch_devices_supported() ||
> + ((fflags & RFTYPE_CTRL) &&
> + !resctrl_arch_alloc_capable()) ||
> + ((fflags & RFTYPE_MON) &&
> + !resctrl_arch_mon_capable())))
> + continue;
> ret = rdtgroup_add_file(kn, rft);
> if (ret)
> goto error;
^ permalink raw reply [flat|nested] 7+ messages in thread