From: Andrew Jones <andrew.jones@oss.qualcomm.com>
To: iommu@lists.linux.dev, kvm-riscv@lists.infradead.org,
kvm@vger.kernel.org, linux-riscv@lists.infradead.org,
linux-kernel@vger.kernel.org
Cc: tomasz.jeznach@linux.dev, jgg@ziepe.ca, jgg@nvidia.com,
joro@8bytes.org, will@kernel.org, robin.murphy@arm.com,
pjw@kernel.org, palmer@dabbelt.com, tglx@kernel.org,
anup@brainfault.org, atish.patra@linux.dev,
fangyu.yu@linux.alibaba.com, zhangzhanpeng.jasper@bytedance.com,
zong.li@sifive.com
Subject: [RFC PATCH v3 13/14] iommu/riscv: Validate IRQ forwarding requests
Date: Mon, 28 Sep 2026 16:31:12 +0200 [thread overview]
Message-ID: <20260928143113.49838-14-andrew.jones@oss.qualcomm.com> (raw)
In-Reply-To: <20260928143113.49838-1-andrew.jones@oss.qualcomm.com>
irq_set_vcpu_affinity() receives an architecture-specific description
of a guest MSI topology. Invalid address layouts or targets could select
the wrong MSI PTE, while accepting a mode unsupported by one of the
IOMMUs serving a shared domain could make interrupt delivery fail only
for some devices.
Validate the request before the table can be activated or updated and
distinguish malformed input from a valid configuration that hardware
cannot support. Check capabilities across all IOMMUs bonded to the
domain, and retain the requirements of an active table so later target
updates and device attachments cannot weaken them.
Signed-off-by: Andrew Jones <andrew.jones@oss.qualcomm.com>
---
drivers/iommu/riscv/iommu-ir.c | 127 ++++++++++++++++++++++++++++++++-
drivers/iommu/riscv/iommu.c | 31 ++++++++
drivers/iommu/riscv/iommu.h | 2 +
3 files changed, 159 insertions(+), 1 deletion(-)
diff --git a/drivers/iommu/riscv/iommu-ir.c b/drivers/iommu/riscv/iommu-ir.c
index 22da76987c7c..12a5d0bc77f2 100644
--- a/drivers/iommu/riscv/iommu-ir.c
+++ b/drivers/iommu/riscv/iommu-ir.c
@@ -20,9 +20,120 @@ static int riscv_iommu_ir_irq_set_affinity(struct irq_data *data,
return irq_chip_set_affinity_parent(data, mask, force);
}
+/* IOMMU extract (RISC-V IOMMU spec section 2.3.3) applied to the two-run IMSIC topology */
+static size_t riscv_iommu_ir_extract(u64 value, u64 mask)
+{
+ /* Separate the optional group index from the low guest and hart indices. */
+ u64 upper_mask = mask & (mask + 1);
+ u64 lower_mask = mask ^ upper_mask;
+ unsigned int shift;
+
+ if (!upper_mask)
+ return value & lower_mask;
+
+ shift = __ffs64(upper_mask) - fls64(lower_mask);
+ return (value & lower_mask) | ((value & upper_mask) >> shift);
+}
+
+static size_t riscv_iommu_ir_nr_ptes(const struct riscv_iommu_ir_vcpu_info *vcpu_info)
+{
+ return riscv_iommu_ir_extract(vcpu_info->msi_addr_mask, vcpu_info->msi_addr_mask) + 1;
+}
+
+static int riscv_iommu_ir_validate_target(const struct riscv_iommu_ir_vcpu_info *vcpu_info,
+ const struct riscv_iommu_ir_target *target,
+ u64 *required_caps)
+{
+ u64 addr = target->gpa >> IMSIC_MMIO_PAGE_SHIFT;
+
+ if (!IS_ALIGNED(target->gpa, IMSIC_MMIO_PAGE_SZ) ||
+ (addr & ~vcpu_info->msi_addr_mask) != vcpu_info->msi_addr_pattern)
+ return -EINVAL;
+
+ switch (target->type) {
+ case RISCV_IOMMU_IR_TARGET_IMSIC:
+ if (!IS_ALIGNED(target->hpa, IMSIC_MMIO_PAGE_SZ) ||
+ (target->hpa >> IMSIC_MMIO_PAGE_SHIFT) > FIELD_MAX(RISCV_IOMMU_MSIPTE_PPN))
+ return -EINVAL;
+ *required_caps |= RISCV_IOMMU_CAPABILITIES_MSI_FLAT;
+ break;
+ case RISCV_IOMMU_IR_TARGET_MRIF:
+ if (!IS_ALIGNED(target->mrif_hpa, SZ_512) ||
+ (target->mrif_hpa >> 9) > FIELD_MAX(RISCV_IOMMU_MSIPTE_MRIF_ADDR) ||
+ !IS_ALIGNED(target->notice_hpa, IMSIC_MMIO_PAGE_SZ) ||
+ (target->notice_hpa >> IMSIC_MMIO_PAGE_SHIFT) >
+ FIELD_MAX(RISCV_IOMMU_MSIPTE_MRIF_NPPN) ||
+ target->notice_id >= IMSIC_MAX_ID)
+ return -EINVAL;
+ *required_caps |= RISCV_IOMMU_CAPABILITIES_MSI_FLAT |
+ RISCV_IOMMU_CAPABILITIES_MSI_MRIF;
+ break;
+ default:
+ return -EINVAL;
+ }
+
+ return 0;
+}
+
+static int riscv_iommu_ir_validate_targets(const struct riscv_iommu_ir_vcpu_info *vcpu_info,
+ size_t nr_ptes, u64 *required_caps)
+{
+ int ret;
+
+ if (!vcpu_info->targets || !vcpu_info->nr_targets || vcpu_info->nr_targets > nr_ptes)
+ return -EINVAL;
+
+ for (unsigned int i = 0; i < vcpu_info->nr_targets; i++) {
+ ret = riscv_iommu_ir_validate_target(vcpu_info, &vcpu_info->targets[i],
+ required_caps);
+ if (ret)
+ return ret;
+ }
+
+ return 0;
+}
+
static int riscv_iommu_ir_activate(struct riscv_iommu_msi_table *msi_table,
+ struct riscv_iommu_device *iommu,
struct riscv_iommu_ir_vcpu_info *vcpu_info)
{
+ u64 pattern = vcpu_info->msi_addr_pattern;
+ u64 mask = vcpu_info->msi_addr_mask;
+ u64 required_caps = 0;
+ size_t nr_ptes;
+ int ret;
+
+ if (!vcpu_info->owner || (pattern & mask) ||
+ ((pattern | mask) & ~RISCV_IOMMU_DC_MSI_ADDR_MASK))
+ return -EINVAL;
+
+ /*
+ * riscv_iommu_ir_extract() requires the mask to contain the low
+ * guest/hart index run and at most one additional group index run.
+ */
+ mask &= mask + 1;
+ if (mask) {
+ mask >>= __ffs64(mask);
+ if (mask & (mask + 1))
+ return -EINVAL;
+ }
+
+ nr_ptes = riscv_iommu_ir_nr_ptes(vcpu_info);
+
+ ret = riscv_iommu_ir_validate_targets(vcpu_info, nr_ptes, &required_caps);
+ if (ret)
+ return ret;
+
+ if (!riscv_iommu_msi_table_check_caps(msi_table, required_caps))
+ return -EOPNOTSUPP;
+
+ if (nr_ptes > msi_table->nr_ptes) {
+ dev_warn_once(iommu->dev,
+ "guest MSI topology requires %zu PTEs, but the IOMMU domain only supports %u; using host IRQ delivery\n",
+ nr_ptes, msi_table->nr_ptes);
+ return -EOPNOTSUPP;
+ }
+
return -EOPNOTSUPP;
}
@@ -34,6 +145,18 @@ static int riscv_iommu_ir_deactivate(struct riscv_iommu_msi_table *msi_table)
static int riscv_iommu_ir_update_target(struct riscv_iommu_msi_table *msi_table,
struct riscv_iommu_ir_vcpu_info *vcpu_info)
{
+ const struct riscv_iommu_ir_target *target = &vcpu_info->target;
+ u64 required_caps = 0;
+ int ret;
+
+ ret = riscv_iommu_ir_validate_target(vcpu_info, target, &required_caps);
+ if (ret)
+ return ret;
+
+ required_caps |= msi_table->required_caps;
+ if (!riscv_iommu_msi_table_check_caps(msi_table, required_caps))
+ return -EOPNOTSUPP;
+
return -EOPNOTSUPP;
}
@@ -42,6 +165,7 @@ static int riscv_iommu_ir_irq_set_vcpu_affinity_locked(struct irq_data *data,
struct riscv_iommu_ir_vcpu_info *vcpu_info,
struct riscv_iommu_msi_table *msi_table)
{
+ struct riscv_iommu_device *iommu = data->domain->host_data;
int ret;
if (!vcpu_info) {
@@ -59,6 +183,7 @@ static int riscv_iommu_ir_irq_set_vcpu_affinity_locked(struct irq_data *data,
irqd_clr_forwarded_to_vcpu(data);
if (!msi_table->nr_forwarded_irqs) {
+ msi_table->required_caps = 0;
msi_table->owner = NULL;
msi_table->msi_addr_mask = 0;
msi_table->msi_addr_pattern = 0;
@@ -72,7 +197,7 @@ static int riscv_iommu_ir_irq_set_vcpu_affinity_locked(struct irq_data *data,
irqd_is_forwarded_to_vcpu(data))
return -EINVAL;
- ret = riscv_iommu_ir_activate(msi_table, vcpu_info);
+ ret = riscv_iommu_ir_activate(msi_table, iommu, vcpu_info);
if (ret)
return ret;
diff --git a/drivers/iommu/riscv/iommu.c b/drivers/iommu/riscv/iommu.c
index 514470291fd9..55c167df2c2e 100644
--- a/drivers/iommu/riscv/iommu.c
+++ b/drivers/iommu/riscv/iommu.c
@@ -1441,6 +1441,33 @@ static void riscv_iommu_free_paging_domain(struct iommu_domain *iommu_domain)
kfree(domain);
}
+bool riscv_iommu_msi_table_check_caps(struct riscv_iommu_msi_table *msi_table, u64 iommu_caps)
+{
+ struct riscv_iommu_domain *domain;
+ struct riscv_iommu_device *iommu, *prev = NULL;
+ struct riscv_iommu_bond *bond;
+ bool supported = true;
+
+ /* The MSI table lock excludes bond updates for this domain. */
+ lockdep_assert_held(&msi_table->lock);
+
+ domain = container_of(msi_table, struct riscv_iommu_domain, msi_table);
+
+ /* Bonds are grouped by IOMMU, so validate each IOMMU once. */
+ list_for_each_entry(bond, &domain->bonds, list) {
+ iommu = dev_to_iommu(bond->dev);
+ if (iommu == prev)
+ continue;
+ if ((iommu->caps & iommu_caps) != iommu_caps) {
+ supported = false;
+ break;
+ }
+ prev = iommu;
+ }
+
+ return supported;
+}
+
static bool riscv_iommu_fsc_supported(struct riscv_iommu_device *iommu,
int mode)
{
@@ -1512,6 +1539,8 @@ static bool riscv_iommu_can_attach_paging_domain(struct iommu_domain *iommu_doma
{
struct riscv_iommu_domain *domain = iommu_domain_to_riscv(iommu_domain);
struct riscv_iommu_info *info = dev_iommu_priv_get(dev);
+ struct riscv_iommu_device *iommu = dev_to_iommu(dev);
+ u64 required_caps = domain->msi_table.required_caps;
bool new_is_s2 = domain->gscid;
struct riscv_iommu_msi_table *new_msi_table, *old_msi_table;
@@ -1532,6 +1561,8 @@ static bool riscv_iommu_can_attach_paging_domain(struct iommu_domain *iommu_doma
*/
if (new_is_s2 && old_msi_table && info->nr_forwarded_irqs)
return false;
+ if (new_is_s2 && required_caps && (iommu->caps & required_caps) != required_caps)
+ return false;
return true;
}
diff --git a/drivers/iommu/riscv/iommu.h b/drivers/iommu/riscv/iommu.h
index a42f0b6a88d4..9852962e245b 100644
--- a/drivers/iommu/riscv/iommu.h
+++ b/drivers/iommu/riscv/iommu.h
@@ -83,6 +83,7 @@ struct riscv_iommu_msi_table {
u64 msi_addr_mask;
u64 msi_addr_pattern;
const void *owner;
+ u64 required_caps; /* RISCV_IOMMU_CAPABILITIES_* required by active MSI PTEs */
};
/* Private IOMMU data for managed devices, dev_iommu_priv_* */
@@ -104,6 +105,7 @@ void riscv_iommu_disable(struct riscv_iommu_device *iommu);
/* Caller must hold rcu_read_lock() while using the returned pointer. */
struct riscv_iommu_msi_table *riscv_iommu_msi_table_rcu(struct riscv_iommu_info *info);
+bool riscv_iommu_msi_table_check_caps(struct riscv_iommu_msi_table *msi_table, u64 iommu_caps);
void riscv_iommu_msi_table_inval(struct riscv_iommu_msi_table *msi_table, unsigned long addr);
void riscv_iommu_msi_table_inval_all(struct riscv_iommu_msi_table *msi_table);
void riscv_iommu_msi_table_update(struct riscv_iommu_msi_table *msi_table, bool activate);
--
2.43.0
next prev parent reply other threads:[~2026-09-28 14:32 UTC|newest]
Thread overview: 17+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-28 14:30 [RFC PATCH v3 00/14] iommu/riscv: Add irqbypass support Andrew Jones
2026-09-28 14:31 ` [RFC PATCH v3 01/14] iommu/riscv: Allocate MSI tables for second-stage domains Andrew Jones
2026-10-05 13:25 ` Andrew Jones
2026-09-28 14:31 ` [RFC PATCH v3 02/14] iommu/riscv: Prepare domain bonds for outer locking Andrew Jones
2026-09-28 14:31 ` [RFC PATCH v3 03/14] iommu/riscv: Serialize MSI table publication with domain attachment Andrew Jones
2026-09-28 14:31 ` [RFC PATCH v3 04/14] iommu/riscv: Reject live S2 replacement with forwarded IRQs Andrew Jones
2026-10-09 9:05 ` Gong Shuai
2026-09-28 14:31 ` [RFC PATCH v3 05/14] iommu/riscv: Derive the IOMMU from the device in IODIR updates Andrew Jones
2026-09-28 14:31 ` [RFC PATCH v3 06/14] iommu/riscv: Cache the programmed device context Andrew Jones
2026-09-28 14:31 ` [RFC PATCH v3 07/14] iommu/riscv: Prepare MSI table updates for interrupt remapping Andrew Jones
2026-09-28 14:31 ` [RFC PATCH v3 08/14] irqchip/riscv-imsic: Define IOMMU IRQ bypass protocol Andrew Jones
2026-09-28 14:31 ` [RFC PATCH v3 09/14] genirq/msi: Provide DOMAIN_BUS_MSI_REMAP Andrew Jones
2026-09-28 14:31 ` [RFC PATCH v3 10/14] iommu/riscv: Add IRQ domain for interrupt remapping Andrew Jones
2026-09-28 14:31 ` [RFC PATCH v3 11/14] iommu/riscv: Prepare info->domain for concurrent RCU access Andrew Jones
2026-09-28 14:31 ` [RFC PATCH v3 12/14] iommu/riscv: Prepare interrupt remapping for IRQ bypass Andrew Jones
2026-09-28 14:31 ` Andrew Jones [this message]
2026-09-28 14:31 ` [RFC PATCH v3 14/14] iommu/riscv: Implement IRQ forwarding Andrew Jones
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260928143113.49838-14-andrew.jones@oss.qualcomm.com \
--to=andrew.jones@oss.qualcomm.com \
--cc=anup@brainfault.org \
--cc=atish.patra@linux.dev \
--cc=fangyu.yu@linux.alibaba.com \
--cc=iommu@lists.linux.dev \
--cc=jgg@nvidia.com \
--cc=jgg@ziepe.ca \
--cc=joro@8bytes.org \
--cc=kvm-riscv@lists.infradead.org \
--cc=kvm@vger.kernel.org \
--cc=linux-kernel@vger.kernel.org \
--cc=linux-riscv@lists.infradead.org \
--cc=palmer@dabbelt.com \
--cc=pjw@kernel.org \
--cc=robin.murphy@arm.com \
--cc=tglx@kernel.org \
--cc=tomasz.jeznach@linux.dev \
--cc=will@kernel.org \
--cc=zhangzhanpeng.jasper@bytedance.com \
--cc=zong.li@sifive.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox