* [RFC PATCH 1/3] iommu/riscv: Add guest IMSIC GPA mapping helpers
2026-10-08 13:59 [RFC PATCH 0/3] iommu/riscv: Add irqbypass support without MSI page table fangyu.yu
@ 2026-10-08 13:59 ` fangyu.yu
2026-10-08 14:00 ` [RFC PATCH 2/3] iommu/riscv: Extract IRQ forwarding payload validation fangyu.yu
2026-10-08 14:00 ` [RFC PATCH 3/3] iommu/riscv: Support IRQ forwarding without MSI page tables fangyu.yu
2 siblings, 0 replies; 4+ messages in thread
From: fangyu.yu @ 2026-10-08 13:59 UTC (permalink / raw)
To: tomasz.jeznach, joro, will, robin.murphy, pjw, palmer, aou, alex,
jroedel, zong.li, andrew.jones, jgg, jgg
Cc: fangyu.yu, guoren, iommu, linux-kernel, linux-riscv, kvm-riscv
From: Fangyu Yu <fangyu.yu@linux.alibaba.com>
IOMMU implementations without the MSI_FLAT capability translate MSI
writes through the second-stage page table, so IRQ forwarding on them
will need guest IMSIC pages mapped into the second-stage domain.
Add an xarray to the MSI table that tracks the HPA installed for each
mapped guest IMSIC GPA, and two helpers for use with the MSI table lock
held: riscv_iommu_msi_table_map_gpa() installs the initial 4 KiB
mapping, and riscv_iommu_msi_table_replace_gpa_leaf() atomically swaps
the leaf PTE's PFN when a vCPU's VS-file host page moves, rejecting
leaves that do not map the expected old HPA.
Signed-off-by: Fangyu Yu <fangyu.yu@linux.alibaba.com>
---
drivers/iommu/riscv/iommu.c | 94 +++++++++++++++++++++++++++++++++++++
drivers/iommu/riscv/iommu.h | 12 +++++
2 files changed, 106 insertions(+)
diff --git a/drivers/iommu/riscv/iommu.c b/drivers/iommu/riscv/iommu.c
index e2e77469ea3c..48fc57d6409c 100644
--- a/drivers/iommu/riscv/iommu.c
+++ b/drivers/iommu/riscv/iommu.c
@@ -24,6 +24,7 @@
#include <linux/moduleparam.h>
#include <linux/mutex.h>
#include <linux/pci.h>
+#include <linux/pgtable.h>
#include <linux/generic_pt/iommu.h>
#include "../dma-iommu.h"
@@ -1217,6 +1218,92 @@ void riscv_iommu_msi_table_inval_all(struct riscv_iommu_msi_table *msi_table)
riscv_iommu_iotlb_inval(domain, &gather);
}
+int riscv_iommu_msi_table_map_gpa(struct riscv_iommu_msi_table *msi_table,
+ dma_addr_t gpa, phys_addr_t hpa)
+{
+ struct riscv_iommu_domain *domain =
+ container_of(msi_table, struct riscv_iommu_domain, msi_table);
+ const int prot = IOMMU_WRITE | IOMMU_NOEXEC | IOMMU_MMIO;
+
+ /* Guest IMSIC GPA mapping only exists in second-stage translations. */
+ if (!domain->gscid)
+ return -EOPNOTSUPP;
+
+ return iommu_map(&domain->domain, gpa, hpa, IMSIC_MMIO_PAGE_SZ, prot,
+ GFP_ATOMIC);
+}
+
+int riscv_iommu_msi_table_replace_gpa_leaf(struct riscv_iommu_msi_table *msi_table,
+ dma_addr_t gpa, phys_addr_t old_hpa,
+ phys_addr_t new_hpa)
+{
+ struct riscv_iommu_domain *domain =
+ container_of(msi_table, struct riscv_iommu_domain, msi_table);
+ struct pt_iommu_riscv_64_hw_info pt_info;
+ u64 *root, *table, *ptep;
+ u64 old, new;
+ int top_level, level;
+
+ if (!IS_ALIGNED(gpa | old_hpa | new_hpa, PAGE_SIZE))
+ return -EINVAL;
+ if (!domain->gscid)
+ return -EOPNOTSUPP;
+
+ pt_iommu_riscv_64_hw_info(&domain->riscvpt, &pt_info);
+ switch (pt_info.iohgatp_mode) {
+ case RISCV_IOMMU_DC_IOHGATP_MODE_SV39X4:
+ top_level = 2;
+ break;
+ case RISCV_IOMMU_DC_IOHGATP_MODE_SV48X4:
+ top_level = 3;
+ break;
+ case RISCV_IOMMU_DC_IOHGATP_MODE_SV57X4:
+ top_level = 4;
+ break;
+ default:
+ return -EINVAL;
+ }
+
+ root = phys_to_virt(pt_info.ppn << PAGE_SHIFT);
+ for (;;) {
+ table = root;
+ for (level = top_level; level >= 0; level--) {
+ unsigned int shift = PAGE_SHIFT + level * 9;
+ unsigned int index = gpa >> shift;
+
+ if (level == top_level)
+ index &= GENMASK(10, 0);
+ else
+ index &= GENMASK(8, 0);
+ ptep = &table[index];
+ old = READ_ONCE(*ptep);
+
+ if (level) {
+ /* A valid non-leaf PTE has R/W/X clear. */
+ if ((old & (_PAGE_PRESENT | _PAGE_LEAF)) !=
+ _PAGE_PRESENT)
+ return -EADDRINUSE;
+ table = phys_to_virt(FIELD_GET(_PAGE_PFN_MASK,
+ old) << PAGE_SHIFT);
+ continue;
+ }
+
+ /* Replace only the L0 leaf previously installed for this GPA. */
+ if (!(old & _PAGE_PRESENT) || !(old & _PAGE_LEAF) ||
+ FIELD_GET(_PAGE_PFN_MASK, old) !=
+ old_hpa >> PAGE_SHIFT)
+ return -EADDRINUSE;
+
+ new = (old & ~_PAGE_PFN_MASK) |
+ FIELD_PREP(_PAGE_PFN_MASK,
+ new_hpa >> PAGE_SHIFT);
+ if (cmpxchg64(ptep, old, new) == old)
+ return 0;
+ break;
+ }
+ }
+}
+
#define RISCV_IOMMU_FSC_BARE 0
/*
* This function sends IOTINVAL commands as required by the RISC-V
@@ -1425,6 +1512,8 @@ static void riscv_iommu_iotlb_sync(struct iommu_domain *iommu_domain,
static void riscv_iommu_free_paging_domain(struct iommu_domain *iommu_domain)
{
struct riscv_iommu_domain *domain = iommu_domain_to_riscv(iommu_domain);
+ struct riscv_iommu_noflat_imsic *imsic;
+ unsigned long index;
WARN_ON(!list_empty(&domain->bonds));
@@ -1435,6 +1524,10 @@ static void riscv_iommu_free_paging_domain(struct iommu_domain *iommu_domain)
if (domain->gscid > 0)
ida_free(&riscv_iommu_gscids, domain->gscid);
+ xa_for_each(&domain->msi_table.noflat_imsics, index, imsic)
+ kfree(imsic);
+ xa_destroy(&domain->msi_table.noflat_imsics);
+
pt_iommu_deinit(&domain->riscvpt.iommu);
iommu_free_pages(domain->msi_table.root);
kfree(domain);
@@ -1676,6 +1769,7 @@ riscv_iommu_domain_alloc_paging_flags(struct device *dev, u32 flags,
INIT_LIST_HEAD_RCU(&domain->bonds);
raw_spin_lock_init(&domain->lock);
raw_spin_lock_init(&domain->msi_table.lock);
+ xa_init(&domain->msi_table.noflat_imsics);
mutex_init(&domain->mutex);
iommu = dev_to_iommu(dev);
cfg.common.hw_max_oasz_lg2 = 56;
diff --git a/drivers/iommu/riscv/iommu.h b/drivers/iommu/riscv/iommu.h
index 9852962e245b..53a368fbbdf2 100644
--- a/drivers/iommu/riscv/iommu.h
+++ b/drivers/iommu/riscv/iommu.h
@@ -18,6 +18,7 @@
#include <linux/irqdomain.h>
#include <linux/rcupdate.h>
#include <linux/sizes.h>
+#include <linux/xarray.h>
#include "iommu-bits.h"
@@ -74,6 +75,11 @@ struct riscv_iommu_device {
struct irq_domain *irqdomain;
};
+/* Tracks a guest IMSIC GPA mapped into an S2 domain on IOMMUs without MSI_FLAT. */
+struct riscv_iommu_noflat_imsic {
+ phys_addr_t hpa;
+};
+
struct riscv_iommu_msi_table {
/* Protects attachment, interrupt forwarding state, and MSI PTE updates. */
raw_spinlock_t lock;
@@ -84,6 +90,7 @@ struct riscv_iommu_msi_table {
u64 msi_addr_pattern;
const void *owner;
u64 required_caps; /* RISCV_IOMMU_CAPABILITIES_* required by active MSI PTEs */
+ struct xarray noflat_imsics;
};
/* Private IOMMU data for managed devices, dev_iommu_priv_* */
@@ -109,6 +116,11 @@ bool riscv_iommu_msi_table_check_caps(struct riscv_iommu_msi_table *msi_table, u
void riscv_iommu_msi_table_inval(struct riscv_iommu_msi_table *msi_table, unsigned long addr);
void riscv_iommu_msi_table_inval_all(struct riscv_iommu_msi_table *msi_table);
void riscv_iommu_msi_table_update(struct riscv_iommu_msi_table *msi_table, bool activate);
+int riscv_iommu_msi_table_map_gpa(struct riscv_iommu_msi_table *msi_table,
+ dma_addr_t gpa, phys_addr_t hpa);
+int riscv_iommu_msi_table_replace_gpa_leaf(struct riscv_iommu_msi_table *msi_table,
+ dma_addr_t gpa, phys_addr_t old_hpa,
+ phys_addr_t new_hpa);
#ifdef CONFIG_RISCV_IMSIC
void riscv_iommu_ir_irq_domain_remove(struct riscv_iommu_device *iommu);
--
2.50.1
^ permalink raw reply [flat|nested] 4+ messages in thread* [RFC PATCH 2/3] iommu/riscv: Extract IRQ forwarding payload validation
2026-10-08 13:59 [RFC PATCH 0/3] iommu/riscv: Add irqbypass support without MSI page table fangyu.yu
2026-10-08 13:59 ` [RFC PATCH 1/3] iommu/riscv: Add guest IMSIC GPA mapping helpers fangyu.yu
@ 2026-10-08 14:00 ` fangyu.yu
2026-10-08 14:00 ` [RFC PATCH 3/3] iommu/riscv: Support IRQ forwarding without MSI page tables fangyu.yu
2 siblings, 0 replies; 4+ messages in thread
From: fangyu.yu @ 2026-10-08 14:00 UTC (permalink / raw)
To: tomasz.jeznach, joro, will, robin.murphy, pjw, palmer, aou, alex,
jroedel, zong.li, andrew.jones, jgg, jgg
Cc: fangyu.yu, guoren, iommu, linux-kernel, linux-riscv, kvm-riscv
From: Fangyu Yu <fangyu.yu@linux.alibaba.com>
Move the owner and MSI address mask/pattern checks from
riscv_iommu_ir_activate() into a separate helper so a second forwarding
backend can validate the same fields. No functional change.
Signed-off-by: Fangyu Yu <fangyu.yu@linux.alibaba.com>
---
drivers/iommu/riscv/iommu-ir.c | 22 ++++++++++++++++------
1 file changed, 16 insertions(+), 6 deletions(-)
diff --git a/drivers/iommu/riscv/iommu-ir.c b/drivers/iommu/riscv/iommu-ir.c
index b3f0a56475ed..b96baff9986d 100644
--- a/drivers/iommu/riscv/iommu-ir.c
+++ b/drivers/iommu/riscv/iommu-ir.c
@@ -104,15 +104,10 @@ static void riscv_iommu_ir_set_target(struct riscv_iommu_msipte *msipte,
}
}
-static int riscv_iommu_ir_activate(struct riscv_iommu_msi_table *msi_table,
- struct riscv_iommu_device *iommu,
- struct riscv_iommu_ir_vcpu_info *vcpu_info)
+static int riscv_iommu_ir_validate_vcpu_info(const struct riscv_iommu_ir_vcpu_info *vcpu_info)
{
u64 pattern = vcpu_info->msi_addr_pattern;
u64 mask = vcpu_info->msi_addr_mask;
- u64 required_caps = 0;
- size_t nr_ptes;
- int ret;
if (!vcpu_info->owner || (pattern & mask) ||
((pattern | mask) & ~RISCV_IOMMU_DC_MSI_ADDR_MASK))
@@ -129,6 +124,21 @@ static int riscv_iommu_ir_activate(struct riscv_iommu_msi_table *msi_table,
return -EINVAL;
}
+ return 0;
+}
+
+static int riscv_iommu_ir_activate(struct riscv_iommu_msi_table *msi_table,
+ struct riscv_iommu_device *iommu,
+ struct riscv_iommu_ir_vcpu_info *vcpu_info)
+{
+ u64 required_caps = 0;
+ size_t nr_ptes;
+ int ret;
+
+ ret = riscv_iommu_ir_validate_vcpu_info(vcpu_info);
+ if (ret)
+ return ret;
+
nr_ptes = riscv_iommu_ir_nr_ptes(vcpu_info);
ret = riscv_iommu_ir_validate_targets(vcpu_info, nr_ptes, &required_caps);
--
2.50.1
^ permalink raw reply [flat|nested] 4+ messages in thread* [RFC PATCH 3/3] iommu/riscv: Support IRQ forwarding without MSI page tables
2026-10-08 13:59 [RFC PATCH 0/3] iommu/riscv: Add irqbypass support without MSI page table fangyu.yu
2026-10-08 13:59 ` [RFC PATCH 1/3] iommu/riscv: Add guest IMSIC GPA mapping helpers fangyu.yu
2026-10-08 14:00 ` [RFC PATCH 2/3] iommu/riscv: Extract IRQ forwarding payload validation fangyu.yu
@ 2026-10-08 14:00 ` fangyu.yu
2 siblings, 0 replies; 4+ messages in thread
From: fangyu.yu @ 2026-10-08 14:00 UTC (permalink / raw)
To: tomasz.jeznach, joro, will, robin.murphy, pjw, palmer, aou, alex,
jroedel, zong.li, andrew.jones, jgg, jgg
Cc: fangyu.yu, guoren, iommu, linux-kernel, linux-riscv, kvm-riscv
From: Fangyu Yu <fangyu.yu@linux.alibaba.com>
IOMMU implementations without the MSI_FLAT capability have no MSI page
table, and their MSI writes are translated by the second-stage page
table. Forward interrupts there by mapping the guest IMSIC GPA to the
host IMSIC HPA in the second-stage domain, so device MSI writes reach
the vCPU's interrupt file directly.
Dispatch on the IOMMU capabilities in the IRQ forwarding entry: with
MSI_FLAT nothing changes; without it, validate that every target is an
IMSIC page in the guest MSI address window, install the GPA mappings,
and replace the leaf PTE when a vCPU migration moves the VS-file host
page. Mappings persist for the lifetime of the domain; disabling
forwarding or a failed operation only drops the per-IRQ state so the
IRQ falls back to host delivery.
Signed-off-by: Fangyu Yu <fangyu.yu@linux.alibaba.com>
---
drivers/iommu/riscv/iommu-ir.c | 157 ++++++++++++++++++++++++++++++++-
1 file changed, 154 insertions(+), 3 deletions(-)
diff --git a/drivers/iommu/riscv/iommu-ir.c b/drivers/iommu/riscv/iommu-ir.c
index b96baff9986d..5f3a909c2206 100644
--- a/drivers/iommu/riscv/iommu-ir.c
+++ b/drivers/iommu/riscv/iommu-ir.c
@@ -271,14 +271,160 @@ static int riscv_iommu_ir_update_target(struct riscv_iommu_msi_table *msi_table,
return 0;
}
-static int riscv_iommu_ir_irq_set_vcpu_affinity_locked(struct irq_data *data,
+static int riscv_iommu_ir_validate_noflat_target(const struct riscv_iommu_ir_vcpu_info *vcpu_info,
+ const struct riscv_iommu_ir_target *target)
+{
+ u64 addr = target->gpa >> IMSIC_MMIO_PAGE_SHIFT;
+
+ if (!IS_ALIGNED(target->gpa, IMSIC_MMIO_PAGE_SZ) ||
+ (addr & ~vcpu_info->msi_addr_mask) != vcpu_info->msi_addr_pattern)
+ return -EINVAL;
+
+ /* Without MSI_FLAT there is no MSI page table, so only IMSIC targets work. */
+ if (target->type != RISCV_IOMMU_IR_TARGET_IMSIC)
+ return -EOPNOTSUPP;
+
+ if (!IS_ALIGNED(target->hpa, IMSIC_MMIO_PAGE_SZ) ||
+ (target->hpa >> IMSIC_MMIO_PAGE_SHIFT) > FIELD_MAX(RISCV_IOMMU_MSIPTE_PPN))
+ return -EINVAL;
+
+ return 0;
+}
+
+/* Caller must hold msi_table->lock. */
+static int riscv_iommu_ir_map_guest_imsic(struct riscv_iommu_msi_table *msi_table,
+ const struct riscv_iommu_ir_target *target)
+{
+ unsigned long index = target->gpa >> IMSIC_MMIO_PAGE_SHIFT;
+ struct riscv_iommu_noflat_imsic *imsic;
+ int ret;
+
+ imsic = xa_load(&msi_table->noflat_imsics, index);
+ if (imsic) {
+ if (imsic->hpa == target->hpa)
+ return 0;
+
+ ret = riscv_iommu_msi_table_replace_gpa_leaf(msi_table, target->gpa,
+ imsic->hpa, target->hpa);
+ if (ret)
+ return ret;
+ } else {
+ imsic = kzalloc_obj(*imsic, GFP_ATOMIC);
+ if (!imsic)
+ return -ENOMEM;
+
+ ret = xa_err(xa_store(&msi_table->noflat_imsics, index, imsic,
+ GFP_ATOMIC));
+ if (ret) {
+ kfree(imsic);
+ return ret;
+ }
+
+ ret = riscv_iommu_msi_table_map_gpa(msi_table, target->gpa, target->hpa);
+ if (ret) {
+ xa_erase(&msi_table->noflat_imsics, index);
+ kfree(imsic);
+ return ret;
+ }
+ }
+
+ imsic->hpa = target->hpa;
+ riscv_iommu_msi_table_inval(msi_table, target->gpa);
+ return 0;
+}
+
+static int riscv_iommu_ir_activate_noflat(struct riscv_iommu_msi_table *msi_table,
+ struct riscv_iommu_ir_vcpu_info *vcpu_info)
+{
+ int ret;
+
+ ret = riscv_iommu_ir_validate_vcpu_info(vcpu_info);
+ if (ret)
+ return ret;
+
+ if (!vcpu_info->targets || !vcpu_info->nr_targets)
+ return -EINVAL;
+
+ for (unsigned int i = 0; i < vcpu_info->nr_targets; i++) {
+ ret = riscv_iommu_ir_validate_noflat_target(vcpu_info, &vcpu_info->targets[i]);
+ if (ret)
+ return ret;
+ }
+
+ for (unsigned int i = 0; i < vcpu_info->nr_targets; i++) {
+ ret = riscv_iommu_ir_map_guest_imsic(msi_table, &vcpu_info->targets[i]);
+ if (ret)
+ return ret;
+ }
+
+ return 0;
+}
+
+static int riscv_iommu_ir_update_target_noflat(struct riscv_iommu_msi_table *msi_table,
+ struct riscv_iommu_ir_vcpu_info *vcpu_info)
+{
+ const struct riscv_iommu_ir_target *target = &vcpu_info->target;
+ int ret;
+
+ ret = riscv_iommu_ir_validate_vcpu_info(vcpu_info);
+ if (ret)
+ return ret;
+
+ ret = riscv_iommu_ir_validate_noflat_target(vcpu_info, target);
+ if (ret)
+ return ret;
+
+ if (!xa_load(&msi_table->noflat_imsics, target->gpa >> IMSIC_MMIO_PAGE_SHIFT))
+ return -EINVAL;
+
+ return riscv_iommu_ir_map_guest_imsic(msi_table, target);
+}
+
+static int riscv_iommu_ir_irq_set_vcpu_affinity_noflat(struct irq_data *data,
struct riscv_iommu_info *info,
struct riscv_iommu_ir_vcpu_info *vcpu_info,
struct riscv_iommu_msi_table *msi_table)
+{
+ int ret;
+
+ if (!vcpu_info) {
+ /* Mappings persist; only the per-IRQ forwarding state is dropped. */
+ if (irqd_is_forwarded_to_vcpu(data)) {
+ irqd_clr_forwarded_to_vcpu(data);
+ info->nr_forwarded_irqs--;
+ }
+ return 0;
+ }
+
+ ret = vcpu_info->cmd == RISCV_IOMMU_IR_FORWARD ?
+ riscv_iommu_ir_activate_noflat(msi_table, vcpu_info) :
+ riscv_iommu_ir_update_target_noflat(msi_table, vcpu_info);
+ if (!ret) {
+ if (!irqd_is_forwarded_to_vcpu(data)) {
+ irqd_set_forwarded_to_vcpu(data);
+ info->nr_forwarded_irqs++;
+ }
+ } else if (irqd_is_forwarded_to_vcpu(data)) {
+ irqd_clr_forwarded_to_vcpu(data);
+ info->nr_forwarded_irqs--;
+ }
+
+ return ret;
+}
+
+static int riscv_iommu_ir_irq_set_vcpu_affinity_locked(struct irq_data *data,
+ struct riscv_iommu_info *info,
+ struct riscv_iommu_ir_vcpu_info *vcpu_info,
+ struct riscv_iommu_msi_table *msi_table,
+ bool noflat)
{
struct riscv_iommu_device *iommu = data->domain->host_data;
int ret;
+ if (noflat)
+ return riscv_iommu_ir_irq_set_vcpu_affinity_noflat(data, info,
+ vcpu_info, msi_table);
+
if (!vcpu_info) {
if (WARN_ON_ONCE(!msi_table->nr_forwarded_irqs || !info->nr_forwarded_irqs))
return -EINVAL;
@@ -332,10 +478,12 @@ static int riscv_iommu_ir_irq_set_vcpu_affinity_locked(struct irq_data *data,
static int riscv_iommu_ir_irq_set_vcpu_affinity(struct irq_data *data, void *arg)
{
struct riscv_iommu_ir_vcpu_info *vcpu_info = arg;
+ struct riscv_iommu_device *iommu = data->domain->host_data;
struct riscv_iommu_msi_table *msi_table;
struct riscv_iommu_info *info;
struct msi_desc *desc;
struct device *dev;
+ bool noflat;
int ret;
if (!vcpu_info && !irqd_is_forwarded_to_vcpu(data))
@@ -354,6 +502,8 @@ static int riscv_iommu_ir_irq_set_vcpu_affinity(struct irq_data *data, void *arg
if (WARN_ON_ONCE(!info))
return -EINVAL;
+ noflat = !(iommu->caps & RISCV_IOMMU_CAPABILITIES_MSI_FLAT);
+
scoped_guard(rcu) {
/*
* RCU keeps the table alive, but the device may switch domains before
@@ -361,7 +511,7 @@ static int riscv_iommu_ir_irq_set_vcpu_affinity(struct irq_data *data, void *arg
*/
for (;;) {
msi_table = riscv_iommu_msi_table_rcu(info);
- if (!msi_table || !msi_table->root)
+ if (!msi_table || (!noflat && !msi_table->root))
return -EOPNOTSUPP;
raw_spin_lock(&msi_table->lock);
@@ -371,7 +521,8 @@ static int riscv_iommu_ir_irq_set_vcpu_affinity(struct irq_data *data, void *arg
}
}
- ret = riscv_iommu_ir_irq_set_vcpu_affinity_locked(data, info, vcpu_info, msi_table);
+ ret = riscv_iommu_ir_irq_set_vcpu_affinity_locked(data, info, vcpu_info,
+ msi_table, noflat);
raw_spin_unlock(&msi_table->lock);
return ret;
--
2.50.1
^ permalink raw reply [flat|nested] 4+ messages in thread