From: fangyu.yu@linux.alibaba.com
To: tomasz.jeznach@linux.dev, joro@8bytes.org, will@kernel.org,
robin.murphy@arm.com, pjw@kernel.org, palmer@dabbelt.com,
aou@eecs.berkeley.edu, alex@ghiti.fr, jroedel@suse.de,
zong.li@sifive.com, andrew.jones@oss.qualcomm.com,
jgg@nvidia.com, jgg@ziepe.ca
Cc: fangyu.yu@linux.alibaba.com, guoren@kernel.org,
iommu@lists.linux.dev, linux-kernel@vger.kernel.org,
linux-riscv@lists.infradead.org, kvm-riscv@lists.infradead.org
Subject: [RFC PATCH 3/3] iommu/riscv: Support IRQ forwarding without MSI page tables
Date: Thu, 8 Oct 2026 22:00:01 +0800 [thread overview]
Message-ID: <20261008140001.94508-4-fangyu.yu@linux.alibaba.com> (raw)
In-Reply-To: <20261008140001.94508-1-fangyu.yu@linux.alibaba.com>
From: Fangyu Yu <fangyu.yu@linux.alibaba.com>
IOMMU implementations without the MSI_FLAT capability have no MSI page
table, and their MSI writes are translated by the second-stage page
table. Forward interrupts there by mapping the guest IMSIC GPA to the
host IMSIC HPA in the second-stage domain, so device MSI writes reach
the vCPU's interrupt file directly.
Dispatch on the IOMMU capabilities in the IRQ forwarding entry: with
MSI_FLAT nothing changes; without it, validate that every target is an
IMSIC page in the guest MSI address window, install the GPA mappings,
and replace the leaf PTE when a vCPU migration moves the VS-file host
page. Mappings persist for the lifetime of the domain; disabling
forwarding or a failed operation only drops the per-IRQ state so the
IRQ falls back to host delivery.
Signed-off-by: Fangyu Yu <fangyu.yu@linux.alibaba.com>
---
drivers/iommu/riscv/iommu-ir.c | 157 ++++++++++++++++++++++++++++++++-
1 file changed, 154 insertions(+), 3 deletions(-)
diff --git a/drivers/iommu/riscv/iommu-ir.c b/drivers/iommu/riscv/iommu-ir.c
index b96baff9986d..5f3a909c2206 100644
--- a/drivers/iommu/riscv/iommu-ir.c
+++ b/drivers/iommu/riscv/iommu-ir.c
@@ -271,14 +271,160 @@ static int riscv_iommu_ir_update_target(struct riscv_iommu_msi_table *msi_table,
return 0;
}
-static int riscv_iommu_ir_irq_set_vcpu_affinity_locked(struct irq_data *data,
+static int riscv_iommu_ir_validate_noflat_target(const struct riscv_iommu_ir_vcpu_info *vcpu_info,
+ const struct riscv_iommu_ir_target *target)
+{
+ u64 addr = target->gpa >> IMSIC_MMIO_PAGE_SHIFT;
+
+ if (!IS_ALIGNED(target->gpa, IMSIC_MMIO_PAGE_SZ) ||
+ (addr & ~vcpu_info->msi_addr_mask) != vcpu_info->msi_addr_pattern)
+ return -EINVAL;
+
+ /* Without MSI_FLAT there is no MSI page table, so only IMSIC targets work. */
+ if (target->type != RISCV_IOMMU_IR_TARGET_IMSIC)
+ return -EOPNOTSUPP;
+
+ if (!IS_ALIGNED(target->hpa, IMSIC_MMIO_PAGE_SZ) ||
+ (target->hpa >> IMSIC_MMIO_PAGE_SHIFT) > FIELD_MAX(RISCV_IOMMU_MSIPTE_PPN))
+ return -EINVAL;
+
+ return 0;
+}
+
+/* Caller must hold msi_table->lock. */
+static int riscv_iommu_ir_map_guest_imsic(struct riscv_iommu_msi_table *msi_table,
+ const struct riscv_iommu_ir_target *target)
+{
+ unsigned long index = target->gpa >> IMSIC_MMIO_PAGE_SHIFT;
+ struct riscv_iommu_noflat_imsic *imsic;
+ int ret;
+
+ imsic = xa_load(&msi_table->noflat_imsics, index);
+ if (imsic) {
+ if (imsic->hpa == target->hpa)
+ return 0;
+
+ ret = riscv_iommu_msi_table_replace_gpa_leaf(msi_table, target->gpa,
+ imsic->hpa, target->hpa);
+ if (ret)
+ return ret;
+ } else {
+ imsic = kzalloc_obj(*imsic, GFP_ATOMIC);
+ if (!imsic)
+ return -ENOMEM;
+
+ ret = xa_err(xa_store(&msi_table->noflat_imsics, index, imsic,
+ GFP_ATOMIC));
+ if (ret) {
+ kfree(imsic);
+ return ret;
+ }
+
+ ret = riscv_iommu_msi_table_map_gpa(msi_table, target->gpa, target->hpa);
+ if (ret) {
+ xa_erase(&msi_table->noflat_imsics, index);
+ kfree(imsic);
+ return ret;
+ }
+ }
+
+ imsic->hpa = target->hpa;
+ riscv_iommu_msi_table_inval(msi_table, target->gpa);
+ return 0;
+}
+
+static int riscv_iommu_ir_activate_noflat(struct riscv_iommu_msi_table *msi_table,
+ struct riscv_iommu_ir_vcpu_info *vcpu_info)
+{
+ int ret;
+
+ ret = riscv_iommu_ir_validate_vcpu_info(vcpu_info);
+ if (ret)
+ return ret;
+
+ if (!vcpu_info->targets || !vcpu_info->nr_targets)
+ return -EINVAL;
+
+ for (unsigned int i = 0; i < vcpu_info->nr_targets; i++) {
+ ret = riscv_iommu_ir_validate_noflat_target(vcpu_info, &vcpu_info->targets[i]);
+ if (ret)
+ return ret;
+ }
+
+ for (unsigned int i = 0; i < vcpu_info->nr_targets; i++) {
+ ret = riscv_iommu_ir_map_guest_imsic(msi_table, &vcpu_info->targets[i]);
+ if (ret)
+ return ret;
+ }
+
+ return 0;
+}
+
+static int riscv_iommu_ir_update_target_noflat(struct riscv_iommu_msi_table *msi_table,
+ struct riscv_iommu_ir_vcpu_info *vcpu_info)
+{
+ const struct riscv_iommu_ir_target *target = &vcpu_info->target;
+ int ret;
+
+ ret = riscv_iommu_ir_validate_vcpu_info(vcpu_info);
+ if (ret)
+ return ret;
+
+ ret = riscv_iommu_ir_validate_noflat_target(vcpu_info, target);
+ if (ret)
+ return ret;
+
+ if (!xa_load(&msi_table->noflat_imsics, target->gpa >> IMSIC_MMIO_PAGE_SHIFT))
+ return -EINVAL;
+
+ return riscv_iommu_ir_map_guest_imsic(msi_table, target);
+}
+
+static int riscv_iommu_ir_irq_set_vcpu_affinity_noflat(struct irq_data *data,
struct riscv_iommu_info *info,
struct riscv_iommu_ir_vcpu_info *vcpu_info,
struct riscv_iommu_msi_table *msi_table)
+{
+ int ret;
+
+ if (!vcpu_info) {
+ /* Mappings persist; only the per-IRQ forwarding state is dropped. */
+ if (irqd_is_forwarded_to_vcpu(data)) {
+ irqd_clr_forwarded_to_vcpu(data);
+ info->nr_forwarded_irqs--;
+ }
+ return 0;
+ }
+
+ ret = vcpu_info->cmd == RISCV_IOMMU_IR_FORWARD ?
+ riscv_iommu_ir_activate_noflat(msi_table, vcpu_info) :
+ riscv_iommu_ir_update_target_noflat(msi_table, vcpu_info);
+ if (!ret) {
+ if (!irqd_is_forwarded_to_vcpu(data)) {
+ irqd_set_forwarded_to_vcpu(data);
+ info->nr_forwarded_irqs++;
+ }
+ } else if (irqd_is_forwarded_to_vcpu(data)) {
+ irqd_clr_forwarded_to_vcpu(data);
+ info->nr_forwarded_irqs--;
+ }
+
+ return ret;
+}
+
+static int riscv_iommu_ir_irq_set_vcpu_affinity_locked(struct irq_data *data,
+ struct riscv_iommu_info *info,
+ struct riscv_iommu_ir_vcpu_info *vcpu_info,
+ struct riscv_iommu_msi_table *msi_table,
+ bool noflat)
{
struct riscv_iommu_device *iommu = data->domain->host_data;
int ret;
+ if (noflat)
+ return riscv_iommu_ir_irq_set_vcpu_affinity_noflat(data, info,
+ vcpu_info, msi_table);
+
if (!vcpu_info) {
if (WARN_ON_ONCE(!msi_table->nr_forwarded_irqs || !info->nr_forwarded_irqs))
return -EINVAL;
@@ -332,10 +478,12 @@ static int riscv_iommu_ir_irq_set_vcpu_affinity_locked(struct irq_data *data,
static int riscv_iommu_ir_irq_set_vcpu_affinity(struct irq_data *data, void *arg)
{
struct riscv_iommu_ir_vcpu_info *vcpu_info = arg;
+ struct riscv_iommu_device *iommu = data->domain->host_data;
struct riscv_iommu_msi_table *msi_table;
struct riscv_iommu_info *info;
struct msi_desc *desc;
struct device *dev;
+ bool noflat;
int ret;
if (!vcpu_info && !irqd_is_forwarded_to_vcpu(data))
@@ -354,6 +502,8 @@ static int riscv_iommu_ir_irq_set_vcpu_affinity(struct irq_data *data, void *arg
if (WARN_ON_ONCE(!info))
return -EINVAL;
+ noflat = !(iommu->caps & RISCV_IOMMU_CAPABILITIES_MSI_FLAT);
+
scoped_guard(rcu) {
/*
* RCU keeps the table alive, but the device may switch domains before
@@ -361,7 +511,7 @@ static int riscv_iommu_ir_irq_set_vcpu_affinity(struct irq_data *data, void *arg
*/
for (;;) {
msi_table = riscv_iommu_msi_table_rcu(info);
- if (!msi_table || !msi_table->root)
+ if (!msi_table || (!noflat && !msi_table->root))
return -EOPNOTSUPP;
raw_spin_lock(&msi_table->lock);
@@ -371,7 +521,8 @@ static int riscv_iommu_ir_irq_set_vcpu_affinity(struct irq_data *data, void *arg
}
}
- ret = riscv_iommu_ir_irq_set_vcpu_affinity_locked(data, info, vcpu_info, msi_table);
+ ret = riscv_iommu_ir_irq_set_vcpu_affinity_locked(data, info, vcpu_info,
+ msi_table, noflat);
raw_spin_unlock(&msi_table->lock);
return ret;
--
2.50.1
prev parent reply other threads:[~2026-10-08 14:00 UTC|newest]
Thread overview: 5+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-10-08 13:59 [RFC PATCH 0/3] iommu/riscv: Add irqbypass support without MSI page table fangyu.yu
2026-10-08 13:59 ` [RFC PATCH 1/3] iommu/riscv: Add guest IMSIC GPA mapping helpers fangyu.yu
2026-10-09 8:50 ` Gong Shuai
2026-10-08 14:00 ` [RFC PATCH 2/3] iommu/riscv: Extract IRQ forwarding payload validation fangyu.yu
2026-10-08 14:00 ` fangyu.yu [this message]
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20261008140001.94508-4-fangyu.yu@linux.alibaba.com \
--to=fangyu.yu@linux.alibaba.com \
--cc=alex@ghiti.fr \
--cc=andrew.jones@oss.qualcomm.com \
--cc=aou@eecs.berkeley.edu \
--cc=guoren@kernel.org \
--cc=iommu@lists.linux.dev \
--cc=jgg@nvidia.com \
--cc=jgg@ziepe.ca \
--cc=joro@8bytes.org \
--cc=jroedel@suse.de \
--cc=kvm-riscv@lists.infradead.org \
--cc=linux-kernel@vger.kernel.org \
--cc=linux-riscv@lists.infradead.org \
--cc=palmer@dabbelt.com \
--cc=pjw@kernel.org \
--cc=robin.murphy@arm.com \
--cc=tomasz.jeznach@linux.dev \
--cc=will@kernel.org \
--cc=zong.li@sifive.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox
all inboxes | Powered by JetHome®