* [PATCH v6 01/16] iommu/dma: Prepare MSI physical address lists
2026-09-25 15:16 [PATCH v6 00/16] iommu/riscv: Enable MSI remapping, IOMMU_DMA and VFIO Andrew Jones
@ 2026-09-25 15:16 ` Andrew Jones
2026-10-02 16:21 ` Robin Murphy
2026-09-25 15:16 ` [PATCH v6 02/16] iommufd: Convert struct iommufd_sw_msi_maps to a growable bitmap Andrew Jones
` (15 subsequent siblings)
16 siblings, 1 reply; 31+ messages in thread
From: Andrew Jones @ 2026-09-25 15:16 UTC (permalink / raw)
To: linux-riscv, iommu
Cc: linux-kernel, tomasz.jeznach, tjeznach, jgg, jgg, joro, will,
robin.murphy, pjw, palmer, anup, tglx, kevin.tian, fangyu.yu
Software MSI mappings may cover an ordered list of physical addresses
which must be mapped into one contiguous IOVA range. Extend the internal
DMA-IOMMU mapping helpers to accept an address array, address count, and
required mapping granule.
Teach iommu_dma_get_msi_page() to map the list into one contiguous IOVA
allocation. Keep the existing per-page cache entries and mark the first
entry with the size of the allocation so an identical list can reuse the
mapping without changing the existing cache representation.
This prepares for the forthcoming iommu_dma_prepare_msi_list() API.
Signed-off-by: Andrew Jones <andrew.jones@oss.qualcomm.com>
Tested-by: Fangyu Yu <fangyu.yu@linux.alibaba.com>
---
drivers/iommu/dma-iommu.c | 116 +++++++++++++++++++++++++++++---------
1 file changed, 88 insertions(+), 28 deletions(-)
diff --git a/drivers/iommu/dma-iommu.c b/drivers/iommu/dma-iommu.c
index 58c624513cd4..0b4eb47d1a95 100644
--- a/drivers/iommu/dma-iommu.c
+++ b/drivers/iommu/dma-iommu.c
@@ -42,6 +42,7 @@ struct iommu_dma_msi_page {
struct list_head list;
dma_addr_t iova;
phys_addr_t phys;
+ size_t range_size; /* IOVA range size, or 0 if not the first map */
};
enum iommu_dma_queue_type {
@@ -481,7 +482,7 @@ static int cookie_init_hw_msi_region(struct iommu_dma_cookie *cookie,
num_pages = iova_align(iovad, end - start) >> iova_shift(iovad);
for (i = 0; i < num_pages; i++) {
- msi_page = kmalloc_obj(*msi_page);
+ msi_page = kzalloc_obj(*msi_page);
if (!msi_page)
return -ENOMEM;
@@ -2191,15 +2192,42 @@ static struct list_head *cookie_msi_pages(const struct iommu_domain *domain)
}
}
+static bool iommu_dma_msi_range_matches(struct list_head *msi_page_list,
+ const struct iommu_dma_msi_page *base_page,
+ const phys_addr_t *phys_addrs,
+ unsigned int nr_addrs, size_t granule)
+{
+ const struct iommu_dma_msi_page *msi_page;
+ unsigned int nr_found = 0;
+ dma_addr_t offset;
+
+ list_for_each_entry(msi_page, msi_page_list, list) {
+ if (msi_page->iova < base_page->iova)
+ continue;
+ offset = msi_page->iova - base_page->iova;
+ if (offset >= base_page->range_size)
+ continue;
+ if (!IS_ALIGNED((size_t)offset, granule) ||
+ msi_page->phys != phys_addrs[(size_t)offset / granule])
+ return false;
+ nr_found++;
+ }
+
+ return nr_found == nr_addrs;
+}
+
static struct iommu_dma_msi_page *iommu_dma_get_msi_page(struct device *dev,
- phys_addr_t msi_addr, struct iommu_domain *domain)
+ const phys_addr_t *phys_addrs, unsigned int nr_addrs, size_t granule,
+ struct iommu_domain *domain)
{
struct list_head *msi_page_list = cookie_msi_pages(domain);
- struct iommu_dma_msi_page *msi_page;
- dma_addr_t iova;
+ struct iommu_dma_msi_page *msi_page, *first_page = NULL;
int prot = IOMMU_WRITE | IOMMU_NOEXEC | IOMMU_MMIO;
- size_t size = cookie_msi_granule(domain);
static DEFINE_MUTEX(msi_prepare_lock);
+ LIST_HEAD(new_msi_pages);
+ dma_addr_t base_iova;
+ unsigned int i;
+ size_t size;
/*
* Normally a device's default domain is only ever attached to that
@@ -2213,32 +2241,60 @@ static struct iommu_dma_msi_page *iommu_dma_get_msi_page(struct device *dev,
*/
guard(mutex)(&msi_prepare_lock);
- msi_addr &= ~(phys_addr_t)(size - 1);
- list_for_each_entry(msi_page, msi_page_list, list)
- if (msi_page->phys == msi_addr)
+ if (!nr_addrs || nr_addrs > SIZE_MAX / granule)
+ return NULL;
+ size = nr_addrs * granule;
+
+ list_for_each_entry(msi_page, msi_page_list, list) {
+ if (msi_page->phys != phys_addrs[0])
+ continue;
+ if (nr_addrs == 1)
+ return msi_page;
+ if (msi_page->range_size == size &&
+ iommu_dma_msi_range_matches(msi_page_list, msi_page, phys_addrs,
+ nr_addrs, granule))
return msi_page;
+ }
- msi_page = kzalloc_obj(*msi_page);
- if (!msi_page)
- return NULL;
+ for (i = 0; i < nr_addrs; i++) {
+ msi_page = kzalloc_obj(*msi_page);
+ if (!msi_page)
+ goto out_free_pages;
+ list_add_tail(&msi_page->list, &new_msi_pages);
+ }
- iova = iommu_dma_alloc_iova(domain, size, dma_get_mask(dev), dev);
- if (!iova)
- goto out_free_page;
+ base_iova = iommu_dma_alloc_iova(domain, size, dma_get_mask(dev), dev);
+ if (!base_iova)
+ goto out_free_pages;
- if (iommu_map(domain, iova, msi_addr, size, prot, GFP_KERNEL))
- goto out_free_iova;
+ i = 0;
+ list_for_each_entry(msi_page, &new_msi_pages, list) {
+ msi_page->phys = phys_addrs[i];
+ msi_page->iova = base_iova + i * granule;
+ if (!i)
+ first_page = msi_page;
+ if (iommu_map(domain, msi_page->iova, msi_page->phys, granule, prot, GFP_KERNEL))
+ goto out_unmap;
+ i++;
+ }
- INIT_LIST_HEAD(&msi_page->list);
- msi_page->phys = msi_addr;
- msi_page->iova = iova;
- list_add(&msi_page->list, msi_page_list);
- return msi_page;
+ first_page->range_size = size;
+ list_splice(&new_msi_pages, msi_page_list);
+ return first_page;
-out_free_iova:
- iommu_dma_free_iova(domain, iova, size, NULL);
-out_free_page:
- kfree(msi_page);
+out_unmap:
+ if (i) {
+ size_t unmap = iommu_unmap(domain, base_iova, i * granule);
+
+ WARN_ON_ONCE(unmap != i * granule);
+ }
+ iommu_dma_free_iova(domain, base_iova, size, NULL);
+out_free_pages:
+ while (!list_empty(&new_msi_pages)) {
+ msi_page = list_first_entry(&new_msi_pages, typeof(*msi_page), list);
+ list_del(&msi_page->list);
+ kfree(msi_page);
+ }
return NULL;
}
@@ -2247,19 +2303,23 @@ int iommu_dma_sw_msi(struct iommu_domain *domain, struct msi_desc *desc,
{
struct device *dev = msi_desc_to_dev(desc);
const struct iommu_dma_msi_page *msi_page;
+ phys_addr_t phys_addr;
+ size_t granule;
if (!has_msi_cookie(domain)) {
msi_desc_set_iommu_msi_iova(desc, 0, 0);
return 0;
}
+ granule = cookie_msi_granule(domain);
+ phys_addr = ALIGN_DOWN(msi_addr, granule);
+
iommu_group_mutex_assert(dev);
- msi_page = iommu_dma_get_msi_page(dev, msi_addr, domain);
+ msi_page = iommu_dma_get_msi_page(dev, &phys_addr, 1, granule, domain);
if (!msi_page)
return -ENOMEM;
- msi_desc_set_iommu_msi_iova(desc, msi_page->iova,
- ilog2(cookie_msi_granule(domain)));
+ msi_desc_set_iommu_msi_iova(desc, msi_page->iova, ilog2(granule));
return 0;
}
--
2.43.0
^ permalink raw reply [flat|nested] 31+ messages in thread* Re: [PATCH v6 01/16] iommu/dma: Prepare MSI physical address lists
2026-09-25 15:16 ` [PATCH v6 01/16] iommu/dma: Prepare MSI physical address lists Andrew Jones
@ 2026-10-02 16:21 ` Robin Murphy
2026-10-03 12:03 ` Andrew Jones
0 siblings, 1 reply; 31+ messages in thread
From: Robin Murphy @ 2026-10-02 16:21 UTC (permalink / raw)
To: Andrew Jones, linux-riscv, iommu
Cc: linux-kernel, tomasz.jeznach, tjeznach, jgg, jgg, joro, will,
pjw, palmer, anup, tglx, kevin.tian, fangyu.yu
On 25/09/2026 4:16 pm, Andrew Jones wrote:
> Software MSI mappings may cover an ordered list of physical addresses
> which must be mapped into one contiguous IOVA range. Extend the internal
> DMA-IOMMU mapping helpers to accept an address array, address count, and
> required mapping granule.
>
> Teach iommu_dma_get_msi_page() to map the list into one contiguous IOVA
> allocation. Keep the existing per-page cache entries and mark the first
> entry with the size of the allocation so an identical list can reuse the
> mapping without changing the existing cache representation.
>
> This prepares for the forthcoming iommu_dma_prepare_msi_list() API.
Sorry, but this looks pretty bonkers, even before we get to patch #8.
You already end up adding what is effectively a RISC-V-specific
entrypoint, so you may as well just carry that all the way through to
its own effectively RISC-V-specific implementation, without making an
unmaintainable mess of the existing code.
The current design for both DMA_IOVA and DMA_MSI cookies is based on the
(Arm-centric) notion that different devices may be associated with
different MSI controllers that are independent of each other, so we only
map what we know we need (and have a way to get the address of at all),
and the list lookup is to save redundant mappings and IOVA space when
devices do happen to share. Your requirement is almost the complete
opposite, where *any* device needs *every* possible MSI controller
mapped up-front, plus the MSI controller driver has to be in on this
notion too, so it really doesn't fit the same logic well at all. In fact
IIUC it should be far simpler - you shouldn't need a list, nor even
really care about the addresses, it should merely be a case of whether
a) this is the first call for the given cookie so everything needs
mapping, or b) it's not the first call, so everything must already be
mapped and we can just return the IOVA. If trying to cram these opposing
notions down the same path results in a bunch of weird complexity for
pretending to support distinct and overlapping values of "everything",
which neither case needs, that seems like a pretty clear sign of it
being a bad idea IMO.
Note that since the core cookie rework, untangling the DMA_MSI
implementation from DMA_IOVA has been on the table, at which point
adding more top-level types of MSI-only cookies would clearly be
straightforward. However for DMA_IOVA that could end up getting a bit
combinatorial, and keeping an internal sub-type (like for flush queues)
would probably be simpler, so it may well make sense to take the latter
approach for both, at least to start with. But getting rid of those odd
special cases in iommu_dma_{alloc,free}_iova() and the clunkiness of
cookie_msi_{granule,pages}() would still be nice either way...
At very worst, a separate hook to just pre-populate msi_page_list with a
regular page for each IMSIC address, such that the existing reuse
mechanism keeps working as-is, could probably suffice without any other
major structural changes; it's only really the IOVA allocation and the
fact that it all has to be done under a single lock acquisition that's
special.
Thanks,
Robin.
> Signed-off-by: Andrew Jones <andrew.jones@oss.qualcomm.com>
> Tested-by: Fangyu Yu <fangyu.yu@linux.alibaba.com>
> ---
> drivers/iommu/dma-iommu.c | 116 +++++++++++++++++++++++++++++---------
> 1 file changed, 88 insertions(+), 28 deletions(-)
>
> diff --git a/drivers/iommu/dma-iommu.c b/drivers/iommu/dma-iommu.c
> index 58c624513cd4..0b4eb47d1a95 100644
> --- a/drivers/iommu/dma-iommu.c
> +++ b/drivers/iommu/dma-iommu.c
> @@ -42,6 +42,7 @@ struct iommu_dma_msi_page {
> struct list_head list;
> dma_addr_t iova;
> phys_addr_t phys;
> + size_t range_size; /* IOVA range size, or 0 if not the first map */
> };
>
> enum iommu_dma_queue_type {
> @@ -481,7 +482,7 @@ static int cookie_init_hw_msi_region(struct iommu_dma_cookie *cookie,
> num_pages = iova_align(iovad, end - start) >> iova_shift(iovad);
>
> for (i = 0; i < num_pages; i++) {
> - msi_page = kmalloc_obj(*msi_page);
> + msi_page = kzalloc_obj(*msi_page);
> if (!msi_page)
> return -ENOMEM;
>
> @@ -2191,15 +2192,42 @@ static struct list_head *cookie_msi_pages(const struct iommu_domain *domain)
> }
> }
>
> +static bool iommu_dma_msi_range_matches(struct list_head *msi_page_list,
> + const struct iommu_dma_msi_page *base_page,
> + const phys_addr_t *phys_addrs,
> + unsigned int nr_addrs, size_t granule)
> +{
> + const struct iommu_dma_msi_page *msi_page;
> + unsigned int nr_found = 0;
> + dma_addr_t offset;
> +
> + list_for_each_entry(msi_page, msi_page_list, list) {
> + if (msi_page->iova < base_page->iova)
> + continue;
> + offset = msi_page->iova - base_page->iova;
> + if (offset >= base_page->range_size)
> + continue;
> + if (!IS_ALIGNED((size_t)offset, granule) ||
> + msi_page->phys != phys_addrs[(size_t)offset / granule])
> + return false;
> + nr_found++;
> + }
> +
> + return nr_found == nr_addrs;
> +}
> +
> static struct iommu_dma_msi_page *iommu_dma_get_msi_page(struct device *dev,
> - phys_addr_t msi_addr, struct iommu_domain *domain)
> + const phys_addr_t *phys_addrs, unsigned int nr_addrs, size_t granule,
> + struct iommu_domain *domain)
> {
> struct list_head *msi_page_list = cookie_msi_pages(domain);
> - struct iommu_dma_msi_page *msi_page;
> - dma_addr_t iova;
> + struct iommu_dma_msi_page *msi_page, *first_page = NULL;
> int prot = IOMMU_WRITE | IOMMU_NOEXEC | IOMMU_MMIO;
> - size_t size = cookie_msi_granule(domain);
> static DEFINE_MUTEX(msi_prepare_lock);
> + LIST_HEAD(new_msi_pages);
> + dma_addr_t base_iova;
> + unsigned int i;
> + size_t size;
>
> /*
> * Normally a device's default domain is only ever attached to that
> @@ -2213,32 +2241,60 @@ static struct iommu_dma_msi_page *iommu_dma_get_msi_page(struct device *dev,
> */
> guard(mutex)(&msi_prepare_lock);
>
> - msi_addr &= ~(phys_addr_t)(size - 1);
> - list_for_each_entry(msi_page, msi_page_list, list)
> - if (msi_page->phys == msi_addr)
> + if (!nr_addrs || nr_addrs > SIZE_MAX / granule)
> + return NULL;
> + size = nr_addrs * granule;
> +
> + list_for_each_entry(msi_page, msi_page_list, list) {
> + if (msi_page->phys != phys_addrs[0])
> + continue;
> + if (nr_addrs == 1)
> + return msi_page;
> + if (msi_page->range_size == size &&
> + iommu_dma_msi_range_matches(msi_page_list, msi_page, phys_addrs,
> + nr_addrs, granule))
> return msi_page;
> + }
>
> - msi_page = kzalloc_obj(*msi_page);
> - if (!msi_page)
> - return NULL;
> + for (i = 0; i < nr_addrs; i++) {
> + msi_page = kzalloc_obj(*msi_page);
> + if (!msi_page)
> + goto out_free_pages;
> + list_add_tail(&msi_page->list, &new_msi_pages);
> + }
>
> - iova = iommu_dma_alloc_iova(domain, size, dma_get_mask(dev), dev);
> - if (!iova)
> - goto out_free_page;
> + base_iova = iommu_dma_alloc_iova(domain, size, dma_get_mask(dev), dev);
> + if (!base_iova)
> + goto out_free_pages;
>
> - if (iommu_map(domain, iova, msi_addr, size, prot, GFP_KERNEL))
> - goto out_free_iova;
> + i = 0;
> + list_for_each_entry(msi_page, &new_msi_pages, list) {
> + msi_page->phys = phys_addrs[i];
> + msi_page->iova = base_iova + i * granule;
> + if (!i)
> + first_page = msi_page;
> + if (iommu_map(domain, msi_page->iova, msi_page->phys, granule, prot, GFP_KERNEL))
> + goto out_unmap;
> + i++;
> + }
>
> - INIT_LIST_HEAD(&msi_page->list);
> - msi_page->phys = msi_addr;
> - msi_page->iova = iova;
> - list_add(&msi_page->list, msi_page_list);
> - return msi_page;
> + first_page->range_size = size;
> + list_splice(&new_msi_pages, msi_page_list);
> + return first_page;
>
> -out_free_iova:
> - iommu_dma_free_iova(domain, iova, size, NULL);
> -out_free_page:
> - kfree(msi_page);
> +out_unmap:
> + if (i) {
> + size_t unmap = iommu_unmap(domain, base_iova, i * granule);
> +
> + WARN_ON_ONCE(unmap != i * granule);
> + }
> + iommu_dma_free_iova(domain, base_iova, size, NULL);
> +out_free_pages:
> + while (!list_empty(&new_msi_pages)) {
> + msi_page = list_first_entry(&new_msi_pages, typeof(*msi_page), list);
> + list_del(&msi_page->list);
> + kfree(msi_page);
> + }
> return NULL;
> }
>
> @@ -2247,19 +2303,23 @@ int iommu_dma_sw_msi(struct iommu_domain *domain, struct msi_desc *desc,
> {
> struct device *dev = msi_desc_to_dev(desc);
> const struct iommu_dma_msi_page *msi_page;
> + phys_addr_t phys_addr;
> + size_t granule;
>
> if (!has_msi_cookie(domain)) {
> msi_desc_set_iommu_msi_iova(desc, 0, 0);
> return 0;
> }
>
> + granule = cookie_msi_granule(domain);
> + phys_addr = ALIGN_DOWN(msi_addr, granule);
> +
> iommu_group_mutex_assert(dev);
> - msi_page = iommu_dma_get_msi_page(dev, msi_addr, domain);
> + msi_page = iommu_dma_get_msi_page(dev, &phys_addr, 1, granule, domain);
> if (!msi_page)
> return -ENOMEM;
>
> - msi_desc_set_iommu_msi_iova(desc, msi_page->iova,
> - ilog2(cookie_msi_granule(domain)));
> + msi_desc_set_iommu_msi_iova(desc, msi_page->iova, ilog2(granule));
> return 0;
> }
>
^ permalink raw reply [flat|nested] 31+ messages in thread* Re: [PATCH v6 01/16] iommu/dma: Prepare MSI physical address lists
2026-10-02 16:21 ` Robin Murphy
@ 2026-10-03 12:03 ` Andrew Jones
2026-10-03 12:10 ` Jason Gunthorpe
0 siblings, 1 reply; 31+ messages in thread
From: Andrew Jones @ 2026-10-03 12:03 UTC (permalink / raw)
To: Robin Murphy
Cc: linux-riscv, iommu, linux-kernel, tomasz.jeznach, tjeznach, jgg,
jgg, joro, will, pjw, palmer, anup, tglx, kevin.tian, fangyu.yu
Hi Robin,
On Fri, Oct 02, 2026 at 05:21:18PM +0100, Robin Murphy wrote:
> You already end up adding what is effectively a RISC-V-specific
> entrypoint, so you may as well just carry that all the way through to
> its own effectively RISC-V-specific implementation
The list API came out of the v4[1] discussion with Jason. Preparing the
targets as one contiguous IOVA range lets the MSI descriptor cache the
base and IMSIC compose the address as base + CPU offset. That got
rid of the separate domain-local lookup table and its lifetime handling,
which I think was a worthwhile improvement.
I'd like to keep that model. I'm open to splitting the implementation,
but most of the added code seems necessary to prepare multiple targets
together, regardless of whether the list is fixed.
> it should merely be a case of whether a) this is the first call for
> the given cookie so everything needs mapping, or b) it's not the
> first call, so everything must already be mapped and we can just
> return the IOVA.
For DMA-IOMMU, yes, remembering the base would let us drop the range
matching. We'd still need the contiguous IOVA allocation, mapping loop,
rollback on failure, and locking across the whole operation.
For iommufd, there are also separate things to track: whether the context
has allocated IOVAs for the set, whether a particular HWPT has those
mappings installed, and whether a group needs them installed when its
domain is replaced. Having one fixed list doesn't by itself remove
that bookkeeping.
The current IMSIC caller does always supply the same list. We could make
that a requirement, but we'd need to define its scope: per cookie for
DMA-IOMMU, and how it applies across groups and HWPTs for iommufd. The
current interface instead takes an ordered list and provides contiguous
IOVAs for it; the range matching makes reuse honor that contract.
> At very worst, a separate hook to just pre-populate msi_page_list
> with a regular page for each IMSIC address [...] could probably
> suffice without any other major structural changes
Is the main concern the range matching, or the shared mapping path
itself? Apart from lookup, both paths need much the same allocation,
mapping, locking, cleanup, and descriptor handling. Keeping those
separate seems likely to duplicate code, while factoring them into
shared helpers seems likely to bring us back towards a list-based
mapping helper.
We could handle lookup of existing mappings separately for the single-page
and list cases while sharing the allocation and mapping code. But I'm not
yet seeing a substantial overall simplification from requiring a fixed
list. Is there more bookkeeping you have in mind that could go away with
that restriction?
[1] https://lore.kernel.org/all/20260820214150.545737-3-andrew.jones@oss.qualcomm.com/
Thanks,
drew
^ permalink raw reply [flat|nested] 31+ messages in thread
* Re: [PATCH v6 01/16] iommu/dma: Prepare MSI physical address lists
2026-10-03 12:03 ` Andrew Jones
@ 2026-10-03 12:10 ` Jason Gunthorpe
0 siblings, 0 replies; 31+ messages in thread
From: Jason Gunthorpe @ 2026-10-03 12:10 UTC (permalink / raw)
To: Andrew Jones
Cc: Robin Murphy, linux-riscv, iommu, linux-kernel, tomasz.jeznach,
tjeznach, joro, will, pjw, palmer, anup, tglx, kevin.tian,
fangyu.yu
On Sat, Oct 03, 2026 at 02:03:35PM +0200, Andrew Jones wrote:
>
> Hi Robin,
>
> On Fri, Oct 02, 2026 at 05:21:18PM +0100, Robin Murphy wrote:
> > You already end up adding what is effectively a RISC-V-specific
> > entrypoint, so you may as well just carry that all the way through to
> > its own effectively RISC-V-specific implementation
>
> The list API came out of the v4[1] discussion with Jason. Preparing the
> targets as one contiguous IOVA range lets the MSI descriptor cache the
> base and IMSIC compose the address as base + CPU offset. That got
> rid of the separate domain-local lookup table and its lifetime handling,
> which I think was a worthwhile improvement.
Yeah, I prefer this idea to what you had earlier, however it is
accomodated.
> > it should merely be a case of whether a) this is the first call for
> > the given cookie so everything needs mapping, or b) it's not the
> > first call, so everything must already be mapped and we can just
> > return the IOVA.
>
> For DMA-IOMMU, yes, remembering the base would let us drop the range
> matching. We'd still need the contiguous IOVA allocation, mapping loop,
> rollback on failure, and locking across the whole operation.
I fell like dma-iommu should have a way to map a page list to iova and
maybe that is all the shared code with arm there is.
Jason
^ permalink raw reply [flat|nested] 31+ messages in thread
* [PATCH v6 02/16] iommufd: Convert struct iommufd_sw_msi_maps to a growable bitmap
2026-09-25 15:16 [PATCH v6 00/16] iommu/riscv: Enable MSI remapping, IOMMU_DMA and VFIO Andrew Jones
2026-09-25 15:16 ` [PATCH v6 01/16] iommu/dma: Prepare MSI physical address lists Andrew Jones
@ 2026-09-25 15:16 ` Andrew Jones
2026-09-25 15:16 ` [PATCH v6 03/16] iommufd: Split software MSI map lookup and allocation Andrew Jones
` (14 subsequent siblings)
16 siblings, 0 replies; 31+ messages in thread
From: Andrew Jones @ 2026-09-25 15:16 UTC (permalink / raw)
To: linux-riscv, iommu
Cc: linux-kernel, tomasz.jeznach, tjeznach, jgg, jgg, joro, will,
robin.murphy, pjw, palmer, anup, tglx, kevin.tian, fangyu.yu,
Nutty Liu
struct iommufd_sw_msi_maps uses a fixed 64-bit bitmap, limiting each
group and hardware page table to 64 software MSI mappings. RISC-V
interrupt remapping needs a mapping for every possible CPU, so this
limit is insufficient.
Make the bitmap grow on demand and treat IDs beyond its current size
as absent. Cap it at 16K entries to bound allocation size while leaving
ample room for expected software MSI users.
Signed-off-by: Andrew Jones <andrew.jones@oss.qualcomm.com>
Reviewed-by: Nutty Liu <nutty.liu@hotmail.com>
Tested-by: Fangyu Yu <fangyu.yu@linux.alibaba.com>
---
drivers/iommu/iommufd/device.c | 3 +-
drivers/iommu/iommufd/driver.c | 37 ++++++++++++++--------
drivers/iommu/iommufd/hw_pagetable.c | 1 +
drivers/iommu/iommufd/iommufd_private.h | 41 +++++++++++++++++++++++--
4 files changed, 66 insertions(+), 16 deletions(-)
diff --git a/drivers/iommu/iommufd/device.c b/drivers/iommu/iommufd/device.c
index a664c70a6fe7..28e2c98ef953 100644
--- a/drivers/iommu/iommufd/device.c
+++ b/drivers/iommu/iommufd/device.c
@@ -49,6 +49,7 @@ static void iommufd_group_release(struct kref *kref)
iommu_group_put(igroup->group);
}
mutex_destroy(&igroup->lock);
+ kfree(igroup->required_sw_msi.bitmap);
kfree(igroup);
}
@@ -450,7 +451,7 @@ static int iommufd_group_setup_msi(struct iommufd_group *igroup,
int rc;
if (cur->sw_msi_start != igroup->sw_msi_start ||
- !test_bit(cur->id, igroup->required_sw_msi.bitmap))
+ !iommufd_sw_msi_maps_test_bit(&igroup->required_sw_msi, cur->id))
continue;
rc = iommufd_sw_msi_install(ictx, hwpt_paging, cur);
diff --git a/drivers/iommu/iommufd/driver.c b/drivers/iommu/iommufd/driver.c
index e4d17a748178..ea705e416139 100644
--- a/drivers/iommu/iommufd/driver.c
+++ b/drivers/iommu/iommufd/driver.c
@@ -196,13 +196,15 @@ iommufd_sw_msi_get_map(struct iommufd_ctx *ictx, phys_addr_t msi_addr,
list_for_each_entry(cur, &ictx->sw_msi_list, sw_msi_item) {
if (cur->sw_msi_start != sw_msi_start)
continue;
+ if (cur->pgoff == UINT_MAX)
+ return ERR_PTR(-EOVERFLOW);
max_pgoff = max(max_pgoff, cur->pgoff + 1);
if (cur->msi_addr == msi_addr)
return cur;
}
- if (ictx->sw_msi_id >=
- BITS_PER_BYTE * sizeof_field(struct iommufd_sw_msi_maps, bitmap))
+ if (ictx->sw_msi_id > IOMMUFD_SW_MSI_MAX_ID ||
+ max_pgoff > (ULONG_MAX - sw_msi_start) / PAGE_SIZE)
return ERR_PTR(-EOVERFLOW);
cur = kzalloc_obj(*cur);
@@ -222,21 +224,25 @@ int iommufd_sw_msi_install(struct iommufd_ctx *ictx,
struct iommufd_sw_msi_map *msi_map)
{
unsigned long iova;
+ int rc;
lockdep_assert_held(&ictx->sw_msi_lock);
+ if (iommufd_sw_msi_maps_test_bit(&hwpt_paging->present_sw_msi, msi_map->id))
+ return 0;
+
iova = msi_map->sw_msi_start + msi_map->pgoff * PAGE_SIZE;
- if (!test_bit(msi_map->id, hwpt_paging->present_sw_msi.bitmap)) {
- int rc;
-
- rc = iommu_map(hwpt_paging->common.domain, iova,
- msi_map->msi_addr, PAGE_SIZE,
- IOMMU_WRITE | IOMMU_READ | IOMMU_MMIO,
- GFP_KERNEL_ACCOUNT);
- if (rc)
- return rc;
- __set_bit(msi_map->id, hwpt_paging->present_sw_msi.bitmap);
- }
+ rc = iommufd_sw_msi_maps_ensure(&hwpt_paging->present_sw_msi, msi_map->id);
+ if (rc)
+ return rc;
+
+ rc = iommu_map(hwpt_paging->common.domain, iova,
+ msi_map->msi_addr, PAGE_SIZE,
+ IOMMU_WRITE | IOMMU_READ | IOMMU_MMIO,
+ GFP_KERNEL_ACCOUNT);
+ if (rc)
+ return rc;
+ __set_bit(msi_map->id, hwpt_paging->present_sw_msi.bitmap);
return 0;
}
EXPORT_SYMBOL_NS_GPL(iommufd_sw_msi_install, "IOMMUFD_INTERNAL");
@@ -290,6 +296,11 @@ int iommufd_sw_msi(struct iommu_domain *domain, struct msi_desc *desc,
if (IS_ERR(msi_map))
return PTR_ERR(msi_map);
+ rc = iommufd_sw_msi_maps_ensure(&handle->idev->igroup->required_sw_msi,
+ msi_map->id);
+ if (rc)
+ return rc;
+
rc = iommufd_sw_msi_install(ictx, hwpt_paging, msi_map);
if (rc)
return rc;
diff --git a/drivers/iommu/iommufd/hw_pagetable.c b/drivers/iommu/iommufd/hw_pagetable.c
index ef6e119c2a75..81f41d437c4c 100644
--- a/drivers/iommu/iommufd/hw_pagetable.c
+++ b/drivers/iommu/iommufd/hw_pagetable.c
@@ -41,6 +41,7 @@ void iommufd_hwpt_paging_destroy(struct iommufd_object *obj)
}
__iommufd_hwpt_destroy(&hwpt_paging->common);
+ kfree(hwpt_paging->present_sw_msi.bitmap);
refcount_dec(&hwpt_paging->ioas->obj.users);
}
diff --git a/drivers/iommu/iommufd/iommufd_private.h b/drivers/iommu/iommufd/iommufd_private.h
index eb2e85b27e42..774cbe28cbaf 100644
--- a/drivers/iommu/iommufd/iommufd_private.h
+++ b/drivers/iommu/iommufd/iommufd_private.h
@@ -9,6 +9,7 @@
#include <linux/iova_bitmap.h>
#include <linux/maple_tree.h>
#include <linux/rwsem.h>
+#include <linux/slab.h>
#include <linux/uaccess.h>
#include <linux/xarray.h>
#include <uapi/linux/iommufd.h>
@@ -29,11 +30,47 @@ struct iommufd_sw_msi_map {
unsigned int id;
};
-/* Bitmap of struct iommufd_sw_msi_map::id */
+/* Bitmap of struct iommufd_sw_msi_map::id; starts empty, grows on demand. */
struct iommufd_sw_msi_maps {
- DECLARE_BITMAP(bitmap, 64);
+ unsigned long *bitmap;
+ unsigned int nbits;
};
+/* Large enough for foreseeable SW MSI users while bounding bitmap growth. */
+#define IOMMUFD_SW_MSI_MAX_ID (16U * 1024 - 1)
+
+/* Grow bitmap to accommodate id. Must be called under ictx->sw_msi_lock. */
+static inline int iommufd_sw_msi_maps_ensure(struct iommufd_sw_msi_maps *maps,
+ unsigned int id)
+{
+ unsigned long *new_bitmap;
+ unsigned int new_nbits;
+
+ if (id < maps->nbits)
+ return 0;
+ if (id > IOMMUFD_SW_MSI_MAX_ID)
+ return -EOVERFLOW;
+
+ new_nbits = max(ALIGN(id + 1, BITS_PER_LONG), 64U);
+ new_bitmap = krealloc(maps->bitmap,
+ BITS_TO_LONGS(new_nbits) * sizeof(unsigned long),
+ GFP_KERNEL_ACCOUNT);
+ if (!new_bitmap)
+ return -ENOMEM;
+ bitmap_clear(new_bitmap, maps->nbits, new_nbits - maps->nbits);
+ maps->bitmap = new_bitmap;
+ maps->nbits = new_nbits;
+ return 0;
+}
+
+static inline bool iommufd_sw_msi_maps_test_bit(const struct iommufd_sw_msi_maps *maps,
+ unsigned int id)
+{
+ if (id >= maps->nbits)
+ return false;
+ return test_bit(id, maps->bitmap);
+}
+
#ifdef CONFIG_IRQ_MSI_IOMMU
int iommufd_sw_msi_install(struct iommufd_ctx *ictx,
struct iommufd_hwpt_paging *hwpt_paging,
--
2.43.0
^ permalink raw reply [flat|nested] 31+ messages in thread* [PATCH v6 03/16] iommufd: Split software MSI map lookup and allocation
2026-09-25 15:16 [PATCH v6 00/16] iommu/riscv: Enable MSI remapping, IOMMU_DMA and VFIO Andrew Jones
2026-09-25 15:16 ` [PATCH v6 01/16] iommu/dma: Prepare MSI physical address lists Andrew Jones
2026-09-25 15:16 ` [PATCH v6 02/16] iommufd: Convert struct iommufd_sw_msi_maps to a growable bitmap Andrew Jones
@ 2026-09-25 15:16 ` Andrew Jones
2026-09-25 15:16 ` [PATCH v6 04/16] iommufd: Bound software MSI mappings to the reserved range Andrew Jones
` (13 subsequent siblings)
16 siblings, 0 replies; 31+ messages in thread
From: Andrew Jones @ 2026-09-25 15:16 UTC (permalink / raw)
To: linux-riscv, iommu
Cc: linux-kernel, tomasz.jeznach, tjeznach, jgg, jgg, joro, will,
robin.murphy, pjw, palmer, anup, tglx, kevin.tian, fangyu.yu
Separate lookup of an existing software MSI mapping from allocation of a
new mapping. This prepares the code for allocating multiple mappings for
an MSI physical address list.
Signed-off-by: Andrew Jones <andrew.jones@oss.qualcomm.com>
Tested-by: Fangyu Yu <fangyu.yu@linux.alibaba.com>
---
drivers/iommu/iommufd/driver.c | 23 +++++++++++++++++++++--
1 file changed, 21 insertions(+), 2 deletions(-)
diff --git a/drivers/iommu/iommufd/driver.c b/drivers/iommu/iommufd/driver.c
index ea705e416139..501f1e1f2797 100644
--- a/drivers/iommu/iommufd/driver.c
+++ b/drivers/iommu/iommufd/driver.c
@@ -189,6 +189,23 @@ iommufd_sw_msi_get_map(struct iommufd_ctx *ictx, phys_addr_t msi_addr,
phys_addr_t sw_msi_start)
{
struct iommufd_sw_msi_map *cur;
+
+ lockdep_assert_held(&ictx->sw_msi_lock);
+
+ list_for_each_entry(cur, &ictx->sw_msi_list, sw_msi_item) {
+ if (cur->sw_msi_start != sw_msi_start)
+ continue;
+ if (cur->msi_addr == msi_addr)
+ return cur;
+ }
+ return NULL;
+}
+
+static struct iommufd_sw_msi_map *
+iommufd_sw_msi_alloc_map(struct iommufd_ctx *ictx, phys_addr_t msi_addr,
+ phys_addr_t sw_msi_start)
+{
+ struct iommufd_sw_msi_map *cur;
unsigned int max_pgoff = 0;
lockdep_assert_held(&ictx->sw_msi_lock);
@@ -199,8 +216,6 @@ iommufd_sw_msi_get_map(struct iommufd_ctx *ictx, phys_addr_t msi_addr,
if (cur->pgoff == UINT_MAX)
return ERR_PTR(-EOVERFLOW);
max_pgoff = max(max_pgoff, cur->pgoff + 1);
- if (cur->msi_addr == msi_addr)
- return cur;
}
if (ictx->sw_msi_id > IOMMUFD_SW_MSI_MAX_ID ||
@@ -293,6 +308,10 @@ int iommufd_sw_msi(struct iommu_domain *domain, struct msi_desc *desc,
msi_map = iommufd_sw_msi_get_map(handle->idev->ictx,
msi_addr & PAGE_MASK,
handle->idev->igroup->sw_msi_start);
+ if (!msi_map)
+ msi_map = iommufd_sw_msi_alloc_map(handle->idev->ictx,
+ msi_addr & PAGE_MASK,
+ handle->idev->igroup->sw_msi_start);
if (IS_ERR(msi_map))
return PTR_ERR(msi_map);
--
2.43.0
^ permalink raw reply [flat|nested] 31+ messages in thread* [PATCH v6 04/16] iommufd: Bound software MSI mappings to the reserved range
2026-09-25 15:16 [PATCH v6 00/16] iommu/riscv: Enable MSI remapping, IOMMU_DMA and VFIO Andrew Jones
` (2 preceding siblings ...)
2026-09-25 15:16 ` [PATCH v6 03/16] iommufd: Split software MSI map lookup and allocation Andrew Jones
@ 2026-09-25 15:16 ` Andrew Jones
2026-09-25 15:16 ` [PATCH v6 05/16] iommufd: Prepare software MSI maps for address lists Andrew Jones
` (12 subsequent siblings)
16 siblings, 0 replies; 31+ messages in thread
From: Andrew Jones @ 2026-09-25 15:16 UTC (permalink / raw)
To: linux-riscv, iommu
Cc: linux-kernel, tomasz.jeznach, tjeznach, jgg, jgg, joro, will,
robin.murphy, pjw, palmer, anup, tglx, kevin.tian, fangyu.yu
Track the start and length of the software MSI reserved region as one
range and pass it to the mapping lookup and allocation helpers.
Reject cached mappings outside the region and new mappings when no
complete page remains. This enforces the reserved region bounds and
prepares the helpers to allocate contiguous MSI address lists.
Signed-off-by: Andrew Jones <andrew.jones@oss.qualcomm.com>
Tested-by: Fangyu Yu <fangyu.yu@linux.alibaba.com>
---
drivers/iommu/iommufd/device.c | 17 +++++++++++------
drivers/iommu/iommufd/driver.c | 21 ++++++++++++---------
drivers/iommu/iommufd/io_pagetable.c | 9 +++++----
drivers/iommu/iommufd/iommufd_private.h | 9 +++++++--
4 files changed, 35 insertions(+), 21 deletions(-)
diff --git a/drivers/iommu/iommufd/device.c b/drivers/iommu/iommufd/device.c
index 28e2c98ef953..2b12b25a74da 100644
--- a/drivers/iommu/iommufd/device.c
+++ b/drivers/iommu/iommufd/device.c
@@ -84,7 +84,7 @@ static struct iommufd_group *iommufd_alloc_group(struct iommufd_ctx *ictx,
kref_init(&new_igroup->ref);
mutex_init(&new_igroup->lock);
xa_init(&new_igroup->pasid_attach);
- new_igroup->sw_msi_start = PHYS_ADDR_MAX;
+ new_igroup->sw_msi_range.start = PHYS_ADDR_MAX;
/* group reference moves into new_igroup */
new_igroup->group = group;
@@ -440,7 +440,7 @@ static int iommufd_group_setup_msi(struct iommufd_group *igroup,
struct iommufd_ctx *ictx = igroup->ictx;
struct iommufd_sw_msi_map *cur;
- if (igroup->sw_msi_start == PHYS_ADDR_MAX)
+ if (igroup->sw_msi_range.start == PHYS_ADDR_MAX)
return 0;
/*
@@ -450,7 +450,7 @@ static int iommufd_group_setup_msi(struct iommufd_group *igroup,
list_for_each_entry(cur, &ictx->sw_msi_list, sw_msi_item) {
int rc;
- if (cur->sw_msi_start != igroup->sw_msi_start ||
+ if (cur->sw_msi_start != igroup->sw_msi_range.start ||
!iommufd_sw_msi_maps_test_bit(&igroup->required_sw_msi, cur->id))
continue;
@@ -480,18 +480,23 @@ static int
iommufd_device_attach_reserved_iova(struct iommufd_device *idev,
struct iommufd_hwpt_paging *hwpt_paging)
{
+ struct iommufd_sw_msi_range sw_msi_range = {
+ .start = PHYS_ADDR_MAX,
+ };
struct iommufd_group *igroup = idev->igroup;
+ bool first_attach;
int rc;
lockdep_assert_held(&igroup->lock);
+ first_attach = iommufd_group_first_attach(igroup, IOMMU_NO_PASID);
rc = iopt_table_enforce_dev_resv_regions(&hwpt_paging->ioas->iopt,
- idev->dev,
- &igroup->sw_msi_start);
+ idev->dev, &sw_msi_range);
if (rc)
return rc;
- if (iommufd_group_first_attach(igroup, IOMMU_NO_PASID)) {
+ if (first_attach) {
+ igroup->sw_msi_range = sw_msi_range;
rc = iommufd_group_setup_msi(igroup, hwpt_paging);
if (rc) {
iopt_remove_reserved_iova(&hwpt_paging->ioas->iopt,
diff --git a/drivers/iommu/iommufd/driver.c b/drivers/iommu/iommufd/driver.c
index 501f1e1f2797..c2ef15c2ca51 100644
--- a/drivers/iommu/iommufd/driver.c
+++ b/drivers/iommu/iommufd/driver.c
@@ -186,14 +186,15 @@ EXPORT_SYMBOL_NS_GPL(iommufd_viommu_report_event, "IOMMUFD");
*/
static struct iommufd_sw_msi_map *
iommufd_sw_msi_get_map(struct iommufd_ctx *ictx, phys_addr_t msi_addr,
- phys_addr_t sw_msi_start)
+ const struct iommufd_sw_msi_range *sw_msi_range)
{
struct iommufd_sw_msi_map *cur;
lockdep_assert_held(&ictx->sw_msi_lock);
list_for_each_entry(cur, &ictx->sw_msi_list, sw_msi_item) {
- if (cur->sw_msi_start != sw_msi_start)
+ if (cur->sw_msi_start != sw_msi_range->start ||
+ cur->pgoff >= sw_msi_range->length / PAGE_SIZE)
continue;
if (cur->msi_addr == msi_addr)
return cur;
@@ -203,7 +204,7 @@ iommufd_sw_msi_get_map(struct iommufd_ctx *ictx, phys_addr_t msi_addr,
static struct iommufd_sw_msi_map *
iommufd_sw_msi_alloc_map(struct iommufd_ctx *ictx, phys_addr_t msi_addr,
- phys_addr_t sw_msi_start)
+ const struct iommufd_sw_msi_range *sw_msi_range)
{
struct iommufd_sw_msi_map *cur;
unsigned int max_pgoff = 0;
@@ -211,7 +212,7 @@ iommufd_sw_msi_alloc_map(struct iommufd_ctx *ictx, phys_addr_t msi_addr,
lockdep_assert_held(&ictx->sw_msi_lock);
list_for_each_entry(cur, &ictx->sw_msi_list, sw_msi_item) {
- if (cur->sw_msi_start != sw_msi_start)
+ if (cur->sw_msi_start != sw_msi_range->start)
continue;
if (cur->pgoff == UINT_MAX)
return ERR_PTR(-EOVERFLOW);
@@ -219,14 +220,16 @@ iommufd_sw_msi_alloc_map(struct iommufd_ctx *ictx, phys_addr_t msi_addr,
}
if (ictx->sw_msi_id > IOMMUFD_SW_MSI_MAX_ID ||
- max_pgoff > (ULONG_MAX - sw_msi_start) / PAGE_SIZE)
+ max_pgoff > (ULONG_MAX - sw_msi_range->start) / PAGE_SIZE)
return ERR_PTR(-EOVERFLOW);
+ if (max_pgoff >= sw_msi_range->length / PAGE_SIZE)
+ return ERR_PTR(-ENOSPC);
cur = kzalloc_obj(*cur);
if (!cur)
return ERR_PTR(-ENOMEM);
- cur->sw_msi_start = sw_msi_start;
+ cur->sw_msi_start = sw_msi_range->start;
cur->msi_addr = msi_addr;
cur->pgoff = max_pgoff;
cur->id = ictx->sw_msi_id++;
@@ -295,7 +298,7 @@ int iommufd_sw_msi(struct iommu_domain *domain, struct msi_desc *desc,
handle = to_iommufd_handle(raw_handle);
/* No IOMMU_RESV_SW_MSI means no change to the msi_msg */
- if (handle->idev->igroup->sw_msi_start == PHYS_ADDR_MAX)
+ if (handle->idev->igroup->sw_msi_range.start == PHYS_ADDR_MAX)
return 0;
ictx = handle->idev->ictx;
@@ -307,11 +310,11 @@ int iommufd_sw_msi(struct iommu_domain *domain, struct msi_desc *desc,
*/
msi_map = iommufd_sw_msi_get_map(handle->idev->ictx,
msi_addr & PAGE_MASK,
- handle->idev->igroup->sw_msi_start);
+ &handle->idev->igroup->sw_msi_range);
if (!msi_map)
msi_map = iommufd_sw_msi_alloc_map(handle->idev->ictx,
msi_addr & PAGE_MASK,
- handle->idev->igroup->sw_msi_start);
+ &handle->idev->igroup->sw_msi_range);
if (IS_ERR(msi_map))
return PTR_ERR(msi_map);
diff --git a/drivers/iommu/iommufd/io_pagetable.c b/drivers/iommu/iommufd/io_pagetable.c
index 4e447ce74cf6..0336d8f0908f 100644
--- a/drivers/iommu/iommufd/io_pagetable.c
+++ b/drivers/iommu/iommufd/io_pagetable.c
@@ -1579,7 +1579,7 @@ void iopt_remove_access(struct io_pagetable *iopt,
/* Narrow the valid_iova_itree to include reserved ranges from a device. */
int iopt_table_enforce_dev_resv_regions(struct io_pagetable *iopt,
struct device *dev,
- phys_addr_t *sw_msi_start)
+ struct iommufd_sw_msi_range *sw_msi_range)
{
struct iommu_resv_region *resv;
LIST_HEAD(resv_regions);
@@ -1598,10 +1598,11 @@ int iopt_table_enforce_dev_resv_regions(struct io_pagetable *iopt,
if (resv->type == IOMMU_RESV_DIRECT_RELAXABLE)
continue;
- if (sw_msi_start && resv->type == IOMMU_RESV_MSI)
+ if (sw_msi_range && resv->type == IOMMU_RESV_MSI)
num_hw_msi++;
- if (sw_msi_start && resv->type == IOMMU_RESV_SW_MSI) {
- *sw_msi_start = resv->start;
+ if (sw_msi_range && resv->type == IOMMU_RESV_SW_MSI) {
+ sw_msi_range->start = resv->start;
+ sw_msi_range->length = resv->length;
num_sw_msi++;
}
diff --git a/drivers/iommu/iommufd/iommufd_private.h b/drivers/iommu/iommufd/iommufd_private.h
index 774cbe28cbaf..af7df868dd4f 100644
--- a/drivers/iommu/iommufd/iommufd_private.h
+++ b/drivers/iommu/iommufd/iommufd_private.h
@@ -22,6 +22,11 @@ struct iommu_option;
struct iommufd_device;
struct dma_buf_attachment;
+struct iommufd_sw_msi_range {
+ phys_addr_t start;
+ size_t length;
+};
+
struct iommufd_sw_msi_map {
struct list_head sw_msi_item;
phys_addr_t sw_msi_start;
@@ -181,7 +186,7 @@ void iopt_table_remove_domain(struct io_pagetable *iopt,
struct iommu_domain *domain);
int iopt_table_enforce_dev_resv_regions(struct io_pagetable *iopt,
struct device *dev,
- phys_addr_t *sw_msi_start);
+ struct iommufd_sw_msi_range *sw_msi_range);
int iopt_set_allow_iova(struct io_pagetable *iopt,
struct rb_root_cached *allowed_iova);
int iopt_reserve_iova(struct io_pagetable *iopt, unsigned long start,
@@ -531,7 +536,7 @@ struct iommufd_group {
struct iommu_group *group;
struct xarray pasid_attach;
struct iommufd_sw_msi_maps required_sw_msi;
- phys_addr_t sw_msi_start;
+ struct iommufd_sw_msi_range sw_msi_range;
};
/*
--
2.43.0
^ permalink raw reply [flat|nested] 31+ messages in thread* [PATCH v6 05/16] iommufd: Prepare software MSI maps for address lists
2026-09-25 15:16 [PATCH v6 00/16] iommu/riscv: Enable MSI remapping, IOMMU_DMA and VFIO Andrew Jones
` (3 preceding siblings ...)
2026-09-25 15:16 ` [PATCH v6 04/16] iommufd: Bound software MSI mappings to the reserved range Andrew Jones
@ 2026-09-25 15:16 ` Andrew Jones
2026-09-25 15:16 ` [PATCH v6 06/16] iommufd: Install software MSI map ranges atomically Andrew Jones
` (11 subsequent siblings)
16 siblings, 0 replies; 31+ messages in thread
From: Andrew Jones @ 2026-09-25 15:16 UTC (permalink / raw)
To: linux-riscv, iommu
Cc: linux-kernel, tomasz.jeznach, tjeznach, jgg, jgg, joro, will,
robin.murphy, pjw, palmer, anup, tglx, kevin.tian, fangyu.yu
Teach iommufd_sw_msi_get_map() to match an ordered physical address list
against an existing contiguous IOVA range. Mark the first map with the
range size so an identical list can reuse the allocation.
Teach iommufd_sw_msi_alloc_map() to reserve identifiers and offsets for
the complete list and build its maps on a temporary list. Keep the
existing scalar installation behavior and only publish newly allocated
maps after installation succeeds.
Signed-off-by: Andrew Jones <andrew.jones@oss.qualcomm.com>
Tested-by: Fangyu Yu <fangyu.yu@linux.alibaba.com>
---
drivers/iommu/iommufd/driver.c | 161 ++++++++++++++++++------
drivers/iommu/iommufd/iommufd_private.h | 1 +
2 files changed, 123 insertions(+), 39 deletions(-)
diff --git a/drivers/iommu/iommufd/driver.c b/drivers/iommu/iommufd/driver.c
index c2ef15c2ca51..8b8f132ef3c5 100644
--- a/drivers/iommu/iommufd/driver.c
+++ b/drivers/iommu/iommufd/driver.c
@@ -179,35 +179,91 @@ EXPORT_SYMBOL_NS_GPL(iommufd_viommu_report_event, "IOMMUFD");
#ifdef CONFIG_IRQ_MSI_IOMMU
/*
- * Get a iommufd_sw_msi_map for the msi physical address requested by the irq
+ * Get an iommufd_sw_msi_map for the msi physical addresses requested by the irq
* layer. The mapping to IOVA is global to the iommufd file descriptor, every
* domain that is attached to a device using the same MSI parameters will use
- * the same IOVA.
+ * the same contiguous IOVA range.
*/
static struct iommufd_sw_msi_map *
-iommufd_sw_msi_get_map(struct iommufd_ctx *ictx, phys_addr_t msi_addr,
- const struct iommufd_sw_msi_range *sw_msi_range)
+iommufd_sw_msi_get_map(struct iommufd_ctx *ictx, const phys_addr_t *phys_addrs,
+ unsigned int nr_addrs, const struct iommufd_sw_msi_range *sw_msi_range)
{
- struct iommufd_sw_msi_map *cur;
+ struct iommufd_sw_msi_map *cur, *msi_map;
+ unsigned int nr_found;
lockdep_assert_held(&ictx->sw_msi_lock);
- list_for_each_entry(cur, &ictx->sw_msi_list, sw_msi_item) {
- if (cur->sw_msi_start != sw_msi_range->start ||
- cur->pgoff >= sw_msi_range->length / PAGE_SIZE)
+ list_for_each_entry(msi_map, &ictx->sw_msi_list, sw_msi_item) {
+ if (msi_map->sw_msi_start != sw_msi_range->start ||
+ msi_map->msi_addr != phys_addrs[0])
+ continue;
+ if (msi_map->range_size != nr_addrs * PAGE_SIZE)
continue;
- if (cur->msi_addr == msi_addr)
- return cur;
+ if (msi_map->pgoff > sw_msi_range->length / PAGE_SIZE ||
+ nr_addrs > sw_msi_range->length / PAGE_SIZE - msi_map->pgoff)
+ continue;
+
+ nr_found = 0;
+ list_for_each_entry(cur, &ictx->sw_msi_list, sw_msi_item) {
+ unsigned int index;
+
+ if (cur->sw_msi_start != sw_msi_range->start ||
+ cur->pgoff < msi_map->pgoff)
+ continue;
+ index = cur->pgoff - msi_map->pgoff;
+ if (index >= nr_addrs)
+ continue;
+ if (cur->msi_addr != phys_addrs[index])
+ break;
+ nr_found++;
+ }
+ if (nr_found == nr_addrs)
+ return msi_map;
}
return NULL;
}
+static int iommufd_sw_msi_check_alloc(struct iommufd_ctx *ictx,
+ const struct iommufd_sw_msi_range *sw_msi_range,
+ unsigned int first_pgoff, unsigned int nr_addrs,
+ size_t *range_size)
+{
+ unsigned long max_iova_pgoff;
+ unsigned int last_pgoff;
+ unsigned int last_id;
+ size_t range_pages;
+
+ if (!nr_addrs)
+ return -EINVAL;
+ if (sw_msi_range->start > ULONG_MAX)
+ return -EOVERFLOW;
+
+ range_pages = sw_msi_range->length / PAGE_SIZE;
+ max_iova_pgoff = (ULONG_MAX - sw_msi_range->start) / PAGE_SIZE;
+
+ if (check_add_overflow(ictx->sw_msi_id, nr_addrs - 1, &last_id) ||
+ last_id > IOMMUFD_SW_MSI_MAX_ID ||
+ check_add_overflow(first_pgoff, nr_addrs - 1, &last_pgoff) ||
+ last_pgoff > max_iova_pgoff ||
+ check_mul_overflow((size_t)nr_addrs, PAGE_SIZE, range_size))
+ return -EOVERFLOW;
+
+ if (last_pgoff >= range_pages)
+ return -ENOSPC;
+
+ return 0;
+}
+
static struct iommufd_sw_msi_map *
-iommufd_sw_msi_alloc_map(struct iommufd_ctx *ictx, phys_addr_t msi_addr,
- const struct iommufd_sw_msi_range *sw_msi_range)
+iommufd_sw_msi_alloc_map(struct iommufd_ctx *ictx, const phys_addr_t *phys_addrs,
+ unsigned int nr_addrs, const struct iommufd_sw_msi_range *sw_msi_range,
+ struct list_head *new_msi_maps)
{
- struct iommufd_sw_msi_map *cur;
- unsigned int max_pgoff = 0;
+ struct iommufd_sw_msi_map *cur, *first_map = NULL;
+ unsigned int next_pgoff = 0;
+ unsigned int i;
+ size_t size;
+ int rc;
lockdep_assert_held(&ictx->sw_msi_lock);
@@ -216,25 +272,38 @@ iommufd_sw_msi_alloc_map(struct iommufd_ctx *ictx, phys_addr_t msi_addr,
continue;
if (cur->pgoff == UINT_MAX)
return ERR_PTR(-EOVERFLOW);
- max_pgoff = max(max_pgoff, cur->pgoff + 1);
+ next_pgoff = max(next_pgoff, cur->pgoff + 1);
}
- if (ictx->sw_msi_id > IOMMUFD_SW_MSI_MAX_ID ||
- max_pgoff > (ULONG_MAX - sw_msi_range->start) / PAGE_SIZE)
- return ERR_PTR(-EOVERFLOW);
- if (max_pgoff >= sw_msi_range->length / PAGE_SIZE)
- return ERR_PTR(-ENOSPC);
-
- cur = kzalloc_obj(*cur);
- if (!cur)
- return ERR_PTR(-ENOMEM);
-
- cur->sw_msi_start = sw_msi_range->start;
- cur->msi_addr = msi_addr;
- cur->pgoff = max_pgoff;
- cur->id = ictx->sw_msi_id++;
- list_add_tail(&cur->sw_msi_item, &ictx->sw_msi_list);
- return cur;
+ rc = iommufd_sw_msi_check_alloc(ictx, sw_msi_range, next_pgoff, nr_addrs, &size);
+ if (rc)
+ return ERR_PTR(rc);
+
+ for (i = 0; i < nr_addrs; i++) {
+ cur = kzalloc_obj(*cur);
+ if (!cur)
+ goto err_free;
+
+ cur->sw_msi_start = sw_msi_range->start;
+ cur->msi_addr = phys_addrs[i];
+ cur->pgoff = next_pgoff + i;
+ cur->id = ictx->sw_msi_id + i;
+ if (!i) {
+ cur->range_size = size;
+ first_map = cur;
+ }
+ list_add_tail(&cur->sw_msi_item, new_msi_maps);
+ }
+
+ return first_map;
+
+err_free:
+ while (!list_empty(new_msi_maps)) {
+ cur = list_first_entry(new_msi_maps, typeof(*cur), sw_msi_item);
+ list_del(&cur->sw_msi_item);
+ kfree(cur);
+ }
+ return ERR_PTR(-ENOMEM);
}
int iommufd_sw_msi_install(struct iommufd_ctx *ictx,
@@ -280,7 +349,9 @@ int iommufd_sw_msi(struct iommu_domain *domain, struct msi_desc *desc,
struct iommufd_attach_handle *handle;
struct iommufd_sw_msi_map *msi_map;
struct iommufd_ctx *ictx;
+ LIST_HEAD(new_msi_maps);
unsigned long iova;
+ phys_addr_t phys_addr;
int rc;
/*
@@ -308,29 +379,41 @@ int iommufd_sw_msi(struct iommu_domain *domain, struct msi_desc *desc,
* assume the caller has checked that it is contained with a MMIO region
* that is secure to map at PAGE_SIZE.
*/
- msi_map = iommufd_sw_msi_get_map(handle->idev->ictx,
- msi_addr & PAGE_MASK,
- &handle->idev->igroup->sw_msi_range);
+ phys_addr = msi_addr & PAGE_MASK;
+ msi_map = iommufd_sw_msi_get_map(ictx, &phys_addr, 1, &handle->idev->igroup->sw_msi_range);
if (!msi_map)
- msi_map = iommufd_sw_msi_alloc_map(handle->idev->ictx,
- msi_addr & PAGE_MASK,
- &handle->idev->igroup->sw_msi_range);
+ msi_map = iommufd_sw_msi_alloc_map(ictx, &phys_addr, 1,
+ &handle->idev->igroup->sw_msi_range,
+ &new_msi_maps);
if (IS_ERR(msi_map))
return PTR_ERR(msi_map);
rc = iommufd_sw_msi_maps_ensure(&handle->idev->igroup->required_sw_msi,
msi_map->id);
if (rc)
- return rc;
+ goto err_free;
rc = iommufd_sw_msi_install(ictx, hwpt_paging, msi_map);
if (rc)
- return rc;
+ goto err_free;
__set_bit(msi_map->id, handle->idev->igroup->required_sw_msi.bitmap);
+ if (!list_empty(&new_msi_maps)) {
+ list_splice_tail_init(&new_msi_maps, &ictx->sw_msi_list);
+ ictx->sw_msi_id++;
+ }
+
iova = msi_map->sw_msi_start + msi_map->pgoff * PAGE_SIZE;
msi_desc_set_iommu_msi_iova(desc, iova, PAGE_SHIFT);
return 0;
+
+err_free:
+ while (!list_empty(&new_msi_maps)) {
+ msi_map = list_first_entry(&new_msi_maps, typeof(*msi_map), sw_msi_item);
+ list_del(&msi_map->sw_msi_item);
+ kfree(msi_map);
+ }
+ return rc;
}
EXPORT_SYMBOL_NS_GPL(iommufd_sw_msi, "IOMMUFD");
#endif
diff --git a/drivers/iommu/iommufd/iommufd_private.h b/drivers/iommu/iommufd/iommufd_private.h
index af7df868dd4f..b54eba5c490b 100644
--- a/drivers/iommu/iommufd/iommufd_private.h
+++ b/drivers/iommu/iommufd/iommufd_private.h
@@ -33,6 +33,7 @@ struct iommufd_sw_msi_map {
phys_addr_t msi_addr;
unsigned int pgoff;
unsigned int id;
+ size_t range_size; /* IOVA range size, or 0 if not the first map */
};
/* Bitmap of struct iommufd_sw_msi_map::id; starts empty, grows on demand. */
--
2.43.0
^ permalink raw reply [flat|nested] 31+ messages in thread* [PATCH v6 06/16] iommufd: Install software MSI map ranges atomically
2026-09-25 15:16 [PATCH v6 00/16] iommu/riscv: Enable MSI remapping, IOMMU_DMA and VFIO Andrew Jones
` (4 preceding siblings ...)
2026-09-25 15:16 ` [PATCH v6 05/16] iommufd: Prepare software MSI maps for address lists Andrew Jones
@ 2026-09-25 15:16 ` Andrew Jones
2026-09-25 15:16 ` [PATCH v6 07/16] iommufd: Prepare software MSI installation for address lists Andrew Jones
` (10 subsequent siblings)
16 siblings, 0 replies; 31+ messages in thread
From: Andrew Jones @ 2026-09-25 15:16 UTC (permalink / raw)
To: linux-riscv, iommu
Cc: linux-kernel, tomasz.jeznach, tjeznach, jgg, jgg, joro, will,
robin.murphy, pjw, palmer, anup, tglx, kevin.tian, fangyu.yu
Add a range installation helper which installs every map in a contiguous
software MSI range and rolls back only mappings added by the failed
operation. Use the helper when replaying range mappings into a new
paging domain.
This prepares iommufd to publish and install physical address lists
without exposing partially installed ranges.
Signed-off-by: Andrew Jones <andrew.jones@oss.qualcomm.com>
Tested-by: Fangyu Yu <fangyu.yu@linux.alibaba.com>
---
drivers/iommu/iommufd/driver.c | 76 ++++++++++++++++++++++++++++++----
1 file changed, 68 insertions(+), 8 deletions(-)
diff --git a/drivers/iommu/iommufd/driver.c b/drivers/iommu/iommufd/driver.c
index 8b8f132ef3c5..3aff04d9dd04 100644
--- a/drivers/iommu/iommufd/driver.c
+++ b/drivers/iommu/iommufd/driver.c
@@ -178,6 +178,17 @@ int iommufd_viommu_report_event(struct iommufd_viommu *viommu,
EXPORT_SYMBOL_NS_GPL(iommufd_viommu_report_event, "IOMMUFD");
#ifdef CONFIG_IRQ_MSI_IOMMU
+static bool iommufd_sw_msi_range_index(const struct iommufd_sw_msi_map *map,
+ const struct iommufd_sw_msi_map *base_map,
+ unsigned int nr_addrs, unsigned int *index)
+{
+ if (map->sw_msi_start != base_map->sw_msi_start || map->pgoff < base_map->pgoff)
+ return false;
+
+ *index = map->pgoff - base_map->pgoff;
+ return *index < nr_addrs;
+}
+
/*
* Get an iommufd_sw_msi_map for the msi physical addresses requested by the irq
* layer. The mapping to IOVA is global to the iommufd file descriptor, every
@@ -207,11 +218,7 @@ iommufd_sw_msi_get_map(struct iommufd_ctx *ictx, const phys_addr_t *phys_addrs,
list_for_each_entry(cur, &ictx->sw_msi_list, sw_msi_item) {
unsigned int index;
- if (cur->sw_msi_start != sw_msi_range->start ||
- cur->pgoff < msi_map->pgoff)
- continue;
- index = cur->pgoff - msi_map->pgoff;
- if (index >= nr_addrs)
+ if (!iommufd_sw_msi_range_index(cur, msi_map, nr_addrs, &index))
continue;
if (cur->msi_addr != phys_addrs[index])
break;
@@ -306,9 +313,9 @@ iommufd_sw_msi_alloc_map(struct iommufd_ctx *ictx, const phys_addr_t *phys_addrs
return ERR_PTR(-ENOMEM);
}
-int iommufd_sw_msi_install(struct iommufd_ctx *ictx,
- struct iommufd_hwpt_paging *hwpt_paging,
- struct iommufd_sw_msi_map *msi_map)
+static int iommufd_sw_msi_install_one(struct iommufd_ctx *ictx,
+ struct iommufd_hwpt_paging *hwpt_paging,
+ struct iommufd_sw_msi_map *msi_map)
{
unsigned long iova;
int rc;
@@ -332,6 +339,59 @@ int iommufd_sw_msi_install(struct iommufd_ctx *ictx,
__set_bit(msi_map->id, hwpt_paging->present_sw_msi.bitmap);
return 0;
}
+
+int iommufd_sw_msi_install(struct iommufd_ctx *ictx,
+ struct iommufd_hwpt_paging *hwpt_paging,
+ struct iommufd_sw_msi_map *base_map)
+{
+ struct iommufd_sw_msi_map *msi_map;
+ struct list_head *msi_maps = &ictx->sw_msi_list;
+ unsigned long *newly_mapped;
+ unsigned int nr_addrs = base_map->range_size ? base_map->range_size / PAGE_SIZE : 1;
+ unsigned int nr_found = 0;
+ unsigned int index;
+ int rc = 0;
+
+ newly_mapped = bitmap_zalloc(nr_addrs, GFP_KERNEL_ACCOUNT);
+ if (!newly_mapped)
+ return -ENOMEM;
+
+ list_for_each_entry(msi_map, msi_maps, sw_msi_item) {
+ if (!iommufd_sw_msi_range_index(msi_map, base_map, nr_addrs, &index))
+ continue;
+ if (iommufd_sw_msi_maps_test_bit(&hwpt_paging->present_sw_msi, msi_map->id)) {
+ nr_found++;
+ continue;
+ }
+ rc = iommufd_sw_msi_install_one(ictx, hwpt_paging, msi_map);
+ if (rc)
+ goto err_unmap;
+ __set_bit(index, newly_mapped);
+ nr_found++;
+ }
+ if (nr_found != nr_addrs) {
+ rc = -EINVAL;
+ goto err_unmap;
+ }
+
+ bitmap_free(newly_mapped);
+ return 0;
+
+err_unmap:
+ list_for_each_entry(msi_map, msi_maps, sw_msi_item) {
+ unsigned long iova;
+
+ if (!iommufd_sw_msi_range_index(msi_map, base_map, nr_addrs, &index) ||
+ !test_bit(index, newly_mapped))
+ continue;
+
+ iova = msi_map->sw_msi_start + msi_map->pgoff * PAGE_SIZE;
+ WARN_ON_ONCE(iommu_unmap(hwpt_paging->common.domain, iova, PAGE_SIZE) != PAGE_SIZE);
+ __clear_bit(msi_map->id, hwpt_paging->present_sw_msi.bitmap);
+ }
+ bitmap_free(newly_mapped);
+ return rc;
+}
EXPORT_SYMBOL_NS_GPL(iommufd_sw_msi_install, "IOMMUFD_INTERNAL");
/*
--
2.43.0
^ permalink raw reply [flat|nested] 31+ messages in thread* [PATCH v6 07/16] iommufd: Prepare software MSI installation for address lists
2026-09-25 15:16 [PATCH v6 00/16] iommu/riscv: Enable MSI remapping, IOMMU_DMA and VFIO Andrew Jones
` (5 preceding siblings ...)
2026-09-25 15:16 ` [PATCH v6 06/16] iommufd: Install software MSI map ranges atomically Andrew Jones
@ 2026-09-25 15:16 ` Andrew Jones
2026-09-28 10:23 ` Andrew Jones
2026-09-25 15:16 ` [PATCH v6 08/16] iommu/dma: Introduce iommu_dma_prepare_msi_list() Andrew Jones
` (9 subsequent siblings)
16 siblings, 1 reply; 31+ messages in thread
From: Andrew Jones @ 2026-09-25 15:16 UTC (permalink / raw)
To: linux-riscv, iommu
Cc: linux-kernel, tomasz.jeznach, tjeznach, jgg, jgg, joro, will,
robin.murphy, pjw, palmer, anup, tglx, kevin.tian, fangyu.yu
Teach the software MSI path to install every mapping in an address list
before publishing newly allocated maps. Mark the complete list as
required by the group and roll back mappings installed by a failed
operation.
Keep iommufd_sw_msi() as a one-address wrapper so existing callers
retain their current behavior.
Signed-off-by: Andrew Jones <andrew.jones@oss.qualcomm.com>
Tested-by: Fangyu Yu <fangyu.yu@linux.alibaba.com>
---
drivers/iommu/iommufd/driver.c | 101 +++++++++++++++++++++++----------
1 file changed, 72 insertions(+), 29 deletions(-)
diff --git a/drivers/iommu/iommufd/driver.c b/drivers/iommu/iommufd/driver.c
index 3aff04d9dd04..f4e812e43822 100644
--- a/drivers/iommu/iommufd/driver.c
+++ b/drivers/iommu/iommufd/driver.c
@@ -340,14 +340,14 @@ static int iommufd_sw_msi_install_one(struct iommufd_ctx *ictx,
return 0;
}
-int iommufd_sw_msi_install(struct iommufd_ctx *ictx,
- struct iommufd_hwpt_paging *hwpt_paging,
- struct iommufd_sw_msi_map *base_map)
+static int __iommufd_sw_msi_install_range(struct iommufd_ctx *ictx,
+ struct iommufd_hwpt_paging *hwpt_paging,
+ struct list_head *msi_maps,
+ const struct iommufd_sw_msi_map *base_map,
+ unsigned int nr_addrs)
{
struct iommufd_sw_msi_map *msi_map;
- struct list_head *msi_maps = &ictx->sw_msi_list;
unsigned long *newly_mapped;
- unsigned int nr_addrs = base_map->range_size ? base_map->range_size / PAGE_SIZE : 1;
unsigned int nr_found = 0;
unsigned int index;
int rc = 0;
@@ -392,16 +392,36 @@ int iommufd_sw_msi_install(struct iommufd_ctx *ictx,
bitmap_free(newly_mapped);
return rc;
}
+
+int iommufd_sw_msi_install(struct iommufd_ctx *ictx,
+ struct iommufd_hwpt_paging *hwpt_paging,
+ struct iommufd_sw_msi_map *base_map)
+{
+ unsigned int nr_addrs = base_map->range_size ? base_map->range_size / PAGE_SIZE : 1;
+
+ return __iommufd_sw_msi_install_range(ictx, hwpt_paging, &ictx->sw_msi_list, base_map,
+ nr_addrs);
+}
EXPORT_SYMBOL_NS_GPL(iommufd_sw_msi_install, "IOMMUFD_INTERNAL");
-/*
- * Called by the irq code if the platform translates the MSI address through the
- * IOMMU. msi_addr is the physical address of the MSI page. iommufd will
- * allocate a fd global iova for the physical page that is the same on all
- * domains and devices.
- */
-int iommufd_sw_msi(struct iommu_domain *domain, struct msi_desc *desc,
- phys_addr_t msi_addr)
+static void iommufd_sw_msi_set_required(struct iommufd_group *igroup,
+ struct list_head *msi_maps,
+ const struct iommufd_sw_msi_map *base_map,
+ unsigned int nr_addrs)
+{
+ struct iommufd_sw_msi_map *msi_map;
+ unsigned int index;
+
+ list_for_each_entry(msi_map, msi_maps, sw_msi_item) {
+ if (!iommufd_sw_msi_range_index(msi_map, base_map, nr_addrs, &index))
+ continue;
+ __set_bit(msi_map->id, igroup->required_sw_msi.bitmap);
+ }
+}
+
+static int iommufd_sw_msi_list(struct iommu_domain *domain, struct msi_desc *desc,
+ const phys_addr_t *phys_addrs, unsigned int nr_addrs,
+ size_t granule)
{
struct device *dev = msi_desc_to_dev(desc);
struct iommufd_hwpt_paging *hwpt_paging;
@@ -410,10 +430,19 @@ int iommufd_sw_msi(struct iommu_domain *domain, struct msi_desc *desc,
struct iommufd_sw_msi_map *msi_map;
struct iommufd_ctx *ictx;
LIST_HEAD(new_msi_maps);
+ struct list_head *msi_maps;
unsigned long iova;
- phys_addr_t phys_addr;
+ unsigned int i;
int rc;
+ if (granule != PAGE_SIZE)
+ return -EOPNOTSUPP;
+ if (!nr_addrs || nr_addrs > SIZE_MAX / PAGE_SIZE)
+ return -EINVAL;
+ for (i = 0; i < nr_addrs; i++)
+ if (!IS_ALIGNED(phys_addrs[i], PAGE_SIZE))
+ return -EINVAL;
+
/*
* It is safe to call iommu_attach_handle_get() here because the iommu
* core code invokes this under the group mutex which also prevents any
@@ -434,33 +463,33 @@ int iommufd_sw_msi(struct iommu_domain *domain, struct msi_desc *desc,
ictx = handle->idev->ictx;
guard(mutex)(&ictx->sw_msi_lock);
- /*
- * The input msi_addr is the exact byte offset of the MSI doorbell, we
- * assume the caller has checked that it is contained with a MMIO region
- * that is secure to map at PAGE_SIZE.
- */
- phys_addr = msi_addr & PAGE_MASK;
- msi_map = iommufd_sw_msi_get_map(ictx, &phys_addr, 1, &handle->idev->igroup->sw_msi_range);
- if (!msi_map)
- msi_map = iommufd_sw_msi_alloc_map(ictx, &phys_addr, 1,
+ msi_map = iommufd_sw_msi_get_map(ictx, phys_addrs, nr_addrs,
+ &handle->idev->igroup->sw_msi_range);
+ if (msi_map) {
+ msi_maps = &ictx->sw_msi_list;
+ } else {
+ msi_map = iommufd_sw_msi_alloc_map(ictx, phys_addrs, nr_addrs,
&handle->idev->igroup->sw_msi_range,
&new_msi_maps);
- if (IS_ERR(msi_map))
- return PTR_ERR(msi_map);
+ if (IS_ERR(msi_map))
+ return PTR_ERR(msi_map);
+ msi_maps = &new_msi_maps;
+ }
rc = iommufd_sw_msi_maps_ensure(&handle->idev->igroup->required_sw_msi,
- msi_map->id);
+ msi_map->id + nr_addrs - 1);
if (rc)
goto err_free;
- rc = iommufd_sw_msi_install(ictx, hwpt_paging, msi_map);
+ rc = __iommufd_sw_msi_install_range(ictx, hwpt_paging, msi_maps, msi_map, nr_addrs);
if (rc)
goto err_free;
- __set_bit(msi_map->id, handle->idev->igroup->required_sw_msi.bitmap);
+
+ iommufd_sw_msi_set_required(handle->idev->igroup, msi_maps, msi_map, nr_addrs);
if (!list_empty(&new_msi_maps)) {
list_splice_tail_init(&new_msi_maps, &ictx->sw_msi_list);
- ictx->sw_msi_id++;
+ ictx->sw_msi_id += nr_addrs;
}
iova = msi_map->sw_msi_start + msi_map->pgoff * PAGE_SIZE;
@@ -475,6 +504,20 @@ int iommufd_sw_msi(struct iommu_domain *domain, struct msi_desc *desc,
}
return rc;
}
+
+/*
+ * Called by the irq code if the platform translates the MSI address through the
+ * IOMMU. msi_addr is the physical address of the MSI page. iommufd will
+ * allocate a fd global iova for the physical page that is the same on all
+ * domains and devices.
+ */
+int iommufd_sw_msi(struct iommu_domain *domain, struct msi_desc *desc,
+ phys_addr_t msi_addr)
+{
+ phys_addr_t phys_addr = msi_addr & PAGE_MASK;
+
+ return iommufd_sw_msi_list(domain, desc, &phys_addr, 1, PAGE_SIZE);
+}
EXPORT_SYMBOL_NS_GPL(iommufd_sw_msi, "IOMMUFD");
#endif
--
2.43.0
^ permalink raw reply [flat|nested] 31+ messages in thread* Re: [PATCH v6 07/16] iommufd: Prepare software MSI installation for address lists
2026-09-25 15:16 ` [PATCH v6 07/16] iommufd: Prepare software MSI installation for address lists Andrew Jones
@ 2026-09-28 10:23 ` Andrew Jones
0 siblings, 0 replies; 31+ messages in thread
From: Andrew Jones @ 2026-09-28 10:23 UTC (permalink / raw)
To: linux-riscv, iommu
Cc: linux-kernel, tomasz.jeznach, tjeznach, jgg, jgg, joro, will,
robin.murphy, pjw, palmer, anup, tglx, kevin.tian, fangyu.yu
On Fri, Sep 25, 2026 at 05:16:50PM +0200, Andrew Jones wrote:
> Teach the software MSI path to install every mapping in an address list
> before publishing newly allocated maps. Mark the complete list as
> required by the group and roll back mappings installed by a failed
> operation.
>
> Keep iommufd_sw_msi() as a one-address wrapper so existing callers
> retain their current behavior.
>
> Signed-off-by: Andrew Jones <andrew.jones@oss.qualcomm.com>
> Tested-by: Fangyu Yu <fangyu.yu@linux.alibaba.com>
> ---
> drivers/iommu/iommufd/driver.c | 101 +++++++++++++++++++++++----------
> 1 file changed, 72 insertions(+), 29 deletions(-)
>
Regarding the Sashiko finding for this patch. It's complaining about the
mixing of goto and cleanup helpers in the same function. However, there
is already precedent for that type of mixing in iommufd, so I'm inclined
to leave it unless somebody other than Sashiko shouts.
Thanks,
drew
^ permalink raw reply [flat|nested] 31+ messages in thread
* [PATCH v6 08/16] iommu/dma: Introduce iommu_dma_prepare_msi_list()
2026-09-25 15:16 [PATCH v6 00/16] iommu/riscv: Enable MSI remapping, IOMMU_DMA and VFIO Andrew Jones
` (6 preceding siblings ...)
2026-09-25 15:16 ` [PATCH v6 07/16] iommufd: Prepare software MSI installation for address lists Andrew Jones
@ 2026-09-25 15:16 ` Andrew Jones
2026-09-25 15:16 ` [PATCH v6 09/16] iommu/riscv: Reserve an MSI IOVA window for iommufd Andrew Jones
` (8 subsequent siblings)
16 siblings, 0 replies; 31+ messages in thread
From: Andrew Jones @ 2026-09-25 15:16 UTC (permalink / raw)
To: linux-riscv, iommu
Cc: linux-kernel, tomasz.jeznach, tjeznach, jgg, jgg, joro, will,
robin.murphy, pjw, palmer, anup, tglx, kevin.tian, fangyu.yu
Add iommu_dma_prepare_msi_list() to map an ordered physical address list
through the current domain backend. Validate that the requested granule
is a power of two no smaller than PAGE_SIZE and that every address is
aligned to it.
Refactor iommu_dma_prepare_msi() to obtain the backend granule and align
its single address before using the same list dispatcher. Convert the
DMA-IOMMU and iommufd backend entry points to consume address lists
directly.
Signed-off-by: Andrew Jones <andrew.jones@oss.qualcomm.com>
Tested-by: Fangyu Yu <fangyu.yu@linux.alibaba.com>
---
drivers/iommu/dma-iommu.c | 13 ++--
drivers/iommu/dma-iommu.h | 13 +++-
drivers/iommu/iommu-priv.h | 7 ++-
drivers/iommu/iommu.c | 109 +++++++++++++++++++++++++++------
drivers/iommu/iommufd/driver.c | 24 +++-----
include/linux/iommu.h | 8 +++
6 files changed, 126 insertions(+), 48 deletions(-)
diff --git a/drivers/iommu/dma-iommu.c b/drivers/iommu/dma-iommu.c
index 0b4eb47d1a95..dc7b369792e4 100644
--- a/drivers/iommu/dma-iommu.c
+++ b/drivers/iommu/dma-iommu.c
@@ -2168,7 +2168,7 @@ static bool has_msi_cookie(const struct iommu_domain *domain)
domain->cookie_type == IOMMU_COOKIE_DMA_MSI);
}
-static size_t cookie_msi_granule(const struct iommu_domain *domain)
+size_t iommu_dma_msi_granule(const struct iommu_domain *domain)
{
switch (domain->cookie_type) {
case IOMMU_COOKIE_DMA_IOVA:
@@ -2299,23 +2299,20 @@ static struct iommu_dma_msi_page *iommu_dma_get_msi_page(struct device *dev,
}
int iommu_dma_sw_msi(struct iommu_domain *domain, struct msi_desc *desc,
- phys_addr_t msi_addr)
+ const phys_addr_t *phys_addrs, unsigned int nr_addrs, size_t granule)
{
struct device *dev = msi_desc_to_dev(desc);
const struct iommu_dma_msi_page *msi_page;
- phys_addr_t phys_addr;
- size_t granule;
if (!has_msi_cookie(domain)) {
msi_desc_set_iommu_msi_iova(desc, 0, 0);
return 0;
}
-
- granule = cookie_msi_granule(domain);
- phys_addr = ALIGN_DOWN(msi_addr, granule);
+ if (granule != iommu_dma_msi_granule(domain))
+ return -EOPNOTSUPP;
iommu_group_mutex_assert(dev);
- msi_page = iommu_dma_get_msi_page(dev, &phys_addr, 1, granule, domain);
+ msi_page = iommu_dma_get_msi_page(dev, phys_addrs, nr_addrs, granule, domain);
if (!msi_page)
return -ENOMEM;
diff --git a/drivers/iommu/dma-iommu.h b/drivers/iommu/dma-iommu.h
index 040d00252563..bf9cd4102d35 100644
--- a/drivers/iommu/dma-iommu.h
+++ b/drivers/iommu/dma-iommu.h
@@ -19,8 +19,9 @@ int iommu_dma_init_fq(struct iommu_domain *domain);
void iommu_dma_get_resv_regions(struct device *dev, struct list_head *list);
+size_t iommu_dma_msi_granule(const struct iommu_domain *domain);
int iommu_dma_sw_msi(struct iommu_domain *domain, struct msi_desc *desc,
- phys_addr_t msi_addr);
+ const phys_addr_t *phys_addrs, unsigned int nr_addrs, size_t granule);
extern bool iommu_dma_forcedac;
@@ -53,8 +54,14 @@ static inline void iommu_dma_get_resv_regions(struct device *dev, struct list_he
{
}
-static inline int iommu_dma_sw_msi(struct iommu_domain *domain,
- struct msi_desc *desc, phys_addr_t msi_addr)
+static inline size_t iommu_dma_msi_granule(const struct iommu_domain *domain)
+{
+ return 0;
+}
+
+static inline int iommu_dma_sw_msi(struct iommu_domain *domain, struct msi_desc *desc,
+ const phys_addr_t *phys_addrs, unsigned int nr_addrs,
+ size_t granule)
{
return -ENODEV;
}
diff --git a/drivers/iommu/iommu-priv.h b/drivers/iommu/iommu-priv.h
index aaffad5854fc..1122c99566d5 100644
--- a/drivers/iommu/iommu-priv.h
+++ b/drivers/iommu/iommu-priv.h
@@ -54,10 +54,11 @@ int iommu_replace_group_handle(struct iommu_group *group,
#if IS_ENABLED(CONFIG_IOMMUFD_DRIVER_CORE) && IS_ENABLED(CONFIG_IRQ_MSI_IOMMU)
int iommufd_sw_msi(struct iommu_domain *domain, struct msi_desc *desc,
- phys_addr_t msi_addr);
+ const phys_addr_t *phys_addrs, unsigned int nr_addrs, size_t granule);
#else /* !CONFIG_IOMMUFD_DRIVER_CORE || !CONFIG_IRQ_MSI_IOMMU */
-static inline int iommufd_sw_msi(struct iommu_domain *domain,
- struct msi_desc *desc, phys_addr_t msi_addr)
+static inline int iommufd_sw_msi(struct iommu_domain *domain, struct msi_desc *desc,
+ const phys_addr_t *phys_addrs, unsigned int nr_addrs,
+ size_t granule)
{
return -EOPNOTSUPP;
}
diff --git a/drivers/iommu/iommu.c b/drivers/iommu/iommu.c
index b486b8bbd1fc..07602b467582 100644
--- a/drivers/iommu/iommu.c
+++ b/drivers/iommu/iommu.c
@@ -4223,21 +4223,89 @@ void pci_dev_reset_iommu_done(struct pci_dev *pdev)
EXPORT_SYMBOL_GPL(pci_dev_reset_iommu_done);
#if IS_ENABLED(CONFIG_IRQ_MSI_IOMMU)
+static int __iommu_dma_prepare_msi_list(struct iommu_group *group, struct msi_desc *desc,
+ const phys_addr_t *phys_addrs, unsigned int nr_addrs,
+ size_t granule)
+{
+ if (!group->domain || group->domain->type == IOMMU_DOMAIN_IDENTITY)
+ return 0;
+
+ switch (group->domain->cookie_type) {
+ case IOMMU_COOKIE_DMA_MSI:
+ case IOMMU_COOKIE_DMA_IOVA:
+ return iommu_dma_sw_msi(group->domain, desc, phys_addrs, nr_addrs, granule);
+ case IOMMU_COOKIE_IOMMUFD:
+ return iommufd_sw_msi(group->domain, desc, phys_addrs, nr_addrs, granule);
+ default:
+ return -EOPNOTSUPP;
+ }
+}
+
+static int iommu_dma_validate_msi_list(const phys_addr_t *phys_addrs, unsigned int nr_addrs,
+ size_t granule)
+{
+ unsigned int i;
+
+ if (!nr_addrs || granule < PAGE_SIZE || !is_power_of_2(granule) ||
+ nr_addrs > SIZE_MAX / granule)
+ return -EINVAL;
+
+ for (i = 0; i < nr_addrs; i++)
+ if (!IS_ALIGNED(phys_addrs[i], granule))
+ return -EINVAL;
+
+ return 0;
+}
+
+/**
+ * iommu_dma_prepare_msi_list() - Map MSI pages in the IOMMU domain
+ * @desc: MSI descriptor to update with the base IOVA
+ * @phys_addrs: Ordered MSI target physical addresses
+ * @nr_addrs: Number of addresses in @phys_addrs
+ * @granule: Mapping granule for every address
+ *
+ * @nr_addrs must be nonzero and @granule must be a power of two no smaller than
+ * PAGE_SIZE. Every address must be aligned to @granule. The addresses are mapped
+ * in order to one contiguous IOVA range and may repeat. The list is consumed
+ * synchronously and is not retained.
+ *
+ * When a software MSI mapping is required, the backend records the base IOVA
+ * and granule shift in @desc. Otherwise, @desc is left unchanged.
+ *
+ * Return: 0 on success, -EINVAL if the parameters are invalid, -EOPNOTSUPP if
+ * the domain backend cannot provide the mapping, or another negative error
+ * code from the backend.
+ */
+int iommu_dma_prepare_msi_list(struct msi_desc *desc, const phys_addr_t *phys_addrs,
+ unsigned int nr_addrs, size_t granule)
+{
+ struct device *dev = msi_desc_to_dev(desc);
+ struct iommu_group *group = dev->iommu_group;
+ int ret;
+
+ ret = iommu_dma_validate_msi_list(phys_addrs, nr_addrs, granule);
+ if (ret || !group)
+ return ret;
+
+ mutex_lock(&group->mutex);
+ ret = __iommu_dma_prepare_msi_list(group, desc, phys_addrs, nr_addrs, granule);
+ mutex_unlock(&group->mutex);
+ return ret;
+}
+
/**
* iommu_dma_prepare_msi() - Map the MSI page in the IOMMU domain
* @desc: MSI descriptor, will store the MSI page
* @msi_addr: MSI target address to be mapped
*
- * The implementation of sw_msi() should take msi_addr and map it to
- * an IOVA in the domain and call msi_desc_set_iommu_msi_iova() with the
- * mapping information.
- *
* Return: 0 on success or negative error code if the mapping failed.
*/
int iommu_dma_prepare_msi(struct msi_desc *desc, phys_addr_t msi_addr)
{
struct device *dev = msi_desc_to_dev(desc);
struct iommu_group *group = dev->iommu_group;
+ phys_addr_t phys_addr;
+ size_t granule;
int ret = 0;
if (!group)
@@ -4248,21 +4316,26 @@ int iommu_dma_prepare_msi(struct msi_desc *desc, phys_addr_t msi_addr)
return ret;
mutex_lock(&group->mutex);
- /* An IDENTITY domain must pass through */
- if (group->domain && group->domain->type != IOMMU_DOMAIN_IDENTITY) {
- switch (group->domain->cookie_type) {
- case IOMMU_COOKIE_DMA_MSI:
- case IOMMU_COOKIE_DMA_IOVA:
- ret = iommu_dma_sw_msi(group->domain, desc, msi_addr);
- break;
- case IOMMU_COOKIE_IOMMUFD:
- ret = iommufd_sw_msi(group->domain, desc, msi_addr);
- break;
- default:
- ret = -EOPNOTSUPP;
- break;
- }
+ if (!group->domain || group->domain->type == IOMMU_DOMAIN_IDENTITY)
+ goto out_unlock;
+
+ switch (group->domain->cookie_type) {
+ case IOMMU_COOKIE_DMA_MSI:
+ case IOMMU_COOKIE_DMA_IOVA:
+ granule = iommu_dma_msi_granule(group->domain);
+ break;
+ case IOMMU_COOKIE_IOMMUFD:
+ granule = PAGE_SIZE;
+ break;
+ default:
+ ret = -EOPNOTSUPP;
+ goto out_unlock;
}
+
+ phys_addr = ALIGN_DOWN(msi_addr, granule);
+ ret = __iommu_dma_prepare_msi_list(group, desc, &phys_addr, 1, granule);
+
+out_unlock:
mutex_unlock(&group->mutex);
return ret;
}
diff --git a/drivers/iommu/iommufd/driver.c b/drivers/iommu/iommufd/driver.c
index f4e812e43822..99104be956ca 100644
--- a/drivers/iommu/iommufd/driver.c
+++ b/drivers/iommu/iommufd/driver.c
@@ -419,9 +419,14 @@ static void iommufd_sw_msi_set_required(struct iommufd_group *igroup,
}
}
-static int iommufd_sw_msi_list(struct iommu_domain *domain, struct msi_desc *desc,
- const phys_addr_t *phys_addrs, unsigned int nr_addrs,
- size_t granule)
+/*
+ * Called by the irq code if the platform translates the MSI addresses through the
+ * IOMMU. phys_addrs are the physical addresses of the MSI pages. iommufd will
+ * allocate contiguous fd global iovas for the physical pages that are the same on
+ * all domains and devices.
+ */
+int iommufd_sw_msi(struct iommu_domain *domain, struct msi_desc *desc,
+ const phys_addr_t *phys_addrs, unsigned int nr_addrs, size_t granule)
{
struct device *dev = msi_desc_to_dev(desc);
struct iommufd_hwpt_paging *hwpt_paging;
@@ -505,19 +510,6 @@ static int iommufd_sw_msi_list(struct iommu_domain *domain, struct msi_desc *des
return rc;
}
-/*
- * Called by the irq code if the platform translates the MSI address through the
- * IOMMU. msi_addr is the physical address of the MSI page. iommufd will
- * allocate a fd global iova for the physical page that is the same on all
- * domains and devices.
- */
-int iommufd_sw_msi(struct iommu_domain *domain, struct msi_desc *desc,
- phys_addr_t msi_addr)
-{
- phys_addr_t phys_addr = msi_addr & PAGE_MASK;
-
- return iommufd_sw_msi_list(domain, desc, &phys_addr, 1, PAGE_SIZE);
-}
EXPORT_SYMBOL_NS_GPL(iommufd_sw_msi, "IOMMUFD");
#endif
diff --git a/include/linux/iommu.h b/include/linux/iommu.h
index ac43b8b93f14..bf08be172e40 100644
--- a/include/linux/iommu.h
+++ b/include/linux/iommu.h
@@ -1561,8 +1561,16 @@ static inline void pci_dev_reset_iommu_done(struct pci_dev *pdev)
#ifdef CONFIG_IRQ_MSI_IOMMU
#ifdef CONFIG_IOMMU_API
+int iommu_dma_prepare_msi_list(struct msi_desc *desc, const phys_addr_t *phys_addrs,
+ unsigned int nr_addrs, size_t granule);
int iommu_dma_prepare_msi(struct msi_desc *desc, phys_addr_t msi_addr);
#else
+static inline int iommu_dma_prepare_msi_list(struct msi_desc *desc, const phys_addr_t *phys_addrs,
+ unsigned int nr_addrs, size_t granule)
+{
+ return 0;
+}
+
static inline int iommu_dma_prepare_msi(struct msi_desc *desc,
phys_addr_t msi_addr)
{
--
2.43.0
^ permalink raw reply [flat|nested] 31+ messages in thread* [PATCH v6 09/16] iommu/riscv: Reserve an MSI IOVA window for iommufd
2026-09-25 15:16 [PATCH v6 00/16] iommu/riscv: Enable MSI remapping, IOMMU_DMA and VFIO Andrew Jones
` (7 preceding siblings ...)
2026-09-25 15:16 ` [PATCH v6 08/16] iommu/dma: Introduce iommu_dma_prepare_msi_list() Andrew Jones
@ 2026-09-25 15:16 ` Andrew Jones
2026-09-25 15:16 ` [PATCH v6 10/16] irqchip/riscv-imsic: Add MSI address list Andrew Jones
` (7 subsequent siblings)
16 siblings, 0 replies; 31+ messages in thread
From: Andrew Jones @ 2026-09-25 15:16 UTC (permalink / raw)
To: linux-riscv, iommu
Cc: linux-kernel, tomasz.jeznach, tjeznach, jgg, jgg, joro, will,
robin.murphy, pjw, palmer, anup, tglx, kevin.tian, fangyu.yu
iommufd requires an IOMMU_RESV_SW_MSI region to allocate stable IOVAs
for MSI targets. Advertise such a region when the device uses an IMSIC
MSI hierarchy so interrupt remapping can map IMSIC pages instead of
falling back to physical addresses.
Reserve one page per possible CPU, sufficient for each supervisor IMSIC
page. Use a 16 MiB base instead of the 128 MiB convention used by ARM
SMMU. kvmtool places its guest IMSIC at 128 MiB, so with IRQ bypass the
domain-wide guest MSI address match can also capture host-delivered
MSI writes to the SW MSI window and misdeliver them to the guest.
16 MiB falls in a hole in the current kvmtool and QEMU virt memory maps,
away from their guest IMSIC windows and RAM. Keep the window below
4 GiB so devices with 32-bit MSI address registers can use it.
This only avoids known conflicts. A VMM can place a guest's IMSIC at
any valid address, including the chosen SW MSI range. As long as a
fixed SW MSI window is required, changing its base cannot solve this
in general or guarantee disjoint host and guest MSI address spaces.
Signed-off-by: Andrew Jones <andrew.jones@oss.qualcomm.com>
Tested-by: Fangyu Yu <fangyu.yu@linux.alibaba.com>
---
drivers/iommu/riscv/iommu.c | 20 ++++++++++++++++++++
drivers/iommu/riscv/iommu.h | 4 ++++
drivers/irqchip/irq-riscv-imsic-state.c | 22 ++++++++++++++++++++++
include/linux/irqchip/riscv-imsic.h | 6 ++++++
4 files changed, 52 insertions(+)
diff --git a/drivers/iommu/riscv/iommu.c b/drivers/iommu/riscv/iommu.c
index fbdca69c436e..dcad99131670 100644
--- a/drivers/iommu/riscv/iommu.c
+++ b/drivers/iommu/riscv/iommu.c
@@ -20,10 +20,12 @@
#include <linux/init.h>
#include <linux/iommu.h>
#include <linux/iopoll.h>
+#include <linux/irqchip/riscv-imsic.h>
#include <linux/kernel.h>
#include <linux/pci.h>
#include <linux/generic_pt/iommu.h>
+#include "../dma-iommu.h"
#include "../iommu-pages.h"
#include "iommu-bits.h"
#include "iommu.h"
@@ -1514,6 +1516,23 @@ static void riscv_iommu_release_device(struct device *dev)
kfree_rcu_mightsleep(info);
}
+static void riscv_iommu_get_resv_regions(struct device *dev, struct list_head *head)
+{
+ struct iommu_resv_region *region;
+
+ if (imsic_dev_has_imsic_msi_parent(dev)) {
+ /* Each hart has one S-mode IMSIC page, a.k.a MSI target page */
+ region = iommu_alloc_resv_region(RISCV_IOMMU_MSI_IOVA_BASE,
+ (size_t)num_possible_cpus() * PAGE_SIZE,
+ IOMMU_WRITE | IOMMU_NOEXEC | IOMMU_MMIO,
+ IOMMU_RESV_SW_MSI, GFP_KERNEL);
+ if (region)
+ list_add_tail(®ion->list, head);
+ }
+
+ iommu_dma_get_resv_regions(dev, head);
+}
+
static const struct iommu_ops riscv_iommu_ops = {
.of_xlate = riscv_iommu_of_xlate,
.capable = riscv_iommu_capable,
@@ -1524,6 +1543,7 @@ static const struct iommu_ops riscv_iommu_ops = {
.device_group = riscv_iommu_device_group,
.probe_device = riscv_iommu_probe_device,
.release_device = riscv_iommu_release_device,
+ .get_resv_regions = riscv_iommu_get_resv_regions,
};
static int riscv_iommu_init_check(struct riscv_iommu_device *iommu)
diff --git a/drivers/iommu/riscv/iommu.h b/drivers/iommu/riscv/iommu.h
index 5676001548cc..6d5c70e9ac6d 100644
--- a/drivers/iommu/riscv/iommu.h
+++ b/drivers/iommu/riscv/iommu.h
@@ -15,9 +15,13 @@
#include <linux/spinlock.h>
#include <linux/types.h>
#include <linux/iopoll.h>
+#include <linux/sizes.h>
#include "iommu-bits.h"
+/* IOVA base for the SW MSI reservation */
+#define RISCV_IOMMU_MSI_IOVA_BASE SZ_16M
+
struct riscv_iommu_device;
struct riscv_iommu_queue {
diff --git a/drivers/irqchip/irq-riscv-imsic-state.c b/drivers/irqchip/irq-riscv-imsic-state.c
index b8d1bbbf42f7..df38a7670a89 100644
--- a/drivers/irqchip/irq-riscv-imsic-state.c
+++ b/drivers/irqchip/irq-riscv-imsic-state.c
@@ -64,6 +64,28 @@ const struct imsic_global_config *imsic_get_global_config(void)
}
EXPORT_SYMBOL_GPL(imsic_get_global_config);
+/**
+ * imsic_dev_has_imsic_msi_parent - Check for an IMSIC MSI parent
+ * @dev: Device to check
+ *
+ * Return: true if @dev's MSI domain or any parent domain is the IMSIC base
+ * domain.
+ */
+bool imsic_dev_has_imsic_msi_parent(struct device *dev)
+{
+ struct irq_domain *domain;
+
+ if (!imsic || !imsic->base_domain)
+ return false;
+
+ for (domain = dev_get_msi_domain(dev); domain; domain = domain->parent)
+ if (domain == imsic->base_domain)
+ return true;
+
+ return false;
+}
+EXPORT_SYMBOL_GPL(imsic_dev_has_imsic_msi_parent);
+
static bool __imsic_eix_read_clear(unsigned long id, bool pend)
{
unsigned long isel, imask;
diff --git a/include/linux/irqchip/riscv-imsic.h b/include/linux/irqchip/riscv-imsic.h
index 61af3a5bea09..662cb0442424 100644
--- a/include/linux/irqchip/riscv-imsic.h
+++ b/include/linux/irqchip/riscv-imsic.h
@@ -81,6 +81,7 @@ struct imsic_global_config {
#ifdef CONFIG_RISCV_IMSIC
const struct imsic_global_config *imsic_get_global_config(void);
+bool imsic_dev_has_imsic_msi_parent(struct device *dev);
#else
@@ -89,6 +90,11 @@ static inline const struct imsic_global_config *imsic_get_global_config(void)
return NULL;
}
+static inline bool imsic_dev_has_imsic_msi_parent(struct device *dev)
+{
+ return false;
+}
+
#endif
#if IS_ENABLED(CONFIG_ACPI) && IS_ENABLED(CONFIG_RISCV_IMSIC)
--
2.43.0
^ permalink raw reply [flat|nested] 31+ messages in thread* [PATCH v6 10/16] irqchip/riscv-imsic: Add MSI address list
2026-09-25 15:16 [PATCH v6 00/16] iommu/riscv: Enable MSI remapping, IOMMU_DMA and VFIO Andrew Jones
` (8 preceding siblings ...)
2026-09-25 15:16 ` [PATCH v6 09/16] iommu/riscv: Reserve an MSI IOVA window for iommufd Andrew Jones
@ 2026-09-25 15:16 ` Andrew Jones
2026-09-25 15:51 ` Anup Patel
` (2 more replies)
2026-09-25 15:16 ` [PATCH v6 11/16] irqchip/riscv-imsic: Support IOMMU MSI address lists Andrew Jones
` (6 subsequent siblings)
16 siblings, 3 replies; 31+ messages in thread
From: Andrew Jones @ 2026-09-25 15:16 UTC (permalink / raw)
To: linux-riscv, iommu
Cc: linux-kernel, tomasz.jeznach, tjeznach, jgg, jgg, joro, will,
robin.murphy, pjw, palmer, anup, tglx, kevin.tian, fangyu.yu
Upcoming RISC-V IOMMU interrupt remapping needs to map every possible
host IMSIC target into one contiguous MSI IOVA range. MSI message
composition then uses the target CPU's position within that range.
Build an IMSIC physical address array indexed by logical CPU.
Require every possible CPU to have an initialized IMSIC page before
publishing the array. This preserves physical address zero as a valid
target and guarantees the array has num_possible_cpus() entries.
The array size and each target's index can therefore be derived
directly, without storing duplicate count or per-CPU index metadata.
Signed-off-by: Andrew Jones <andrew.jones@oss.qualcomm.com>
Tested-by: Fangyu Yu <fangyu.yu@linux.alibaba.com>
---
drivers/irqchip/irq-riscv-imsic-state.c | 32 +++++++++++++++++++++++++
drivers/irqchip/irq-riscv-imsic-state.h | 3 +++
2 files changed, 35 insertions(+)
diff --git a/drivers/irqchip/irq-riscv-imsic-state.c b/drivers/irqchip/irq-riscv-imsic-state.c
index df38a7670a89..abd1cdb640ea 100644
--- a/drivers/irqchip/irq-riscv-imsic-state.c
+++ b/drivers/irqchip/irq-riscv-imsic-state.c
@@ -709,6 +709,31 @@ static int __init imsic_get_mmio_resource(struct fwnode_handle *fwnode,
return of_address_to_resource(to_of_node(fwnode), index, res);
}
+static int __init imsic_init_msi_pa_list(void)
+{
+ struct imsic_global_config *global = &imsic->global;
+ phys_addr_t *msi_pa_list;
+ unsigned int cpu;
+
+ msi_pa_list = kcalloc(num_possible_cpus(), sizeof(*msi_pa_list), GFP_KERNEL);
+ if (!msi_pa_list)
+ return -ENOMEM;
+
+ for_each_possible_cpu(cpu) {
+ struct imsic_local_config *local = per_cpu_ptr(global->local, cpu);
+
+ if (!local->msi_va) {
+ kfree(msi_pa_list);
+ return -ENODEV;
+ }
+
+ msi_pa_list[cpu] = local->msi_pa;
+ }
+
+ imsic->msi_pa_list = msi_pa_list;
+ return 0;
+}
+
static int __init imsic_parse_fwnode(struct fwnode_handle *fwnode,
struct imsic_global_config *global,
u32 *nr_parent_irqs,
@@ -959,6 +984,12 @@ int __init imsic_setup_state(struct fwnode_handle *fwnode, void *opaque)
goto out_local_cleanup;
}
+ rc = imsic_init_msi_pa_list();
+ if (rc) {
+ pr_err("%pfwP: failed to initialize MSI address list\n", fwnode);
+ goto out_local_cleanup;
+ }
+
/* Initialize matrix allocator */
rc = imsic_matrix_init();
if (rc) {
@@ -984,6 +1015,7 @@ int __init imsic_setup_state(struct fwnode_handle *fwnode, void *opaque)
out_free_local:
free_percpu(imsic->global.local);
out_free_priv:
+ kfree(imsic->msi_pa_list);
kfree(imsic);
imsic = NULL;
return rc;
diff --git a/drivers/irqchip/irq-riscv-imsic-state.h b/drivers/irqchip/irq-riscv-imsic-state.h
index c42ee180b305..e51c61dcced3 100644
--- a/drivers/irqchip/irq-riscv-imsic-state.h
+++ b/drivers/irqchip/irq-riscv-imsic-state.h
@@ -50,6 +50,9 @@ struct imsic_priv {
/* Global configuration common for all HARTs */
struct imsic_global_config global;
+ /* MSI target physical addresses indexed by logical CPU for IOMMU mapping */
+ phys_addr_t *msi_pa_list;
+
/* Per-CPU state */
struct imsic_local_priv __percpu *lpriv;
--
2.43.0
^ permalink raw reply [flat|nested] 31+ messages in thread* Re: [PATCH v6 10/16] irqchip/riscv-imsic: Add MSI address list
2026-09-25 15:16 ` [PATCH v6 10/16] irqchip/riscv-imsic: Add MSI address list Andrew Jones
@ 2026-09-25 15:51 ` Anup Patel
2026-09-28 3:27 ` Nutty.Liu
2026-09-28 10:20 ` Andrew Jones
2 siblings, 0 replies; 31+ messages in thread
From: Anup Patel @ 2026-09-25 15:51 UTC (permalink / raw)
To: Andrew Jones
Cc: linux-riscv, iommu, linux-kernel, tomasz.jeznach, tjeznach, jgg,
jgg, joro, will, robin.murphy, pjw, palmer, tglx, kevin.tian,
fangyu.yu
On Fri, Sep 25, 2026 at 8:47 PM Andrew Jones
<andrew.jones@oss.qualcomm.com> wrote:
>
> Upcoming RISC-V IOMMU interrupt remapping needs to map every possible
> host IMSIC target into one contiguous MSI IOVA range. MSI message
> composition then uses the target CPU's position within that range.
>
> Build an IMSIC physical address array indexed by logical CPU.
> Require every possible CPU to have an initialized IMSIC page before
> publishing the array. This preserves physical address zero as a valid
> target and guarantees the array has num_possible_cpus() entries.
>
> The array size and each target's index can therefore be derived
> directly, without storing duplicate count or per-CPU index metadata.
>
> Signed-off-by: Andrew Jones <andrew.jones@oss.qualcomm.com>
> Tested-by: Fangyu Yu <fangyu.yu@linux.alibaba.com>
LGMT.
Reviewed-by: Anup Patel <anup@brainfault.org>
Acked-by: Anup Patel <anup@brainfault.org>
Thanks,
Anup
> ---
> drivers/irqchip/irq-riscv-imsic-state.c | 32 +++++++++++++++++++++++++
> drivers/irqchip/irq-riscv-imsic-state.h | 3 +++
> 2 files changed, 35 insertions(+)
>
> diff --git a/drivers/irqchip/irq-riscv-imsic-state.c b/drivers/irqchip/irq-riscv-imsic-state.c
> index df38a7670a89..abd1cdb640ea 100644
> --- a/drivers/irqchip/irq-riscv-imsic-state.c
> +++ b/drivers/irqchip/irq-riscv-imsic-state.c
> @@ -709,6 +709,31 @@ static int __init imsic_get_mmio_resource(struct fwnode_handle *fwnode,
> return of_address_to_resource(to_of_node(fwnode), index, res);
> }
>
> +static int __init imsic_init_msi_pa_list(void)
> +{
> + struct imsic_global_config *global = &imsic->global;
> + phys_addr_t *msi_pa_list;
> + unsigned int cpu;
> +
> + msi_pa_list = kcalloc(num_possible_cpus(), sizeof(*msi_pa_list), GFP_KERNEL);
> + if (!msi_pa_list)
> + return -ENOMEM;
> +
> + for_each_possible_cpu(cpu) {
> + struct imsic_local_config *local = per_cpu_ptr(global->local, cpu);
> +
> + if (!local->msi_va) {
> + kfree(msi_pa_list);
> + return -ENODEV;
> + }
> +
> + msi_pa_list[cpu] = local->msi_pa;
> + }
> +
> + imsic->msi_pa_list = msi_pa_list;
> + return 0;
> +}
> +
> static int __init imsic_parse_fwnode(struct fwnode_handle *fwnode,
> struct imsic_global_config *global,
> u32 *nr_parent_irqs,
> @@ -959,6 +984,12 @@ int __init imsic_setup_state(struct fwnode_handle *fwnode, void *opaque)
> goto out_local_cleanup;
> }
>
> + rc = imsic_init_msi_pa_list();
> + if (rc) {
> + pr_err("%pfwP: failed to initialize MSI address list\n", fwnode);
> + goto out_local_cleanup;
> + }
> +
> /* Initialize matrix allocator */
> rc = imsic_matrix_init();
> if (rc) {
> @@ -984,6 +1015,7 @@ int __init imsic_setup_state(struct fwnode_handle *fwnode, void *opaque)
> out_free_local:
> free_percpu(imsic->global.local);
> out_free_priv:
> + kfree(imsic->msi_pa_list);
> kfree(imsic);
> imsic = NULL;
> return rc;
> diff --git a/drivers/irqchip/irq-riscv-imsic-state.h b/drivers/irqchip/irq-riscv-imsic-state.h
> index c42ee180b305..e51c61dcced3 100644
> --- a/drivers/irqchip/irq-riscv-imsic-state.h
> +++ b/drivers/irqchip/irq-riscv-imsic-state.h
> @@ -50,6 +50,9 @@ struct imsic_priv {
> /* Global configuration common for all HARTs */
> struct imsic_global_config global;
>
> + /* MSI target physical addresses indexed by logical CPU for IOMMU mapping */
> + phys_addr_t *msi_pa_list;
> +
> /* Per-CPU state */
> struct imsic_local_priv __percpu *lpriv;
>
> --
> 2.43.0
>
^ permalink raw reply [flat|nested] 31+ messages in thread* Re: [PATCH v6 10/16] irqchip/riscv-imsic: Add MSI address list
2026-09-25 15:16 ` [PATCH v6 10/16] irqchip/riscv-imsic: Add MSI address list Andrew Jones
2026-09-25 15:51 ` Anup Patel
@ 2026-09-28 3:27 ` Nutty.Liu
2026-09-28 10:20 ` Andrew Jones
2 siblings, 0 replies; 31+ messages in thread
From: Nutty.Liu @ 2026-09-28 3:27 UTC (permalink / raw)
To: Andrew Jones, linux-riscv, iommu
Cc: linux-kernel, tomasz.jeznach, tjeznach, jgg, jgg, joro, will,
robin.murphy, pjw, palmer, anup, tglx, kevin.tian, fangyu.yu
On 9/25/2026 11:16 PM, Andrew Jones wrote:
> Upcoming RISC-V IOMMU interrupt remapping needs to map every possible
> host IMSIC target into one contiguous MSI IOVA range. MSI message
> composition then uses the target CPU's position within that range.
>
> Build an IMSIC physical address array indexed by logical CPU.
> Require every possible CPU to have an initialized IMSIC page before
> publishing the array. This preserves physical address zero as a valid
> target and guarantees the array has num_possible_cpus() entries.
>
> The array size and each target's index can therefore be derived
> directly, without storing duplicate count or per-CPU index metadata.
>
> Signed-off-by: Andrew Jones <andrew.jones@oss.qualcomm.com>
> Tested-by: Fangyu Yu <fangyu.yu@linux.alibaba.com>
Reviewed-by: Nutty Liu <nutty.liu@hotmail.com>
Thanks,
Nutty
> ---
> drivers/irqchip/irq-riscv-imsic-state.c | 32 +++++++++++++++++++++++++
> drivers/irqchip/irq-riscv-imsic-state.h | 3 +++
> 2 files changed, 35 insertions(+)
>
> diff --git a/drivers/irqchip/irq-riscv-imsic-state.c b/drivers/irqchip/irq-riscv-imsic-state.c
> index df38a7670a89..abd1cdb640ea 100644
> --- a/drivers/irqchip/irq-riscv-imsic-state.c
> +++ b/drivers/irqchip/irq-riscv-imsic-state.c
> @@ -709,6 +709,31 @@ static int __init imsic_get_mmio_resource(struct fwnode_handle *fwnode,
> return of_address_to_resource(to_of_node(fwnode), index, res);
> }
>
> +static int __init imsic_init_msi_pa_list(void)
> +{
> + struct imsic_global_config *global = &imsic->global;
> + phys_addr_t *msi_pa_list;
> + unsigned int cpu;
> +
> + msi_pa_list = kcalloc(num_possible_cpus(), sizeof(*msi_pa_list), GFP_KERNEL);
> + if (!msi_pa_list)
> + return -ENOMEM;
> +
> + for_each_possible_cpu(cpu) {
> + struct imsic_local_config *local = per_cpu_ptr(global->local, cpu);
> +
> + if (!local->msi_va) {
> + kfree(msi_pa_list);
> + return -ENODEV;
> + }
> +
> + msi_pa_list[cpu] = local->msi_pa;
> + }
> +
> + imsic->msi_pa_list = msi_pa_list;
> + return 0;
> +}
> +
> static int __init imsic_parse_fwnode(struct fwnode_handle *fwnode,
> struct imsic_global_config *global,
> u32 *nr_parent_irqs,
> @@ -959,6 +984,12 @@ int __init imsic_setup_state(struct fwnode_handle *fwnode, void *opaque)
> goto out_local_cleanup;
> }
>
> + rc = imsic_init_msi_pa_list();
> + if (rc) {
> + pr_err("%pfwP: failed to initialize MSI address list\n", fwnode);
> + goto out_local_cleanup;
> + }
> +
> /* Initialize matrix allocator */
> rc = imsic_matrix_init();
> if (rc) {
> @@ -984,6 +1015,7 @@ int __init imsic_setup_state(struct fwnode_handle *fwnode, void *opaque)
> out_free_local:
> free_percpu(imsic->global.local);
> out_free_priv:
> + kfree(imsic->msi_pa_list);
> kfree(imsic);
> imsic = NULL;
> return rc;
> diff --git a/drivers/irqchip/irq-riscv-imsic-state.h b/drivers/irqchip/irq-riscv-imsic-state.h
> index c42ee180b305..e51c61dcced3 100644
> --- a/drivers/irqchip/irq-riscv-imsic-state.h
> +++ b/drivers/irqchip/irq-riscv-imsic-state.h
> @@ -50,6 +50,9 @@ struct imsic_priv {
> /* Global configuration common for all HARTs */
> struct imsic_global_config global;
>
> + /* MSI target physical addresses indexed by logical CPU for IOMMU mapping */
> + phys_addr_t *msi_pa_list;
> +
> /* Per-CPU state */
> struct imsic_local_priv __percpu *lpriv;
>
^ permalink raw reply [flat|nested] 31+ messages in thread* Re: [PATCH v6 10/16] irqchip/riscv-imsic: Add MSI address list
2026-09-25 15:16 ` [PATCH v6 10/16] irqchip/riscv-imsic: Add MSI address list Andrew Jones
2026-09-25 15:51 ` Anup Patel
2026-09-28 3:27 ` Nutty.Liu
@ 2026-09-28 10:20 ` Andrew Jones
2 siblings, 0 replies; 31+ messages in thread
From: Andrew Jones @ 2026-09-28 10:20 UTC (permalink / raw)
To: linux-riscv, iommu
Cc: linux-kernel, tomasz.jeznach, tjeznach, jgg, jgg, joro, will,
robin.murphy, pjw, palmer, anup, tglx, kevin.tian, fangyu.yu
On Fri, Sep 25, 2026 at 05:16:53PM +0200, Andrew Jones wrote:
> Upcoming RISC-V IOMMU interrupt remapping needs to map every possible
> host IMSIC target into one contiguous MSI IOVA range. MSI message
> composition then uses the target CPU's position within that range.
>
> Build an IMSIC physical address array indexed by logical CPU.
> Require every possible CPU to have an initialized IMSIC page before
> publishing the array. This preserves physical address zero as a valid
> target and guarantees the array has num_possible_cpus() entries.
>
> The array size and each target's index can therefore be derived
> directly, without storing duplicate count or per-CPU index metadata.
>
> Signed-off-by: Andrew Jones <andrew.jones@oss.qualcomm.com>
> Tested-by: Fangyu Yu <fangyu.yu@linux.alibaba.com>
> ---
> drivers/irqchip/irq-riscv-imsic-state.c | 32 +++++++++++++++++++++++++
> drivers/irqchip/irq-riscv-imsic-state.h | 3 +++
> 2 files changed, 35 insertions(+)
>
Regarding the Sashiko findings for this patch. The first I already
responded to on v5[1] and the second, which is new, is complaining
about the error handing in imsic_setup_state() for the new call to
imsic_init_msi_pa_list(), but the error handling follows the pattern
of all other error handling for that function. So there is nothing
to do for Sashiko here.
[1] https://lore.kernel.org/all/20260901133835.345001-1-andrew.jones@oss.qualcomm.com/
Thanks,
drew
^ permalink raw reply [flat|nested] 31+ messages in thread
* [PATCH v6 11/16] irqchip/riscv-imsic: Support IOMMU MSI address lists
2026-09-25 15:16 [PATCH v6 00/16] iommu/riscv: Enable MSI remapping, IOMMU_DMA and VFIO Andrew Jones
` (9 preceding siblings ...)
2026-09-25 15:16 ` [PATCH v6 10/16] irqchip/riscv-imsic: Add MSI address list Andrew Jones
@ 2026-09-25 15:16 ` Andrew Jones
2026-09-25 15:52 ` Anup Patel
2026-09-28 3:18 ` Nutty.Liu
2026-09-25 15:16 ` [PATCH v6 12/16] iommu/dma: Enable IOMMU_DMA for 64-bit RISC-V Andrew Jones
` (5 subsequent siblings)
16 siblings, 2 replies; 31+ messages in thread
From: Andrew Jones @ 2026-09-25 15:16 UTC (permalink / raw)
To: linux-riscv, iommu
Cc: linux-kernel, tomasz.jeznach, tjeznach, jgg, jgg, joro, will,
robin.murphy, pjw, palmer, anup, tglx, kevin.tian, fangyu.yu
RISC-V IOMMU host interrupt remapping maps every possible host IMSIC
target into one contiguous IOVA range. MSI message composition therefore
needs each target's position within that range.
Prepare the complete IMSIC address list while allocating an IRQ. The
list is guaranteed to exist after IMSIC state initialization. The IOMMU
backend maps the pages and caches the contiguous base IOVA in the MSI
descriptor when translation is needed.
When the descriptor has an IOMMU MSI mapping, use the selected logical
CPU directly as the page index within the IOVA range. Use the same path
for initial composition and affinity updates. A zero iommu_msi_shift
means no MSI address translation is required, so keep physical messages.
Signed-off-by: Andrew Jones <andrew.jones@oss.qualcomm.com>
Tested-by: Fangyu Yu <fangyu.yu@linux.alibaba.com>
---
drivers/irqchip/Kconfig | 1 +
drivers/irqchip/irq-riscv-imsic-platform.c | 24 +++++++++++++++++++---
2 files changed, 22 insertions(+), 3 deletions(-)
diff --git a/drivers/irqchip/Kconfig b/drivers/irqchip/Kconfig
index 20b77fbc51ee..a038e1121be3 100644
--- a/drivers/irqchip/Kconfig
+++ b/drivers/irqchip/Kconfig
@@ -651,6 +651,7 @@ config RISCV_IMSIC
select IRQ_DOMAIN_HIERARCHY
select GENERIC_IRQ_MATRIX_ALLOCATOR
select GENERIC_MSI_IRQ
+ select IRQ_MSI_IOMMU
select IRQ_MSI_LIB
config RISCV_RPMI_SYSMSI
diff --git a/drivers/irqchip/irq-riscv-imsic-platform.c b/drivers/irqchip/irq-riscv-imsic-platform.c
index 643c8e459611..ddb2bb570cdc 100644
--- a/drivers/irqchip/irq-riscv-imsic-platform.c
+++ b/drivers/irqchip/irq-riscv-imsic-platform.c
@@ -10,6 +10,7 @@
#include <linux/cpu.h>
#include <linux/interrupt.h>
#include <linux/io.h>
+#include <linux/iommu.h>
#include <linux/irq.h>
#include <linux/irqchip.h>
#include <linux/irqdomain.h>
@@ -69,8 +70,10 @@ static void imsic_irq_ack(struct irq_data *d)
irq_move_irq(d);
}
-static void imsic_irq_compose_vector_msg(struct imsic_vector *vec, struct msi_msg *msg)
+static void imsic_irq_compose_vector_msg(struct irq_data *d, struct imsic_vector *vec,
+ struct msi_msg *msg)
{
+ struct msi_desc *desc = irq_data_get_msi_desc(d);
phys_addr_t msi_addr;
if (WARN_ON(!vec))
@@ -79,6 +82,12 @@ static void imsic_irq_compose_vector_msg(struct imsic_vector *vec, struct msi_ms
if (WARN_ON(!imsic_cpu_page_phys(vec->cpu, 0, &msi_addr)))
return;
+ /* A zero shift means no IOMMU MSI mapping is needed. */
+ if (desc->iommu_msi_shift) {
+ msi_addr = (desc->iommu_msi_iova << desc->iommu_msi_shift) +
+ vec->cpu * IMSIC_MMIO_PAGE_SZ;
+ }
+
msg->address_hi = upper_32_bits(msi_addr);
msg->address_lo = lower_32_bits(msi_addr);
msg->data = vec->local_id;
@@ -86,7 +95,7 @@ static void imsic_irq_compose_vector_msg(struct imsic_vector *vec, struct msi_ms
static void imsic_irq_compose_msg(struct irq_data *d, struct msi_msg *msg)
{
- imsic_irq_compose_vector_msg(irq_data_get_irq_chip_data(d), msg);
+ imsic_irq_compose_vector_msg(d, irq_data_get_irq_chip_data(d), msg);
}
#ifdef CONFIG_SMP
@@ -94,7 +103,7 @@ static void imsic_msi_update_msg(struct irq_data *d, struct imsic_vector *vec)
{
struct msi_msg msg = { };
- imsic_irq_compose_vector_msg(vec, &msg);
+ imsic_irq_compose_vector_msg(d, vec, &msg);
irq_data_get_irq_chip(d)->irq_write_msi_msg(d, &msg);
}
@@ -225,7 +234,9 @@ static struct irq_chip imsic_irq_base_chip = {
static int imsic_irq_domain_alloc(struct irq_domain *domain, unsigned int virq,
unsigned int nr_irqs, void *args)
{
+ msi_alloc_info_t *info = args;
struct imsic_vector *vec;
+ int ret;
/* Multi-MSI is not supported yet. */
if (nr_irqs > 1)
@@ -235,6 +246,13 @@ static int imsic_irq_domain_alloc(struct irq_domain *domain, unsigned int virq,
if (!vec)
return -ENOSPC;
+ ret = iommu_dma_prepare_msi_list(info->desc, imsic->msi_pa_list,
+ num_possible_cpus(), IMSIC_MMIO_PAGE_SZ);
+ if (ret) {
+ imsic_vector_free(vec);
+ return ret;
+ }
+
irq_domain_set_info(domain, virq, virq, &imsic_irq_base_chip, vec,
handle_edge_irq, NULL, NULL);
irq_set_noprobe(virq);
--
2.43.0
^ permalink raw reply [flat|nested] 31+ messages in thread* Re: [PATCH v6 11/16] irqchip/riscv-imsic: Support IOMMU MSI address lists
2026-09-25 15:16 ` [PATCH v6 11/16] irqchip/riscv-imsic: Support IOMMU MSI address lists Andrew Jones
@ 2026-09-25 15:52 ` Anup Patel
2026-09-28 3:18 ` Nutty.Liu
1 sibling, 0 replies; 31+ messages in thread
From: Anup Patel @ 2026-09-25 15:52 UTC (permalink / raw)
To: Andrew Jones
Cc: linux-riscv, iommu, linux-kernel, tomasz.jeznach, tjeznach, jgg,
jgg, joro, will, robin.murphy, pjw, palmer, tglx, kevin.tian,
fangyu.yu
On Fri, Sep 25, 2026 at 8:47 PM Andrew Jones
<andrew.jones@oss.qualcomm.com> wrote:
>
> RISC-V IOMMU host interrupt remapping maps every possible host IMSIC
> target into one contiguous IOVA range. MSI message composition therefore
> needs each target's position within that range.
>
> Prepare the complete IMSIC address list while allocating an IRQ. The
> list is guaranteed to exist after IMSIC state initialization. The IOMMU
> backend maps the pages and caches the contiguous base IOVA in the MSI
> descriptor when translation is needed.
>
> When the descriptor has an IOMMU MSI mapping, use the selected logical
> CPU directly as the page index within the IOVA range. Use the same path
> for initial composition and affinity updates. A zero iommu_msi_shift
> means no MSI address translation is required, so keep physical messages.
>
> Signed-off-by: Andrew Jones <andrew.jones@oss.qualcomm.com>
> Tested-by: Fangyu Yu <fangyu.yu@linux.alibaba.com>
LGMT.
Reviewed-by: Anup Patel <anup@brainfault.org>
Acked-by: Anup Patel <anup@brainfault.org>
Thanks,
Anup
> ---
> drivers/irqchip/Kconfig | 1 +
> drivers/irqchip/irq-riscv-imsic-platform.c | 24 +++++++++++++++++++---
> 2 files changed, 22 insertions(+), 3 deletions(-)
>
> diff --git a/drivers/irqchip/Kconfig b/drivers/irqchip/Kconfig
> index 20b77fbc51ee..a038e1121be3 100644
> --- a/drivers/irqchip/Kconfig
> +++ b/drivers/irqchip/Kconfig
> @@ -651,6 +651,7 @@ config RISCV_IMSIC
> select IRQ_DOMAIN_HIERARCHY
> select GENERIC_IRQ_MATRIX_ALLOCATOR
> select GENERIC_MSI_IRQ
> + select IRQ_MSI_IOMMU
> select IRQ_MSI_LIB
>
> config RISCV_RPMI_SYSMSI
> diff --git a/drivers/irqchip/irq-riscv-imsic-platform.c b/drivers/irqchip/irq-riscv-imsic-platform.c
> index 643c8e459611..ddb2bb570cdc 100644
> --- a/drivers/irqchip/irq-riscv-imsic-platform.c
> +++ b/drivers/irqchip/irq-riscv-imsic-platform.c
> @@ -10,6 +10,7 @@
> #include <linux/cpu.h>
> #include <linux/interrupt.h>
> #include <linux/io.h>
> +#include <linux/iommu.h>
> #include <linux/irq.h>
> #include <linux/irqchip.h>
> #include <linux/irqdomain.h>
> @@ -69,8 +70,10 @@ static void imsic_irq_ack(struct irq_data *d)
> irq_move_irq(d);
> }
>
> -static void imsic_irq_compose_vector_msg(struct imsic_vector *vec, struct msi_msg *msg)
> +static void imsic_irq_compose_vector_msg(struct irq_data *d, struct imsic_vector *vec,
> + struct msi_msg *msg)
> {
> + struct msi_desc *desc = irq_data_get_msi_desc(d);
> phys_addr_t msi_addr;
>
> if (WARN_ON(!vec))
> @@ -79,6 +82,12 @@ static void imsic_irq_compose_vector_msg(struct imsic_vector *vec, struct msi_ms
> if (WARN_ON(!imsic_cpu_page_phys(vec->cpu, 0, &msi_addr)))
> return;
>
> + /* A zero shift means no IOMMU MSI mapping is needed. */
> + if (desc->iommu_msi_shift) {
> + msi_addr = (desc->iommu_msi_iova << desc->iommu_msi_shift) +
> + vec->cpu * IMSIC_MMIO_PAGE_SZ;
> + }
> +
> msg->address_hi = upper_32_bits(msi_addr);
> msg->address_lo = lower_32_bits(msi_addr);
> msg->data = vec->local_id;
> @@ -86,7 +95,7 @@ static void imsic_irq_compose_vector_msg(struct imsic_vector *vec, struct msi_ms
>
> static void imsic_irq_compose_msg(struct irq_data *d, struct msi_msg *msg)
> {
> - imsic_irq_compose_vector_msg(irq_data_get_irq_chip_data(d), msg);
> + imsic_irq_compose_vector_msg(d, irq_data_get_irq_chip_data(d), msg);
> }
>
> #ifdef CONFIG_SMP
> @@ -94,7 +103,7 @@ static void imsic_msi_update_msg(struct irq_data *d, struct imsic_vector *vec)
> {
> struct msi_msg msg = { };
>
> - imsic_irq_compose_vector_msg(vec, &msg);
> + imsic_irq_compose_vector_msg(d, vec, &msg);
> irq_data_get_irq_chip(d)->irq_write_msi_msg(d, &msg);
> }
>
> @@ -225,7 +234,9 @@ static struct irq_chip imsic_irq_base_chip = {
> static int imsic_irq_domain_alloc(struct irq_domain *domain, unsigned int virq,
> unsigned int nr_irqs, void *args)
> {
> + msi_alloc_info_t *info = args;
> struct imsic_vector *vec;
> + int ret;
>
> /* Multi-MSI is not supported yet. */
> if (nr_irqs > 1)
> @@ -235,6 +246,13 @@ static int imsic_irq_domain_alloc(struct irq_domain *domain, unsigned int virq,
> if (!vec)
> return -ENOSPC;
>
> + ret = iommu_dma_prepare_msi_list(info->desc, imsic->msi_pa_list,
> + num_possible_cpus(), IMSIC_MMIO_PAGE_SZ);
> + if (ret) {
> + imsic_vector_free(vec);
> + return ret;
> + }
> +
> irq_domain_set_info(domain, virq, virq, &imsic_irq_base_chip, vec,
> handle_edge_irq, NULL, NULL);
> irq_set_noprobe(virq);
> --
> 2.43.0
>
^ permalink raw reply [flat|nested] 31+ messages in thread* Re: [PATCH v6 11/16] irqchip/riscv-imsic: Support IOMMU MSI address lists
2026-09-25 15:16 ` [PATCH v6 11/16] irqchip/riscv-imsic: Support IOMMU MSI address lists Andrew Jones
2026-09-25 15:52 ` Anup Patel
@ 2026-09-28 3:18 ` Nutty.Liu
1 sibling, 0 replies; 31+ messages in thread
From: Nutty.Liu @ 2026-09-28 3:18 UTC (permalink / raw)
To: Andrew Jones, linux-riscv, iommu
Cc: linux-kernel, tomasz.jeznach, tjeznach, jgg, jgg, joro, will,
robin.murphy, pjw, palmer, anup, tglx, kevin.tian, fangyu.yu
On 9/25/2026 11:16 PM, Andrew Jones wrote:
> RISC-V IOMMU host interrupt remapping maps every possible host IMSIC
> target into one contiguous IOVA range. MSI message composition therefore
> needs each target's position within that range.
>
> Prepare the complete IMSIC address list while allocating an IRQ. The
> list is guaranteed to exist after IMSIC state initialization. The IOMMU
> backend maps the pages and caches the contiguous base IOVA in the MSI
> descriptor when translation is needed.
>
> When the descriptor has an IOMMU MSI mapping, use the selected logical
> CPU directly as the page index within the IOVA range. Use the same path
> for initial composition and affinity updates. A zero iommu_msi_shift
> means no MSI address translation is required, so keep physical messages.
>
> Signed-off-by: Andrew Jones <andrew.jones@oss.qualcomm.com>
> Tested-by: Fangyu Yu <fangyu.yu@linux.alibaba.com>
Reviewed-by: Nutty Liu <nutty.liu@hotmail.com>
Thanks,
Nutty
> ---
> drivers/irqchip/Kconfig | 1 +
> drivers/irqchip/irq-riscv-imsic-platform.c | 24 +++++++++++++++++++---
> 2 files changed, 22 insertions(+), 3 deletions(-)
>
> diff --git a/drivers/irqchip/Kconfig b/drivers/irqchip/Kconfig
> index 20b77fbc51ee..a038e1121be3 100644
> --- a/drivers/irqchip/Kconfig
> +++ b/drivers/irqchip/Kconfig
> @@ -651,6 +651,7 @@ config RISCV_IMSIC
> select IRQ_DOMAIN_HIERARCHY
> select GENERIC_IRQ_MATRIX_ALLOCATOR
> select GENERIC_MSI_IRQ
> + select IRQ_MSI_IOMMU
> select IRQ_MSI_LIB
>
> config RISCV_RPMI_SYSMSI
> diff --git a/drivers/irqchip/irq-riscv-imsic-platform.c b/drivers/irqchip/irq-riscv-imsic-platform.c
> index 643c8e459611..ddb2bb570cdc 100644
> --- a/drivers/irqchip/irq-riscv-imsic-platform.c
> +++ b/drivers/irqchip/irq-riscv-imsic-platform.c
> @@ -10,6 +10,7 @@
> #include <linux/cpu.h>
> #include <linux/interrupt.h>
> #include <linux/io.h>
> +#include <linux/iommu.h>
> #include <linux/irq.h>
> #include <linux/irqchip.h>
> #include <linux/irqdomain.h>
> @@ -69,8 +70,10 @@ static void imsic_irq_ack(struct irq_data *d)
> irq_move_irq(d);
> }
>
> -static void imsic_irq_compose_vector_msg(struct imsic_vector *vec, struct msi_msg *msg)
> +static void imsic_irq_compose_vector_msg(struct irq_data *d, struct imsic_vector *vec,
> + struct msi_msg *msg)
> {
> + struct msi_desc *desc = irq_data_get_msi_desc(d);
> phys_addr_t msi_addr;
>
> if (WARN_ON(!vec))
> @@ -79,6 +82,12 @@ static void imsic_irq_compose_vector_msg(struct imsic_vector *vec, struct msi_ms
> if (WARN_ON(!imsic_cpu_page_phys(vec->cpu, 0, &msi_addr)))
> return;
>
> + /* A zero shift means no IOMMU MSI mapping is needed. */
> + if (desc->iommu_msi_shift) {
> + msi_addr = (desc->iommu_msi_iova << desc->iommu_msi_shift) +
> + vec->cpu * IMSIC_MMIO_PAGE_SZ;
> + }
> +
> msg->address_hi = upper_32_bits(msi_addr);
> msg->address_lo = lower_32_bits(msi_addr);
> msg->data = vec->local_id;
> @@ -86,7 +95,7 @@ static void imsic_irq_compose_vector_msg(struct imsic_vector *vec, struct msi_ms
>
> static void imsic_irq_compose_msg(struct irq_data *d, struct msi_msg *msg)
> {
> - imsic_irq_compose_vector_msg(irq_data_get_irq_chip_data(d), msg);
> + imsic_irq_compose_vector_msg(d, irq_data_get_irq_chip_data(d), msg);
> }
>
> #ifdef CONFIG_SMP
> @@ -94,7 +103,7 @@ static void imsic_msi_update_msg(struct irq_data *d, struct imsic_vector *vec)
> {
> struct msi_msg msg = { };
>
> - imsic_irq_compose_vector_msg(vec, &msg);
> + imsic_irq_compose_vector_msg(d, vec, &msg);
> irq_data_get_irq_chip(d)->irq_write_msi_msg(d, &msg);
> }
>
> @@ -225,7 +234,9 @@ static struct irq_chip imsic_irq_base_chip = {
> static int imsic_irq_domain_alloc(struct irq_domain *domain, unsigned int virq,
> unsigned int nr_irqs, void *args)
> {
> + msi_alloc_info_t *info = args;
> struct imsic_vector *vec;
> + int ret;
>
> /* Multi-MSI is not supported yet. */
> if (nr_irqs > 1)
> @@ -235,6 +246,13 @@ static int imsic_irq_domain_alloc(struct irq_domain *domain, unsigned int virq,
> if (!vec)
> return -ENOSPC;
>
> + ret = iommu_dma_prepare_msi_list(info->desc, imsic->msi_pa_list,
> + num_possible_cpus(), IMSIC_MMIO_PAGE_SZ);
> + if (ret) {
> + imsic_vector_free(vec);
> + return ret;
> + }
> +
> irq_domain_set_info(domain, virq, virq, &imsic_irq_base_chip, vec,
> handle_edge_irq, NULL, NULL);
> irq_set_noprobe(virq);
^ permalink raw reply [flat|nested] 31+ messages in thread
* [PATCH v6 12/16] iommu/dma: Enable IOMMU_DMA for 64-bit RISC-V
2026-09-25 15:16 [PATCH v6 00/16] iommu/riscv: Enable MSI remapping, IOMMU_DMA and VFIO Andrew Jones
` (10 preceding siblings ...)
2026-09-25 15:16 ` [PATCH v6 11/16] irqchip/riscv-imsic: Support IOMMU MSI address lists Andrew Jones
@ 2026-09-25 15:16 ` Andrew Jones
2026-09-28 3:17 ` Nutty.Liu
2026-09-25 15:16 ` [PATCH v6 13/16] vfio: enable IOMMU_TYPE1 for RISC-V Andrew Jones
` (4 subsequent siblings)
16 siblings, 1 reply; 31+ messages in thread
From: Andrew Jones @ 2026-09-25 15:16 UTC (permalink / raw)
To: linux-riscv, iommu
Cc: linux-kernel, tomasz.jeznach, tjeznach, jgg, jgg, joro, will,
robin.murphy, pjw, palmer, anup, tglx, kevin.tian, fangyu.yu
From: Tomasz Jeznach <tomasz.jeznach@linux.dev>
Enable IOMMU_DMA for 64-bit RISC-V now that the RISC-V
IOMMU driver supports MSI remapping.
Signed-off-by: Tomasz Jeznach <tjeznach@rivosinc.com>
Signed-off-by: Andrew Jones <andrew.jones@oss.qualcomm.com>
Tested-by: Fangyu Yu <fangyu.yu@linux.alibaba.com>
---
drivers/iommu/Kconfig | 2 +-
1 file changed, 1 insertion(+), 1 deletion(-)
diff --git a/drivers/iommu/Kconfig b/drivers/iommu/Kconfig
index 27954f462582..2d38d5b5accd 100644
--- a/drivers/iommu/Kconfig
+++ b/drivers/iommu/Kconfig
@@ -151,7 +151,7 @@ config OF_IOMMU
# IOMMU-agnostic DMA-mapping layer
config IOMMU_DMA
- def_bool ARM64 || X86 || S390
+ def_bool ARM64 || X86 || S390 || (RISCV && 64BIT)
select DMA_OPS_HELPERS
select IOMMU_API
select IOMMU_IOVA
--
2.43.0
^ permalink raw reply [flat|nested] 31+ messages in thread* Re: [PATCH v6 12/16] iommu/dma: Enable IOMMU_DMA for 64-bit RISC-V
2026-09-25 15:16 ` [PATCH v6 12/16] iommu/dma: Enable IOMMU_DMA for 64-bit RISC-V Andrew Jones
@ 2026-09-28 3:17 ` Nutty.Liu
0 siblings, 0 replies; 31+ messages in thread
From: Nutty.Liu @ 2026-09-28 3:17 UTC (permalink / raw)
To: Andrew Jones, linux-riscv, iommu
Cc: linux-kernel, tomasz.jeznach, tjeznach, jgg, jgg, joro, will,
robin.murphy, pjw, palmer, anup, tglx, kevin.tian, fangyu.yu
On 9/25/2026 11:16 PM, Andrew Jones wrote:
> From: Tomasz Jeznach <tomasz.jeznach@linux.dev>
>
> Enable IOMMU_DMA for 64-bit RISC-V now that the RISC-V
> IOMMU driver supports MSI remapping.
>
> Signed-off-by: Tomasz Jeznach <tjeznach@rivosinc.com>
> Signed-off-by: Andrew Jones <andrew.jones@oss.qualcomm.com>
> Tested-by: Fangyu Yu <fangyu.yu@linux.alibaba.com>
Reviewed-by: Nutty Liu <nutty.liu@hotmail.com>
Thanks,
Nutty
> ---
> drivers/iommu/Kconfig | 2 +-
> 1 file changed, 1 insertion(+), 1 deletion(-)
>
> diff --git a/drivers/iommu/Kconfig b/drivers/iommu/Kconfig
> index 27954f462582..2d38d5b5accd 100644
> --- a/drivers/iommu/Kconfig
> +++ b/drivers/iommu/Kconfig
> @@ -151,7 +151,7 @@ config OF_IOMMU
>
> # IOMMU-agnostic DMA-mapping layer
> config IOMMU_DMA
> - def_bool ARM64 || X86 || S390
> + def_bool ARM64 || X86 || S390 || (RISCV && 64BIT)
> select DMA_OPS_HELPERS
> select IOMMU_API
> select IOMMU_IOVA
^ permalink raw reply [flat|nested] 31+ messages in thread
* [PATCH v6 13/16] vfio: enable IOMMU_TYPE1 for RISC-V
2026-09-25 15:16 [PATCH v6 00/16] iommu/riscv: Enable MSI remapping, IOMMU_DMA and VFIO Andrew Jones
` (11 preceding siblings ...)
2026-09-25 15:16 ` [PATCH v6 12/16] iommu/dma: Enable IOMMU_DMA for 64-bit RISC-V Andrew Jones
@ 2026-09-25 15:16 ` Andrew Jones
2026-09-25 15:16 ` [PATCH v6 14/16] RISC-V: KVM: Enable KVM_VFIO interfaces on RISC-V arch Andrew Jones
` (3 subsequent siblings)
16 siblings, 0 replies; 31+ messages in thread
From: Andrew Jones @ 2026-09-25 15:16 UTC (permalink / raw)
To: linux-riscv, iommu
Cc: linux-kernel, tomasz.jeznach, tjeznach, jgg, jgg, joro, will,
robin.murphy, pjw, palmer, anup, tglx, kevin.tian, fangyu.yu,
Nutty Liu
From: Tomasz Jeznach <tjeznach@rivosinc.com>
Enable VFIO support on RISC-V architecture, now that the RISC-V IOMMU
driver reports the IOMMU_CAP_CACHE_COHERENCY capability VFIO_TYPE1 and
iommufd both require before allowing a device to be bound.
Signed-off-by: Tomasz Jeznach <tjeznach@rivosinc.com>
Signed-off-by: Andrew Jones <andrew.jones@oss.qualcomm.com>
Reviewed-by: Nutty Liu <nutty.liu@hotmail.com>
Tested-by: Fangyu Yu <fangyu.yu@linux.alibaba.com>
---
drivers/vfio/Kconfig | 2 +-
1 file changed, 1 insertion(+), 1 deletion(-)
diff --git a/drivers/vfio/Kconfig b/drivers/vfio/Kconfig
index b9d6e1c22aed..f7f5b9c1950d 100644
--- a/drivers/vfio/Kconfig
+++ b/drivers/vfio/Kconfig
@@ -38,7 +38,7 @@ config VFIO_GROUP
config VFIO_CONTAINER
bool "Support for the VFIO container /dev/vfio/vfio"
- select VFIO_IOMMU_TYPE1 if MMU && (X86 || S390 || ARM || ARM64)
+ select VFIO_IOMMU_TYPE1 if MMU && (X86 || S390 || ARM || ARM64 || RISCV)
depends on VFIO_GROUP
default y
help
--
2.43.0
^ permalink raw reply [flat|nested] 31+ messages in thread* [PATCH v6 14/16] RISC-V: KVM: Enable KVM_VFIO interfaces on RISC-V arch
2026-09-25 15:16 [PATCH v6 00/16] iommu/riscv: Enable MSI remapping, IOMMU_DMA and VFIO Andrew Jones
` (12 preceding siblings ...)
2026-09-25 15:16 ` [PATCH v6 13/16] vfio: enable IOMMU_TYPE1 for RISC-V Andrew Jones
@ 2026-09-25 15:16 ` Andrew Jones
2026-09-25 15:51 ` Anup Patel
2026-09-25 15:16 ` [PATCH v6 15/16] riscv: defconfig: Enable IOMMUFD and VFIO Andrew Jones
` (2 subsequent siblings)
16 siblings, 1 reply; 31+ messages in thread
From: Andrew Jones @ 2026-09-25 15:16 UTC (permalink / raw)
To: linux-riscv, iommu
Cc: linux-kernel, tomasz.jeznach, tjeznach, jgg, jgg, joro, will,
robin.murphy, pjw, palmer, anup, tglx, kevin.tian, fangyu.yu,
Nutty Liu
From: Tomasz Jeznach <tjeznach@rivosinc.com>
Enable KVM/VFIO support on RISC-V architecture, now that VFIO device
assignment is available on RISC-V through VFIO_IOMMU_TYPE1, so a
RISC-V KVM guest can be notified about VFIO-assigned devices.
Signed-off-by: Tomasz Jeznach <tjeznach@rivosinc.com>
Signed-off-by: Andrew Jones <andrew.jones@oss.qualcomm.com>
Reviewed-by: Nutty Liu <nutty.liu@hotmail.com>
Tested-by: Fangyu Yu <fangyu.yu@linux.alibaba.com>
---
arch/riscv/kvm/Kconfig | 1 +
1 file changed, 1 insertion(+)
diff --git a/arch/riscv/kvm/Kconfig b/arch/riscv/kvm/Kconfig
index ec2cee0a39e0..49179aae9504 100644
--- a/arch/riscv/kvm/Kconfig
+++ b/arch/riscv/kvm/Kconfig
@@ -29,6 +29,7 @@ config KVM
select KVM_GENERIC_DIRTYLOG_READ_PROTECT
select KVM_GENERIC_HARDWARE_ENABLING
select KVM_MMIO
+ select KVM_VFIO
select VIRT_XFER_TO_GUEST_WORK
select SCHED_INFO
select GUEST_PERF_EVENTS if PERF_EVENTS
--
2.43.0
^ permalink raw reply [flat|nested] 31+ messages in thread* Re: [PATCH v6 14/16] RISC-V: KVM: Enable KVM_VFIO interfaces on RISC-V arch
2026-09-25 15:16 ` [PATCH v6 14/16] RISC-V: KVM: Enable KVM_VFIO interfaces on RISC-V arch Andrew Jones
@ 2026-09-25 15:51 ` Anup Patel
0 siblings, 0 replies; 31+ messages in thread
From: Anup Patel @ 2026-09-25 15:51 UTC (permalink / raw)
To: Andrew Jones
Cc: linux-riscv, iommu, linux-kernel, tomasz.jeznach, tjeznach, jgg,
jgg, joro, will, robin.murphy, pjw, palmer, tglx, kevin.tian,
fangyu.yu, Nutty Liu
On Fri, Sep 25, 2026 at 8:47 PM Andrew Jones
<andrew.jones@oss.qualcomm.com> wrote:
>
> From: Tomasz Jeznach <tjeznach@rivosinc.com>
>
> Enable KVM/VFIO support on RISC-V architecture, now that VFIO device
> assignment is available on RISC-V through VFIO_IOMMU_TYPE1, so a
> RISC-V KVM guest can be notified about VFIO-assigned devices.
>
> Signed-off-by: Tomasz Jeznach <tjeznach@rivosinc.com>
> Signed-off-by: Andrew Jones <andrew.jones@oss.qualcomm.com>
> Reviewed-by: Nutty Liu <nutty.liu@hotmail.com>
> Tested-by: Fangyu Yu <fangyu.yu@linux.alibaba.com>
LGMT.
Reviewed-by: Anup Patel <anup@brainfault.org>
Acked-by: Anup Patel <anup@brainfault.org>
Thanks,
Anup
> ---
> arch/riscv/kvm/Kconfig | 1 +
> 1 file changed, 1 insertion(+)
>
> diff --git a/arch/riscv/kvm/Kconfig b/arch/riscv/kvm/Kconfig
> index ec2cee0a39e0..49179aae9504 100644
> --- a/arch/riscv/kvm/Kconfig
> +++ b/arch/riscv/kvm/Kconfig
> @@ -29,6 +29,7 @@ config KVM
> select KVM_GENERIC_DIRTYLOG_READ_PROTECT
> select KVM_GENERIC_HARDWARE_ENABLING
> select KVM_MMIO
> + select KVM_VFIO
> select VIRT_XFER_TO_GUEST_WORK
> select SCHED_INFO
> select GUEST_PERF_EVENTS if PERF_EVENTS
> --
> 2.43.0
>
^ permalink raw reply [flat|nested] 31+ messages in thread
* [PATCH v6 15/16] riscv: defconfig: Enable IOMMUFD and VFIO
2026-09-25 15:16 [PATCH v6 00/16] iommu/riscv: Enable MSI remapping, IOMMU_DMA and VFIO Andrew Jones
` (13 preceding siblings ...)
2026-09-25 15:16 ` [PATCH v6 14/16] RISC-V: KVM: Enable KVM_VFIO interfaces on RISC-V arch Andrew Jones
@ 2026-09-25 15:16 ` Andrew Jones
2026-09-28 3:17 ` Nutty.Liu
2026-09-25 15:16 ` [PATCH v6 16/16] selftests/vfio: Allow building on RISC-V Andrew Jones
2026-09-28 10:24 ` [PATCH v6 00/16] iommu/riscv: Enable MSI remapping, IOMMU_DMA and VFIO Andrew Jones
16 siblings, 1 reply; 31+ messages in thread
From: Andrew Jones @ 2026-09-25 15:16 UTC (permalink / raw)
To: linux-riscv, iommu
Cc: linux-kernel, tomasz.jeznach, tjeznach, jgg, jgg, joro, will,
robin.murphy, pjw, palmer, anup, tglx, kevin.tian, fangyu.yu
Enable iommufd and VFIO in the default configuration now that the
RISC-V IOMMU provides the DMA and MSI remapping needed for userspace
device access, including assignment to virtual machines.
Signed-off-by: Andrew Jones <andrew.jones@oss.qualcomm.com>
Tested-by: Fangyu Yu <fangyu.yu@linux.alibaba.com>
---
arch/riscv/configs/defconfig | 4 ++++
1 file changed, 4 insertions(+)
diff --git a/arch/riscv/configs/defconfig b/arch/riscv/configs/defconfig
index 04ae305d5511..daf382c05441 100644
--- a/arch/riscv/configs/defconfig
+++ b/arch/riscv/configs/defconfig
@@ -251,6 +251,10 @@ CONFIG_DMADEVICES=y
CONFIG_DMA_SUN6I=m
CONFIG_DW_AXI_DMAC=y
CONFIG_MMP_PDMA=m
+CONFIG_IOMMUFD=m
+CONFIG_VFIO_DEVICE_CDEV=y
+CONFIG_VFIO=m
+CONFIG_VFIO_PCI=m
CONFIG_VIRTIO_PCI=y
CONFIG_VIRTIO_BALLOON=y
CONFIG_VIRTIO_INPUT=y
--
2.43.0
^ permalink raw reply [flat|nested] 31+ messages in thread* Re: [PATCH v6 15/16] riscv: defconfig: Enable IOMMUFD and VFIO
2026-09-25 15:16 ` [PATCH v6 15/16] riscv: defconfig: Enable IOMMUFD and VFIO Andrew Jones
@ 2026-09-28 3:17 ` Nutty.Liu
0 siblings, 0 replies; 31+ messages in thread
From: Nutty.Liu @ 2026-09-28 3:17 UTC (permalink / raw)
To: Andrew Jones, linux-riscv, iommu
Cc: linux-kernel, tomasz.jeznach, tjeznach, jgg, jgg, joro, will,
robin.murphy, pjw, palmer, anup, tglx, kevin.tian, fangyu.yu
On 9/25/2026 11:16 PM, Andrew Jones wrote:
> Enable iommufd and VFIO in the default configuration now that the
> RISC-V IOMMU provides the DMA and MSI remapping needed for userspace
> device access, including assignment to virtual machines.
>
> Signed-off-by: Andrew Jones <andrew.jones@oss.qualcomm.com>
> Tested-by: Fangyu Yu <fangyu.yu@linux.alibaba.com>
Reviewed-by: Nutty Liu <nutty.liu@hotmail.com>
Thanks,
Nutty
> ---
> arch/riscv/configs/defconfig | 4 ++++
> 1 file changed, 4 insertions(+)
>
> diff --git a/arch/riscv/configs/defconfig b/arch/riscv/configs/defconfig
> index 04ae305d5511..daf382c05441 100644
> --- a/arch/riscv/configs/defconfig
> +++ b/arch/riscv/configs/defconfig
> @@ -251,6 +251,10 @@ CONFIG_DMADEVICES=y
> CONFIG_DMA_SUN6I=m
> CONFIG_DW_AXI_DMAC=y
> CONFIG_MMP_PDMA=m
> +CONFIG_IOMMUFD=m
> +CONFIG_VFIO_DEVICE_CDEV=y
> +CONFIG_VFIO=m
> +CONFIG_VFIO_PCI=m
> CONFIG_VIRTIO_PCI=y
> CONFIG_VIRTIO_BALLOON=y
> CONFIG_VIRTIO_INPUT=y
^ permalink raw reply [flat|nested] 31+ messages in thread
* [PATCH v6 16/16] selftests/vfio: Allow building on RISC-V
2026-09-25 15:16 [PATCH v6 00/16] iommu/riscv: Enable MSI remapping, IOMMU_DMA and VFIO Andrew Jones
` (14 preceding siblings ...)
2026-09-25 15:16 ` [PATCH v6 15/16] riscv: defconfig: Enable IOMMUFD and VFIO Andrew Jones
@ 2026-09-25 15:16 ` Andrew Jones
2026-09-28 3:16 ` Nutty.Liu
2026-09-28 10:24 ` [PATCH v6 00/16] iommu/riscv: Enable MSI remapping, IOMMU_DMA and VFIO Andrew Jones
16 siblings, 1 reply; 31+ messages in thread
From: Andrew Jones @ 2026-09-25 15:16 UTC (permalink / raw)
To: linux-riscv, iommu
Cc: linux-kernel, tomasz.jeznach, tjeznach, jgg, jgg, joro, will,
robin.murphy, pjw, palmer, anup, tglx, kevin.tian, fangyu.yu
The generic VFIO selftests, including iommufd coverage, require no
RISC-V-specific source changes. Top-level kselftest builds normalize
riscv64 to riscv, while direct builds may use riscv64. Include both
names in the architecture filter.
Signed-off-by: Andrew Jones <andrew.jones@oss.qualcomm.com>
Tested-by: Fangyu Yu <fangyu.yu@linux.alibaba.com>
---
tools/testing/selftests/vfio/Makefile | 2 +-
1 file changed, 1 insertion(+), 1 deletion(-)
diff --git a/tools/testing/selftests/vfio/Makefile b/tools/testing/selftests/vfio/Makefile
index 2c32c48db509..2c7d8ea81b61 100644
--- a/tools/testing/selftests/vfio/Makefile
+++ b/tools/testing/selftests/vfio/Makefile
@@ -1,6 +1,6 @@
ARCH ?= $(shell uname -m)
-ifeq (,$(filter $(ARCH),aarch64 arm64 x86 x86_64))
+ifeq (,$(filter $(ARCH),aarch64 arm64 x86 x86_64 riscv riscv64))
# Do nothing on unsupported architectures
include ../lib.mk
else
--
2.43.0
^ permalink raw reply [flat|nested] 31+ messages in thread* Re: [PATCH v6 16/16] selftests/vfio: Allow building on RISC-V
2026-09-25 15:16 ` [PATCH v6 16/16] selftests/vfio: Allow building on RISC-V Andrew Jones
@ 2026-09-28 3:16 ` Nutty.Liu
0 siblings, 0 replies; 31+ messages in thread
From: Nutty.Liu @ 2026-09-28 3:16 UTC (permalink / raw)
To: Andrew Jones, linux-riscv, iommu
Cc: linux-kernel, tomasz.jeznach, tjeznach, jgg, jgg, joro, will,
robin.murphy, pjw, palmer, anup, tglx, kevin.tian, fangyu.yu
On 9/25/2026 11:16 PM, Andrew Jones wrote:
> The generic VFIO selftests, including iommufd coverage, require no
> RISC-V-specific source changes. Top-level kselftest builds normalize
> riscv64 to riscv, while direct builds may use riscv64. Include both
> names in the architecture filter.
>
> Signed-off-by: Andrew Jones <andrew.jones@oss.qualcomm.com>
> Tested-by: Fangyu Yu <fangyu.yu@linux.alibaba.com>
Reviewed-by: Nutty Liu <nutty.liu@hotmail.com>
Thanks,
Nutty
> ---
> tools/testing/selftests/vfio/Makefile | 2 +-
> 1 file changed, 1 insertion(+), 1 deletion(-)
>
> diff --git a/tools/testing/selftests/vfio/Makefile b/tools/testing/selftests/vfio/Makefile
> index 2c32c48db509..2c7d8ea81b61 100644
> --- a/tools/testing/selftests/vfio/Makefile
> +++ b/tools/testing/selftests/vfio/Makefile
> @@ -1,6 +1,6 @@
> ARCH ?= $(shell uname -m)
>
> -ifeq (,$(filter $(ARCH),aarch64 arm64 x86 x86_64))
> +ifeq (,$(filter $(ARCH),aarch64 arm64 x86 x86_64 riscv riscv64))
> # Do nothing on unsupported architectures
> include ../lib.mk
> else
^ permalink raw reply [flat|nested] 31+ messages in thread
* Re: [PATCH v6 00/16] iommu/riscv: Enable MSI remapping, IOMMU_DMA and VFIO
2026-09-25 15:16 [PATCH v6 00/16] iommu/riscv: Enable MSI remapping, IOMMU_DMA and VFIO Andrew Jones
` (15 preceding siblings ...)
2026-09-25 15:16 ` [PATCH v6 16/16] selftests/vfio: Allow building on RISC-V Andrew Jones
@ 2026-09-28 10:24 ` Andrew Jones
16 siblings, 0 replies; 31+ messages in thread
From: Andrew Jones @ 2026-09-28 10:24 UTC (permalink / raw)
To: linux-riscv, iommu
Cc: linux-kernel, tomasz.jeznach, tjeznach, jgg, jgg, joro, will,
robin.murphy, pjw, palmer, anup, tglx, kevin.tian, fangyu.yu
On Fri, Sep 25, 2026 at 05:16:43PM +0200, Andrew Jones wrote:
> This series adds MSI remapping for IMSIC so a device's MSI target gets
> translated the same way its DMA does, allowing RISC-V to enable IOMMU_DMA
> and paging domains by default.
>
Sashiko only had findings on two patches, and I've replied them. IMHO,
neither warrant a v7.
Thanks,
drew
^ permalink raw reply [flat|nested] 31+ messages in thread