From: Guanghui Feng <guanghuifeng@linux.alibaba.com>
To: jgg@ziepe.ca
Cc: alex@shazbot.org, guanghuifeng@linux.alibaba.com,
iommu@lists.linux.dev, joro@8bytes.org, kevin.tian@intel.com,
kvm@vger.kernel.org, linux-kernel@vger.kernel.org,
robin.murphy@arm.com, will@kernel.org
Subject: [PATCH v3 2/2] iommufd: Add IOMMU_GET_PCI_MMIO_WINDOWS ioctl
Date: Mon, 5 Oct 2026 00:22:13 +0800 [thread overview]
Message-ID: <20261004162213.3787623-3-guanghuifeng@linux.alibaba.com> (raw)
In-Reply-To: <20261004162213.3787623-1-guanghuifeng@linux.alibaba.com>
Add a new per-device query ioctl that reports the PCI host bridge MMIO
windows behind a given device. This allows userspace (e.g. QEMU) to
learn which address ranges a PCIe switch might misinterpret as
peer-to-peer DMA targets, and voluntarily keep its IOVA allocations away
from them.
The kernel does NOT reserve or enforce these ranges -- it only reports
them. Userspace can choose to opt into avoiding these windows based on
its own topology knowledge and requirements. This follows the principle
established by commit cd2c9fcf5c66 ("iommu/dma: Move PCI window region reservation back into dma specific path.")
that PCI window information should not be forced through the IOMMU
reserved-region API onto userspace.
The ioctl uses the standard "user provides array + count, kernel fills
and returns actual count, -EMSGSIZE if too small" pattern established by
IOMMU_IOAS_IOVA_RANGES. Per-device granularity is chosen because PCI
host bridge windows are a property of the device's bus topology and are
available immediately after VFIO_DEVICE_BIND_IOMMUFD, before any IOAS
attachment.
Signed-off-by: Guanghui Feng <guanghuifeng@linux.alibaba.com>
---
drivers/iommu/iommufd/device.c | 64 +++++++++++++++++++++++++
drivers/iommu/iommufd/iommufd_private.h | 1 +
drivers/iommu/iommufd/main.c | 3 ++
include/uapi/linux/iommufd.h | 55 +++++++++++++++++++++
4 files changed, 123 insertions(+)
diff --git a/drivers/iommu/iommufd/device.c b/drivers/iommu/iommufd/device.c
index a664c70a6fe7..74a0642ff4fb 100644
--- a/drivers/iommu/iommufd/device.c
+++ b/drivers/iommu/iommufd/device.c
@@ -1747,3 +1747,67 @@ int iommufd_get_hw_info(struct iommufd_ucmd *ucmd)
iommufd_put_object(ucmd->ictx, &idev->obj);
return rc;
}
+
+int iommufd_device_get_pci_mmio_windows(struct iommufd_ucmd *ucmd)
+{
+ struct iommu_pci_mmio_window __user *windows;
+ struct iommu_pci_mmio_windows *cmd = ucmd->cmd;
+ struct iommu_resv_region *resv, *next;
+ struct iommufd_device *idev;
+ LIST_HEAD(resv_windows);
+ u32 max_windows;
+ int rc;
+
+ if (cmd->flags)
+ return -EOPNOTSUPP;
+
+ idev = iommufd_get_device(ucmd, cmd->dev_id);
+ if (IS_ERR(idev))
+ return PTR_ERR(idev);
+
+ if (!dev_is_pci(idev->dev)) {
+ rc = -EOPNOTSUPP;
+ goto out_put;
+ }
+
+ rc = iommu_get_pci_resv_windows(to_pci_dev(idev->dev), &resv_windows);
+ if (rc)
+ goto out_put;
+
+ max_windows = cmd->num_windows;
+ windows = u64_to_user_ptr(cmd->windows);
+ cmd->num_windows = 0;
+ list_for_each_entry(resv, &resv_windows, list) {
+ if (!resv->length)
+ continue;
+
+ if (cmd->num_windows < max_windows) {
+ struct iommu_pci_mmio_window elm = {
+ .start = resv->start,
+ .last = resv->start + resv->length - 1,
+ };
+
+ if (copy_to_user(&windows[cmd->num_windows], &elm,
+ sizeof(elm))) {
+ rc = -EFAULT;
+ goto out_free;
+ }
+ }
+ cmd->num_windows++;
+ }
+
+ rc = iommufd_ucmd_respond(ucmd, sizeof(*cmd));
+ if (rc)
+ goto out_free;
+ if (cmd->num_windows > max_windows)
+ rc = -EMSGSIZE;
+
+out_free:
+ list_for_each_entry_safe(resv, next, &resv_windows, list) {
+ list_del(&resv->list);
+ kfree(resv);
+ }
+out_put:
+ iommufd_put_object(ucmd->ictx, &idev->obj);
+ return rc;
+}
diff --git a/drivers/iommu/iommufd/iommufd_private.h b/drivers/iommu/iommufd/iommufd_private.h
index eb2e85b27e42..557f205e7836 100644
--- a/drivers/iommu/iommufd/iommufd_private.h
+++ b/drivers/iommu/iommufd/iommufd_private.h
@@ -535,6 +535,7 @@ iommufd_device_get_iommu_dev(struct iommufd_device *idev)
void iommufd_device_pre_destroy(struct iommufd_object *obj);
void iommufd_device_destroy(struct iommufd_object *obj);
int iommufd_get_hw_info(struct iommufd_ucmd *ucmd);
+int iommufd_device_get_pci_mmio_windows(struct iommufd_ucmd *ucmd);
struct device *iommufd_global_device(void);
diff --git a/drivers/iommu/iommufd/main.c b/drivers/iommu/iommufd/main.c
index 9a921b153162..0d1be82c1ddd 100644
--- a/drivers/iommu/iommufd/main.c
+++ b/drivers/iommu/iommufd/main.c
@@ -449,6 +449,7 @@ union ucmd_buffer {
struct iommu_ioas_map map;
struct iommu_ioas_unmap unmap;
struct iommu_option option;
+ struct iommu_pci_mmio_windows pci_mmio_windows;
struct iommu_vdevice_alloc vdev;
struct iommu_veventq_alloc veventq;
struct iommu_vfio_ioas vfio_ioas;
@@ -480,6 +481,8 @@ static const struct iommufd_ioctl_op iommufd_ioctl_ops[] = {
struct iommu_fault_alloc, out_fault_fd),
IOCTL_OP(IOMMU_GET_HW_INFO, iommufd_get_hw_info, struct iommu_hw_info,
__reserved),
+ IOCTL_OP(IOMMU_GET_PCI_MMIO_WINDOWS, iommufd_device_get_pci_mmio_windows,
+ struct iommu_pci_mmio_windows, windows),
IOCTL_OP(IOMMU_HW_QUEUE_ALLOC, iommufd_hw_queue_alloc_ioctl,
struct iommu_hw_queue_alloc, length),
IOCTL_OP(IOMMU_HWPT_ALLOC, iommufd_hwpt_alloc, struct iommu_hwpt_alloc,
diff --git a/include/uapi/linux/iommufd.h b/include/uapi/linux/iommufd.h
index 206fa667c782..5707eb6130ee 100644
--- a/include/uapi/linux/iommufd.h
+++ b/include/uapi/linux/iommufd.h
@@ -58,6 +58,7 @@ enum {
IOMMUFD_CMD_VEVENTQ_ALLOC = 0x93,
IOMMUFD_CMD_HW_QUEUE_ALLOC = 0x94,
IOMMUFD_CMD_IOAS_NOIOMMU_GET_PA = 0x95,
+ IOMMUFD_CMD_GET_PCI_MMIO_WINDOWS = 0x96,
};
/**
@@ -1390,4 +1391,58 @@ struct iommu_hw_queue_alloc {
__aligned_u64 length;
};
#define IOMMU_HW_QUEUE_ALLOC _IO(IOMMUFD_TYPE, IOMMUFD_CMD_HW_QUEUE_ALLOC)
+
+/**
+ * struct iommu_pci_mmio_window - a PCI host bridge MMIO window
+ * @start: First PCI bus address of the window
+ * @last: Inclusive last PCI bus address of the window
+ *
+ * A memory window claimed by the PCI host bridge that a device sits
+ * behind. Addresses are PCI bus addresses (i.e. what appears in the
+ * TLP), which differ from CPU physical addresses on platforms where the
+ * host bridge applies an address translation offset. For a device
+ * using an identity (1:1) IOVA layout the IOVA equals the bus address,
+ * so these are the IOVAs that a PCIe switch could misinterpret as
+ * peer-to-peer DMA and route to the wrong device instead of memory.
+ */
+struct iommu_pci_mmio_window {
+ __aligned_u64 start;
+ __aligned_u64 last;
+};
+
+/**
+ * struct iommu_pci_mmio_windows - ioctl(IOMMU_GET_PCI_MMIO_WINDOWS)
+ * @size: sizeof(struct iommu_pci_mmio_windows)
+ * @flags: Must be 0
+ * @dev_id: The device bound to iommufd to query
+ * @num_windows: Input/output number of MMIO windows
+ * @windows: Pointer to the output array of struct iommu_pci_mmio_window
+ *
+ * Query the memory windows of the PCI host bridge that @dev_id sits
+ * behind. This is advisory: the kernel does not reserve these ranges, it
+ * only reports them so userspace can choose to keep its IOVA allocations
+ * away from them and avoid accidental peer-to-peer DMA routing. The
+ * reported windows are pessimistic (the whole host bridge window, not the
+ * precise BAR ranges that may actually conflict). If a IOAS has devices
+ * behind multiple host bridges, query each device and take the union.
+ *
+ * On input num_windows is the length of the windows array. On output it
+ * is the total number of windows. The ioctl will return -EMSGSIZE and set
+ * num_windows to the required value if num_windows is too small. In this
+ * case the caller should allocate a larger output array and re-issue the
+ * ioctl.
+ *
+ * Return: 0 on success, -EOPNOTSUPP if @dev_id is not a PCI device,
+ * -ENOENT if @dev_id is invalid, -EMSGSIZE if the output array is too
+ * small.
+ */
+struct iommu_pci_mmio_windows {
+ __u32 size;
+ __u32 flags;
+ __u32 dev_id;
+ __u32 num_windows;
+ __aligned_u64 windows;
+};
+#define IOMMU_GET_PCI_MMIO_WINDOWS \
+ _IO(IOMMUFD_TYPE, IOMMUFD_CMD_GET_PCI_MMIO_WINDOWS)
#endif
--
2.43.7
prev parent reply other threads:[~2026-10-04 16:22 UTC|newest]
Thread overview: 9+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-21 7:02 [PATCH] iommu: Reserve PCI host bridge MMIO windows in group reserved regions Guanghui Feng
2026-09-21 10:39 ` [PATCH v2 0/2] iommu: Reserve PCI host bridge MMIO windows for IOVA Guanghui Feng
2026-09-21 10:39 ` [PATCH v2 1/2] iommu: Reserve PCI host bridge MMIO windows in group reserved regions Guanghui Feng
2026-09-21 10:39 ` [PATCH v2 2/2] iommufd: Reserve PCI host bridge MMIO windows in IOAS " Guanghui Feng
2026-09-21 11:24 ` [PATCH v2 0/2] iommu: Reserve PCI host bridge MMIO windows for IOVA Robin Murphy
2026-09-21 11:55 ` Jason Gunthorpe
2026-10-04 16:22 ` [PATCH v3 0/2] iommu/iommufd: Expose PCI host bridge MMIO windows for opt-in IOVA avoidance Guanghui Feng
2026-10-04 16:22 ` [PATCH v3 1/2] iommu: Add iommu_get_pci_resv_windows() helper Guanghui Feng
2026-10-04 16:22 ` Guanghui Feng [this message]
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20261004162213.3787623-3-guanghuifeng@linux.alibaba.com \
--to=guanghuifeng@linux.alibaba.com \
--cc=alex@shazbot.org \
--cc=iommu@lists.linux.dev \
--cc=jgg@ziepe.ca \
--cc=joro@8bytes.org \
--cc=kevin.tian@intel.com \
--cc=kvm@vger.kernel.org \
--cc=linux-kernel@vger.kernel.org \
--cc=robin.murphy@arm.com \
--cc=will@kernel.org \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox
all inboxes | Powered by JetHome®