mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
From: Guanghui Feng <guanghuifeng@linux.alibaba.com>
To: jgg@ziepe.ca
Cc: alex@shazbot.org, guanghuifeng@linux.alibaba.com,
	iommu@lists.linux.dev, joro@8bytes.org, kevin.tian@intel.com,
	kvm@vger.kernel.org, linux-kernel@vger.kernel.org,
	robin.murphy@arm.com, will@kernel.org
Subject: [PATCH v3 2/2] iommufd: Add IOMMU_GET_PCI_MMIO_WINDOWS ioctl
Date: Mon,  5 Oct 2026 00:22:13 +0800	[thread overview]
Message-ID: <20261004162213.3787623-3-guanghuifeng@linux.alibaba.com> (raw)
In-Reply-To: <20261004162213.3787623-1-guanghuifeng@linux.alibaba.com>

Add a new per-device query ioctl that reports the PCI host bridge MMIO
windows behind a given device.  This allows userspace (e.g. QEMU) to
learn which address ranges a PCIe switch might misinterpret as
peer-to-peer DMA targets, and voluntarily keep its IOVA allocations away
from them.

The kernel does NOT reserve or enforce these ranges -- it only reports
them.  Userspace can choose to opt into avoiding these windows based on
its own topology knowledge and requirements.  This follows the principle
established by commit cd2c9fcf5c66 ("iommu/dma: Move PCI window region reservation back into dma specific path.")
that PCI window information should not be forced through the IOMMU
reserved-region API onto userspace.

The ioctl uses the standard "user provides array + count, kernel fills
and returns actual count, -EMSGSIZE if too small" pattern established by
IOMMU_IOAS_IOVA_RANGES.  Per-device granularity is chosen because PCI
host bridge windows are a property of the device's bus topology and are
available immediately after VFIO_DEVICE_BIND_IOMMUFD, before any IOAS
attachment.

Signed-off-by: Guanghui Feng <guanghuifeng@linux.alibaba.com>
---
 drivers/iommu/iommufd/device.c          | 64 +++++++++++++++++++++++++
 drivers/iommu/iommufd/iommufd_private.h |  1 +
 drivers/iommu/iommufd/main.c            |  3 ++
 include/uapi/linux/iommufd.h            | 55 +++++++++++++++++++++
 4 files changed, 123 insertions(+)

diff --git a/drivers/iommu/iommufd/device.c b/drivers/iommu/iommufd/device.c
index a664c70a6fe7..74a0642ff4fb 100644
--- a/drivers/iommu/iommufd/device.c
+++ b/drivers/iommu/iommufd/device.c
@@ -1747,3 +1747,67 @@ int iommufd_get_hw_info(struct iommufd_ucmd *ucmd)
 	iommufd_put_object(ucmd->ictx, &idev->obj);
 	return rc;
 }
+
+int iommufd_device_get_pci_mmio_windows(struct iommufd_ucmd *ucmd)
+{
+	struct iommu_pci_mmio_window __user *windows;
+	struct iommu_pci_mmio_windows *cmd = ucmd->cmd;
+	struct iommu_resv_region *resv, *next;
+	struct iommufd_device *idev;
+	LIST_HEAD(resv_windows);
+	u32 max_windows;
+	int rc;
+
+	if (cmd->flags)
+		return -EOPNOTSUPP;
+
+	idev = iommufd_get_device(ucmd, cmd->dev_id);
+	if (IS_ERR(idev))
+		return PTR_ERR(idev);
+
+	if (!dev_is_pci(idev->dev)) {
+		rc = -EOPNOTSUPP;
+		goto out_put;
+	}
+
+	rc = iommu_get_pci_resv_windows(to_pci_dev(idev->dev), &resv_windows);
+	if (rc)
+		goto out_put;
+
+	max_windows = cmd->num_windows;
+	windows = u64_to_user_ptr(cmd->windows);
+	cmd->num_windows = 0;
+	list_for_each_entry(resv, &resv_windows, list) {
+		if (!resv->length)
+			continue;
+
+		if (cmd->num_windows < max_windows) {
+			struct iommu_pci_mmio_window elm = {
+				.start = resv->start,
+				.last = resv->start + resv->length - 1,
+			};
+
+			if (copy_to_user(&windows[cmd->num_windows], &elm,
+					 sizeof(elm))) {
+				rc = -EFAULT;
+				goto out_free;
+			}
+		}
+		cmd->num_windows++;
+	}
+
+	rc = iommufd_ucmd_respond(ucmd, sizeof(*cmd));
+	if (rc)
+		goto out_free;
+	if (cmd->num_windows > max_windows)
+		rc = -EMSGSIZE;
+
+out_free:
+	list_for_each_entry_safe(resv, next, &resv_windows, list) {
+		list_del(&resv->list);
+		kfree(resv);
+	}
+out_put:
+	iommufd_put_object(ucmd->ictx, &idev->obj);
+	return rc;
+}
diff --git a/drivers/iommu/iommufd/iommufd_private.h b/drivers/iommu/iommufd/iommufd_private.h
index eb2e85b27e42..557f205e7836 100644
--- a/drivers/iommu/iommufd/iommufd_private.h
+++ b/drivers/iommu/iommufd/iommufd_private.h
@@ -535,6 +535,7 @@ iommufd_device_get_iommu_dev(struct iommufd_device *idev)
 void iommufd_device_pre_destroy(struct iommufd_object *obj);
 void iommufd_device_destroy(struct iommufd_object *obj);
 int iommufd_get_hw_info(struct iommufd_ucmd *ucmd);
+int iommufd_device_get_pci_mmio_windows(struct iommufd_ucmd *ucmd);
 
 struct device *iommufd_global_device(void);
 
diff --git a/drivers/iommu/iommufd/main.c b/drivers/iommu/iommufd/main.c
index 9a921b153162..0d1be82c1ddd 100644
--- a/drivers/iommu/iommufd/main.c
+++ b/drivers/iommu/iommufd/main.c
@@ -449,6 +449,7 @@ union ucmd_buffer {
 	struct iommu_ioas_map map;
 	struct iommu_ioas_unmap unmap;
 	struct iommu_option option;
+	struct iommu_pci_mmio_windows pci_mmio_windows;
 	struct iommu_vdevice_alloc vdev;
 	struct iommu_veventq_alloc veventq;
 	struct iommu_vfio_ioas vfio_ioas;
@@ -480,6 +481,8 @@ static const struct iommufd_ioctl_op iommufd_ioctl_ops[] = {
 		 struct iommu_fault_alloc, out_fault_fd),
 	IOCTL_OP(IOMMU_GET_HW_INFO, iommufd_get_hw_info, struct iommu_hw_info,
 		 __reserved),
+	IOCTL_OP(IOMMU_GET_PCI_MMIO_WINDOWS, iommufd_device_get_pci_mmio_windows,
+		 struct iommu_pci_mmio_windows, windows),
 	IOCTL_OP(IOMMU_HW_QUEUE_ALLOC, iommufd_hw_queue_alloc_ioctl,
 		 struct iommu_hw_queue_alloc, length),
 	IOCTL_OP(IOMMU_HWPT_ALLOC, iommufd_hwpt_alloc, struct iommu_hwpt_alloc,
diff --git a/include/uapi/linux/iommufd.h b/include/uapi/linux/iommufd.h
index 206fa667c782..5707eb6130ee 100644
--- a/include/uapi/linux/iommufd.h
+++ b/include/uapi/linux/iommufd.h
@@ -58,6 +58,7 @@ enum {
 	IOMMUFD_CMD_VEVENTQ_ALLOC = 0x93,
 	IOMMUFD_CMD_HW_QUEUE_ALLOC = 0x94,
 	IOMMUFD_CMD_IOAS_NOIOMMU_GET_PA = 0x95,
+	IOMMUFD_CMD_GET_PCI_MMIO_WINDOWS = 0x96,
 };
 
 /**
@@ -1390,4 +1391,58 @@ struct iommu_hw_queue_alloc {
 	__aligned_u64 length;
 };
 #define IOMMU_HW_QUEUE_ALLOC _IO(IOMMUFD_TYPE, IOMMUFD_CMD_HW_QUEUE_ALLOC)
+
+/**
+ * struct iommu_pci_mmio_window - a PCI host bridge MMIO window
+ * @start: First PCI bus address of the window
+ * @last: Inclusive last PCI bus address of the window
+ *
+ * A memory window claimed by the PCI host bridge that a device sits
+ * behind.  Addresses are PCI bus addresses (i.e. what appears in the
+ * TLP), which differ from CPU physical addresses on platforms where the
+ * host bridge applies an address translation offset.  For a device
+ * using an identity (1:1) IOVA layout the IOVA equals the bus address,
+ * so these are the IOVAs that a PCIe switch could misinterpret as
+ * peer-to-peer DMA and route to the wrong device instead of memory.
+ */
+struct iommu_pci_mmio_window {
+	__aligned_u64 start;
+	__aligned_u64 last;
+};
+
+/**
+ * struct iommu_pci_mmio_windows - ioctl(IOMMU_GET_PCI_MMIO_WINDOWS)
+ * @size: sizeof(struct iommu_pci_mmio_windows)
+ * @flags: Must be 0
+ * @dev_id: The device bound to iommufd to query
+ * @num_windows: Input/output number of MMIO windows
+ * @windows: Pointer to the output array of struct iommu_pci_mmio_window
+ *
+ * Query the memory windows of the PCI host bridge that @dev_id sits
+ * behind.  This is advisory: the kernel does not reserve these ranges, it
+ * only reports them so userspace can choose to keep its IOVA allocations
+ * away from them and avoid accidental peer-to-peer DMA routing.  The
+ * reported windows are pessimistic (the whole host bridge window, not the
+ * precise BAR ranges that may actually conflict).  If a IOAS has devices
+ * behind multiple host bridges, query each device and take the union.
+ *
+ * On input num_windows is the length of the windows array.  On output it
+ * is the total number of windows.  The ioctl will return -EMSGSIZE and set
+ * num_windows to the required value if num_windows is too small.  In this
+ * case the caller should allocate a larger output array and re-issue the
+ * ioctl.
+ *
+ * Return: 0 on success, -EOPNOTSUPP if @dev_id is not a PCI device,
+ * -ENOENT if @dev_id is invalid, -EMSGSIZE if the output array is too
+ * small.
+ */
+struct iommu_pci_mmio_windows {
+	__u32 size;
+	__u32 flags;
+	__u32 dev_id;
+	__u32 num_windows;
+	__aligned_u64 windows;
+};
+#define IOMMU_GET_PCI_MMIO_WINDOWS \
+	_IO(IOMMUFD_TYPE, IOMMUFD_CMD_GET_PCI_MMIO_WINDOWS)
 #endif
-- 
2.43.7


      parent reply	other threads:[~2026-10-04 16:22 UTC|newest]

Thread overview: 9+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-21  7:02 [PATCH] iommu: Reserve PCI host bridge MMIO windows in group reserved regions Guanghui Feng
2026-09-21 10:39 ` [PATCH v2 0/2] iommu: Reserve PCI host bridge MMIO windows for IOVA Guanghui Feng
2026-09-21 10:39   ` [PATCH v2 1/2] iommu: Reserve PCI host bridge MMIO windows in group reserved regions Guanghui Feng
2026-09-21 10:39   ` [PATCH v2 2/2] iommufd: Reserve PCI host bridge MMIO windows in IOAS " Guanghui Feng
2026-09-21 11:24   ` [PATCH v2 0/2] iommu: Reserve PCI host bridge MMIO windows for IOVA Robin Murphy
2026-09-21 11:55     ` Jason Gunthorpe
2026-10-04 16:22       ` [PATCH v3 0/2] iommu/iommufd: Expose PCI host bridge MMIO windows for opt-in IOVA avoidance Guanghui Feng
2026-10-04 16:22         ` [PATCH v3 1/2] iommu: Add iommu_get_pci_resv_windows() helper Guanghui Feng
2026-10-04 16:22         ` Guanghui Feng [this message]

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20261004162213.3787623-3-guanghuifeng@linux.alibaba.com \
    --to=guanghuifeng@linux.alibaba.com \
    --cc=alex@shazbot.org \
    --cc=iommu@lists.linux.dev \
    --cc=jgg@ziepe.ca \
    --cc=joro@8bytes.org \
    --cc=kevin.tian@intel.com \
    --cc=kvm@vger.kernel.org \
    --cc=linux-kernel@vger.kernel.org \
    --cc=robin.murphy@arm.com \
    --cc=will@kernel.org \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®