[PATCH v3 2/2] iommufd: Add IOMMU_GET_PCI_MMIO_WINDOWS ioctl

From: Guanghui Feng

Date: Sun Oct 04 2026 - 12:23:11 EST


Add a new per-device query ioctl that reports the PCI host bridge MMIO
windows behind a given device. This allows userspace (e.g. QEMU) to
learn which address ranges a PCIe switch might misinterpret as
peer-to-peer DMA targets, and voluntarily keep its IOVA allocations away
from them.

The kernel does NOT reserve or enforce these ranges -- it only reports
them. Userspace can choose to opt into avoiding these windows based on
its own topology knowledge and requirements. This follows the principle
established by commit cd2c9fcf5c66 ("iommu/dma: Move PCI window region reservation back into dma specific path.")
that PCI window information should not be forced through the IOMMU
reserved-region API onto userspace.

The ioctl uses the standard "user provides array + count, kernel fills
and returns actual count, -EMSGSIZE if too small" pattern established by
IOMMU_IOAS_IOVA_RANGES. Per-device granularity is chosen because PCI
host bridge windows are a property of the device's bus topology and are
available immediately after VFIO_DEVICE_BIND_IOMMUFD, before any IOAS
attachment.

Signed-off-by: Guanghui Feng <guanghuifeng@xxxxxxxxxxxxxxxxx>
---
drivers/iommu/iommufd/device.c | 64 +++++++++++++++++++++++++
drivers/iommu/iommufd/iommufd_private.h | 1 +
drivers/iommu/iommufd/main.c | 3 ++
include/uapi/linux/iommufd.h | 55 +++++++++++++++++++++
4 files changed, 123 insertions(+)

diff --git a/drivers/iommu/iommufd/device.c b/drivers/iommu/iommufd/device.c
index a664c70a6fe7..74a0642ff4fb 100644
--- a/drivers/iommu/iommufd/device.c
+++ b/drivers/iommu/iommufd/device.c
@@ -1747,3 +1747,67 @@ int iommufd_get_hw_info(struct iommufd_ucmd *ucmd)
iommufd_put_object(ucmd->ictx, &idev->obj);
return rc;
}
+
+int iommufd_device_get_pci_mmio_windows(struct iommufd_ucmd *ucmd)
+{
+ struct iommu_pci_mmio_window __user *windows;
+ struct iommu_pci_mmio_windows *cmd = ucmd->cmd;
+ struct iommu_resv_region *resv, *next;
+ struct iommufd_device *idev;
+ LIST_HEAD(resv_windows);
+ u32 max_windows;
+ int rc;
+
+ if (cmd->flags)
+ return -EOPNOTSUPP;
+
+ idev = iommufd_get_device(ucmd, cmd->dev_id);
+ if (IS_ERR(idev))
+ return PTR_ERR(idev);
+
+ if (!dev_is_pci(idev->dev)) {
+ rc = -EOPNOTSUPP;
+ goto out_put;
+ }
+
+ rc = iommu_get_pci_resv_windows(to_pci_dev(idev->dev), &resv_windows);
+ if (rc)
+ goto out_put;
+
+ max_windows = cmd->num_windows;
+ windows = u64_to_user_ptr(cmd->windows);
+ cmd->num_windows = 0;
+ list_for_each_entry(resv, &resv_windows, list) {
+ if (!resv->length)
+ continue;
+
+ if (cmd->num_windows < max_windows) {
+ struct iommu_pci_mmio_window elm = {
+ .start = resv->start,
+ .last = resv->start + resv->length - 1,
+ };
+
+ if (copy_to_user(&windows[cmd->num_windows], &elm,
+ sizeof(elm))) {
+ rc = -EFAULT;
+ goto out_free;
+ }
+ }
+ cmd->num_windows++;
+ }
+
+ rc = iommufd_ucmd_respond(ucmd, sizeof(*cmd));
+ if (rc)
+ goto out_free;
+ if (cmd->num_windows > max_windows)
+ rc = -EMSGSIZE;
+
+out_free:
+ list_for_each_entry_safe(resv, next, &resv_windows, list) {
+ list_del(&resv->list);
+ kfree(resv);
+ }
+out_put:
+ iommufd_put_object(ucmd->ictx, &idev->obj);
+ return rc;
+}
diff --git a/drivers/iommu/iommufd/iommufd_private.h b/drivers/iommu/iommufd/iommufd_private.h
index eb2e85b27e42..557f205e7836 100644
--- a/drivers/iommu/iommufd/iommufd_private.h
+++ b/drivers/iommu/iommufd/iommufd_private.h
@@ -535,6 +535,7 @@ iommufd_device_get_iommu_dev(struct iommufd_device *idev)
void iommufd_device_pre_destroy(struct iommufd_object *obj);
void iommufd_device_destroy(struct iommufd_object *obj);
int iommufd_get_hw_info(struct iommufd_ucmd *ucmd);
+int iommufd_device_get_pci_mmio_windows(struct iommufd_ucmd *ucmd);

struct device *iommufd_global_device(void);

diff --git a/drivers/iommu/iommufd/main.c b/drivers/iommu/iommufd/main.c
index 9a921b153162..0d1be82c1ddd 100644
--- a/drivers/iommu/iommufd/main.c
+++ b/drivers/iommu/iommufd/main.c
@@ -449,6 +449,7 @@ union ucmd_buffer {
struct iommu_ioas_map map;
struct iommu_ioas_unmap unmap;
struct iommu_option option;
+ struct iommu_pci_mmio_windows pci_mmio_windows;
struct iommu_vdevice_alloc vdev;
struct iommu_veventq_alloc veventq;
struct iommu_vfio_ioas vfio_ioas;
@@ -480,6 +481,8 @@ static const struct iommufd_ioctl_op iommufd_ioctl_ops[] = {
struct iommu_fault_alloc, out_fault_fd),
IOCTL_OP(IOMMU_GET_HW_INFO, iommufd_get_hw_info, struct iommu_hw_info,
__reserved),
+ IOCTL_OP(IOMMU_GET_PCI_MMIO_WINDOWS, iommufd_device_get_pci_mmio_windows,
+ struct iommu_pci_mmio_windows, windows),
IOCTL_OP(IOMMU_HW_QUEUE_ALLOC, iommufd_hw_queue_alloc_ioctl,
struct iommu_hw_queue_alloc, length),
IOCTL_OP(IOMMU_HWPT_ALLOC, iommufd_hwpt_alloc, struct iommu_hwpt_alloc,
diff --git a/include/uapi/linux/iommufd.h b/include/uapi/linux/iommufd.h
index 206fa667c782..5707eb6130ee 100644
--- a/include/uapi/linux/iommufd.h
+++ b/include/uapi/linux/iommufd.h
@@ -58,6 +58,7 @@ enum {
IOMMUFD_CMD_VEVENTQ_ALLOC = 0x93,
IOMMUFD_CMD_HW_QUEUE_ALLOC = 0x94,
IOMMUFD_CMD_IOAS_NOIOMMU_GET_PA = 0x95,
+ IOMMUFD_CMD_GET_PCI_MMIO_WINDOWS = 0x96,
};

/**
@@ -1390,4 +1391,58 @@ struct iommu_hw_queue_alloc {
__aligned_u64 length;
};
#define IOMMU_HW_QUEUE_ALLOC _IO(IOMMUFD_TYPE, IOMMUFD_CMD_HW_QUEUE_ALLOC)
+
+/**
+ * struct iommu_pci_mmio_window - a PCI host bridge MMIO window
+ * @start: First PCI bus address of the window
+ * @last: Inclusive last PCI bus address of the window
+ *
+ * A memory window claimed by the PCI host bridge that a device sits
+ * behind. Addresses are PCI bus addresses (i.e. what appears in the
+ * TLP), which differ from CPU physical addresses on platforms where the
+ * host bridge applies an address translation offset. For a device
+ * using an identity (1:1) IOVA layout the IOVA equals the bus address,
+ * so these are the IOVAs that a PCIe switch could misinterpret as
+ * peer-to-peer DMA and route to the wrong device instead of memory.
+ */
+struct iommu_pci_mmio_window {
+ __aligned_u64 start;
+ __aligned_u64 last;
+};
+
+/**
+ * struct iommu_pci_mmio_windows - ioctl(IOMMU_GET_PCI_MMIO_WINDOWS)
+ * @size: sizeof(struct iommu_pci_mmio_windows)
+ * @flags: Must be 0
+ * @dev_id: The device bound to iommufd to query
+ * @num_windows: Input/output number of MMIO windows
+ * @windows: Pointer to the output array of struct iommu_pci_mmio_window
+ *
+ * Query the memory windows of the PCI host bridge that @dev_id sits
+ * behind. This is advisory: the kernel does not reserve these ranges, it
+ * only reports them so userspace can choose to keep its IOVA allocations
+ * away from them and avoid accidental peer-to-peer DMA routing. The
+ * reported windows are pessimistic (the whole host bridge window, not the
+ * precise BAR ranges that may actually conflict). If a IOAS has devices
+ * behind multiple host bridges, query each device and take the union.
+ *
+ * On input num_windows is the length of the windows array. On output it
+ * is the total number of windows. The ioctl will return -EMSGSIZE and set
+ * num_windows to the required value if num_windows is too small. In this
+ * case the caller should allocate a larger output array and re-issue the
+ * ioctl.
+ *
+ * Return: 0 on success, -EOPNOTSUPP if @dev_id is not a PCI device,
+ * -ENOENT if @dev_id is invalid, -EMSGSIZE if the output array is too
+ * small.
+ */
+struct iommu_pci_mmio_windows {
+ __u32 size;
+ __u32 flags;
+ __u32 dev_id;
+ __u32 num_windows;
+ __aligned_u64 windows;
+};
+#define IOMMU_GET_PCI_MMIO_WINDOWS \
+ _IO(IOMMUFD_TYPE, IOMMUFD_CMD_GET_PCI_MMIO_WINDOWS)
#endif
--
2.43.7