From: <mhonap@nvidia.com>
To: <alex@shazbot.org>, <jgg@ziepe.ca>, <ankita@nvidia.com>,
<jic23@kernel.org>, <dave.jiang@intel.com>,
<alejandro.lucero-palau@amd.com>, <smadhavan@nvidia.com>,
<corbet@lwn.net>, <skhan@linuxfoundation.org>,
<dave@stgolabs.net>, <alison.schofield@intel.com>,
<vishal.l.verma@intel.com>, <iweiny@kernel.org>,
<ming.li@zohomail.com>, <yishaih@nvidia.com>,
<skolothumtho@nvidia.com>, <kevin.tian@intel.com>,
<bhelgaas@google.com>, <dmatlack@google.com>, <kees@kernel.org>,
<gustavoars@kernel.org>
Cc: <cjia@nvidia.com>, <kjaju@nvidia.com>, <vsethi@nvidia.com>,
<zhiw@nvidia.com>, <mhonap@nvidia.com>,
<linux-doc@vger.kernel.org>, <linux-kernel@vger.kernel.org>,
<kvm@vger.kernel.org>, <linux-cxl@vger.kernel.org>,
<linux-pci@vger.kernel.org>, <linux-kselftest@vger.kernel.org>,
<linux-hardening@vger.kernel.org>
Subject: [PATCH v5 09/27] vfio/pci: Add a generic excluded-range list
Date: Thu, 17 Sep 2026 00:05:22 +0530 [thread overview]
Message-ID: <20260916183540.3813685-10-mhonap@nvidia.com> (raw)
In-Reply-To: <20260916183540.3813685-1-mhonap@nvidia.com>
From: Manish Honap <mhonap@nvidia.com>
The MSI-X table is virtualized in vfio_pci_bar_rw() by an open-coded
x_start/x_end window that fills reads with -1 and drops writes. Now that
a generic excluded-range list expresses the same fill/drop behavior,
register the MSI-X table as a read and write excluded range instead of
special-casing it in the read/write path.
Add the range when the MSI-X capability is parsed in
vfio_pci_core_enable() and clear the list in vfio_pci_core_disable()
alongside the config teardown. The read/write path now relies solely on
vfio_pci_bar_find_exclusion(), so MSI-X and a provider's (e.g. vfio-cxl)
trapped registers share one mechanism.
No functional change.
Assisted-by: LLM
Signed-off-by: Manish Honap <mhonap@nvidia.com>
---
drivers/vfio/pci/vfio_pci_core.c | 209 +++++++++++++++++++++++++++++++
drivers/vfio/pci/vfio_pci_priv.h | 9 ++
drivers/vfio/pci/vfio_pci_rdwr.c | 8 ++
include/linux/vfio_pci_core.h | 14 +++
4 files changed, 240 insertions(+)
diff --git a/drivers/vfio/pci/vfio_pci_core.c b/drivers/vfio/pci/vfio_pci_core.c
index 9a75c30b67e2..9e4fa5d088a4 100644
--- a/drivers/vfio/pci/vfio_pci_core.c
+++ b/drivers/vfio/pci/vfio_pci_core.c
@@ -24,6 +24,7 @@
#include <linux/pci.h>
#include <linux/pm_runtime.h>
#include <linux/slab.h>
+#include <linux/sort.h>
#include <linux/types.h>
#include <linux/uaccess.h>
#include <linux/vgaarb.h>
@@ -1012,6 +1013,204 @@ static int msix_mmappable_cap(struct vfio_pci_core_device *vdev,
return vfio_info_add_capability(caps, &header, sizeof(header));
}
+struct vfio_pci_excluded_range {
+ struct list_head entry;
+ int bar;
+ u64 start;
+ u64 size;
+ u32 flags;
+};
+
+int vfio_pci_core_add_excluded_range(struct vfio_pci_core_device *vdev, int bar,
+ u64 start, u64 size, u32 flags)
+{
+ struct vfio_pci_excluded_range *range;
+
+ range = kzalloc_obj(*range);
+ if (!range)
+ return -ENOMEM;
+
+ range->bar = bar;
+ range->start = start;
+ range->size = size;
+ range->flags = flags;
+ list_add_tail(&range->entry, &vdev->excluded_ranges);
+
+ return 0;
+}
+EXPORT_SYMBOL_GPL(vfio_pci_core_add_excluded_range);
+
+static void vfio_pci_free_excluded_ranges(struct vfio_pci_core_device *vdev)
+{
+ struct vfio_pci_excluded_range *range, *tmp;
+
+ list_for_each_entry_safe(range, tmp, &vdev->excluded_ranges, entry) {
+ list_del(&range->entry);
+ kfree(range);
+ }
+}
+
+bool vfio_pci_bar_find_exclusion(struct vfio_pci_core_device *vdev, int bar,
+ loff_t pos, size_t count, bool iswrite,
+ size_t *x_start, size_t *x_end)
+{
+ u32 want = iswrite ? VFIO_PCI_EXCLUDE_WRITE : VFIO_PCI_EXCLUDE_READ;
+ struct vfio_pci_excluded_range *range;
+ bool found = false;
+
+ list_for_each_entry(range, &vdev->excluded_ranges, entry) {
+ if (range->bar != bar || !(range->flags & want))
+ continue;
+ if (pos < range->start + range->size &&
+ pos + count > range->start) {
+ /*
+ * A BAR can carry more than one excluded window (e.g.
+ * the MSI-X table and a CXL HDM decoder block). Return
+ * the overlapping window with the lowest start so the
+ * caller can walk them in order.
+ */
+ if (!found || range->start < *x_start) {
+ *x_start = range->start;
+ *x_end = range->start + range->size;
+ found = true;
+ }
+ }
+ }
+
+ return found;
+}
+
+/* True when [start, start + len) on @bar overlaps an mmap-excluded range. */
+static bool vfio_pci_bar_mmap_excluded(struct vfio_pci_core_device *vdev,
+ int bar, u64 start, u64 len)
+{
+ struct vfio_pci_excluded_range *range;
+
+ list_for_each_entry(range, &vdev->excluded_ranges, entry) {
+ if (range->bar != bar ||
+ !(range->flags & VFIO_PCI_EXCLUDE_MMAP))
+ continue;
+ if (start < range->start + range->size &&
+ start + len > range->start)
+ return true;
+ }
+
+ return false;
+}
+
+/* A page-aligned mmap hole, derived from an mmap-excluded range. */
+struct vfio_pci_mmap_hole {
+ u64 start;
+ u64 end;
+};
+
+static int vfio_pci_mmap_hole_cmp(const void *a, const void *b)
+{
+ const struct vfio_pci_mmap_hole *x = a, *y = b;
+
+ if (x->start < y->start)
+ return -1;
+ return x->start > y->start;
+}
+
+/*
+ * Advertise the BAR as mmappable minus every page-aligned mmap-excluded hole.
+ * A BAR can carry several holes at unrelated offsets (for example an MSI-X
+ * table and one or more trapped CXL component sub-blocks, which the CXL spec
+ * locates by pointer, not at fixed offsets).
+ * Collect the holes, page-align and sort them, coalesce any that overlap or
+ * touch, and advertise the gaps.
+ */
+static int vfio_pci_excluded_sparse_cap(struct vfio_pci_core_device *vdev,
+ int index, struct vfio_info_cap *caps)
+{
+ u64 bar_len = pci_resource_len(vdev->pdev, index);
+ struct vfio_region_info_cap_sparse_mmap *sparse;
+ struct vfio_pci_excluded_range *range;
+ struct vfio_pci_mmap_hole *holes;
+ int nr_holes = 0, nr_areas = 0, i, j;
+ size_t size;
+ u64 pos;
+ int ret;
+
+ list_for_each_entry(range, &vdev->excluded_ranges, entry)
+ if (range->bar == index &&
+ (range->flags & VFIO_PCI_EXCLUDE_MMAP))
+ nr_holes++;
+
+ if (!nr_holes)
+ return 0;
+
+ holes = kmalloc_array(nr_holes, sizeof(*holes), GFP_KERNEL);
+ if (!holes)
+ return -ENOMEM;
+
+ /*
+ * mmap is page granular, so each hole rounds out to the page boundaries
+ * enclosing its excluded sub-range. The byte-granular exclusion still
+ * governs the fault and read/write paths; only the advertised mmap areas
+ * round to whole pages.
+ */
+ i = 0;
+ list_for_each_entry(range, &vdev->excluded_ranges, entry) {
+ if (range->bar != index ||
+ !(range->flags & VFIO_PCI_EXCLUDE_MMAP))
+ continue;
+ holes[i].start = ALIGN_DOWN(range->start, PAGE_SIZE);
+ holes[i].end = ALIGN(range->start + range->size, PAGE_SIZE);
+ i++;
+ }
+
+ sort(holes, nr_holes, sizeof(*holes), vfio_pci_mmap_hole_cmp, NULL);
+
+ /* Coalesce holes that overlap or touch after page alignment. */
+ for (i = 0, j = 0; i < nr_holes; i++) {
+ if (j && holes[i].start <= holes[j - 1].end)
+ holes[j - 1].end = max(holes[j - 1].end, holes[i].end);
+ else
+ holes[j++] = holes[i];
+ }
+ nr_holes = j;
+
+ /* One mmappable area per gap: before, between, and after the holes. */
+ for (i = 0, pos = 0; i < nr_holes; i++) {
+ if (holes[i].start > pos)
+ nr_areas++;
+ pos = holes[i].end;
+ }
+ if (pos < bar_len)
+ nr_areas++;
+
+ size = struct_size(sparse, areas, nr_areas);
+ sparse = kzalloc(size, GFP_KERNEL);
+ if (!sparse) {
+ kfree(holes);
+ return -ENOMEM;
+ }
+
+ sparse->header.id = VFIO_REGION_INFO_CAP_SPARSE_MMAP;
+ sparse->header.version = 1;
+ sparse->nr_areas = nr_areas;
+
+ for (i = 0, j = 0, pos = 0; i < nr_holes; i++) {
+ if (holes[i].start > pos) {
+ sparse->areas[j].offset = pos;
+ sparse->areas[j].size = holes[i].start - pos;
+ j++;
+ }
+ pos = holes[i].end;
+ }
+ if (pos < bar_len) {
+ sparse->areas[j].offset = pos;
+ sparse->areas[j].size = bar_len - pos;
+ }
+
+ kfree(holes);
+ ret = vfio_info_add_capability(caps, &sparse->header, size);
+ kfree(sparse);
+ return ret;
+}
+
int vfio_pci_core_register_dev_region(struct vfio_pci_core_device *vdev,
unsigned int type, unsigned int subtype,
const struct vfio_pci_regops *ops,
@@ -1160,6 +1359,10 @@ int vfio_pci_ioctl_get_region_info(struct vfio_device *core_vdev,
if (ret)
return ret;
}
+ ret = vfio_pci_excluded_sparse_cap(vdev, info->index,
+ caps);
+ if (ret)
+ return ret;
}
break;
@@ -1853,6 +2056,10 @@ int vfio_pci_core_mmap(struct vfio_device *core_vdev, struct vm_area_struct *vma
if (req_start + req_len > phys_len)
return -EINVAL;
+ /* An excluded sub-range is reachable only through its trap, not mmap. */
+ if (vfio_pci_bar_mmap_excluded(vdev, index, req_start, req_len))
+ return -EINVAL;
+
/*
* Ensure the BAR resource region is reserved for use.
*/
@@ -2317,6 +2524,7 @@ int vfio_pci_core_init_dev(struct vfio_device *core_vdev)
INIT_LIST_HEAD(&vdev->dmabufs);
init_rwsem(&vdev->memory_lock);
xa_init(&vdev->ctx);
+ INIT_LIST_HEAD(&vdev->excluded_ranges);
ret = vfio_pci_core_cxl_init(vdev);
if (ret)
@@ -2332,6 +2540,7 @@ void vfio_pci_core_release_dev(struct vfio_device *core_vdev)
container_of(core_vdev, struct vfio_pci_core_device, vdev);
vfio_pci_core_cxl_release(vdev);
+ vfio_pci_free_excluded_ranges(vdev);
mutex_destroy(&vdev->igate);
mutex_destroy(&vdev->ioeventfds_lock);
diff --git a/drivers/vfio/pci/vfio_pci_priv.h b/drivers/vfio/pci/vfio_pci_priv.h
index 4e7162234a2e..c268c99aea82 100644
--- a/drivers/vfio/pci/vfio_pci_priv.h
+++ b/drivers/vfio/pci/vfio_pci_priv.h
@@ -44,6 +44,15 @@ ssize_t vfio_pci_config_rw_single(struct vfio_pci_core_device *vdev,
ssize_t vfio_pci_bar_rw(struct vfio_pci_core_device *vdev, char __user *buf,
size_t count, loff_t *ppos, bool iswrite);
+/*
+ * If a read (or write) to [pos, pos + count) on @bar overlaps an excluded
+ * range, report the byte window do_io_rw() should fill with -1 (or drop) and
+ * return true. A single access spans at most one such window.
+ */
+bool vfio_pci_bar_find_exclusion(struct vfio_pci_core_device *vdev, int bar,
+ loff_t pos, size_t count, bool iswrite,
+ size_t *x_start, size_t *x_end);
+
#ifdef CONFIG_VFIO_PCI_VGA
ssize_t vfio_pci_vga_rw(struct vfio_pci_core_device *vdev, char __user *buf,
size_t count, loff_t *ppos, bool iswrite);
diff --git a/drivers/vfio/pci/vfio_pci_rdwr.c b/drivers/vfio/pci/vfio_pci_rdwr.c
index 7f14dd46de17..48da1cb08296 100644
--- a/drivers/vfio/pci/vfio_pci_rdwr.c
+++ b/drivers/vfio/pci/vfio_pci_rdwr.c
@@ -261,6 +261,14 @@ ssize_t vfio_pci_bar_rw(struct vfio_pci_core_device *vdev, char __user *buf,
x_end = vdev->msix_offset + vdev->msix_size;
}
+ /*
+ * A provider-excluded sub-range is filled with -1 on read and dropped on
+ * write for the same reason: the guest reaches it only through the trap.
+ * An access spans at most one exclusion window.
+ */
+ vfio_pci_bar_find_exclusion(vdev, bar, pos, count, iswrite,
+ &x_start, &x_end);
+
done = vfio_pci_core_do_io_rw(vdev, res->flags & IORESOURCE_MEM, io, buf, pos,
count, x_start, x_end, iswrite, max_width);
diff --git a/include/linux/vfio_pci_core.h b/include/linux/vfio_pci_core.h
index 7f3a2bcb5830..92e3db116068 100644
--- a/include/linux/vfio_pci_core.h
+++ b/include/linux/vfio_pci_core.h
@@ -162,6 +162,7 @@ struct vfio_pci_core_device {
struct notifier_block nb;
struct rw_semaphore memory_lock;
struct list_head dmabufs;
+ struct list_head excluded_ranges;
};
enum vfio_pci_io_width {
@@ -172,6 +173,19 @@ enum vfio_pci_io_width {
};
/* Will be exported for vfio pci drivers usage */
+/*
+ * A provider can keep a BAR sub-range off the direct guest path, reached only
+ * through its own trap. The flags select which paths are excluded: mmap, and
+ * region read and write (an excluded read fills -1, an excluded write is
+ * dropped).
+ */
+#define VFIO_PCI_EXCLUDE_MMAP BIT(0)
+#define VFIO_PCI_EXCLUDE_READ BIT(1)
+#define VFIO_PCI_EXCLUDE_WRITE BIT(2)
+
+int vfio_pci_core_add_excluded_range(struct vfio_pci_core_device *vdev, int bar,
+ u64 start, u64 size, u32 flags);
+
int vfio_pci_core_register_dev_region(struct vfio_pci_core_device *vdev,
unsigned int type, unsigned int subtype,
const struct vfio_pci_regops *ops,
--
2.25.1
next prev parent reply other threads:[~2026-09-16 18:38 UTC|newest]
Thread overview: 31+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-16 18:35 [PATCH v5 00/27] vfio/pci: Add CXL Type-2 device passthrough support mhonap
2026-09-16 18:35 ` [PATCH v5 01/27] cxl/regs: Split the BAR block request and ioremap helpers mhonap
2026-09-16 18:35 ` [PATCH v5 02/27] cxl/regs: Let a BAR-owning driver own the component register block mhonap
2026-09-16 18:35 ` [PATCH v5 03/27] cxl: Move component register defines to uapi/cxl/cxl_regs.h mhonap
2026-09-16 18:35 ` [PATCH v5 04/27] cxl: Add cxl_reset_dvsec_sequence() for vfio-pci mhonap
2026-09-16 18:35 ` [PATCH v5 05/27] vfio/pci: Add the CXL provider ops registration interface mhonap
2026-09-16 18:35 ` [PATCH v5 06/27] vfio/pci: Detect CXL devices and load the CXL provider on demand mhonap
2026-09-16 18:35 ` [PATCH v5 07/27] vfio/pci: Honor -EPROBE_DEFER from CXL provider probe mhonap
2026-09-16 18:35 ` [PATCH v5 08/27] vfio/pci: Fall back to plain vfio-pci when CXL init fails mhonap
2026-09-16 18:35 ` mhonap [this message]
2026-09-16 18:35 ` [PATCH v5 10/27] vfio/pci: Migrate MSI-X exclusion onto the generic excluded-range list mhonap
2026-09-16 18:35 ` [PATCH v5 11/27] vfio/pci: Virtualize the CXL DVSEC in vfio_pci_config.c mhonap
2026-09-16 18:35 ` [PATCH v5 12/27] vfio/pci: Call the CXL open and close hooks around device use mhonap
2026-09-16 18:35 ` [PATCH v5 13/27] vfio/pci: Bracket PCI resets with the CXL reset hooks mhonap
2026-09-16 18:35 ` [PATCH v5 14/27] vfio/pci: Provide an opt-out for the CXL Type-2 extensions mhonap
2026-09-16 18:35 ` [PATCH v5 15/27] vfio/cxl: Add the vfio-cxl provider module skeleton mhonap
2026-09-16 18:35 ` [PATCH v5 16/27] vfio/cxl: Create the CXL memdev and set media ready at bind mhonap
2026-09-16 18:35 ` [PATCH v5 17/27] vfio/cxl: Own the whole component register BAR mhonap
2026-09-16 18:35 ` [PATCH v5 18/27] vfio/cxl: Expose the HDM memory region to the guest mhonap
2026-09-16 18:35 ` [PATCH v5 19/27] vfio/cxl: Contain HDM memory errors with memory_failure() mhonap
2026-09-16 18:35 ` [PATCH v5 20/27] vfio/cxl: Expose the HDM decoder registers read-only to the guest mhonap
2026-09-16 18:35 ` [PATCH v5 21/27] vfio/cxl: Exclude the HDM decoder registers from direct BAR access mhonap
2026-09-17 7:28 ` Richard Cheng
2026-09-16 18:35 ` [PATCH v5 22/27] vfio/cxl: Clear the HDM access gate after a hot reset mhonap
2026-09-16 18:35 ` [PATCH v5 23/27] vfio/cxl: Describe the CXL device and decoder geometry to userspace mhonap
2026-09-16 18:35 ` [PATCH v5 24/27] vfio/cxl: Export the HDM memory region as a dma-buf mhonap
2026-09-17 7:55 ` Richard Cheng
2026-09-16 18:35 ` [PATCH v5 25/27] vfio/cxl: Run the CXL reset at the vfio reset points mhonap
2026-09-16 18:35 ` [PATCH v5 26/27] Documentation: vfio-pci: Document CXL Type-2 device passthrough mhonap
2026-09-16 19:33 ` Gregory Price
2026-09-16 18:35 ` [PATCH v5 27/27] selftests/vfio: Add CXL Type-2 passthrough tests mhonap
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260916183540.3813685-10-mhonap@nvidia.com \
--to=mhonap@nvidia.com \
--cc=alejandro.lucero-palau@amd.com \
--cc=alex@shazbot.org \
--cc=alison.schofield@intel.com \
--cc=ankita@nvidia.com \
--cc=bhelgaas@google.com \
--cc=cjia@nvidia.com \
--cc=corbet@lwn.net \
--cc=dave.jiang@intel.com \
--cc=dave@stgolabs.net \
--cc=dmatlack@google.com \
--cc=gustavoars@kernel.org \
--cc=iweiny@kernel.org \
--cc=jgg@ziepe.ca \
--cc=jic23@kernel.org \
--cc=kees@kernel.org \
--cc=kevin.tian@intel.com \
--cc=kjaju@nvidia.com \
--cc=kvm@vger.kernel.org \
--cc=linux-cxl@vger.kernel.org \
--cc=linux-doc@vger.kernel.org \
--cc=linux-hardening@vger.kernel.org \
--cc=linux-kernel@vger.kernel.org \
--cc=linux-kselftest@vger.kernel.org \
--cc=linux-pci@vger.kernel.org \
--cc=ming.li@zohomail.com \
--cc=skhan@linuxfoundation.org \
--cc=skolothumtho@nvidia.com \
--cc=smadhavan@nvidia.com \
--cc=vishal.l.verma@intel.com \
--cc=vsethi@nvidia.com \
--cc=yishaih@nvidia.com \
--cc=zhiw@nvidia.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox
all inboxes | Powered by JetHome®