mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
From: <mhonap@nvidia.com>
To: <alex@shazbot.org>, <jgg@ziepe.ca>, <ankita@nvidia.com>,
	<jic23@kernel.org>, <dave.jiang@intel.com>,
	<alejandro.lucero-palau@amd.com>, <smadhavan@nvidia.com>,
	<corbet@lwn.net>, <skhan@linuxfoundation.org>,
	<dave@stgolabs.net>, <alison.schofield@intel.com>,
	<vishal.l.verma@intel.com>, <iweiny@kernel.org>,
	<ming.li@zohomail.com>, <yishaih@nvidia.com>,
	<skolothumtho@nvidia.com>, <kevin.tian@intel.com>,
	<bhelgaas@google.com>, <dmatlack@google.com>, <kees@kernel.org>,
	<gustavoars@kernel.org>
Cc: <cjia@nvidia.com>, <kjaju@nvidia.com>, <vsethi@nvidia.com>,
	<zhiw@nvidia.com>, <mhonap@nvidia.com>,
	<linux-doc@vger.kernel.org>, <linux-kernel@vger.kernel.org>,
	<kvm@vger.kernel.org>, <linux-cxl@vger.kernel.org>,
	<linux-pci@vger.kernel.org>, <linux-kselftest@vger.kernel.org>,
	<linux-hardening@vger.kernel.org>
Subject: [PATCH v5 09/27] vfio/pci: Add a generic excluded-range list
Date: Thu, 17 Sep 2026 00:05:22 +0530	[thread overview]
Message-ID: <20260916183540.3813685-10-mhonap@nvidia.com> (raw)
In-Reply-To: <20260916183540.3813685-1-mhonap@nvidia.com>

From: Manish Honap <mhonap@nvidia.com>

The MSI-X table is virtualized in vfio_pci_bar_rw() by an open-coded
x_start/x_end window that fills reads with -1 and drops writes. Now that
a generic excluded-range list expresses the same fill/drop behavior,
register the MSI-X table as a read and write excluded range instead of
special-casing it in the read/write path.

Add the range when the MSI-X capability is parsed in
vfio_pci_core_enable() and clear the list in vfio_pci_core_disable()
alongside the config teardown. The read/write path now relies solely on
vfio_pci_bar_find_exclusion(), so MSI-X and a provider's (e.g. vfio-cxl)
trapped registers share one mechanism.

No functional change.

Assisted-by: LLM
Signed-off-by: Manish Honap <mhonap@nvidia.com>
---
 drivers/vfio/pci/vfio_pci_core.c | 209 +++++++++++++++++++++++++++++++
 drivers/vfio/pci/vfio_pci_priv.h |   9 ++
 drivers/vfio/pci/vfio_pci_rdwr.c |   8 ++
 include/linux/vfio_pci_core.h    |  14 +++
 4 files changed, 240 insertions(+)

diff --git a/drivers/vfio/pci/vfio_pci_core.c b/drivers/vfio/pci/vfio_pci_core.c
index 9a75c30b67e2..9e4fa5d088a4 100644
--- a/drivers/vfio/pci/vfio_pci_core.c
+++ b/drivers/vfio/pci/vfio_pci_core.c
@@ -24,6 +24,7 @@
 #include <linux/pci.h>
 #include <linux/pm_runtime.h>
 #include <linux/slab.h>
+#include <linux/sort.h>
 #include <linux/types.h>
 #include <linux/uaccess.h>
 #include <linux/vgaarb.h>
@@ -1012,6 +1013,204 @@ static int msix_mmappable_cap(struct vfio_pci_core_device *vdev,
 	return vfio_info_add_capability(caps, &header, sizeof(header));
 }
 
+struct vfio_pci_excluded_range {
+	struct list_head	entry;
+	int			bar;
+	u64			start;
+	u64			size;
+	u32			flags;
+};
+
+int vfio_pci_core_add_excluded_range(struct vfio_pci_core_device *vdev, int bar,
+				     u64 start, u64 size, u32 flags)
+{
+	struct vfio_pci_excluded_range *range;
+
+	range = kzalloc_obj(*range);
+	if (!range)
+		return -ENOMEM;
+
+	range->bar = bar;
+	range->start = start;
+	range->size = size;
+	range->flags = flags;
+	list_add_tail(&range->entry, &vdev->excluded_ranges);
+
+	return 0;
+}
+EXPORT_SYMBOL_GPL(vfio_pci_core_add_excluded_range);
+
+static void vfio_pci_free_excluded_ranges(struct vfio_pci_core_device *vdev)
+{
+	struct vfio_pci_excluded_range *range, *tmp;
+
+	list_for_each_entry_safe(range, tmp, &vdev->excluded_ranges, entry) {
+		list_del(&range->entry);
+		kfree(range);
+	}
+}
+
+bool vfio_pci_bar_find_exclusion(struct vfio_pci_core_device *vdev, int bar,
+				 loff_t pos, size_t count, bool iswrite,
+				 size_t *x_start, size_t *x_end)
+{
+	u32 want = iswrite ? VFIO_PCI_EXCLUDE_WRITE : VFIO_PCI_EXCLUDE_READ;
+	struct vfio_pci_excluded_range *range;
+	bool found = false;
+
+	list_for_each_entry(range, &vdev->excluded_ranges, entry) {
+		if (range->bar != bar || !(range->flags & want))
+			continue;
+		if (pos < range->start + range->size &&
+		    pos + count > range->start) {
+			/*
+			 * A BAR can carry more than one excluded window (e.g.
+			 * the MSI-X table and a CXL HDM decoder block). Return
+			 * the overlapping window with the lowest start so the
+			 * caller can walk them in order.
+			 */
+			if (!found || range->start < *x_start) {
+				*x_start = range->start;
+				*x_end = range->start + range->size;
+				found = true;
+			}
+		}
+	}
+
+	return found;
+}
+
+/* True when [start, start + len) on @bar overlaps an mmap-excluded range. */
+static bool vfio_pci_bar_mmap_excluded(struct vfio_pci_core_device *vdev,
+				       int bar, u64 start, u64 len)
+{
+	struct vfio_pci_excluded_range *range;
+
+	list_for_each_entry(range, &vdev->excluded_ranges, entry) {
+		if (range->bar != bar ||
+		    !(range->flags & VFIO_PCI_EXCLUDE_MMAP))
+			continue;
+		if (start < range->start + range->size &&
+		    start + len > range->start)
+			return true;
+	}
+
+	return false;
+}
+
+/* A page-aligned mmap hole, derived from an mmap-excluded range. */
+struct vfio_pci_mmap_hole {
+	u64 start;
+	u64 end;
+};
+
+static int vfio_pci_mmap_hole_cmp(const void *a, const void *b)
+{
+	const struct vfio_pci_mmap_hole *x = a, *y = b;
+
+	if (x->start < y->start)
+		return -1;
+	return x->start > y->start;
+}
+
+/*
+ * Advertise the BAR as mmappable minus every page-aligned mmap-excluded hole.
+ * A BAR can carry several holes at unrelated offsets (for example an MSI-X
+ * table and one or more trapped CXL component sub-blocks, which the CXL spec
+ * locates by pointer, not at fixed offsets).
+ * Collect the holes, page-align and sort them, coalesce any that overlap or
+ * touch, and advertise the gaps.
+ */
+static int vfio_pci_excluded_sparse_cap(struct vfio_pci_core_device *vdev,
+					int index, struct vfio_info_cap *caps)
+{
+	u64 bar_len = pci_resource_len(vdev->pdev, index);
+	struct vfio_region_info_cap_sparse_mmap *sparse;
+	struct vfio_pci_excluded_range *range;
+	struct vfio_pci_mmap_hole *holes;
+	int nr_holes = 0, nr_areas = 0, i, j;
+	size_t size;
+	u64 pos;
+	int ret;
+
+	list_for_each_entry(range, &vdev->excluded_ranges, entry)
+		if (range->bar == index &&
+		    (range->flags & VFIO_PCI_EXCLUDE_MMAP))
+			nr_holes++;
+
+	if (!nr_holes)
+		return 0;
+
+	holes = kmalloc_array(nr_holes, sizeof(*holes), GFP_KERNEL);
+	if (!holes)
+		return -ENOMEM;
+
+	/*
+	 * mmap is page granular, so each hole rounds out to the page boundaries
+	 * enclosing its excluded sub-range. The byte-granular exclusion still
+	 * governs the fault and read/write paths; only the advertised mmap areas
+	 * round to whole pages.
+	 */
+	i = 0;
+	list_for_each_entry(range, &vdev->excluded_ranges, entry) {
+		if (range->bar != index ||
+		    !(range->flags & VFIO_PCI_EXCLUDE_MMAP))
+			continue;
+		holes[i].start = ALIGN_DOWN(range->start, PAGE_SIZE);
+		holes[i].end = ALIGN(range->start + range->size, PAGE_SIZE);
+		i++;
+	}
+
+	sort(holes, nr_holes, sizeof(*holes), vfio_pci_mmap_hole_cmp, NULL);
+
+	/* Coalesce holes that overlap or touch after page alignment. */
+	for (i = 0, j = 0; i < nr_holes; i++) {
+		if (j && holes[i].start <= holes[j - 1].end)
+			holes[j - 1].end = max(holes[j - 1].end, holes[i].end);
+		else
+			holes[j++] = holes[i];
+	}
+	nr_holes = j;
+
+	/* One mmappable area per gap: before, between, and after the holes. */
+	for (i = 0, pos = 0; i < nr_holes; i++) {
+		if (holes[i].start > pos)
+			nr_areas++;
+		pos = holes[i].end;
+	}
+	if (pos < bar_len)
+		nr_areas++;
+
+	size = struct_size(sparse, areas, nr_areas);
+	sparse = kzalloc(size, GFP_KERNEL);
+	if (!sparse) {
+		kfree(holes);
+		return -ENOMEM;
+	}
+
+	sparse->header.id = VFIO_REGION_INFO_CAP_SPARSE_MMAP;
+	sparse->header.version = 1;
+	sparse->nr_areas = nr_areas;
+
+	for (i = 0, j = 0, pos = 0; i < nr_holes; i++) {
+		if (holes[i].start > pos) {
+			sparse->areas[j].offset = pos;
+			sparse->areas[j].size = holes[i].start - pos;
+			j++;
+		}
+		pos = holes[i].end;
+	}
+	if (pos < bar_len) {
+		sparse->areas[j].offset = pos;
+		sparse->areas[j].size = bar_len - pos;
+	}
+
+	kfree(holes);
+	ret = vfio_info_add_capability(caps, &sparse->header, size);
+	kfree(sparse);
+	return ret;
+}
+
 int vfio_pci_core_register_dev_region(struct vfio_pci_core_device *vdev,
 				      unsigned int type, unsigned int subtype,
 				      const struct vfio_pci_regops *ops,
@@ -1160,6 +1359,10 @@ int vfio_pci_ioctl_get_region_info(struct vfio_device *core_vdev,
 				if (ret)
 					return ret;
 			}
+			ret = vfio_pci_excluded_sparse_cap(vdev, info->index,
+							   caps);
+			if (ret)
+				return ret;
 		}
 
 		break;
@@ -1853,6 +2056,10 @@ int vfio_pci_core_mmap(struct vfio_device *core_vdev, struct vm_area_struct *vma
 	if (req_start + req_len > phys_len)
 		return -EINVAL;
 
+	/* An excluded sub-range is reachable only through its trap, not mmap. */
+	if (vfio_pci_bar_mmap_excluded(vdev, index, req_start, req_len))
+		return -EINVAL;
+
 	/*
 	 * Ensure the BAR resource region is reserved for use.
 	 */
@@ -2317,6 +2524,7 @@ int vfio_pci_core_init_dev(struct vfio_device *core_vdev)
 	INIT_LIST_HEAD(&vdev->dmabufs);
 	init_rwsem(&vdev->memory_lock);
 	xa_init(&vdev->ctx);
+	INIT_LIST_HEAD(&vdev->excluded_ranges);
 
 	ret = vfio_pci_core_cxl_init(vdev);
 	if (ret)
@@ -2332,6 +2540,7 @@ void vfio_pci_core_release_dev(struct vfio_device *core_vdev)
 		container_of(core_vdev, struct vfio_pci_core_device, vdev);
 
 	vfio_pci_core_cxl_release(vdev);
+	vfio_pci_free_excluded_ranges(vdev);
 
 	mutex_destroy(&vdev->igate);
 	mutex_destroy(&vdev->ioeventfds_lock);
diff --git a/drivers/vfio/pci/vfio_pci_priv.h b/drivers/vfio/pci/vfio_pci_priv.h
index 4e7162234a2e..c268c99aea82 100644
--- a/drivers/vfio/pci/vfio_pci_priv.h
+++ b/drivers/vfio/pci/vfio_pci_priv.h
@@ -44,6 +44,15 @@ ssize_t vfio_pci_config_rw_single(struct vfio_pci_core_device *vdev,
 ssize_t vfio_pci_bar_rw(struct vfio_pci_core_device *vdev, char __user *buf,
 			size_t count, loff_t *ppos, bool iswrite);
 
+/*
+ * If a read (or write) to [pos, pos + count) on @bar overlaps an excluded
+ * range, report the byte window do_io_rw() should fill with -1 (or drop) and
+ * return true. A single access spans at most one such window.
+ */
+bool vfio_pci_bar_find_exclusion(struct vfio_pci_core_device *vdev, int bar,
+				 loff_t pos, size_t count, bool iswrite,
+				 size_t *x_start, size_t *x_end);
+
 #ifdef CONFIG_VFIO_PCI_VGA
 ssize_t vfio_pci_vga_rw(struct vfio_pci_core_device *vdev, char __user *buf,
 			size_t count, loff_t *ppos, bool iswrite);
diff --git a/drivers/vfio/pci/vfio_pci_rdwr.c b/drivers/vfio/pci/vfio_pci_rdwr.c
index 7f14dd46de17..48da1cb08296 100644
--- a/drivers/vfio/pci/vfio_pci_rdwr.c
+++ b/drivers/vfio/pci/vfio_pci_rdwr.c
@@ -261,6 +261,14 @@ ssize_t vfio_pci_bar_rw(struct vfio_pci_core_device *vdev, char __user *buf,
 		x_end = vdev->msix_offset + vdev->msix_size;
 	}
 
+	/*
+	 * A provider-excluded sub-range is filled with -1 on read and dropped on
+	 * write for the same reason: the guest reaches it only through the trap.
+	 * An access spans at most one exclusion window.
+	 */
+	vfio_pci_bar_find_exclusion(vdev, bar, pos, count, iswrite,
+				    &x_start, &x_end);
+
 	done = vfio_pci_core_do_io_rw(vdev, res->flags & IORESOURCE_MEM, io, buf, pos,
 				      count, x_start, x_end, iswrite, max_width);
 
diff --git a/include/linux/vfio_pci_core.h b/include/linux/vfio_pci_core.h
index 7f3a2bcb5830..92e3db116068 100644
--- a/include/linux/vfio_pci_core.h
+++ b/include/linux/vfio_pci_core.h
@@ -162,6 +162,7 @@ struct vfio_pci_core_device {
 	struct notifier_block	nb;
 	struct rw_semaphore	memory_lock;
 	struct list_head	dmabufs;
+	struct list_head	excluded_ranges;
 };
 
 enum vfio_pci_io_width {
@@ -172,6 +173,19 @@ enum vfio_pci_io_width {
 };
 
 /* Will be exported for vfio pci drivers usage */
+/*
+ * A provider can keep a BAR sub-range off the direct guest path, reached only
+ * through its own trap. The flags select which paths are excluded: mmap, and
+ * region read and write (an excluded read fills -1, an excluded write is
+ * dropped).
+ */
+#define VFIO_PCI_EXCLUDE_MMAP	BIT(0)
+#define VFIO_PCI_EXCLUDE_READ	BIT(1)
+#define VFIO_PCI_EXCLUDE_WRITE	BIT(2)
+
+int vfio_pci_core_add_excluded_range(struct vfio_pci_core_device *vdev, int bar,
+				     u64 start, u64 size, u32 flags);
+
 int vfio_pci_core_register_dev_region(struct vfio_pci_core_device *vdev,
 				      unsigned int type, unsigned int subtype,
 				      const struct vfio_pci_regops *ops,
-- 
2.25.1


  parent reply	other threads:[~2026-09-16 18:38 UTC|newest]

Thread overview: 31+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-16 18:35 [PATCH v5 00/27] vfio/pci: Add CXL Type-2 device passthrough support mhonap
2026-09-16 18:35 ` [PATCH v5 01/27] cxl/regs: Split the BAR block request and ioremap helpers mhonap
2026-09-16 18:35 ` [PATCH v5 02/27] cxl/regs: Let a BAR-owning driver own the component register block mhonap
2026-09-16 18:35 ` [PATCH v5 03/27] cxl: Move component register defines to uapi/cxl/cxl_regs.h mhonap
2026-09-16 18:35 ` [PATCH v5 04/27] cxl: Add cxl_reset_dvsec_sequence() for vfio-pci mhonap
2026-09-16 18:35 ` [PATCH v5 05/27] vfio/pci: Add the CXL provider ops registration interface mhonap
2026-09-16 18:35 ` [PATCH v5 06/27] vfio/pci: Detect CXL devices and load the CXL provider on demand mhonap
2026-09-16 18:35 ` [PATCH v5 07/27] vfio/pci: Honor -EPROBE_DEFER from CXL provider probe mhonap
2026-09-16 18:35 ` [PATCH v5 08/27] vfio/pci: Fall back to plain vfio-pci when CXL init fails mhonap
2026-09-16 18:35 ` mhonap [this message]
2026-09-16 18:35 ` [PATCH v5 10/27] vfio/pci: Migrate MSI-X exclusion onto the generic excluded-range list mhonap
2026-09-16 18:35 ` [PATCH v5 11/27] vfio/pci: Virtualize the CXL DVSEC in vfio_pci_config.c mhonap
2026-09-16 18:35 ` [PATCH v5 12/27] vfio/pci: Call the CXL open and close hooks around device use mhonap
2026-09-16 18:35 ` [PATCH v5 13/27] vfio/pci: Bracket PCI resets with the CXL reset hooks mhonap
2026-09-16 18:35 ` [PATCH v5 14/27] vfio/pci: Provide an opt-out for the CXL Type-2 extensions mhonap
2026-09-16 18:35 ` [PATCH v5 15/27] vfio/cxl: Add the vfio-cxl provider module skeleton mhonap
2026-09-16 18:35 ` [PATCH v5 16/27] vfio/cxl: Create the CXL memdev and set media ready at bind mhonap
2026-09-16 18:35 ` [PATCH v5 17/27] vfio/cxl: Own the whole component register BAR mhonap
2026-09-16 18:35 ` [PATCH v5 18/27] vfio/cxl: Expose the HDM memory region to the guest mhonap
2026-09-16 18:35 ` [PATCH v5 19/27] vfio/cxl: Contain HDM memory errors with memory_failure() mhonap
2026-09-16 18:35 ` [PATCH v5 20/27] vfio/cxl: Expose the HDM decoder registers read-only to the guest mhonap
2026-09-16 18:35 ` [PATCH v5 21/27] vfio/cxl: Exclude the HDM decoder registers from direct BAR access mhonap
2026-09-17  7:28   ` Richard Cheng
2026-09-16 18:35 ` [PATCH v5 22/27] vfio/cxl: Clear the HDM access gate after a hot reset mhonap
2026-09-16 18:35 ` [PATCH v5 23/27] vfio/cxl: Describe the CXL device and decoder geometry to userspace mhonap
2026-09-16 18:35 ` [PATCH v5 24/27] vfio/cxl: Export the HDM memory region as a dma-buf mhonap
2026-09-17  7:55   ` Richard Cheng
2026-09-16 18:35 ` [PATCH v5 25/27] vfio/cxl: Run the CXL reset at the vfio reset points mhonap
2026-09-16 18:35 ` [PATCH v5 26/27] Documentation: vfio-pci: Document CXL Type-2 device passthrough mhonap
2026-09-16 19:33   ` Gregory Price
2026-09-16 18:35 ` [PATCH v5 27/27] selftests/vfio: Add CXL Type-2 passthrough tests mhonap

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260916183540.3813685-10-mhonap@nvidia.com \
    --to=mhonap@nvidia.com \
    --cc=alejandro.lucero-palau@amd.com \
    --cc=alex@shazbot.org \
    --cc=alison.schofield@intel.com \
    --cc=ankita@nvidia.com \
    --cc=bhelgaas@google.com \
    --cc=cjia@nvidia.com \
    --cc=corbet@lwn.net \
    --cc=dave.jiang@intel.com \
    --cc=dave@stgolabs.net \
    --cc=dmatlack@google.com \
    --cc=gustavoars@kernel.org \
    --cc=iweiny@kernel.org \
    --cc=jgg@ziepe.ca \
    --cc=jic23@kernel.org \
    --cc=kees@kernel.org \
    --cc=kevin.tian@intel.com \
    --cc=kjaju@nvidia.com \
    --cc=kvm@vger.kernel.org \
    --cc=linux-cxl@vger.kernel.org \
    --cc=linux-doc@vger.kernel.org \
    --cc=linux-hardening@vger.kernel.org \
    --cc=linux-kernel@vger.kernel.org \
    --cc=linux-kselftest@vger.kernel.org \
    --cc=linux-pci@vger.kernel.org \
    --cc=ming.li@zohomail.com \
    --cc=skhan@linuxfoundation.org \
    --cc=skolothumtho@nvidia.com \
    --cc=smadhavan@nvidia.com \
    --cc=vishal.l.verma@intel.com \
    --cc=vsethi@nvidia.com \
    --cc=yishaih@nvidia.com \
    --cc=zhiw@nvidia.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®