* [RFC PATCH 1/4] PCI/TSM: Create MMIO descriptors via TDISP Report
2026-10-05 7:02 [RFC PATCH 0/4] PCI/TSM: Resolve TDISP coherent (CXL) ranges from precommitted HDM decoders Ankit Agrawal
@ 2026-10-05 7:02 ` Ankit Agrawal
2026-10-05 11:17 ` Bradley Morgan
2026-10-05 7:02 ` [RFC PATCH 2/4] PCI/CXL: Populate and insert/remove pdev->coh_resource[] Ankit Agrawal
` (2 subsequent siblings)
3 siblings, 1 reply; 6+ messages in thread
From: Ankit Agrawal @ 2026-10-05 7:02 UTC (permalink / raw)
To: linux-cxl, linux-pci, linux-kernel
Cc: dave, jic23, dave.jiang, alison.schofield, vishal.l.verma, jgg,
aneesh.kumar, aik, yilun.xu, iweiny, ming.li, icheng, bhelgaas,
smadhavan, ankita, ilpo.jarvinen, Smita.KoralahalliChannabasappa,
andriy.shevchenko, linux-coco
After pci_tsm_bind() and pci_tsm_lock() the low level TSM driver is
expected to populate the TDISP GET_DEVICE_INTERFACE_REPORT response
payload for the device.
Add pci_tsm_mmio_alloc()/pci_tsm_mmio_free() to parse that report into
a set of encrypted MMIO ranges, and pci_tsm_mmio_setup()/
pci_tsm_mmio_teardown() to mark those ranges as encrypted in iomem via
a new encrypted_iomem_resource tree (IORES_DESC_ENCRYPTED). With those
descriptors the TSM driver can use pci_tsm_mmio_setup() to inform
ioremap() how to map the device per the device expectations. The VM
is expected to validate the interface with the relying party before
accepting the device for operation.
pci_tsm_mmio_alloc() also recovers the obfuscated starting address for
each encrypted MMIO range, since the VM is never disclosed the HPA
that correlates to the GPA of the device's MMIO. The obfuscated
address is BAR aligned.
Based on an original patch by Aneesh Kumar K.V [1], and cherry-picked
from Dan Williams' upstream commit [2]. Major functional differences
from Dan's posting is due to the lack of evidence-store infrastructure
yet (device_evidence / DEVICE_EVIDENCE_TYPE_REPORT). The
pci_tsm_mmio_alloc() takes the raw report bytes directly from the caller
instead of pulling them from a device evidence store.
Link: https://lore.kernel.org/linux-coco/20251117140007.122062-8-aneesh.kumar@kernel.org/ [1]
Link: https://lore.kernel.org/all/20260705220819.2472765-15-djbw@kernel.org/ [2]
Signed-off-by: Ankit Agrawal <ankita@nvidia.com>
Assisted-by: Claude:sonnet-5
---
drivers/pci/tsm.c | 236 ++++++++++++++++++++++++++++++++++++++++
include/linux/ioport.h | 2 +
include/linux/pci-tsm.h | 33 ++++++
kernel/resource.c | 8 ++
4 files changed, 279 insertions(+)
diff --git a/drivers/pci/tsm.c b/drivers/pci/tsm.c
index 5fdcd7f2e820..c8cfe89b6224 100644
--- a/drivers/pci/tsm.c
+++ b/drivers/pci/tsm.c
@@ -9,11 +9,16 @@
#define dev_fmt(fmt) "PCI/TSM: " fmt
#include <linux/bitfield.h>
+#include <linux/overflow.h>
#include <linux/pci.h>
#include <linux/pci-doe.h>
#include <linux/pci-tsm.h>
+#include <linux/range.h>
+#include <linux/sizes.h>
+#include <linux/slab.h>
#include <linux/sysfs.h>
#include <linux/tsm.h>
+#include <linux/unaligned.h>
#include <linux/xarray.h>
#include "pci.h"
@@ -898,3 +903,234 @@ int pci_tsm_doe_transfer(struct pci_dev *pdev, u8 type, const void *req,
resp, resp_sz);
}
EXPORT_SYMBOL_GPL(pci_tsm_doe_transfer);
+
+static void mmio_teardown(struct pci_tsm_mmio *mmio, int nr)
+{
+ while (nr--) {
+ struct resource *res = pci_tsm_mmio_resource(mmio, nr);
+
+ /*
+ * Entries past the point where pci_tsm_mmio_setup() stopped
+ * (or that were never run through setup() at all, e.g. a
+ * caller that only wants the resolved range from
+ * pci_tsm_mmio_alloc()) were never insert_resource()'d, so
+ * res->parent is NULL. remove_resource() dereferences
+ * res->parent unconditionally.
+ */
+ if (res->parent)
+ remove_resource(res);
+ }
+}
+
+/**
+ * pci_tsm_mmio_setup() - mark device MMIO as encrypted in iomem
+ * @pdev: device owner of MMIO resources
+ * @mmio: container of an array of resources to mark encrypted
+ */
+int pci_tsm_mmio_setup(struct pci_dev *pdev, struct pci_tsm_mmio *mmio)
+{
+ int i;
+
+ device_lock_assert(&pdev->dev);
+ if (pdev->dev.driver)
+ return -EBUSY;
+
+ for (i = 0; i < mmio->nr; i++) {
+ struct resource *res = pci_tsm_mmio_resource(mmio, i);
+ int j;
+
+ if (resource_size(res) == 0 || !(res->flags & IORESOURCE_MEM))
+ break;
+
+ /* Only require the caller to set the range, init remainder */
+ *res = DEFINE_RES_NAMED_DESC(res->start, resource_size(res),
+ pci_name(pdev), IORESOURCE_MEM,
+ IORES_DESC_ENCRYPTED);
+
+ for (j = 0; j < PCI_NUM_RESOURCES; j++)
+ if (resource_contains(pci_resource_n(pdev, j), res))
+ break;
+
+ /* Request is outside of device MMIO */
+ if (j >= PCI_NUM_RESOURCES)
+ break;
+
+ if (insert_resource(&encrypted_iomem_resource, res) != 0)
+ break;
+ }
+
+ if (i >= mmio->nr)
+ return 0;
+
+ mmio_teardown(mmio, i);
+
+ return -EINVAL;
+}
+EXPORT_SYMBOL_GPL(pci_tsm_mmio_setup);
+
+void pci_tsm_mmio_teardown(struct pci_tsm_mmio *mmio)
+{
+ mmio_teardown(mmio, mmio->nr);
+}
+EXPORT_SYMBOL_GPL(pci_tsm_mmio_teardown);
+
+/*
+ * PCIe ECN TEE Device Interface Security Protocol (TDISP)
+ *
+ * Device Interface Report data object layout as defined by PCIe r7.0 section
+ * 11.3.11
+ */
+#define PCI_TSM_DEVIF_REPORT_MMIO_ATTR_MSIX_TABLE BIT(0)
+#define PCI_TSM_DEVIF_REPORT_MMIO_ATTR_MSIX_PBA BIT(1)
+#define PCI_TSM_DEVIF_REPORT_MMIO_ATTR_IS_NON_TEE BIT(2)
+#define PCI_TSM_DEVIF_REPORT_MMIO_ATTR_IS_UPDATABLE BIT(3)
+#define PCI_TSM_DEVIF_REPORT_MMIO_ATTR_RANGE_ID GENMASK(31, 16)
+
+/* An interface report 'pfn' is 4K in size */
+struct pci_tsm_devif_mmio {
+ __le64 phys;
+ __le32 nr_pfns;
+ __le32 attributes;
+};
+
+struct pci_tsm_devif_report {
+ __le16 interface_info;
+ __le16 reserved;
+ __le16 msi_x_message_control;
+ __le16 lnr_control;
+ __le32 tph_control;
+ __le32 mmio_range_count;
+ struct pci_tsm_devif_mmio mmio[];
+};
+
+/**
+ * pci_tsm_mmio_alloc() - allocate encrypted MMIO range descriptor
+ * @pdev: device owner of MMIO ranges
+ * @report: raw TDISP Device Interface Report bytes (PCIe r7.0 section
+ * 11.3.11), e.g. as returned by GET_DEVICE_INTERFACE_REPORT
+ * @report_len: length of @report in bytes
+ *
+ * Return: the encrypted MMIO range descriptor on success, NULL on failure
+ *
+ * Unlike upstream, @report is supplied directly by the caller rather than
+ * pulled from a device evidence store, which this tree does not yet have.
+ */
+struct pci_tsm_mmio *pci_tsm_mmio_alloc(struct pci_dev *pdev,
+ const void *report, size_t report_len)
+{
+ const struct pci_tsm_devif_report *devif_report = report;
+ u64 reporting_bar_base, last_reporting_end;
+ u32 mmio_range_count;
+ int last_bar = -1;
+ int i;
+
+ if (report_len < sizeof(*devif_report))
+ return NULL;
+
+ mmio_range_count = __le32_to_cpu(devif_report->mmio_range_count);
+
+ /* check that the report is self-consistent on mmio entries */
+ if (report_len < struct_size(devif_report, mmio, mmio_range_count))
+ return NULL;
+
+ /* create pci_tsm_mmio descriptors from the report data */
+ struct pci_tsm_mmio *mmio __free(kfree) =
+ kzalloc_flex(*mmio, mmio, mmio_range_count);
+ if (!mmio)
+ return NULL;
+
+ for (i = 0; i < mmio_range_count; i++) {
+ u64 range_off;
+ struct range range;
+ const struct pci_tsm_devif_mmio *mmio_data = &devif_report->mmio[i];
+ struct pci_tsm_mmio_entry *entry =
+ pci_tsm_mmio_entry(mmio, mmio->nr);
+ u64 tsm_offset = __le64_to_cpu(mmio_data->phys);
+ u64 size = (u64)__le32_to_cpu(mmio_data->nr_pfns) * SZ_4K;
+ u32 attr = __le32_to_cpu(mmio_data->attributes);
+ int bar = FIELD_GET(PCI_TSM_DEVIF_REPORT_MMIO_ATTR_RANGE_ID,
+ attr);
+
+ if (bar >= PCI_STD_NUM_BARS ||
+ !(pci_resource_flags(pdev, bar) & IORESOURCE_MEM) ||
+ (pci_resource_flags(pdev, bar) & IORESOURCE_UNSET)) {
+ pci_dbg(pdev, "Invalid reporting bar ID %d\n", bar);
+ return NULL;
+ }
+
+ if (last_bar > bar) {
+ pci_dbg(pdev, "Reporting bar ID not in ascending order\n");
+ return NULL;
+ }
+
+ if (last_bar < bar) {
+ resource_size_t mask = pci_resource_len(pdev, bar) - 1;
+
+ /* Transition to a new bar */
+ last_bar = bar;
+
+ /*
+ * Determine the obfuscated base of the BAR. BAR
+ * offsets are never obfuscated.
+ */
+ reporting_bar_base = tsm_offset & ~mask;
+ } else if (tsm_offset < last_reporting_end) {
+ pci_dbg(pdev, "Reporting ranges within BAR not in ascending order\n");
+ return NULL;
+ }
+
+ /* Per spec the tsm_offset never results in overflow / underflow */
+ last_reporting_end = tsm_offset + size;
+ if (last_reporting_end < tsm_offset) {
+ pci_dbg(pdev, "Reporting range overflow\n");
+ return NULL;
+ }
+
+ range_off = tsm_offset - reporting_bar_base;
+ if (pci_resource_len(pdev, bar) < range_off + size) {
+ pci_dbg(pdev, "Reporting range larger than BAR size\n");
+ return NULL;
+ }
+
+ range.start = pci_resource_start(pdev, bar) + range_off;
+ range.end = range.start + size - 1;
+
+ /* Only record the TEE ranges for later consideration by ioremap() */
+ if (FIELD_GET(PCI_TSM_DEVIF_REPORT_MMIO_ATTR_IS_NON_TEE,
+ attr)) {
+ pci_dbg(pdev, "Skipping non-TEE range, BAR%d %pra\n",
+ bar, &range);
+ continue;
+ }
+
+ entry->res.start = range.start;
+ entry->res.end = range.end;
+ entry->res.flags = IORESOURCE_MEM;
+ entry->tsm_offset = tsm_offset;
+ mmio->nr++;
+ }
+
+ return_ptr(mmio);
+}
+EXPORT_SYMBOL_GPL(pci_tsm_mmio_alloc);
+
+/**
+ * pci_tsm_mmio_free() - free a pci_tsm_mmio instance
+ * @pdev: device owner of MMIO ranges
+ * @mmio: instance to free
+ *
+ * Returns 0 if @mmio was idle on entry, -EBUSY otherwise
+ */
+int pci_tsm_mmio_free(struct pci_dev *pdev, struct pci_tsm_mmio *mmio)
+{
+ for (int i = 0; i < mmio->nr; i++) {
+ struct resource *res = pci_tsm_mmio_resource(mmio, i);
+
+ if (dev_WARN_ONCE(&pdev->dev, resource_assigned(res),
+ "MMIO resource still assigned %pr\n", res))
+ return -EBUSY;
+ }
+ kfree(mmio);
+ return 0;
+}
+EXPORT_SYMBOL_GPL(pci_tsm_mmio_free);
diff --git a/include/linux/ioport.h b/include/linux/ioport.h
index f7930b3dfd0a..122f1eefb4b9 100644
--- a/include/linux/ioport.h
+++ b/include/linux/ioport.h
@@ -144,6 +144,7 @@ enum {
IORES_DESC_RESERVED = 7,
IORES_DESC_SOFT_RESERVED = 8,
IORES_DESC_CXL = 9,
+ IORES_DESC_ENCRYPTED = 10,
};
/*
@@ -240,6 +241,7 @@ struct resource_constraint {
extern struct resource ioport_resource;
extern struct resource iomem_resource;
extern struct resource soft_reserve_resource;
+extern struct resource encrypted_iomem_resource;
extern struct resource *request_resource_conflict(struct resource *root, struct resource *new);
extern int request_resource(struct resource *root, struct resource *new);
diff --git a/include/linux/pci-tsm.h b/include/linux/pci-tsm.h
index a6435aba03f9..e54ad057fa8e 100644
--- a/include/linux/pci-tsm.h
+++ b/include/linux/pci-tsm.h
@@ -199,6 +199,34 @@ enum pci_tsm_req_scope {
PCI_TSM_REQ_DEBUG_WRITE = 3,
};
+/**
+ * struct pci_tsm_mmio_entry - an encrypted MMIO range
+ * @res: MMIO address range (typically Guest Physical Address, GPA)
+ * @tsm_offset: Host Physical Address, HPA obfuscation offset added by the TSM.
+ * Translates report addresses to GPA.
+ */
+struct pci_tsm_mmio_entry {
+ struct resource res;
+ u64 tsm_offset;
+};
+
+struct pci_tsm_mmio {
+ int nr;
+ struct pci_tsm_mmio_entry mmio[];
+};
+
+static inline struct pci_tsm_mmio_entry *
+pci_tsm_mmio_entry(struct pci_tsm_mmio *mmio, int idx)
+{
+ return &mmio->mmio[idx];
+}
+
+static inline struct resource *pci_tsm_mmio_resource(struct pci_tsm_mmio *mmio,
+ int idx)
+{
+ return &mmio->mmio[idx].res;
+}
+
#ifdef CONFIG_PCI_TSM
int pci_tsm_register(struct tsm_dev *tsm_dev);
void pci_tsm_unregister(struct tsm_dev *tsm_dev);
@@ -216,6 +244,11 @@ void pci_tsm_tdi_constructor(struct pci_dev *pdev, struct pci_tdi *tdi,
ssize_t pci_tsm_guest_req(struct pci_dev *pdev, enum pci_tsm_req_scope scope,
sockptr_t req_in, size_t in_len, sockptr_t req_out,
size_t out_len, u64 *tsm_code);
+int pci_tsm_mmio_setup(struct pci_dev *pdev, struct pci_tsm_mmio *mmio);
+void pci_tsm_mmio_teardown(struct pci_tsm_mmio *mmio);
+struct pci_tsm_mmio *pci_tsm_mmio_alloc(struct pci_dev *pdev,
+ const void *report, size_t report_len);
+int pci_tsm_mmio_free(struct pci_dev *pdev, struct pci_tsm_mmio *mmio);
#else
static inline int pci_tsm_register(struct tsm_dev *tsm_dev)
{
diff --git a/kernel/resource.c b/kernel/resource.c
index cfc1a00e86aa..5500f7828b2d 100644
--- a/kernel/resource.c
+++ b/kernel/resource.c
@@ -56,6 +56,14 @@ struct resource soft_reserve_resource = {
.flags = IORESOURCE_MEM,
};
+struct resource encrypted_iomem_resource = {
+ .name = "Encrypted MMIO",
+ .start = 0,
+ .end = -1,
+ .desc = IORES_DESC_ENCRYPTED,
+ .flags = IORESOURCE_MEM,
+};
+
static DEFINE_RWLOCK(resource_lock);
/*
--
2.43.0
^ permalink raw reply [flat|nested] 6+ messages in thread* Re: [RFC PATCH 1/4] PCI/TSM: Create MMIO descriptors via TDISP Report
2026-10-05 7:02 ` [RFC PATCH 1/4] PCI/TSM: Create MMIO descriptors via TDISP Report Ankit Agrawal
@ 2026-10-05 11:17 ` Bradley Morgan
0 siblings, 0 replies; 6+ messages in thread
From: Bradley Morgan @ 2026-10-05 11:17 UTC (permalink / raw)
To: ankita
Cc: Smita.KoralahalliChannabasappa, aik, alison.schofield,
andriy.shevchenko, aneesh.kumar, bhelgaas, dave.jiang, dave,
icheng, ilpo.jarvinen, iweiny, jgg, jic23, linux-coco, linux-cxl,
linux-kernel, linux-pci, ming.li, smadhavan, vishal.l.verma,
yilun.xu
On 5 October 2026 08:02:49 BST, Ankit Agrawal <ankita@nvidia.com> wrote:
>After pci_tsm_bind() and pci_tsm_lock() the low level TSM driver is
>expected to populate the TDISP GET_DEVICE_INTERFACE_REPORT response
>payload for the device.
>
>Add pci_tsm_mmio_alloc()/pci_tsm_mmio_free() to parse that report into
>a set of encrypted MMIO ranges, and pci_tsm_mmio_setup()/
>pci_tsm_mmio_teardown() to mark those ranges as encrypted in iomem via
>a new encrypted_iomem_resource tree (IORES_DESC_ENCRYPTED). With those
>descriptors the TSM driver can use pci_tsm_mmio_setup() to inform
>ioremap() how to map the device per the device expectations. The VM
>is expected to validate the interface with the relying party before
>accepting the device for operation.
>
>pci_tsm_mmio_alloc() also recovers the obfuscated starting address for
>each encrypted MMIO range, since the VM is never disclosed the HPA
>that correlates to the GPA of the device's MMIO. The obfuscated
>address is BAR aligned.
>
>Based on an original patch by Aneesh Kumar K.V [1], and cherry-picked
>from Dan Williams' upstream commit [2]. Major functional differences
>from Dan's posting is due to the lack of evidence-store infrastructure
>yet (device_evidence / DEVICE_EVIDENCE_TYPE_REPORT). The
>pci_tsm_mmio_alloc() takes the raw report bytes directly from the caller
>instead of pulling them from a device evidence store.
>
>Link: https://lore.kernel.org/linux-coco/20251117140007.122062-8-aneesh.kumar@kernel.org/ [1]
>Link: https://lore.kernel.org/all/20260705220819.2472765-15-djbw@kernel.org/ [2]
LGTM from a kernel/resource.c standpoint:
Reviewed-by: Bradley Morgan <brads@mainlining.org> # kernel/
>
>Signed-off-by: Ankit Agrawal <ankita@nvidia.com>
>Assisted-by: Claude:sonnet-5
>---
> drivers/pci/tsm.c | 236 ++++++++++++++++++++++++++++++++++++++++
> include/linux/ioport.h | 2 +
> include/linux/pci-tsm.h | 33 ++++++
> kernel/resource.c | 8 ++
> 4 files changed, 279 insertions(+)
>
>diff --git a/drivers/pci/tsm.c b/drivers/pci/tsm.c
>index 5fdcd7f2e820..c8cfe89b6224 100644
>--- a/drivers/pci/tsm.c
>+++ b/drivers/pci/tsm.c
>@@ -9,11 +9,16 @@
> #define dev_fmt(fmt) "PCI/TSM: " fmt
>
> #include <linux/bitfield.h>
>+#include <linux/overflow.h>
> #include <linux/pci.h>
> #include <linux/pci-doe.h>
> #include <linux/pci-tsm.h>
>+#include <linux/range.h>
>+#include <linux/sizes.h>
>+#include <linux/slab.h>
> #include <linux/sysfs.h>
> #include <linux/tsm.h>
>+#include <linux/unaligned.h>
> #include <linux/xarray.h>
> #include "pci.h"
>
>@@ -898,3 +903,234 @@ int pci_tsm_doe_transfer(struct pci_dev *pdev, u8 type, const void *req,
> resp, resp_sz);
> }
> EXPORT_SYMBOL_GPL(pci_tsm_doe_transfer);
>+
>+static void mmio_teardown(struct pci_tsm_mmio *mmio, int nr)
>+{
>+ while (nr--) {
>+ struct resource *res = pci_tsm_mmio_resource(mmio, nr);
>+
>+ /*
>+ * Entries past the point where pci_tsm_mmio_setup() stopped
>+ * (or that were never run through setup() at all, e.g. a
>+ * caller that only wants the resolved range from
>+ * pci_tsm_mmio_alloc()) were never insert_resource()'d, so
>+ * res->parent is NULL. remove_resource() dereferences
>+ * res->parent unconditionally.
>+ */
>+ if (res->parent)
>+ remove_resource(res);
>+ }
>+}
>+
>+/**
>+ * pci_tsm_mmio_setup() - mark device MMIO as encrypted in iomem
>+ * @pdev: device owner of MMIO resources
>+ * @mmio: container of an array of resources to mark encrypted
>+ */
>+int pci_tsm_mmio_setup(struct pci_dev *pdev, struct pci_tsm_mmio *mmio)
>+{
>+ int i;
>+
>+ device_lock_assert(&pdev->dev);
>+ if (pdev->dev.driver)
>+ return -EBUSY;
>+
>+ for (i = 0; i < mmio->nr; i++) {
>+ struct resource *res = pci_tsm_mmio_resource(mmio, i);
>+ int j;
>+
>+ if (resource_size(res) == 0 || !(res->flags & IORESOURCE_MEM))
>+ break;
>+
>+ /* Only require the caller to set the range, init remainder */
>+ *res = DEFINE_RES_NAMED_DESC(res->start, resource_size(res),
>+ pci_name(pdev), IORESOURCE_MEM,
>+ IORES_DESC_ENCRYPTED);
>+
>+ for (j = 0; j < PCI_NUM_RESOURCES; j++)
>+ if (resource_contains(pci_resource_n(pdev, j), res))
>+ break;
>+
>+ /* Request is outside of device MMIO */
>+ if (j >= PCI_NUM_RESOURCES)
>+ break;
>+
>+ if (insert_resource(&encrypted_iomem_resource, res) != 0)
>+ break;
>+ }
>+
>+ if (i >= mmio->nr)
>+ return 0;
>+
>+ mmio_teardown(mmio, i);
>+
>+ return -EINVAL;
>+}
>+EXPORT_SYMBOL_GPL(pci_tsm_mmio_setup);
>+
>+void pci_tsm_mmio_teardown(struct pci_tsm_mmio *mmio)
>+{
>+ mmio_teardown(mmio, mmio->nr);
>+}
>+EXPORT_SYMBOL_GPL(pci_tsm_mmio_teardown);
>+
>+/*
>+ * PCIe ECN TEE Device Interface Security Protocol (TDISP)
>+ *
>+ * Device Interface Report data object layout as defined by PCIe r7.0 section
>+ * 11.3.11
>+ */
>+#define PCI_TSM_DEVIF_REPORT_MMIO_ATTR_MSIX_TABLE BIT(0)
>+#define PCI_TSM_DEVIF_REPORT_MMIO_ATTR_MSIX_PBA BIT(1)
>+#define PCI_TSM_DEVIF_REPORT_MMIO_ATTR_IS_NON_TEE BIT(2)
>+#define PCI_TSM_DEVIF_REPORT_MMIO_ATTR_IS_UPDATABLE BIT(3)
>+#define PCI_TSM_DEVIF_REPORT_MMIO_ATTR_RANGE_ID GENMASK(31, 16)
>+
>+/* An interface report 'pfn' is 4K in size */
>+struct pci_tsm_devif_mmio {
>+ __le64 phys;
>+ __le32 nr_pfns;
>+ __le32 attributes;
>+};
>+
>+struct pci_tsm_devif_report {
>+ __le16 interface_info;
>+ __le16 reserved;
>+ __le16 msi_x_message_control;
>+ __le16 lnr_control;
>+ __le32 tph_control;
>+ __le32 mmio_range_count;
>+ struct pci_tsm_devif_mmio mmio[];
>+};
>+
>+/**
>+ * pci_tsm_mmio_alloc() - allocate encrypted MMIO range descriptor
>+ * @pdev: device owner of MMIO ranges
>+ * @report: raw TDISP Device Interface Report bytes (PCIe r7.0 section
>+ * 11.3.11), e.g. as returned by GET_DEVICE_INTERFACE_REPORT
>+ * @report_len: length of @report in bytes
>+ *
>+ * Return: the encrypted MMIO range descriptor on success, NULL on failure
>+ *
>+ * Unlike upstream, @report is supplied directly by the caller rather than
>+ * pulled from a device evidence store, which this tree does not yet have.
>+ */
>+struct pci_tsm_mmio *pci_tsm_mmio_alloc(struct pci_dev *pdev,
>+ const void *report, size_t report_len)
>+{
>+ const struct pci_tsm_devif_report *devif_report = report;
>+ u64 reporting_bar_base, last_reporting_end;
>+ u32 mmio_range_count;
>+ int last_bar = -1;
>+ int i;
>+
>+ if (report_len < sizeof(*devif_report))
>+ return NULL;
>+
>+ mmio_range_count = __le32_to_cpu(devif_report->mmio_range_count);
>+
>+ /* check that the report is self-consistent on mmio entries */
>+ if (report_len < struct_size(devif_report, mmio, mmio_range_count))
>+ return NULL;
>+
>+ /* create pci_tsm_mmio descriptors from the report data */
>+ struct pci_tsm_mmio *mmio __free(kfree) =
>+ kzalloc_flex(*mmio, mmio, mmio_range_count);
>+ if (!mmio)
>+ return NULL;
>+
>+ for (i = 0; i < mmio_range_count; i++) {
>+ u64 range_off;
>+ struct range range;
>+ const struct pci_tsm_devif_mmio *mmio_data = &devif_report->mmio[i];
>+ struct pci_tsm_mmio_entry *entry =
>+ pci_tsm_mmio_entry(mmio, mmio->nr);
>+ u64 tsm_offset = __le64_to_cpu(mmio_data->phys);
>+ u64 size = (u64)__le32_to_cpu(mmio_data->nr_pfns) * SZ_4K;
>+ u32 attr = __le32_to_cpu(mmio_data->attributes);
>+ int bar = FIELD_GET(PCI_TSM_DEVIF_REPORT_MMIO_ATTR_RANGE_ID,
>+ attr);
>+
>+ if (bar >= PCI_STD_NUM_BARS ||
>+ !(pci_resource_flags(pdev, bar) & IORESOURCE_MEM) ||
>+ (pci_resource_flags(pdev, bar) & IORESOURCE_UNSET)) {
>+ pci_dbg(pdev, "Invalid reporting bar ID %d\n", bar);
>+ return NULL;
>+ }
>+
>+ if (last_bar > bar) {
>+ pci_dbg(pdev, "Reporting bar ID not in ascending order\n");
>+ return NULL;
>+ }
>+
>+ if (last_bar < bar) {
>+ resource_size_t mask = pci_resource_len(pdev, bar) - 1;
>+
>+ /* Transition to a new bar */
>+ last_bar = bar;
>+
>+ /*
>+ * Determine the obfuscated base of the BAR. BAR
>+ * offsets are never obfuscated.
>+ */
>+ reporting_bar_base = tsm_offset & ~mask;
>+ } else if (tsm_offset < last_reporting_end) {
>+ pci_dbg(pdev, "Reporting ranges within BAR not in ascending order\n");
>+ return NULL;
>+ }
>+
>+ /* Per spec the tsm_offset never results in overflow / underflow */
>+ last_reporting_end = tsm_offset + size;
>+ if (last_reporting_end < tsm_offset) {
>+ pci_dbg(pdev, "Reporting range overflow\n");
>+ return NULL;
>+ }
>+
>+ range_off = tsm_offset - reporting_bar_base;
>+ if (pci_resource_len(pdev, bar) < range_off + size) {
>+ pci_dbg(pdev, "Reporting range larger than BAR size\n");
>+ return NULL;
>+ }
>+
>+ range.start = pci_resource_start(pdev, bar) + range_off;
>+ range.end = range.start + size - 1;
>+
>+ /* Only record the TEE ranges for later consideration by ioremap() */
>+ if (FIELD_GET(PCI_TSM_DEVIF_REPORT_MMIO_ATTR_IS_NON_TEE,
>+ attr)) {
>+ pci_dbg(pdev, "Skipping non-TEE range, BAR%d %pra\n",
>+ bar, &range);
>+ continue;
>+ }
>+
>+ entry->res.start = range.start;
>+ entry->res.end = range.end;
>+ entry->res.flags = IORESOURCE_MEM;
>+ entry->tsm_offset = tsm_offset;
>+ mmio->nr++;
>+ }
>+
>+ return_ptr(mmio);
>+}
>+EXPORT_SYMBOL_GPL(pci_tsm_mmio_alloc);
>+
>+/**
>+ * pci_tsm_mmio_free() - free a pci_tsm_mmio instance
>+ * @pdev: device owner of MMIO ranges
>+ * @mmio: instance to free
>+ *
>+ * Returns 0 if @mmio was idle on entry, -EBUSY otherwise
>+ */
>+int pci_tsm_mmio_free(struct pci_dev *pdev, struct pci_tsm_mmio *mmio)
>+{
>+ for (int i = 0; i < mmio->nr; i++) {
>+ struct resource *res = pci_tsm_mmio_resource(mmio, i);
>+
>+ if (dev_WARN_ONCE(&pdev->dev, resource_assigned(res),
>+ "MMIO resource still assigned %pr\n", res))
>+ return -EBUSY;
>+ }
>+ kfree(mmio);
>+ return 0;
>+}
>+EXPORT_SYMBOL_GPL(pci_tsm_mmio_free);
>diff --git a/include/linux/ioport.h b/include/linux/ioport.h
>index f7930b3dfd0a..122f1eefb4b9 100644
>--- a/include/linux/ioport.h
>+++ b/include/linux/ioport.h
>@@ -144,6 +144,7 @@ enum {
> IORES_DESC_RESERVED = 7,
> IORES_DESC_SOFT_RESERVED = 8,
> IORES_DESC_CXL = 9,
>+ IORES_DESC_ENCRYPTED = 10,
> };
>
> /*
>@@ -240,6 +241,7 @@ struct resource_constraint {
> extern struct resource ioport_resource;
> extern struct resource iomem_resource;
> extern struct resource soft_reserve_resource;
>+extern struct resource encrypted_iomem_resource;
>
> extern struct resource *request_resource_conflict(struct resource *root,
> struct resource *new);
> extern int request_resource(struct resource *root, struct resource *new);
>diff --git a/include/linux/pci-tsm.h b/include/linux/pci-tsm.h
>index a6435aba03f9..e54ad057fa8e 100644
>--- a/include/linux/pci-tsm.h
>+++ b/include/linux/pci-tsm.h
>@@ -199,6 +199,34 @@ enum pci_tsm_req_scope {
> PCI_TSM_REQ_DEBUG_WRITE = 3,
> };
>
>+/**
>+ * struct pci_tsm_mmio_entry - an encrypted MMIO range
>+ * @res: MMIO address range (typically Guest Physical Address, GPA)
>+ * @tsm_offset: Host Physical Address, HPA obfuscation offset added by the TSM.
>+ * Translates report addresses to GPA.
>+ */
>+struct pci_tsm_mmio_entry {
>+ struct resource res;
>+ u64 tsm_offset;
>+};
>+
>+struct pci_tsm_mmio {
>+ int nr;
>+ struct pci_tsm_mmio_entry mmio[];
>+};
>+
>+static inline struct pci_tsm_mmio_entry *
>+pci_tsm_mmio_entry(struct pci_tsm_mmio *mmio, int idx)
>+{
>+ return &mmio->mmio[idx];
>+}
>+
>+static inline struct resource *pci_tsm_mmio_resource(struct pci_tsm_mmio *mmio,
>+ int idx)
>+{
>+ return &mmio->mmio[idx].res;
>+}
>+
> #ifdef CONFIG_PCI_TSM
> int pci_tsm_register(struct tsm_dev *tsm_dev);
> void pci_tsm_unregister(struct tsm_dev *tsm_dev);
>@@ -216,6 +244,11 @@ void pci_tsm_tdi_constructor(struct pci_dev *pdev, struct pci_tdi *tdi,
> ssize_t pci_tsm_guest_req(struct pci_dev *pdev, enum pci_tsm_req_scope
> scope,
> sockptr_t req_in, size_t in_len, sockptr_t req_out,
> size_t out_len, u64 *tsm_code);
>+int pci_tsm_mmio_setup(struct pci_dev *pdev, struct pci_tsm_mmio *mmio);
>+void pci_tsm_mmio_teardown(struct pci_tsm_mmio *mmio);
>+struct pci_tsm_mmio *pci_tsm_mmio_alloc(struct pci_dev *pdev,
>+ const void *report, size_t report_len);
>+int pci_tsm_mmio_free(struct pci_dev *pdev, struct pci_tsm_mmio *mmio);
> #else
> static inline int pci_tsm_register(struct tsm_dev *tsm_dev)
> {
>diff --git a/kernel/resource.c b/kernel/resource.c
>index cfc1a00e86aa..5500f7828b2d 100644
>--- a/kernel/resource.c
>+++ b/kernel/resource.c
>@@ -56,6 +56,14 @@ struct resource soft_reserve_resource = {
> .flags = IORESOURCE_MEM,
> };
>
>+struct resource encrypted_iomem_resource = {
>+ .name = "Encrypted MMIO",
>+ .start = 0,
>+ .end = -1,
>+ .desc = IORES_DESC_ENCRYPTED,
>+ .flags = IORESOURCE_MEM,
>+};
>+
> static DEFINE_RWLOCK(resource_lock);
>
> /*
>
--- Thanks!
"I'm not a very positive person" - Linus torvalds
^ permalink raw reply [flat|nested] 6+ messages in thread
* [RFC PATCH 2/4] PCI/CXL: Populate and insert/remove pdev->coh_resource[]
2026-10-05 7:02 [RFC PATCH 0/4] PCI/TSM: Resolve TDISP coherent (CXL) ranges from precommitted HDM decoders Ankit Agrawal
2026-10-05 7:02 ` [RFC PATCH 1/4] PCI/TSM: Create MMIO descriptors via TDISP Report Ankit Agrawal
@ 2026-10-05 7:02 ` Ankit Agrawal
2026-10-05 7:02 ` [RFC PATCH 3/4] PCI/TSM: Derive the coherent-range IPA from CXL Ankit Agrawal
2026-10-05 7:02 ` [RFC PATCH 4/4] PCI/TSM: Support multiple coherent ranges via coh_idx Ankit Agrawal
3 siblings, 0 replies; 6+ messages in thread
From: Ankit Agrawal @ 2026-10-05 7:02 UTC (permalink / raw)
To: linux-cxl, linux-pci, linux-kernel
Cc: dave, jic23, dave.jiang, alison.schofield, vishal.l.verma, jgg,
aneesh.kumar, aik, yilun.xu, iweiny, ming.li, icheng, bhelgaas,
smadhavan, ankita, ilpo.jarvinen, Smita.KoralahalliChannabasappa,
andriy.shevchenko, linux-coco
On systems with precommitted HDM decoders, the firmware programs
the decoders before any software runs. This range also stays constant
over the runs.
Snapshot precommitted CXL HDM decoder ranges into a new
pdev->coh_resource[] array, so later code has a known-good source for
a device coherent (CXL) windows without re-scanning decoder
registers.
Signed-off-by: Ankit Agrawal <ankita@nvidia.com>
Assisted-by: Claude:sonnet-5
---
drivers/cxl/core/resource.c | 52 ++++++++++++++++++++++++++++++++++++-
include/linux/pci.h | 14 ++++++++++
2 files changed, 65 insertions(+), 1 deletion(-)
diff --git a/drivers/cxl/core/resource.c b/drivers/cxl/core/resource.c
index 3422139ae3ab..4d2082c8e321 100644
--- a/drivers/cxl/core/resource.c
+++ b/drivers/cxl/core/resource.c
@@ -404,6 +404,46 @@ static struct cxl_hdm_info *cxl_pci_hdm_read_info(struct pci_dev *pdev,
return ERR_PTR(pcibios_err_to_errno(rc));
}
+/*
+ * Snapshot committed decoders into pdev->coh_resource[] which is called
+ * once per pdev before any driver can bind, so a populated entry is
+ * always firmware-committed and not driver-committed.
+ *
+ * Insert each entry into iomem_resource for /proc/iomem visibility and
+ * conflict detection.
+ */
+static void cxl_populate_coh_resource(struct pci_dev *pdev,
+ struct cxl_hdm_info *info)
+{
+ int count = min(info->decoder_count, PCI_CXL_MAX_COHERENT_RANGES);
+
+ for (int i = 0; i < count; i++) {
+ struct cxl_decoder_config *config = &info->settings[i].config;
+ struct resource *res = &pdev->coh_resource[i];
+ struct resource *conflict;
+
+ if (!(config->flags & CXL_DECODER_F_ENABLE))
+ continue;
+
+ *res = DEFINE_RES_NAMED_DESC(config->hpa_range.start,
+ range_len(&config->hpa_range),
+ "CXL coherent memory",
+ IORESOURCE_MEM, IORES_DESC_NONE);
+
+ conflict = insert_resource_conflict(&iomem_resource, res);
+ if (conflict) {
+ pci_warn(pdev,
+ "CXL coherent range: decoder%d %pR conflicts with %s %pR, leaving unpopulated\n",
+ i, res, conflict->name, conflict);
+ memset(res, 0, sizeof(*res));
+ continue;
+ }
+
+ pci_info(pdev, "CXL coherent range: decoder%d precommitted, populated %pR\n",
+ i, res);
+ }
+}
+
static int __pci_cxl_hdm_cache_init(struct pci_dev *pdev)
{
struct cxl_register_map map = { };
@@ -442,8 +482,10 @@ static int __pci_cxl_hdm_cache_init(struct pci_dev *pdev)
struct cxl_hdm_info *info __free(kfree) = read_info;
guard(rwsem_write)(&cxl_rwsem.dpa);
/* Another initializer may have published while we read MMIO. */
- if (!pdev->hdm)
+ if (!pdev->hdm) {
+ cxl_populate_coh_resource(pdev, info);
pdev->hdm = no_free_ptr(info);
+ }
return 0;
}
@@ -466,6 +508,14 @@ void pci_cxl_hdm_cache_release(struct pci_dev *pdev)
info = pdev->hdm;
/* Unpublish before freeing so subsequent readers cannot use stale state. */
pdev->hdm = NULL;
+
+ for (int i = 0; i < PCI_CXL_MAX_COHERENT_RANGES; i++) {
+ struct resource *res = &pdev->coh_resource[i];
+
+ if (res->parent)
+ remove_resource(res);
+ }
+
kfree(info);
}
diff --git a/include/linux/pci.h b/include/linux/pci.h
index b00f5f38f77d..ce724d067fd2 100644
--- a/include/linux/pci.h
+++ b/include/linux/pci.h
@@ -148,6 +148,13 @@ enum {
DEVICE_COUNT_RESOURCE = PCI_NUM_RESOURCES,
};
+/*
+ * Max CXL HDM decoders tracked per endpoint in pci_dev->coh_resource[].
+ * CXL 3.1 spec section 8.2.4.20.1 - CXL HDM Decoder Capability Register,
+ * mention a max of 10 decoders for a CXL device.
+ */
+#define PCI_CXL_MAX_COHERENT_RANGES 10
+
/**
* enum pci_interrupt_pin - PCI INTx interrupt values
* @PCI_INTERRUPT_UNKNOWN: Unknown or unassigned interrupt
@@ -572,6 +579,13 @@ struct pci_dev {
#endif
#ifdef CONFIG_CXL_RESET
struct cxl_hdm_info *hdm; /* CXL HDM decoder state */
+ /*
+ * Precommitted CXL HDM decoder coherent ranges, snapshotted before
+ * any driver binds. One entry per decoder index. Only written by
+ * pci_cxl_hdm_cache_init()/_release(), outside any bound driver's
+ * read window.
+ */
+ struct resource coh_resource[PCI_CXL_MAX_COHERENT_RANGES];
#endif
#ifdef CONFIG_PCI_NPEM
struct npem *npem; /* Native PCIe Enclosure Management */
--
2.43.0
^ permalink raw reply [flat|nested] 6+ messages in thread* [RFC PATCH 3/4] PCI/TSM: Derive the coherent-range IPA from CXL
2026-10-05 7:02 [RFC PATCH 0/4] PCI/TSM: Resolve TDISP coherent (CXL) ranges from precommitted HDM decoders Ankit Agrawal
2026-10-05 7:02 ` [RFC PATCH 1/4] PCI/TSM: Create MMIO descriptors via TDISP Report Ankit Agrawal
2026-10-05 7:02 ` [RFC PATCH 2/4] PCI/CXL: Populate and insert/remove pdev->coh_resource[] Ankit Agrawal
@ 2026-10-05 7:02 ` Ankit Agrawal
2026-10-05 7:02 ` [RFC PATCH 4/4] PCI/TSM: Support multiple coherent ranges via coh_idx Ankit Agrawal
3 siblings, 0 replies; 6+ messages in thread
From: Ankit Agrawal @ 2026-10-05 7:02 UTC (permalink / raw)
To: linux-cxl, linux-pci, linux-kernel
Cc: dave, jic23, dave.jiang, alison.schofield, vishal.l.verma, jgg,
aneesh.kumar, aik, yilun.xu, iweiny, ming.li, icheng, bhelgaas,
smadhavan, ankita, ilpo.jarvinen, Smita.KoralahalliChannabasappa,
andriy.shevchenko, linux-coco
The DevIf report range with range ID 0xffff describes the device's
coherent (CXL) memory window and unlike every other reported range
carries no BAR number. Recognize
PCI_TSM_DEVIF_REPORT_MMIO_RANGE_ID_COHERENT and recover that window's
address from the CXL side of the device instead of treating it as a
BAR-relative offset.
pci_tsm_coherent_range() resolves the window from pdev->coh_resource[],
snapshotted once with the committed CXL HDM decoders range. The
range to IPA conversion is applied onto range_base/range_len using
pdev->coh_resource[]. The ascending BAR order checks stay on the BAR
path only. The coherent range is required to be the last report entry
and is not part of the BAR sequence.
An HDM decoded window lives in the CXL Fixed Memory Window and is not
part of a BAR. So by construction, no pci_dev resource contains it. So
the function pci_tsm_coh_resource_contains() validates the range against
pdev->coh_resource[] instead; the same known-good snapshot
pci_tsm_coherent_range() already trusts.
Signed-off-by: Ankit Agrawal <ankita@nvidia.com>
Assisted-by: Claude:sonnet-5
---
drivers/pci/tsm.c | 192 ++++++++++++++++++++++++++++++++--------
include/linux/pci-tsm.h | 4 +
2 files changed, 160 insertions(+), 36 deletions(-)
diff --git a/drivers/pci/tsm.c b/drivers/pci/tsm.c
index c8cfe89b6224..1d736b606a7c 100644
--- a/drivers/pci/tsm.c
+++ b/drivers/pci/tsm.c
@@ -904,6 +904,41 @@ int pci_tsm_doe_transfer(struct pci_dev *pdev, u8 type, const void *req,
}
EXPORT_SYMBOL_GPL(pci_tsm_doe_transfer);
+#ifdef CONFIG_CXL_RESET
+/*
+ * pci_tsm_coh_resource_contains() - check a coherent range against the
+ * PCI-core coh_resource[] snapshot
+ * @pdev: device owner of @res
+ * @res: candidate coherent MMIO range to validate
+ *
+ * The coherent (CXL.mem) range isn't BAR-backed. So pci_resource_n()
+ * containment doesn't apply. Check against pdev->coh_resource[] instead.
+ *
+ * Return: true if @res falls within a populated coh_resource[] entry.
+ */
+static bool pci_tsm_coh_resource_contains(struct pci_dev *pdev,
+ const struct resource *res)
+{
+ int i;
+
+ for (i = 0; i < PCI_CXL_MAX_COHERENT_RANGES; i++) {
+ struct resource *coh_res = &pdev->coh_resource[i];
+
+ if ((coh_res->flags & IORESOURCE_MEM) &&
+ resource_contains(coh_res, res))
+ return true;
+ }
+
+ return false;
+}
+#else
+static bool pci_tsm_coh_resource_contains(struct pci_dev *pdev,
+ const struct resource *res)
+{
+ return false;
+}
+#endif /* CONFIG_CXL_RESET */
+
static void mmio_teardown(struct pci_tsm_mmio *mmio, int nr)
{
while (nr--) {
@@ -936,7 +971,9 @@ int pci_tsm_mmio_setup(struct pci_dev *pdev, struct pci_tsm_mmio *mmio)
return -EBUSY;
for (i = 0; i < mmio->nr; i++) {
- struct resource *res = pci_tsm_mmio_resource(mmio, i);
+ struct pci_tsm_mmio_entry *entry = pci_tsm_mmio_entry(mmio, i);
+ struct resource *res = &entry->res;
+ bool coherent = entry->flags & PCI_TSM_MMIO_F_COHERENT;
int j;
if (resource_size(res) == 0 || !(res->flags & IORESOURCE_MEM))
@@ -947,13 +984,22 @@ int pci_tsm_mmio_setup(struct pci_dev *pdev, struct pci_tsm_mmio *mmio)
pci_name(pdev), IORESOURCE_MEM,
IORES_DESC_ENCRYPTED);
- for (j = 0; j < PCI_NUM_RESOURCES; j++)
- if (resource_contains(pci_resource_n(pdev, j), res))
+ /*
+ * The coherent (CXL.mem) range isn't BAR-backed, so check it
+ * against pdev->coh_resource[] instead of pci_resource_n().
+ */
+ if (coherent) {
+ if (!pci_tsm_coh_resource_contains(pdev, res))
break;
+ } else {
+ for (j = 0; j < PCI_NUM_RESOURCES; j++)
+ if (resource_contains(pci_resource_n(pdev, j), res))
+ break;
- /* Request is outside of device MMIO */
- if (j >= PCI_NUM_RESOURCES)
- break;
+ /* Request is outside of device MMIO */
+ if (j >= PCI_NUM_RESOURCES)
+ break;
+ }
if (insert_resource(&encrypted_iomem_resource, res) != 0)
break;
@@ -985,6 +1031,7 @@ EXPORT_SYMBOL_GPL(pci_tsm_mmio_teardown);
#define PCI_TSM_DEVIF_REPORT_MMIO_ATTR_IS_NON_TEE BIT(2)
#define PCI_TSM_DEVIF_REPORT_MMIO_ATTR_IS_UPDATABLE BIT(3)
#define PCI_TSM_DEVIF_REPORT_MMIO_ATTR_RANGE_ID GENMASK(31, 16)
+#define PCI_TSM_DEVIF_REPORT_MMIO_RANGE_ID_COHERENT 0xffff
/* An interface report 'pfn' is 4K in size */
struct pci_tsm_devif_mmio {
@@ -1003,6 +1050,44 @@ struct pci_tsm_devif_report {
struct pci_tsm_devif_mmio mmio[];
};
+#ifdef CONFIG_CXL_RESET
+/*
+ * pci_tsm_coherent_range() - resolve the device's coherent CXL window
+ * @pdev: device owner of the reported ranges
+ * @out_base: host physical (guest IPA) base of the coherent window
+ * @out_size: size of the coherent window
+ *
+ * The coherent range carries no BAR number, so its address comes from
+ * pdev->coh_resource[], a pre-driver-bind snapshot of committed CXL HDM
+ * decoders.
+ *
+ * Return: 0 with *@out_base / *@out_size set from the first populated entry,
+ * or -ENODEV if no entry is populated.
+ */
+static int pci_tsm_coherent_range(struct pci_dev *pdev, u64 *out_base,
+ u64 *out_size)
+{
+ for (int i = 0; i < PCI_CXL_MAX_COHERENT_RANGES; i++) {
+ struct resource *res = &pdev->coh_resource[i];
+
+ if (!(res->flags & IORESOURCE_MEM))
+ continue;
+
+ *out_base = res->start;
+ *out_size = resource_size(res);
+ return 0;
+ }
+
+ return -ENODEV;
+}
+#else
+static int pci_tsm_coherent_range(struct pci_dev *pdev, u64 *out_base,
+ u64 *out_size)
+{
+ return -ENODEV;
+}
+#endif /* CONFIG_CXL_RESET */
+
/**
* pci_tsm_mmio_alloc() - allocate encrypted MMIO range descriptor
* @pdev: device owner of MMIO ranges
@@ -1040,7 +1125,7 @@ struct pci_tsm_mmio *pci_tsm_mmio_alloc(struct pci_dev *pdev,
return NULL;
for (i = 0; i < mmio_range_count; i++) {
- u64 range_off;
+ u64 range_base, range_len, range_off;
struct range range;
const struct pci_tsm_devif_mmio *mmio_data = &devif_report->mmio[i];
struct pci_tsm_mmio_entry *entry =
@@ -1050,33 +1135,58 @@ struct pci_tsm_mmio *pci_tsm_mmio_alloc(struct pci_dev *pdev,
u32 attr = __le32_to_cpu(mmio_data->attributes);
int bar = FIELD_GET(PCI_TSM_DEVIF_REPORT_MMIO_ATTR_RANGE_ID,
attr);
+ bool coherent =
+ bar == PCI_TSM_DEVIF_REPORT_MMIO_RANGE_ID_COHERENT;
- if (bar >= PCI_STD_NUM_BARS ||
- !(pci_resource_flags(pdev, bar) & IORESOURCE_MEM) ||
- (pci_resource_flags(pdev, bar) & IORESOURCE_UNSET)) {
- pci_dbg(pdev, "Invalid reporting bar ID %d\n", bar);
- return NULL;
- }
-
- if (last_bar > bar) {
- pci_dbg(pdev, "Reporting bar ID not in ascending order\n");
- return NULL;
- }
-
- if (last_bar < bar) {
- resource_size_t mask = pci_resource_len(pdev, bar) - 1;
-
- /* Transition to a new bar */
- last_bar = bar;
+ if (coherent) {
+ if (i != mmio_range_count - 1) {
+ pci_dbg(pdev, "Coherent reporting range is not last\n");
+ return NULL;
+ }
/*
- * Determine the obfuscated base of the BAR. BAR
- * offsets are never obfuscated.
+ * No BAR names the coherent range, so fail closed if
+ * neither a CXL region nor a committed HDM decoder
+ * resolves its address.
*/
- reporting_bar_base = tsm_offset & ~mask;
- } else if (tsm_offset < last_reporting_end) {
- pci_dbg(pdev, "Reporting ranges within BAR not in ascending order\n");
- return NULL;
+ if (pci_tsm_coherent_range(pdev, &range_base,
+ &range_len)) {
+ pci_dbg(pdev, "No CXL region or committed HDM decoder for coherent reporting range\n");
+ return NULL;
+ }
+
+ /* Coherent range is last and not part of the BAR sequence. */
+ } else {
+ if (bar >= PCI_STD_NUM_BARS ||
+ !(pci_resource_flags(pdev, bar) & IORESOURCE_MEM) ||
+ (pci_resource_flags(pdev, bar) & IORESOURCE_UNSET)) {
+ pci_dbg(pdev, "Invalid reporting bar ID %d\n", bar);
+ return NULL;
+ }
+
+ if (last_bar > bar) {
+ pci_dbg(pdev, "Reporting bar ID not in ascending order\n");
+ return NULL;
+ }
+
+ if (last_bar < bar) {
+ resource_size_t mask = pci_resource_len(pdev, bar) - 1;
+
+ /* Transition to a new bar */
+ last_bar = bar;
+
+ /*
+ * Determine the obfuscated base of the BAR. BAR
+ * offsets are never obfuscated.
+ */
+ reporting_bar_base = tsm_offset & ~mask;
+ } else if (tsm_offset < last_reporting_end) {
+ pci_dbg(pdev, "Reporting ranges within BAR not in ascending order\n");
+ return NULL;
+ }
+
+ range_base = pci_resource_start(pdev, bar);
+ range_len = pci_resource_len(pdev, bar);
}
/* Per spec the tsm_offset never results in overflow / underflow */
@@ -1086,20 +1196,28 @@ struct pci_tsm_mmio *pci_tsm_mmio_alloc(struct pci_dev *pdev,
return NULL;
}
- range_off = tsm_offset - reporting_bar_base;
- if (pci_resource_len(pdev, bar) < range_off + size) {
- pci_dbg(pdev, "Reporting range larger than BAR size\n");
+ /*
+ * Use tsm_offset directly for the coherent range instead of
+ * masking: an HDM decoder's size - unlike a BAR's - need not be
+ * a power of two. So masking could wrap an out-of-range
+ * offset back in-bounds. The bounds check below still catches it.
+ */
+ range_off = coherent ? tsm_offset :
+ tsm_offset - reporting_bar_base;
+ if (range_len < range_off + size) {
+ pci_dbg(pdev, "Reporting range larger than %s size\n",
+ coherent ? "coherent range" : "BAR");
return NULL;
}
- range.start = pci_resource_start(pdev, bar) + range_off;
+ range.start = range_base + range_off;
range.end = range.start + size - 1;
/* Only record the TEE ranges for later consideration by ioremap() */
if (FIELD_GET(PCI_TSM_DEVIF_REPORT_MMIO_ATTR_IS_NON_TEE,
attr)) {
- pci_dbg(pdev, "Skipping non-TEE range, BAR%d %pra\n",
- bar, &range);
+ pci_dbg(pdev, "Skipping non-TEE range, %s %pra\n",
+ coherent ? "coherent range" : "BAR", &range);
continue;
}
@@ -1107,6 +1225,8 @@ struct pci_tsm_mmio *pci_tsm_mmio_alloc(struct pci_dev *pdev,
entry->res.end = range.end;
entry->res.flags = IORESOURCE_MEM;
entry->tsm_offset = tsm_offset;
+ if (coherent)
+ entry->flags |= PCI_TSM_MMIO_F_COHERENT;
mmio->nr++;
}
diff --git a/include/linux/pci-tsm.h b/include/linux/pci-tsm.h
index e54ad057fa8e..90bd8d1dd729 100644
--- a/include/linux/pci-tsm.h
+++ b/include/linux/pci-tsm.h
@@ -1,6 +1,7 @@
/* SPDX-License-Identifier: GPL-2.0 */
#ifndef __PCI_TSM_H
#define __PCI_TSM_H
+#include <linux/bits.h>
#include <linux/mutex.h>
#include <linux/pci.h>
#include <linux/sockptr.h>
@@ -204,10 +205,13 @@ enum pci_tsm_req_scope {
* @res: MMIO address range (typically Guest Physical Address, GPA)
* @tsm_offset: Host Physical Address, HPA obfuscation offset added by the TSM.
* Translates report addresses to GPA.
+ * @flags: PCI_TSM_MMIO_F_* attributes for the range
*/
+#define PCI_TSM_MMIO_F_COHERENT BIT(0)
struct pci_tsm_mmio_entry {
struct resource res;
u64 tsm_offset;
+ u32 flags;
};
struct pci_tsm_mmio {
--
2.43.0
^ permalink raw reply [flat|nested] 6+ messages in thread* [RFC PATCH 4/4] PCI/TSM: Support multiple coherent ranges via coh_idx
2026-10-05 7:02 [RFC PATCH 0/4] PCI/TSM: Resolve TDISP coherent (CXL) ranges from precommitted HDM decoders Ankit Agrawal
` (2 preceding siblings ...)
2026-10-05 7:02 ` [RFC PATCH 3/4] PCI/TSM: Derive the coherent-range IPA from CXL Ankit Agrawal
@ 2026-10-05 7:02 ` Ankit Agrawal
3 siblings, 0 replies; 6+ messages in thread
From: Ankit Agrawal @ 2026-10-05 7:02 UTC (permalink / raw)
To: linux-cxl, linux-pci, linux-kernel
Cc: dave, jic23, dave.jiang, alison.schofield, vishal.l.verma, jgg,
aneesh.kumar, aik, yilun.xu, iweiny, ming.li, icheng, bhelgaas,
smadhavan, ankita, ilpo.jarvinen, Smita.KoralahalliChannabasappa,
andriy.shevchenko, linux-coco
Devices with multiple decoders can report more than one coherent
window in a single interface report. Generalize the single
0xffff sentinel into a small id space:
PCI_TSM_DEVIF_REPORT_MMIO_RANGE_ID_TO_COH_IDX() treats any range id in
the top PCI_CXL_MAX_COHERENT_RANGES values (counting down from ~0) as
a coherent range index coh_idx. This keeps id 0xffff (~0) backward
compatible with the single-range case (coh_idx 0) while allowing ~1,
~2, and so on for additional windows.
Only id 0xffff (~0) is spec-defined: Arm RME System Architecture
(DEN0129) section B2.3.5.2 reserves range ID 0xFFFF for a device's
region range. ~1, ~2, and so on are this tree's own extension of that
idea, invented to cover multi-range reporting.
Use coh_idx through pci_tsm_coherent_range() to select the coh_idx
entry of pdev->coh_resource[], the precommitted-decoder snapshot and
relax pci_tsm_mmio_alloc() to accept more than one coherent region.
The coherent-range entries follow all BAR entries in the report
and allow multiple trailing coherent ranges.
Signed-off-by: Ankit Agrawal <ankita@nvidia.com>
Assisted-by: Claude:sonnet-5
---
drivers/pci/tsm.c | 78 ++++++++++++++++++++++++++++-------------------
1 file changed, 46 insertions(+), 32 deletions(-)
diff --git a/drivers/pci/tsm.c b/drivers/pci/tsm.c
index 1d736b606a7c..b4f18a7c7187 100644
--- a/drivers/pci/tsm.c
+++ b/drivers/pci/tsm.c
@@ -1031,7 +1031,14 @@ EXPORT_SYMBOL_GPL(pci_tsm_mmio_teardown);
#define PCI_TSM_DEVIF_REPORT_MMIO_ATTR_IS_NON_TEE BIT(2)
#define PCI_TSM_DEVIF_REPORT_MMIO_ATTR_IS_UPDATABLE BIT(3)
#define PCI_TSM_DEVIF_REPORT_MMIO_ATTR_RANGE_ID GENMASK(31, 16)
-#define PCI_TSM_DEVIF_REPORT_MMIO_RANGE_ID_COHERENT 0xffff
+
+/*
+ * Only id ~0 (0xffff) is spec-defined (Arm RME DEN0129 section B2.3.5.2);
+ * ~1, ~2, etc. extend it for multi-range reporting: id ~N selects coherent
+ * window N, keeping coherent ids disjoint from BAR ids
+ * (0..PCI_STD_NUM_BARS-1) without a separate flag.
+ */
+#define PCI_TSM_DEVIF_REPORT_MMIO_RANGE_ID_TO_COH_IDX(id) ((u16)~(id))
/* An interface report 'pfn' is 4K in size */
struct pci_tsm_devif_mmio {
@@ -1052,37 +1059,39 @@ struct pci_tsm_devif_report {
#ifdef CONFIG_CXL_RESET
/*
- * pci_tsm_coherent_range() - resolve the device's coherent CXL window
+ * pci_tsm_coherent_range() - resolve the device coherent CXL window
* @pdev: device owner of the reported ranges
+ * @coh_idx: which coherent range to resolve, per
+ * %PCI_TSM_DEVIF_REPORT_MMIO_RANGE_ID_TO_COH_IDX()
* @out_base: host physical (guest IPA) base of the coherent window
* @out_size: size of the coherent window
*
- * The coherent range carries no BAR number, so its address comes from
- * pdev->coh_resource[], a pre-driver-bind snapshot of committed CXL HDM
- * decoders.
+ * Each @coh_idx names one of the device coherent (CXL) windows. It carries
+ * no BAR number and so its address comes from pdev->coh_resource[@coh_idx],
+ * which is a snapshot of committed CXL HDM decoders regions.
*
- * Return: 0 with *@out_base / *@out_size set from the first populated entry,
- * or -ENODEV if no entry is populated.
+ * Return: 0 with *@out_base / *@out_size set, or -ENODEV if @coh_idx is out
+ * of range or has no populated coh_resource[] entry.
*/
-static int pci_tsm_coherent_range(struct pci_dev *pdev, u64 *out_base,
- u64 *out_size)
+static int pci_tsm_coherent_range(struct pci_dev *pdev, unsigned int coh_idx,
+ u64 *out_base, u64 *out_size)
{
- for (int i = 0; i < PCI_CXL_MAX_COHERENT_RANGES; i++) {
- struct resource *res = &pdev->coh_resource[i];
+ struct resource *res;
- if (!(res->flags & IORESOURCE_MEM))
- continue;
+ if (coh_idx >= PCI_CXL_MAX_COHERENT_RANGES)
+ return -ENODEV;
- *out_base = res->start;
- *out_size = resource_size(res);
- return 0;
- }
+ res = &pdev->coh_resource[coh_idx];
+ if (!(res->flags & IORESOURCE_MEM))
+ return -ENODEV;
- return -ENODEV;
+ *out_base = res->start;
+ *out_size = resource_size(res);
+ return 0;
}
#else
-static int pci_tsm_coherent_range(struct pci_dev *pdev, u64 *out_base,
- u64 *out_size)
+static int pci_tsm_coherent_range(struct pci_dev *pdev, unsigned int coh_idx,
+ u64 *out_base, u64 *out_size)
{
return -ENODEV;
}
@@ -1107,6 +1116,7 @@ struct pci_tsm_mmio *pci_tsm_mmio_alloc(struct pci_dev *pdev,
u64 reporting_bar_base, last_reporting_end;
u32 mmio_range_count;
int last_bar = -1;
+ bool seen_coherent = false;
int i;
if (report_len < sizeof(*devif_report))
@@ -1135,28 +1145,32 @@ struct pci_tsm_mmio *pci_tsm_mmio_alloc(struct pci_dev *pdev,
u32 attr = __le32_to_cpu(mmio_data->attributes);
int bar = FIELD_GET(PCI_TSM_DEVIF_REPORT_MMIO_ATTR_RANGE_ID,
attr);
- bool coherent =
- bar == PCI_TSM_DEVIF_REPORT_MMIO_RANGE_ID_COHERENT;
+ unsigned int coh_idx =
+ PCI_TSM_DEVIF_REPORT_MMIO_RANGE_ID_TO_COH_IDX(bar);
+ bool coherent = coh_idx < PCI_CXL_MAX_COHERENT_RANGES;
if (coherent) {
- if (i != mmio_range_count - 1) {
- pci_dbg(pdev, "Coherent reporting range is not last\n");
- return NULL;
- }
+ seen_coherent = true;
/*
- * No BAR names the coherent range, so fail closed if
- * neither a CXL region nor a committed HDM decoder
- * resolves its address.
+ * Fail closed if pdev->coh_resource[coh_idx] isn't
+ * populated: nothing else in the report identifies
+ * the window's address.
*/
- if (pci_tsm_coherent_range(pdev, &range_base,
+ if (pci_tsm_coherent_range(pdev, coh_idx, &range_base,
&range_len)) {
- pci_dbg(pdev, "No CXL region or committed HDM decoder for coherent reporting range\n");
+ pci_dbg(pdev, "No populated coh_resource[] entry for coherent range %u\n",
+ coh_idx);
return NULL;
}
- /* Coherent range is last and not part of the BAR sequence. */
+ /* Coherent ranges are not part of the BAR sequence. */
} else {
+ if (seen_coherent) {
+ pci_dbg(pdev, "BAR reporting range follows a coherent range\n");
+ return NULL;
+ }
+
if (bar >= PCI_STD_NUM_BARS ||
!(pci_resource_flags(pdev, bar) & IORESOURCE_MEM) ||
(pci_resource_flags(pdev, bar) & IORESOURCE_UNSET)) {
--
2.43.0
^ permalink raw reply [flat|nested] 6+ messages in thread