* [RFC PATCH v2 1/3] cxl/mem: Separate provider registration from region attachment
2026-10-07 9:05 [RFC PATCH v2 0/3] cxl: Provide explicit RAM region creation for Type-2 providers Richard Cheng
@ 2026-10-07 9:05 ` Richard Cheng
2026-10-07 9:05 ` [RFC PATCH v2 2/3] cxl/mem: Add explicit RAM region creation for providers Richard Cheng
2026-10-07 9:05 ` [RFC PATCH v2 3/3] cxl/test: Exercise explicit Type-2 RAM region creation Richard Cheng
2 siblings, 0 replies; 4+ messages in thread
From: Richard Cheng @ 2026-10-07 9:05 UTC (permalink / raw)
To: jic23, dave, dave.jiang, alison.schofield, vishal.l.verma,
iweiny, ming.li, icheng
Cc: kaihengf, kobak, newtonl, kristinc, linux-cxl, linux-kernel
devm_cxl_probe_mem() requires a committed region when registering a
provider-owned memdev. If FW has not created a region, registration
fails before the provider can use the endpoint topology to provision
one.
Make the attachment probe callback optional while retaining a non-NULL
descriptor to preserve provider ownership and handling of CXL link loss.
Install endpoint cleanup for FW-discovered provider regions even when
the provider never requests attachment.
This establishes the registration and attachment APIs needed for
explicit region provisioning without introducing region creation or
allocation policy.
Signed-off-by: Richard Cheng <icheng@nvidia.com>
---
Changelog:
v1 -> v2:
- Rework the patch into a registration/attachment refactor, following
Alejandro's request to separate region provisioning from memdev
registration.
- Add devm_cxl_register_mem() to establish a provider-owned memdev and
its endpoint without requiring a comitted region.
- Add devm_cxl_attach_mem_region() to obtain the HPA range of an
existing committed, single-target region after registration.
- Make the region probe callback optional while preserving provider
ownership, synchronous endpoint setup, and hdling of CXL link loss.
---
drivers/cxl/core/region.c | 146 ++++++++++++++++++++++++++------------
drivers/cxl/cxlmem.h | 22 ++++--
drivers/cxl/mem.c | 69 +++++++++++++++++-
include/cxl/cxl.h | 2 +
4 files changed, 188 insertions(+), 51 deletions(-)
diff --git a/drivers/cxl/core/region.c b/drivers/cxl/core/region.c
index 5ef0ca0694ff..7a64a730587d 100644
--- a/drivers/cxl/core/region.c
+++ b/drivers/cxl/core/region.c
@@ -4111,67 +4111,123 @@ static int first_mapped_decoder(struct device *dev, const void *data)
return 0;
}
-/*
- * Runs in cxl_mem_probe context after successful endpoint probe, assumes the
- * simple case of single mapped decoder per memdev.
- */
-int cxl_memdev_attach_region(struct cxl_memdev *cxlmd)
+static int unregister_memdev_region(struct device *dev, void *data)
{
- struct cxl_attach_region *attach =
- container_of(cxlmd->attach, typeof(*attach), attach);
- struct cxl_port *endpoint = cxlmd->endpoint;
struct cxl_endpoint_decoder *cxled;
struct cxl_region *cxlr;
- int rc;
- /* hold endpoint lock to setup autoremove of the region */
+ if (!is_endpoint_decoder(dev))
+ return 0;
+
+ cxled = to_cxl_endpoint_decoder(dev);
+ scoped_guard(rwsem_read, &cxl_rwsem.region) {
+ cxlr = cxled->cxld.region;
+ if (!cxlr)
+ return 0;
+ get_device(&cxlr->dev);
+ }
+
+ /* Unregistration needs the region write lock. */
+ endpoint_unregister_region(cxlr);
+ return 0;
+}
+
+static void endpoint_unregister_regions(void *data)
+{
+ struct cxl_port *endpoint = data;
+
+ device_for_each_child(&endpoint->dev, NULL, unregister_memdev_region);
+}
+
+int cxl_memdev_setup_region_cleanup(struct cxl_memdev *cxlmd)
+{
+ struct cxl_port *endpoint = cxlmd->endpoint;
+
+ device_lock_assert(&cxlmd->dev);
+ if (IS_ERR_OR_NULL(endpoint))
+ return -ENXIO;
+
guard(device)(&endpoint->dev);
- if (!endpoint->dev.driver)
+ if (!endpoint->dev.driver || endpoint->dead)
return -ENXIO;
- guard(rwsem_read)(&cxl_rwsem.region);
- guard(rwsem_read)(&cxl_rwsem.dpa);
/*
- * TODO auto-instantiate a region, for now assume this will find an
- * auto-region
+ * Endpoint probe may discover provider-owned firmware regions even if
+ * the provider never requests their HPA range. Run before decoder
+ * teardown so those regions are unregistered, not just detached.
*/
- struct device *dev __free(put_device) =
- device_find_child(&endpoint->dev, NULL, first_mapped_decoder);
-
- if (!dev) {
- dev_dbg(cxlmd->cxlds->dev, "no region found for memdev %s\n",
- dev_name(&cxlmd->dev));
- return -ENXIO;
- }
+ return devm_add_action_or_reset(&endpoint->dev,
+ endpoint_unregister_regions, endpoint);
+}
+EXPORT_SYMBOL_FOR_MODULES(cxl_memdev_setup_region_cleanup, "cxl_mem");
- cxled = to_cxl_endpoint_decoder(dev);
- cxlr = cxled->cxld.region;
+/* Caller holds the memdev lock; attach to a single mapped decoder. */
+int cxl_memdev_attach_region(struct cxl_memdev *cxlmd, struct range *hpa_range)
+{
+ struct cxl_port *endpoint = cxlmd->endpoint;
+ struct cxl_endpoint_decoder *cxled;
+ struct cxl_region *cxlr;
+ int rc;
- if (cxlr->params.state < CXL_CONFIG_COMMIT) {
- dev_dbg(cxlmd->cxlds->dev,
- "region %s not committed for memdev %s\n",
- dev_name(&cxlr->dev), dev_name(&cxlmd->dev));
+ device_lock_assert(&cxlmd->dev);
+ if (IS_ERR_OR_NULL(endpoint))
return -ENXIO;
- }
- if (cxlr->params.nr_targets > 1) {
- dev_dbg(cxlmd->cxlds->dev,
- "Only attach to local non-interleaved region\n");
+ /* hold endpoint lock to setup autoremove of the region */
+ guard(device)(&endpoint->dev);
+ if (!endpoint->dev.driver || endpoint->dead)
return -ENXIO;
- }
- /* Only teardown regions that pass validation, ignore the rest */
- get_device(&cxlr->dev);
- rc = devm_add_action_or_reset(&endpoint->dev,
- endpoint_unregister_region, cxlr);
- if (rc)
- return rc;
+ scoped_guard(rwsem_read, &cxl_rwsem.region) {
+ guard(rwsem_read)(&cxl_rwsem.dpa);
- attach->hpa_range = (struct range) {
- .start = cxlr->params.res->start,
- .end = cxlr->params.res->end,
- };
- return 0;
+ struct device *dev __free(put_device) =
+ device_find_child(&endpoint->dev, NULL,
+ first_mapped_decoder);
+ if (!dev) {
+ dev_dbg(cxlmd->cxlds->dev,
+ "no region found for memdev %s\n",
+ dev_name(&cxlmd->dev));
+ return -ENXIO;
+ }
+
+ cxled = to_cxl_endpoint_decoder(dev);
+ cxlr = cxled->cxld.region;
+ if (cxlr->params.state < CXL_CONFIG_COMMIT) {
+ dev_dbg(cxlmd->cxlds->dev,
+ "region %s not committed for memdev %s\n",
+ dev_name(&cxlr->dev), dev_name(&cxlmd->dev));
+ return -ENXIO;
+ }
+
+ if (cxlr->params.nr_targets > 1) {
+ dev_dbg(cxlmd->cxlds->dev,
+ "Only attach to local non-interleaved region\n");
+ return -ENXIO;
+ }
+ if (!cxlr->params.res)
+ return -ENXIO;
+
+ /* Only teardown regions that pass validation, ignore the rest. */
+ if (!devm_is_action_added(&endpoint->dev,
+ endpoint_unregister_region, cxlr)) {
+ get_device(&cxlr->dev);
+ rc = devm_add_action(&endpoint->dev,
+ endpoint_unregister_region, cxlr);
+ if (rc)
+ break;
+ }
+
+ *hpa_range = (struct range) {
+ .start = cxlr->params.res->start,
+ .end = cxlr->params.res->end,
+ };
+ return 0;
+ }
+
+ /* devm_add_action() failed; teardown needs the region write lock. */
+ endpoint_unregister_region(cxlr);
+ return rc;
}
EXPORT_SYMBOL_FOR_MODULES(cxl_memdev_attach_region, "cxl_mem");
diff --git a/drivers/cxl/cxlmem.h b/drivers/cxl/cxlmem.h
index c401e3a1af06..8c050bc308bd 100644
--- a/drivers/cxl/cxlmem.h
+++ b/drivers/cxl/cxlmem.h
@@ -97,6 +97,13 @@ static inline bool is_cxl_endpoint(struct cxl_port *port)
return is_cxl_memdev(port->uport_dev);
}
+/**
+ * struct cxl_memdev_attach - provider ownership and CXL link requirements
+ * @probe: optional region probe callback, called with the memdev locked
+ *
+ * A non-NULL descriptor requires successful synchronous endpoint setup and
+ * preserves provider ownership even when no region probe is requested.
+ */
struct cxl_memdev_attach {
int (*probe)(struct cxl_memdev *cxlmd);
};
@@ -107,8 +114,8 @@ struct cxl_memdev_attach {
* @hpa_range: physical address range of the region
*
* For the common simple case of a CXL device with private (non-general purpose
- * / "accelerator") memory, enumerate firmware instantiated region, or
- * instantiate a region for the device's capacity. Destroy the region on detach.
+ * / "accelerator") memory, enumerate a firmware-instantiated region and
+ * report its range. Destroy the region on detach.
*/
struct cxl_attach_region {
struct cxl_memdev_attach attach;
@@ -116,12 +123,19 @@ struct cxl_attach_region {
};
#ifdef CONFIG_CXL_REGION
-int cxl_memdev_attach_region(struct cxl_memdev *cxlmd);
+int cxl_memdev_attach_region(struct cxl_memdev *cxlmd, struct range *hpa_range);
+int cxl_memdev_setup_region_cleanup(struct cxl_memdev *cxlmd);
#else
-static inline int cxl_memdev_attach_region(struct cxl_memdev *cxlmd)
+static inline int cxl_memdev_attach_region(struct cxl_memdev *cxlmd,
+ struct range *hpa_range)
{
return -EOPNOTSUPP;
}
+
+static inline int cxl_memdev_setup_region_cleanup(struct cxl_memdev *cxlmd)
+{
+ return 0;
+}
#endif
struct cxl_memdev *devm_cxl_add_classdev(struct cxl_dev_state *cxlds);
diff --git a/drivers/cxl/mem.c b/drivers/cxl/mem.c
index 3959ec963026..aa08d88ab104 100644
--- a/drivers/cxl/mem.c
+++ b/drivers/cxl/mem.c
@@ -172,7 +172,10 @@ static int cxl_mem_probe(struct device *dev)
}
if (cxlmd->attach) {
- rc = cxlmd->attach->probe(cxlmd);
+ if (cxlmd->attach->probe)
+ rc = cxlmd->attach->probe(cxlmd);
+ else
+ rc = cxl_memdev_setup_region_cleanup(cxlmd);
if (rc)
return rc;
}
@@ -215,6 +218,68 @@ struct cxl_memdev *devm_cxl_add_classdev(struct cxl_dev_state *cxlds)
}
EXPORT_SYMBOL_NS_GPL(devm_cxl_add_classdev, "CXL");
+/**
+ * devm_cxl_register_mem - Register a provider-owned CXL memory device
+ * @cxlds: CXL device state to associate with the memdev
+ *
+ * Establish the CXL port topology and endpoint synchronously, without requiring
+ * a committed region. The provider retains ownership of its memory, including
+ * any firmware-discovered regions, and must detach if the CXL link is lost.
+ *
+ * The parent of the resulting device and the devm context for allocations is
+ * @cxlds->dev. Returns the registered memdev or an ERR_PTR() on failure.
+ */
+struct cxl_memdev *devm_cxl_register_mem(struct cxl_dev_state *cxlds)
+{
+ struct cxl_memdev_attach *attach;
+
+ attach = devm_kzalloc(cxlds->dev, sizeof(*attach), GFP_KERNEL);
+ if (!attach)
+ return ERR_PTR(-ENOMEM);
+
+ return __devm_cxl_add_memdev(cxlds, attach);
+}
+EXPORT_SYMBOL_NS_GPL(devm_cxl_register_mem, "CXL");
+
+/**
+ * devm_cxl_attach_mem_region - Attach a registered memdev to its region
+ * @cxlmd: provider-owned memdev returned by devm_cxl_register_mem()
+ * @hpa_range: CXL.mem physical address range result
+ *
+ * Attach to an existing committed, single-target region. This does not create
+ * or program a region. Repeated attachment to the same region returns the same
+ * range without adding another cleanup action. Failure leaves the memdev
+ * registered so that the provider can decide how to proceed.
+ *
+ * The region is removed when the endpoint detaches. Returns zero on success or
+ * a negative errno; @hpa_range is empty on failure.
+ */
+int devm_cxl_attach_mem_region(struct cxl_memdev *cxlmd,
+ struct range *hpa_range)
+{
+ if (!hpa_range)
+ return -EINVAL;
+ *hpa_range = DEFINE_RANGE(0, -1);
+
+ if (!cxlmd->attach)
+ return -EINVAL;
+
+ guard(device)(&cxlmd->dev);
+ if (!cxlmd->dev.driver || !cxlmd->cxlds)
+ return -ENXIO;
+
+ return cxl_memdev_attach_region(cxlmd, hpa_range);
+}
+EXPORT_SYMBOL_NS_GPL(devm_cxl_attach_mem_region, "CXL");
+
+static int cxl_probe_mem_region(struct cxl_memdev *cxlmd)
+{
+ struct cxl_attach_region *attach =
+ container_of(cxlmd->attach, typeof(*attach), attach);
+
+ return cxl_memdev_attach_region(cxlmd, &attach->hpa_range);
+}
+
/**
* devm_cxl_probe_mem - Add a CXL memory device and probe its region
* @cxlds: CXL device state to associate with the memdev
@@ -242,7 +307,7 @@ struct cxl_memdev *devm_cxl_probe_mem(struct cxl_dev_state *cxlds,
*attach = (struct cxl_attach_region) {
.attach = {
- .probe = cxl_memdev_attach_region,
+ .probe = cxl_probe_mem_region,
},
.hpa_range = { 0, -1 },
};
diff --git a/include/cxl/cxl.h b/include/cxl/cxl.h
index 802b143de83d..3019e3ea5f09 100644
--- a/include/cxl/cxl.h
+++ b/include/cxl/cxl.h
@@ -224,6 +224,8 @@ struct cxl_dev_state *_devm_cxl_dev_state_create(struct device *dev,
sizeof(drv_struct), mbox); \
})
+struct cxl_memdev *devm_cxl_register_mem(struct cxl_dev_state *cxlds);
+int devm_cxl_attach_mem_region(struct cxl_memdev *cxlmd, struct range *hpa_range);
struct cxl_memdev *devm_cxl_probe_mem(struct cxl_dev_state *cxlds,
struct range *range);
--
2.43.0
^ permalink raw reply [flat|nested] 4+ messages in thread* [RFC PATCH v2 2/3] cxl/mem: Add explicit RAM region creation for providers
2026-10-07 9:05 [RFC PATCH v2 0/3] cxl: Provide explicit RAM region creation for Type-2 providers Richard Cheng
2026-10-07 9:05 ` [RFC PATCH v2 1/3] cxl/mem: Separate provider registration from region attachment Richard Cheng
@ 2026-10-07 9:05 ` Richard Cheng
2026-10-07 9:05 ` [RFC PATCH v2 3/3] cxl/test: Exercise explicit Type-2 RAM region creation Richard Cheng
2 siblings, 0 replies; 4+ messages in thread
From: Richard Cheng @ 2026-10-07 9:05 UTC (permalink / raw)
To: jic23, dave, dave.jiang, alison.schofield, vishal.l.verma,
iweiny, ming.li, icheng
Cc: kaihengf, kobak, newtonl, kristinc, linux-cxl, linux-kernel
Type-2 providers may need a region when FW supplies none.
Add devm_cxl_create_ram_region() so providers choose the size and
creation time after registering their memdev.
Reuse core DPA/HPA allocation, decoder setup and commit. Serialize
creation against sysfs decoder writes, unwind failures, and reset
SW-created mappings on detach.
Keep provider attachment separate.
CXL_REGION_F_LOCK also reflects FIXED CFMWS root windows. Restricting
the reset exemption to locked AUTO regions allows teardown of
user-created regions under those windows to reset programmable
downstream decoders, which previously remained committed. The root
window and HW-locked decoders remain protected.
Signed-off-by: Richard Cheng <icheng@nvidia.com>
---
Changelog:
v1 -> v2:
- Replace implicit attach-time creation with an explicit API
- Accept caller-selected size instead of allocating all volatile
capacity
---
drivers/cxl/core/port.c | 6 +
drivers/cxl/core/region.c | 223 +++++++++++++++++++++++++++++++++++++-
drivers/cxl/cxlmem.h | 7 ++
drivers/cxl/mem.c | 31 ++++++
include/cxl/cxl.h | 1 +
5 files changed, 263 insertions(+), 5 deletions(-)
diff --git a/drivers/cxl/core/port.c b/drivers/cxl/core/port.c
index 6024bc9c1376..741240021cc9 100644
--- a/drivers/cxl/core/port.c
+++ b/drivers/cxl/core/port.c
@@ -235,6 +235,9 @@ static ssize_t mode_store(struct device *dev, struct device_attribute *attr,
else
return -EINVAL;
+ /* Serialize partition selection with kernel region provisioning. */
+ guard(rwsem_write)(&cxl_rwsem.region);
+
rc = cxl_dpa_set_part(cxled, mode);
if (rc)
return rc;
@@ -276,6 +279,9 @@ static ssize_t dpa_size_store(struct device *dev, struct device_attribute *attr,
if (!IS_ALIGNED(size, SZ_256M))
return -EINVAL;
+ /* Keep the free/allocate sequence atomic against region setup. */
+ guard(rwsem_write)(&cxl_rwsem.region);
+
rc = cxl_dpa_free(cxled);
if (rc)
return rc;
diff --git a/drivers/cxl/core/region.c b/drivers/cxl/core/region.c
index 7a64a730587d..4cfa0a644481 100644
--- a/drivers/cxl/core/region.c
+++ b/drivers/cxl/core/region.c
@@ -250,7 +250,9 @@ static void cxl_region_decode_reset(struct cxl_region *cxlr, int count)
struct cxl_region_params *p = &cxlr->params;
int i;
- if (test_bit(CXL_REGION_F_LOCK, &cxlr->flags))
+ /* Provider ownership must not suppress reset of software-created regions. */
+ if (test_bit(CXL_REGION_F_LOCK, &cxlr->flags) &&
+ test_bit(CXL_REGION_F_AUTO, &cxlr->flags))
return;
/*
@@ -374,14 +376,12 @@ static int queue_reset(struct cxl_region *cxlr)
return 0;
}
-static int __commit(struct cxl_region *cxlr)
+static int cxl_region_commit(struct cxl_region *cxlr)
{
struct cxl_region_params *p = &cxlr->params;
int rc;
- ACQUIRE(rwsem_write_kill, rwsem)(&cxl_rwsem.region);
- if ((rc = ACQUIRE_ERR(rwsem_write_kill, &rwsem)))
- return rc;
+ lockdep_assert_held_write(&cxl_rwsem.region);
/* Already in the requested state? */
if (p->state >= CXL_CONFIG_COMMIT)
@@ -408,6 +408,18 @@ static int __commit(struct cxl_region *cxlr)
return 0;
}
+static int __commit(struct cxl_region *cxlr)
+{
+ int rc;
+
+ ACQUIRE(rwsem_write_kill, rwsem)(&cxl_rwsem.region);
+ rc = ACQUIRE_ERR(rwsem_write_kill, &rwsem);
+ if (rc)
+ return rc;
+
+ return cxl_region_commit(cxlr);
+}
+
static ssize_t commit_store(struct device *dev, struct device_attribute *attr,
const char *buf, size_t len)
{
@@ -4111,6 +4123,207 @@ static int first_mapped_decoder(struct device *dev, const void *data)
return 0;
}
+static int match_free_endpoint_decoder(struct device *dev, const void *data)
+{
+ struct cxl_port *port = to_cxl_port(dev->parent);
+ struct cxl_endpoint_decoder *cxled;
+ struct cxl_decoder *cxld;
+
+ if (!is_endpoint_decoder(dev))
+ return 0;
+
+ lockdep_assert_held_write(&cxl_rwsem.region);
+ lockdep_assert_held(&cxl_rwsem.dpa);
+ cxled = to_cxl_endpoint_decoder(dev);
+ cxld = &cxled->cxld;
+ if (cxled->state != CXL_DECODER_STATE_MANUAL || cxled->dpa_res ||
+ cxld->region || (cxld->flags & CXL_DECODER_F_RESET_MASK) ||
+ !cxld->commit || !cxld->reset)
+ return 0;
+
+ if (cxld->id != port->hdm_end + 1 ||
+ cxld->id != cxl_num_decoders_committed(port))
+ return 0;
+
+ return !device_for_each_child_reverse_from(dev->parent, dev, NULL,
+ check_commit_order);
+}
+
+static int cxl_configure_ram_region(struct cxl_region *cxlr,
+ struct cxl_memdev *cxlmd, u64 size)
+{
+ struct cxl_port *endpoint = cxlmd->endpoint;
+ struct cxl_endpoint_decoder *cxled;
+ struct cxl_decoder *cxld;
+ struct cxl_region *detach;
+ struct device *dev;
+ struct range hpa_range;
+ int part, ways, granularity, pos, rc;
+
+ down_write(&cxl_rwsem.region);
+
+ dev = device_find_child(&endpoint->dev, NULL, first_mapped_decoder);
+ if (dev) {
+ rc = -EBUSY;
+ goto out;
+ }
+
+ down_read(&cxl_rwsem.dpa);
+ dev = device_find_child(&endpoint->dev, NULL,
+ match_free_endpoint_decoder);
+ up_read(&cxl_rwsem.dpa);
+ if (!dev) {
+ rc = -ENOSPC;
+ goto out;
+ }
+
+ cxled = to_cxl_endpoint_decoder(dev);
+ cxld = &cxled->cxld;
+ part = cxled->part;
+ pos = cxled->pos;
+ ways = cxld->interleave_ways;
+ granularity = cxld->interleave_granularity;
+ hpa_range = cxld->hpa_range;
+
+ rc = cxl_dpa_set_part(cxled, CXL_PARTMODE_RAM);
+ if (rc)
+ goto out;
+ rc = cxl_dpa_alloc(cxled, size);
+ if (rc)
+ goto restore;
+
+ rc = set_interleave_ways(cxlr, 1);
+ if (rc)
+ goto free_dpa;
+ rc = set_interleave_granularity(cxlr,
+ cxlr->cxlrd->cxlsd.cxld.interleave_granularity);
+ if (rc)
+ goto free_dpa;
+ rc = alloc_hpa(cxlr, size);
+ if (rc)
+ goto free_dpa;
+
+ down_read(&cxl_rwsem.dpa);
+ rc = cxl_region_attach(cxlr, cxled, 0);
+ up_read(&cxl_rwsem.dpa);
+ if (rc)
+ goto detach;
+ rc = cxl_region_commit(cxlr);
+ if (rc)
+ goto detach;
+
+ /* Added after DPA reservation so decode is reset before DPA release. */
+ get_device(&cxlr->dev);
+ rc = devm_add_action(&endpoint->dev, endpoint_unregister_region, cxlr);
+ if (!rc)
+ goto out;
+ put_device(&cxlr->dev);
+
+detach:
+ down_read(&cxl_rwsem.dpa);
+ detach = __cxl_decoder_detach(cxlr, cxled, 0, DETACH_ONLY);
+ if (detach)
+ put_device(&detach->dev);
+ up_read(&cxl_rwsem.dpa);
+free_dpa:
+ cxl_dpa_free(cxled);
+restore:
+ down_write(&cxl_rwsem.dpa);
+ cxled->part = part;
+ up_write(&cxl_rwsem.dpa);
+ cxled->pos = pos;
+ cxld->interleave_ways = ways;
+ cxld->interleave_granularity = granularity;
+ cxld->hpa_range = hpa_range;
+out:
+ put_device(dev);
+ up_write(&cxl_rwsem.region);
+ return rc;
+}
+
+struct cxl_ram_region_context {
+ struct cxl_memdev *cxlmd;
+ u64 size;
+ int rc;
+};
+
+static int create_memdev_ram_region(struct device *dev, void *data)
+{
+ struct cxl_ram_region_context *ctx = data;
+ struct cxl_port *endpoint = ctx->cxlmd->endpoint;
+ struct cxl_root_decoder *cxlrd;
+ struct cxl_switch_decoder *cxlsd;
+ struct cxl_decoder *cxld;
+ struct cxl_region *cxlr;
+ enum cxl_decoder_type type;
+ unsigned long caps;
+
+ if (!is_root_decoder(dev))
+ return 0;
+
+ cxlrd = to_cxl_root_decoder(dev);
+ cxlsd = &cxlrd->cxlsd;
+ cxld = &cxlsd->cxld;
+ type = ctx->cxlmd->cxlds->type == CXL_DEVTYPE_CLASSMEM ?
+ CXL_DECODER_HOSTONLYMEM : CXL_DECODER_DEVMEM;
+ caps = CXL_DECODER_F_RAM | (type == CXL_DECODER_DEVMEM ?
+ CXL_DECODER_F_TYPE2 : CXL_DECODER_F_TYPE3);
+ if ((cxld->flags & caps) != caps || cxld->interleave_ways != 1 ||
+ cxlsd->nr_targets != 1 || !cxlsd->target[0] ||
+ cxlsd->target[0]->dport_dev != endpoint->host_bridge ||
+ !cxlrd->res || cxlrd->cache_size ||
+ (cxld->flags & CXL_DECODER_F_NORMALIZED_ADDRESSING))
+ return 0;
+
+ guard(mutex)(&cxlrd->regions_lock);
+ do {
+ cxlr = __create_region(cxlrd, CXL_PARTMODE_RAM,
+ atomic_read(&cxlrd->region_id), type);
+ } while (IS_ERR(cxlr) && PTR_ERR(cxlr) == -EBUSY);
+ if (IS_ERR(cxlr)) {
+ ctx->rc = PTR_ERR(cxlr);
+ return ctx->rc == -ENXIO ? 0 : ctx->rc;
+ }
+
+ ctx->rc = cxl_configure_ram_region(cxlr, ctx->cxlmd, ctx->size);
+ if (!ctx->rc)
+ return 1;
+
+ /* Region unregistration acquires cxl_rwsem.region internally. */
+ unregister_region(cxlr);
+
+ /* Retry HPA window exhaustion; endpoint and DPA failures are terminal. */
+ return ctx->rc == -ERANGE ? 0 : ctx->rc;
+}
+
+int cxl_memdev_create_ram_region(struct cxl_memdev *cxlmd, u64 size)
+{
+ struct cxl_port *endpoint = cxlmd->endpoint;
+ struct cxl_ram_region_context ctx = {
+ .cxlmd = cxlmd,
+ .size = size,
+ .rc = -ENXIO,
+ };
+
+ device_lock_assert(&cxlmd->dev);
+ if (!size || !IS_ALIGNED(size, SZ_256M))
+ return -EINVAL;
+ if (IS_ERR_OR_NULL(endpoint))
+ return -ENXIO;
+
+ guard(device)(&endpoint->dev);
+ if (!endpoint->dev.driver || endpoint->dead)
+ return -ENXIO;
+
+ struct cxl_root *root __free(put_cxl_root) = find_cxl_root(endpoint);
+ if (!root)
+ return -ENXIO;
+
+ device_for_each_child(&root->port.dev, &ctx, create_memdev_ram_region);
+ return ctx.rc;
+}
+EXPORT_SYMBOL_FOR_MODULES(cxl_memdev_create_ram_region, "cxl_mem");
+
static int unregister_memdev_region(struct device *dev, void *data)
{
struct cxl_endpoint_decoder *cxled;
diff --git a/drivers/cxl/cxlmem.h b/drivers/cxl/cxlmem.h
index 8c050bc308bd..9f074007dbf8 100644
--- a/drivers/cxl/cxlmem.h
+++ b/drivers/cxl/cxlmem.h
@@ -123,9 +123,16 @@ struct cxl_attach_region {
};
#ifdef CONFIG_CXL_REGION
+int cxl_memdev_create_ram_region(struct cxl_memdev *cxlmd, u64 size);
int cxl_memdev_attach_region(struct cxl_memdev *cxlmd, struct range *hpa_range);
int cxl_memdev_setup_region_cleanup(struct cxl_memdev *cxlmd);
#else
+static inline int cxl_memdev_create_ram_region(struct cxl_memdev *cxlmd,
+ u64 size)
+{
+ return -EOPNOTSUPP;
+}
+
static inline int cxl_memdev_attach_region(struct cxl_memdev *cxlmd,
struct range *hpa_range)
{
diff --git a/drivers/cxl/mem.c b/drivers/cxl/mem.c
index aa08d88ab104..00bd7d9fe561 100644
--- a/drivers/cxl/mem.c
+++ b/drivers/cxl/mem.c
@@ -241,6 +241,37 @@ struct cxl_memdev *devm_cxl_register_mem(struct cxl_dev_state *cxlds)
}
EXPORT_SYMBOL_NS_GPL(devm_cxl_register_mem, "CXL");
+/**
+ * devm_cxl_create_ram_region - Create a provider-owned RAM region
+ * @cxlmd: provider-owned memdev returned by devm_cxl_register_mem()
+ * @size: requested region size in bytes, a nonzero multiple of 256 MiB
+ *
+ * Reserve @size bytes of volatile DPA and HPA, establish a single-target
+ * decoder path, and commit it. The caller chooses the size; this helper does
+ * not consume the device's entire capacity or replace an existing region.
+ * Obtain the resulting HPA range separately with devm_cxl_attach_mem_region().
+ *
+ * Requires an unused programmable endpoint decoder and a compatible
+ * non-interleaved root window without normalized addressing or an extended
+ * linear cache. Persistent memory provisioning is outside this helper.
+ *
+ * The region is removed and its software-programmed decoders are reset when
+ * the endpoint detaches. Failure unwinds the new allocations and leaves the
+ * memdev registered. Returns zero on success or a negative errno.
+ */
+int devm_cxl_create_ram_region(struct cxl_memdev *cxlmd, u64 size)
+{
+ if (!cxlmd->attach || !size)
+ return -EINVAL;
+
+ guard(device)(&cxlmd->dev);
+ if (!cxlmd->dev.driver || !cxlmd->cxlds)
+ return -ENXIO;
+
+ return cxl_memdev_create_ram_region(cxlmd, size);
+}
+EXPORT_SYMBOL_NS_GPL(devm_cxl_create_ram_region, "CXL");
+
/**
* devm_cxl_attach_mem_region - Attach a registered memdev to its region
* @cxlmd: provider-owned memdev returned by devm_cxl_register_mem()
diff --git a/include/cxl/cxl.h b/include/cxl/cxl.h
index 3019e3ea5f09..b873bd0cfacd 100644
--- a/include/cxl/cxl.h
+++ b/include/cxl/cxl.h
@@ -225,6 +225,7 @@ struct cxl_dev_state *_devm_cxl_dev_state_create(struct device *dev,
})
struct cxl_memdev *devm_cxl_register_mem(struct cxl_dev_state *cxlds);
+int devm_cxl_create_ram_region(struct cxl_memdev *cxlmd, u64 size);
int devm_cxl_attach_mem_region(struct cxl_memdev *cxlmd, struct range *hpa_range);
struct cxl_memdev *devm_cxl_probe_mem(struct cxl_dev_state *cxlds,
struct range *range);
--
2.43.0
^ permalink raw reply [flat|nested] 4+ messages in thread* [RFC PATCH v2 3/3] cxl/test: Exercise explicit Type-2 RAM region creation
2026-10-07 9:05 [RFC PATCH v2 0/3] cxl: Provide explicit RAM region creation for Type-2 providers Richard Cheng
2026-10-07 9:05 ` [RFC PATCH v2 1/3] cxl/mem: Separate provider registration from region attachment Richard Cheng
2026-10-07 9:05 ` [RFC PATCH v2 2/3] cxl/mem: Add explicit RAM region creation for providers Richard Cheng
@ 2026-10-07 9:05 ` Richard Cheng
2 siblings, 0 replies; 4+ messages in thread
From: Richard Cheng @ 2026-10-07 9:05 UTC (permalink / raw)
To: jic23, dave, dave.jiang, alison.schofield, vishal.l.verma,
iweiny, ming.li, icheng
Cc: kaihengf, kobak, newtonl, kristinc, linux-cxl, linux-kernel
Add a second mock Type-2 accelerator without a FW-provided region.
Register its memdev, try attachment, and explicitly create a 256 MB RAM
region when needed.
Check the returned HPA range and retain the original accelerator on
devm_cxl_probe_mem(). Preserve MANUAL DEVMEM configuration after
disabled reply or failed saved-state restoration.
Signed-off-by: Richard Cheng <icheng@nvidia.com>
---
Changelog:
v1 -> v2:
- Use separate registration, creation and attachment APIs
- Request 256 MB from 512 MB capacity
- Preserve DEVMEM config after failed restoration
---
tools/testing/cxl/test/accel.c | 28 +++++++++++++++++++-
tools/testing/cxl/test/cxl.c | 47 +++++++++++++++++++++++++++++-----
2 files changed, 68 insertions(+), 7 deletions(-)
diff --git a/tools/testing/cxl/test/accel.c b/tools/testing/cxl/test/accel.c
index 8e6f4687ca02..83962702145d 100644
--- a/tools/testing/cxl/test/accel.c
+++ b/tools/testing/cxl/test/accel.c
@@ -20,6 +20,7 @@ static int cxl_mock_accel_probe(struct platform_device *pdev)
struct cxl_dev_state *cxlds;
struct cxl_memdev *cxlmd;
struct range mock_range;
+ u64 region_size = pdev->id == 0 ? SZ_512M : SZ_256M;
int rc;
cxl_accel = devm_cxl_dev_state_create(&pdev->dev, CXL_DEVTYPE_DEVMEM,
@@ -35,9 +36,34 @@ static int cxl_mock_accel_probe(struct platform_device *pdev)
if (rc)
return rc;
- cxlmd = devm_cxl_probe_mem(cxlds, &mock_range);
+ /* Keep the firmware-configured accelerator on the legacy API. */
+ if (pdev->id == 0)
+ cxlmd = devm_cxl_probe_mem(cxlds, &mock_range);
+ else
+ cxlmd = devm_cxl_register_mem(cxlds);
if (IS_ERR(cxlmd))
return PTR_ERR(cxlmd);
+
+ if (pdev->id != 0) {
+ rc = devm_cxl_attach_mem_region(cxlmd, &mock_range);
+ if (rc == -ENXIO) {
+ /* The provider chooses to map half its volatile capacity. */
+ rc = devm_cxl_create_ram_region(cxlmd, region_size);
+ if (rc)
+ return rc;
+ rc = devm_cxl_attach_mem_region(cxlmd, &mock_range);
+ }
+ if (rc)
+ return rc;
+ }
+
+ if (mock_range.start > mock_range.end ||
+ range_len(&mock_range) != region_size) {
+ dev_err(dev,
+ "accelerator%d returned invalid HPA range %pra (expected %llu bytes)\n",
+ pdev->id, &mock_range, region_size);
+ return -ERANGE;
+ }
cxl_accel->cxlmd = cxlmd;
dev_dbg(dev, "Probed mock accelerator with range %pra\n", &mock_range);
diff --git a/tools/testing/cxl/test/cxl.c b/tools/testing/cxl/test/cxl.c
index 3eb1051b1ee8..4fab2572a1aa 100644
--- a/tools/testing/cxl/test/cxl.c
+++ b/tools/testing/cxl/test/cxl.c
@@ -29,7 +29,7 @@ static bool mock_zero_size_decoders;
#define NR_CXL_SWITCH_PORTS 2
#define NR_CXL_PORT_DECODERS 8
#define NR_BRIDGES (NR_CXL_HOST_BRIDGES + NR_CXL_SINGLE_HOST + NR_CXL_RCH)
-#define NR_CXL_TYPE2_ACCEL 1
+#define NR_CXL_TYPE2_ACCEL 2
#define MOCK_AUTO_REGION_SIZE_DEFAULT SZ_512M
static int mock_auto_region_size = MOCK_AUTO_REGION_SIZE_DEFAULT;
@@ -494,7 +494,17 @@ static void cfmws_elc_update(struct acpi_cedt_cfmws *window, int index)
static void update_type2_cfmws(void)
{
+ struct acpi_cedt_cfmws *window = &mock_cedt.cfmws1.cfmws;
+
memcpy(&mock_cedt.cfmws0.cfmws, &type2_cfmws0, sizeof(type2_cfmws0));
+
+ /* Give the unconfigured accelerator its own single-target window. */
+ window->header.length = sizeof(*window) +
+ sizeof(mock_cedt.cfmws1.target[0]);
+ window->interleave_ways = 0;
+ window->restrictions = ACPI_CEDT_CFMWS_RESTRICT_DEVMEM |
+ ACPI_CEDT_CFMWS_RESTRICT_VOLATILE;
+ mock_cedt.cfmws1.target[0] = 1;
}
static int populate_cedt(void)
@@ -1122,6 +1132,7 @@ enum cxld_init_type {
MOCK_DECODER_INIT_SAVED,
MOCK_DECODER_INIT_TYPE3_AUTO,
MOCK_DECODER_INIT_TYPE2_AUTO,
+ MOCK_DECODER_INIT_TYPE2_MANUAL,
};
static enum cxld_init_type get_decoder_init_type(struct cxl_decoder *cxld,
@@ -1131,12 +1142,13 @@ static enum cxld_init_type get_decoder_init_type(struct cxl_decoder *cxld,
{
struct cxl_test_decoder *found_td = cxld_registry_find(cxld);
- if (found_td) {
- *td = found_td;
- return MOCK_DECODER_INIT_SAVED;
- }
+ *td = found_td;
+ if (type2_test && is_endpoint_decoder(&cxld->dev) && pdev->id == 1 &&
+ cxld->id == 0)
+ return MOCK_DECODER_INIT_TYPE2_MANUAL;
- *td = NULL;
+ if (found_td)
+ return MOCK_DECODER_INIT_SAVED;
/*
* The first decoder on the first 2 devices on the first switch
@@ -1170,6 +1182,27 @@ static bool mock_decoder_handle_saved(struct cxl_decoder *cxld, struct cxl_test_
return false;
}
+static bool mock_init_hdm_type2_manual(struct cxl_endpoint_decoder *cxled,
+ struct cxl_test_decoder *td)
+{
+ struct cxl_decoder *cxld = &cxled->cxld;
+
+ if (td) {
+ if (mock_decoder_handle_saved(cxld, td))
+ return true;
+ } else {
+ init_disabled_mock_decoder(cxld);
+ }
+
+ /* Preserve DEVMEM on disabled replay and failed saved-state restore. */
+ cxld->target_type = CXL_DECODER_DEVMEM;
+ cxld->interleave_granularity = CXL_DECODER_MIN_GRANULARITY;
+ if (!td)
+ WARN_ON_ONCE(!cxld_registry_new(cxld));
+
+ return false;
+}
+
static void mock_init_hdm_type2_cxled(struct cxl_endpoint_decoder *cxled,
struct cxl_port *port)
{
@@ -1445,6 +1478,8 @@ static bool mock_init_hdm_decoder(struct cxl_decoder *cxld)
case MOCK_DECODER_INIT_TYPE2_AUTO:
mock_init_hdm_type2_cxled(cxled, port);
return false;
+ case MOCK_DECODER_INIT_TYPE2_MANUAL:
+ return mock_init_hdm_type2_manual(cxled, td);
default:
return false;
}
--
2.43.0
^ permalink raw reply [flat|nested] 4+ messages in thread