From: Richard Cheng <icheng@nvidia.com>
To: jic23@kernel.org, dave@stgolabs.net, dave.jiang@intel.com,
alison.schofield@intel.com, vishal.l.verma@intel.com,
iweiny@kernel.org, ming.li@zohomail.com, icheng@nvidia.com
Cc: kaihengf@nvidia.com, kobak@nvidia.com, newtonl@nvidia.com,
kristinc@nvidia.com, linux-cxl@vger.kernel.org,
linux-kernel@vger.kernel.org
Subject: [RFC PATCH v2 2/3] cxl/mem: Add explicit RAM region creation for providers
Date: Wed, 7 Oct 2026 17:05:38 +0800 [thread overview]
Message-ID: <20261007090540.43817-3-icheng@nvidia.com> (raw)
In-Reply-To: <20261007090540.43817-1-icheng@nvidia.com>
Type-2 providers may need a region when FW supplies none.
Add devm_cxl_create_ram_region() so providers choose the size and
creation time after registering their memdev.
Reuse core DPA/HPA allocation, decoder setup and commit. Serialize
creation against sysfs decoder writes, unwind failures, and reset
SW-created mappings on detach.
Keep provider attachment separate.
CXL_REGION_F_LOCK also reflects FIXED CFMWS root windows. Restricting
the reset exemption to locked AUTO regions allows teardown of
user-created regions under those windows to reset programmable
downstream decoders, which previously remained committed. The root
window and HW-locked decoders remain protected.
Signed-off-by: Richard Cheng <icheng@nvidia.com>
---
Changelog:
v1 -> v2:
- Replace implicit attach-time creation with an explicit API
- Accept caller-selected size instead of allocating all volatile
capacity
---
drivers/cxl/core/port.c | 6 +
drivers/cxl/core/region.c | 223 +++++++++++++++++++++++++++++++++++++-
drivers/cxl/cxlmem.h | 7 ++
drivers/cxl/mem.c | 31 ++++++
include/cxl/cxl.h | 1 +
5 files changed, 263 insertions(+), 5 deletions(-)
diff --git a/drivers/cxl/core/port.c b/drivers/cxl/core/port.c
index 6024bc9c1376..741240021cc9 100644
--- a/drivers/cxl/core/port.c
+++ b/drivers/cxl/core/port.c
@@ -235,6 +235,9 @@ static ssize_t mode_store(struct device *dev, struct device_attribute *attr,
else
return -EINVAL;
+ /* Serialize partition selection with kernel region provisioning. */
+ guard(rwsem_write)(&cxl_rwsem.region);
+
rc = cxl_dpa_set_part(cxled, mode);
if (rc)
return rc;
@@ -276,6 +279,9 @@ static ssize_t dpa_size_store(struct device *dev, struct device_attribute *attr,
if (!IS_ALIGNED(size, SZ_256M))
return -EINVAL;
+ /* Keep the free/allocate sequence atomic against region setup. */
+ guard(rwsem_write)(&cxl_rwsem.region);
+
rc = cxl_dpa_free(cxled);
if (rc)
return rc;
diff --git a/drivers/cxl/core/region.c b/drivers/cxl/core/region.c
index 7a64a730587d..4cfa0a644481 100644
--- a/drivers/cxl/core/region.c
+++ b/drivers/cxl/core/region.c
@@ -250,7 +250,9 @@ static void cxl_region_decode_reset(struct cxl_region *cxlr, int count)
struct cxl_region_params *p = &cxlr->params;
int i;
- if (test_bit(CXL_REGION_F_LOCK, &cxlr->flags))
+ /* Provider ownership must not suppress reset of software-created regions. */
+ if (test_bit(CXL_REGION_F_LOCK, &cxlr->flags) &&
+ test_bit(CXL_REGION_F_AUTO, &cxlr->flags))
return;
/*
@@ -374,14 +376,12 @@ static int queue_reset(struct cxl_region *cxlr)
return 0;
}
-static int __commit(struct cxl_region *cxlr)
+static int cxl_region_commit(struct cxl_region *cxlr)
{
struct cxl_region_params *p = &cxlr->params;
int rc;
- ACQUIRE(rwsem_write_kill, rwsem)(&cxl_rwsem.region);
- if ((rc = ACQUIRE_ERR(rwsem_write_kill, &rwsem)))
- return rc;
+ lockdep_assert_held_write(&cxl_rwsem.region);
/* Already in the requested state? */
if (p->state >= CXL_CONFIG_COMMIT)
@@ -408,6 +408,18 @@ static int __commit(struct cxl_region *cxlr)
return 0;
}
+static int __commit(struct cxl_region *cxlr)
+{
+ int rc;
+
+ ACQUIRE(rwsem_write_kill, rwsem)(&cxl_rwsem.region);
+ rc = ACQUIRE_ERR(rwsem_write_kill, &rwsem);
+ if (rc)
+ return rc;
+
+ return cxl_region_commit(cxlr);
+}
+
static ssize_t commit_store(struct device *dev, struct device_attribute *attr,
const char *buf, size_t len)
{
@@ -4111,6 +4123,207 @@ static int first_mapped_decoder(struct device *dev, const void *data)
return 0;
}
+static int match_free_endpoint_decoder(struct device *dev, const void *data)
+{
+ struct cxl_port *port = to_cxl_port(dev->parent);
+ struct cxl_endpoint_decoder *cxled;
+ struct cxl_decoder *cxld;
+
+ if (!is_endpoint_decoder(dev))
+ return 0;
+
+ lockdep_assert_held_write(&cxl_rwsem.region);
+ lockdep_assert_held(&cxl_rwsem.dpa);
+ cxled = to_cxl_endpoint_decoder(dev);
+ cxld = &cxled->cxld;
+ if (cxled->state != CXL_DECODER_STATE_MANUAL || cxled->dpa_res ||
+ cxld->region || (cxld->flags & CXL_DECODER_F_RESET_MASK) ||
+ !cxld->commit || !cxld->reset)
+ return 0;
+
+ if (cxld->id != port->hdm_end + 1 ||
+ cxld->id != cxl_num_decoders_committed(port))
+ return 0;
+
+ return !device_for_each_child_reverse_from(dev->parent, dev, NULL,
+ check_commit_order);
+}
+
+static int cxl_configure_ram_region(struct cxl_region *cxlr,
+ struct cxl_memdev *cxlmd, u64 size)
+{
+ struct cxl_port *endpoint = cxlmd->endpoint;
+ struct cxl_endpoint_decoder *cxled;
+ struct cxl_decoder *cxld;
+ struct cxl_region *detach;
+ struct device *dev;
+ struct range hpa_range;
+ int part, ways, granularity, pos, rc;
+
+ down_write(&cxl_rwsem.region);
+
+ dev = device_find_child(&endpoint->dev, NULL, first_mapped_decoder);
+ if (dev) {
+ rc = -EBUSY;
+ goto out;
+ }
+
+ down_read(&cxl_rwsem.dpa);
+ dev = device_find_child(&endpoint->dev, NULL,
+ match_free_endpoint_decoder);
+ up_read(&cxl_rwsem.dpa);
+ if (!dev) {
+ rc = -ENOSPC;
+ goto out;
+ }
+
+ cxled = to_cxl_endpoint_decoder(dev);
+ cxld = &cxled->cxld;
+ part = cxled->part;
+ pos = cxled->pos;
+ ways = cxld->interleave_ways;
+ granularity = cxld->interleave_granularity;
+ hpa_range = cxld->hpa_range;
+
+ rc = cxl_dpa_set_part(cxled, CXL_PARTMODE_RAM);
+ if (rc)
+ goto out;
+ rc = cxl_dpa_alloc(cxled, size);
+ if (rc)
+ goto restore;
+
+ rc = set_interleave_ways(cxlr, 1);
+ if (rc)
+ goto free_dpa;
+ rc = set_interleave_granularity(cxlr,
+ cxlr->cxlrd->cxlsd.cxld.interleave_granularity);
+ if (rc)
+ goto free_dpa;
+ rc = alloc_hpa(cxlr, size);
+ if (rc)
+ goto free_dpa;
+
+ down_read(&cxl_rwsem.dpa);
+ rc = cxl_region_attach(cxlr, cxled, 0);
+ up_read(&cxl_rwsem.dpa);
+ if (rc)
+ goto detach;
+ rc = cxl_region_commit(cxlr);
+ if (rc)
+ goto detach;
+
+ /* Added after DPA reservation so decode is reset before DPA release. */
+ get_device(&cxlr->dev);
+ rc = devm_add_action(&endpoint->dev, endpoint_unregister_region, cxlr);
+ if (!rc)
+ goto out;
+ put_device(&cxlr->dev);
+
+detach:
+ down_read(&cxl_rwsem.dpa);
+ detach = __cxl_decoder_detach(cxlr, cxled, 0, DETACH_ONLY);
+ if (detach)
+ put_device(&detach->dev);
+ up_read(&cxl_rwsem.dpa);
+free_dpa:
+ cxl_dpa_free(cxled);
+restore:
+ down_write(&cxl_rwsem.dpa);
+ cxled->part = part;
+ up_write(&cxl_rwsem.dpa);
+ cxled->pos = pos;
+ cxld->interleave_ways = ways;
+ cxld->interleave_granularity = granularity;
+ cxld->hpa_range = hpa_range;
+out:
+ put_device(dev);
+ up_write(&cxl_rwsem.region);
+ return rc;
+}
+
+struct cxl_ram_region_context {
+ struct cxl_memdev *cxlmd;
+ u64 size;
+ int rc;
+};
+
+static int create_memdev_ram_region(struct device *dev, void *data)
+{
+ struct cxl_ram_region_context *ctx = data;
+ struct cxl_port *endpoint = ctx->cxlmd->endpoint;
+ struct cxl_root_decoder *cxlrd;
+ struct cxl_switch_decoder *cxlsd;
+ struct cxl_decoder *cxld;
+ struct cxl_region *cxlr;
+ enum cxl_decoder_type type;
+ unsigned long caps;
+
+ if (!is_root_decoder(dev))
+ return 0;
+
+ cxlrd = to_cxl_root_decoder(dev);
+ cxlsd = &cxlrd->cxlsd;
+ cxld = &cxlsd->cxld;
+ type = ctx->cxlmd->cxlds->type == CXL_DEVTYPE_CLASSMEM ?
+ CXL_DECODER_HOSTONLYMEM : CXL_DECODER_DEVMEM;
+ caps = CXL_DECODER_F_RAM | (type == CXL_DECODER_DEVMEM ?
+ CXL_DECODER_F_TYPE2 : CXL_DECODER_F_TYPE3);
+ if ((cxld->flags & caps) != caps || cxld->interleave_ways != 1 ||
+ cxlsd->nr_targets != 1 || !cxlsd->target[0] ||
+ cxlsd->target[0]->dport_dev != endpoint->host_bridge ||
+ !cxlrd->res || cxlrd->cache_size ||
+ (cxld->flags & CXL_DECODER_F_NORMALIZED_ADDRESSING))
+ return 0;
+
+ guard(mutex)(&cxlrd->regions_lock);
+ do {
+ cxlr = __create_region(cxlrd, CXL_PARTMODE_RAM,
+ atomic_read(&cxlrd->region_id), type);
+ } while (IS_ERR(cxlr) && PTR_ERR(cxlr) == -EBUSY);
+ if (IS_ERR(cxlr)) {
+ ctx->rc = PTR_ERR(cxlr);
+ return ctx->rc == -ENXIO ? 0 : ctx->rc;
+ }
+
+ ctx->rc = cxl_configure_ram_region(cxlr, ctx->cxlmd, ctx->size);
+ if (!ctx->rc)
+ return 1;
+
+ /* Region unregistration acquires cxl_rwsem.region internally. */
+ unregister_region(cxlr);
+
+ /* Retry HPA window exhaustion; endpoint and DPA failures are terminal. */
+ return ctx->rc == -ERANGE ? 0 : ctx->rc;
+}
+
+int cxl_memdev_create_ram_region(struct cxl_memdev *cxlmd, u64 size)
+{
+ struct cxl_port *endpoint = cxlmd->endpoint;
+ struct cxl_ram_region_context ctx = {
+ .cxlmd = cxlmd,
+ .size = size,
+ .rc = -ENXIO,
+ };
+
+ device_lock_assert(&cxlmd->dev);
+ if (!size || !IS_ALIGNED(size, SZ_256M))
+ return -EINVAL;
+ if (IS_ERR_OR_NULL(endpoint))
+ return -ENXIO;
+
+ guard(device)(&endpoint->dev);
+ if (!endpoint->dev.driver || endpoint->dead)
+ return -ENXIO;
+
+ struct cxl_root *root __free(put_cxl_root) = find_cxl_root(endpoint);
+ if (!root)
+ return -ENXIO;
+
+ device_for_each_child(&root->port.dev, &ctx, create_memdev_ram_region);
+ return ctx.rc;
+}
+EXPORT_SYMBOL_FOR_MODULES(cxl_memdev_create_ram_region, "cxl_mem");
+
static int unregister_memdev_region(struct device *dev, void *data)
{
struct cxl_endpoint_decoder *cxled;
diff --git a/drivers/cxl/cxlmem.h b/drivers/cxl/cxlmem.h
index 8c050bc308bd..9f074007dbf8 100644
--- a/drivers/cxl/cxlmem.h
+++ b/drivers/cxl/cxlmem.h
@@ -123,9 +123,16 @@ struct cxl_attach_region {
};
#ifdef CONFIG_CXL_REGION
+int cxl_memdev_create_ram_region(struct cxl_memdev *cxlmd, u64 size);
int cxl_memdev_attach_region(struct cxl_memdev *cxlmd, struct range *hpa_range);
int cxl_memdev_setup_region_cleanup(struct cxl_memdev *cxlmd);
#else
+static inline int cxl_memdev_create_ram_region(struct cxl_memdev *cxlmd,
+ u64 size)
+{
+ return -EOPNOTSUPP;
+}
+
static inline int cxl_memdev_attach_region(struct cxl_memdev *cxlmd,
struct range *hpa_range)
{
diff --git a/drivers/cxl/mem.c b/drivers/cxl/mem.c
index aa08d88ab104..00bd7d9fe561 100644
--- a/drivers/cxl/mem.c
+++ b/drivers/cxl/mem.c
@@ -241,6 +241,37 @@ struct cxl_memdev *devm_cxl_register_mem(struct cxl_dev_state *cxlds)
}
EXPORT_SYMBOL_NS_GPL(devm_cxl_register_mem, "CXL");
+/**
+ * devm_cxl_create_ram_region - Create a provider-owned RAM region
+ * @cxlmd: provider-owned memdev returned by devm_cxl_register_mem()
+ * @size: requested region size in bytes, a nonzero multiple of 256 MiB
+ *
+ * Reserve @size bytes of volatile DPA and HPA, establish a single-target
+ * decoder path, and commit it. The caller chooses the size; this helper does
+ * not consume the device's entire capacity or replace an existing region.
+ * Obtain the resulting HPA range separately with devm_cxl_attach_mem_region().
+ *
+ * Requires an unused programmable endpoint decoder and a compatible
+ * non-interleaved root window without normalized addressing or an extended
+ * linear cache. Persistent memory provisioning is outside this helper.
+ *
+ * The region is removed and its software-programmed decoders are reset when
+ * the endpoint detaches. Failure unwinds the new allocations and leaves the
+ * memdev registered. Returns zero on success or a negative errno.
+ */
+int devm_cxl_create_ram_region(struct cxl_memdev *cxlmd, u64 size)
+{
+ if (!cxlmd->attach || !size)
+ return -EINVAL;
+
+ guard(device)(&cxlmd->dev);
+ if (!cxlmd->dev.driver || !cxlmd->cxlds)
+ return -ENXIO;
+
+ return cxl_memdev_create_ram_region(cxlmd, size);
+}
+EXPORT_SYMBOL_NS_GPL(devm_cxl_create_ram_region, "CXL");
+
/**
* devm_cxl_attach_mem_region - Attach a registered memdev to its region
* @cxlmd: provider-owned memdev returned by devm_cxl_register_mem()
diff --git a/include/cxl/cxl.h b/include/cxl/cxl.h
index 3019e3ea5f09..b873bd0cfacd 100644
--- a/include/cxl/cxl.h
+++ b/include/cxl/cxl.h
@@ -225,6 +225,7 @@ struct cxl_dev_state *_devm_cxl_dev_state_create(struct device *dev,
})
struct cxl_memdev *devm_cxl_register_mem(struct cxl_dev_state *cxlds);
+int devm_cxl_create_ram_region(struct cxl_memdev *cxlmd, u64 size);
int devm_cxl_attach_mem_region(struct cxl_memdev *cxlmd, struct range *hpa_range);
struct cxl_memdev *devm_cxl_probe_mem(struct cxl_dev_state *cxlds,
struct range *range);
--
2.43.0
next prev parent reply other threads:[~2026-10-07 9:06 UTC|newest]
Thread overview: 4+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-10-07 9:05 [RFC PATCH v2 0/3] cxl: Provide explicit RAM region creation for Type-2 providers Richard Cheng
2026-10-07 9:05 ` [RFC PATCH v2 1/3] cxl/mem: Separate provider registration from region attachment Richard Cheng
2026-10-07 9:05 ` Richard Cheng [this message]
2026-10-07 9:05 ` [RFC PATCH v2 3/3] cxl/test: Exercise explicit Type-2 RAM region creation Richard Cheng
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20261007090540.43817-3-icheng@nvidia.com \
--to=icheng@nvidia.com \
--cc=alison.schofield@intel.com \
--cc=dave.jiang@intel.com \
--cc=dave@stgolabs.net \
--cc=iweiny@kernel.org \
--cc=jic23@kernel.org \
--cc=kaihengf@nvidia.com \
--cc=kobak@nvidia.com \
--cc=kristinc@nvidia.com \
--cc=linux-cxl@vger.kernel.org \
--cc=linux-kernel@vger.kernel.org \
--cc=ming.li@zohomail.com \
--cc=newtonl@nvidia.com \
--cc=vishal.l.verma@intel.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox
all inboxes | Powered by JetHome®