From: Max Zhen <max.zhen@amd.com>
To: Lizhi Hou <lizhi.hou@amd.com>, <ogabbay@kernel.org>,
<quic_jhugo@quicinc.com>, <dri-devel@lists.freedesktop.org>,
<mario.limonciello@amd.com>, <karol.wachowski@linux.intel.com>
Cc: <linux-kernel@vger.kernel.org>, <sonal.santan@amd.com>,
Jonghyuk Kim <malhyuk97@gmail.com>
Subject: Re: [PATCH V1] accel/amdxdna: Preallocate hardware context schedulers
Date: Wed, 23 Sep 2026 13:58:47 -0700 [thread overview]
Message-ID: <a570abe2-4da7-4a0f-8c7f-4c227ac924cd@amd.com> (raw)
In-Reply-To: <20260923200501.281208-1-lizhi.hou@amd.com>
On 9/23/2026 Wed 13:05, Lizhi Hou wrote:
> The hardware context scheduler fence is exposed through
> drm_syncobj_add_point(). After a hardware context is destroyed,
> userspace can still call drm_sched_fence_get_timeline_name() on the
> fence through SYNC_IOC_FILE_INFO. This results in a use-after-free
> when accessing fence->sched->name.
>
> Preallocate hardware context schedulers and free them during device
> teardown instead. A destroyed hardware context no longer frees its
> scheduler, ensuring that the scheduler remains valid while its fences
> are still accessible.
>
> Fixes: aac243092b70 ("accel/amdxdna: Add command execution")
> Reported-by: Jonghyuk Kim(MalHyuk) <malhyuk97@gmail.com>
> Link: https://lore.kernel.org/dri-devel/20260902012712.880520-1-malhyuk97@gmail.com/
> Signed-off-by: Lizhi Hou <lizhi.hou@amd.com>
Reviewed-by: Max Zhen <max.zhen@amd.com>
> ---
> drivers/accel/amdxdna/aie2_ctx.c | 117 +++++++++++++++++++++++--------
> drivers/accel/amdxdna/aie2_pci.c | 15 +++-
> drivers/accel/amdxdna/aie2_pci.h | 6 +-
> 3 files changed, 105 insertions(+), 33 deletions(-)
>
> diff --git a/drivers/accel/amdxdna/aie2_ctx.c b/drivers/accel/amdxdna/aie2_ctx.c
> index 164441e43940..d927c8c9d557 100644
> --- a/drivers/accel/amdxdna/aie2_ctx.c
> +++ b/drivers/accel/amdxdna/aie2_ctx.c
> @@ -7,6 +7,7 @@
> #include <drm/drm_device.h>
> #include <drm/drm_gem.h>
> #include <drm/drm_gem_shmem_helper.h>
> +#include <drm/drm_managed.h>
> #include <drm/drm_print.h>
> #include <drm/drm_syncobj.h>
> #include <linux/hmm.h>
> @@ -99,9 +100,9 @@ static void aie2_job_put(struct amdxdna_sched_job *job)
> static void aie2_hwctx_stop(struct amdxdna_dev *xdna, struct amdxdna_hwctx *hwctx,
> struct drm_sched_job *bad_job)
> {
> - drm_sched_stop(&hwctx->priv->sched, bad_job);
> + drm_sched_stop(hwctx->priv->sched, bad_job);
> aie2_destroy_context(xdna->dev_handle, hwctx);
> - drm_sched_start(&hwctx->priv->sched, 0);
> + drm_sched_start(hwctx->priv->sched, 0);
> }
>
> static int aie2_hwctx_restart(struct amdxdna_dev *xdna, struct amdxdna_hwctx *hwctx)
> @@ -659,10 +660,19 @@ static void aie2_ctx_syncobj_destroy(struct amdxdna_hwctx *hwctx)
> drm_syncobj_put(hwctx->priv->syncobj);
> }
>
> -int aie2_hwctx_init(struct amdxdna_hwctx *hwctx)
> +void aie2_hwctx_sched_fini(struct amdxdna_dev_hdl *ndev)
> {
> - struct amdxdna_client *client = hwctx->client;
> - struct amdxdna_dev *xdna = client->xdna;
> + int i;
> +
> + for (i = 0; i < ndev->priv->hwctx_limit; i++)
> + drm_sched_fini(&ndev->hwctx_sched[i]);
> +
> + ida_destroy(&ndev->hwctx_sched_ida);
> +}
> +
> +int aie2_hwctx_sched_init(struct amdxdna_dev_hdl *ndev)
> +{
> + struct amdxdna_dev *xdna = ndev->aie.xdna;
> const struct drm_sched_init_args args = {
> .ops = &sched_ops,
> .num_rqs = DRM_SCHED_PRIORITY_COUNT,
> @@ -673,7 +683,55 @@ int aie2_hwctx_init(struct amdxdna_hwctx *hwctx)
> .name = "amdxdna_js",
> .dev = xdna->ddev.dev,
> };
> - struct drm_gpu_scheduler *sched;
> + int i, ret;
> +
> + ndev->hwctx_sched = drmm_kcalloc(&xdna->ddev, ndev->priv->hwctx_limit,
> + sizeof(*ndev->hwctx_sched), GFP_KERNEL);
> + if (!ndev->hwctx_sched)
> + return -ENOMEM;
> +
> + for (i = 0; i < ndev->priv->hwctx_limit; i++) {
> + ret = drm_sched_init(&ndev->hwctx_sched[i], &args);
> + if (ret) {
> + XDNA_ERR(xdna, "Failed to init DRM scheduler. ret %d", ret);
> + goto fini_sched;
> + }
> + }
> +
> + ida_init(&ndev->hwctx_sched_ida);
> +
> + return 0;
> +
> +fini_sched:
> + while (--i >= 0)
> + drm_sched_fini(&ndev->hwctx_sched[i]);
> +
> + return ret;
> +}
> +
> +static struct drm_gpu_scheduler *aie2_hwctx_sched_alloc(struct amdxdna_dev_hdl *ndev)
> +{
> + int ret;
> +
> + ret = ida_alloc_range(&ndev->hwctx_sched_ida, 0,
> + ndev->priv->hwctx_limit - 1, GFP_KERNEL);
> + if (ret < 0)
> + return ERR_PTR(ret);
> +
> + return &ndev->hwctx_sched[ret];
> +}
> +
> +static void aie2_hwctx_sched_free(struct amdxdna_dev_hdl *ndev,
> + struct drm_gpu_scheduler *sched)
> +{
> + ida_free(&ndev->hwctx_sched_ida, sched - ndev->hwctx_sched);
> +}
> +
> +int aie2_hwctx_init(struct amdxdna_hwctx *hwctx)
> +{
> + struct amdxdna_client *client = hwctx->client;
> + struct amdxdna_dev *xdna = client->xdna;
> + struct amdxdna_dev_hdl *ndev = xdna->dev_handle;
> struct amdxdna_hwctx_priv *priv;
> struct amdxdna_gem_obj *heap;
> int i, ret;
> @@ -683,13 +741,26 @@ int aie2_hwctx_init(struct amdxdna_hwctx *hwctx)
> return -ENOMEM;
> hwctx->priv = priv;
>
> + priv->sched = aie2_hwctx_sched_alloc(ndev);
> + if (IS_ERR(priv->sched)) {
> + ret = PTR_ERR(priv->sched);
> + goto free_priv;
> + }
> +
> + ret = drm_sched_entity_init(&priv->entity, DRM_SCHED_PRIORITY_NORMAL,
> + &priv->sched, 1, NULL);
> + if (ret) {
> + XDNA_ERR(xdna, "Failed to initial sched entity. ret %d", ret);
> + goto free_sched;
> + }
> +
> mutex_lock(&client->mm_lock);
> heap = xa_load(&client->dev_heap_xa, 0);
> if (!heap) {
> XDNA_ERR(xdna, "The client dev heap object not exist");
> mutex_unlock(&client->mm_lock);
> ret = -ENOENT;
> - goto free_priv;
> + goto free_entity;
> }
> drm_gem_object_get(to_gobj(heap));
> mutex_unlock(&client->mm_lock);
> @@ -722,30 +793,16 @@ int aie2_hwctx_init(struct amdxdna_hwctx *hwctx)
> priv->cmd_buf[i] = abo;
> }
>
> - sched = &priv->sched;
> mutex_init(&priv->io_lock);
>
> fs_reclaim_acquire(GFP_KERNEL);
> might_lock(&priv->io_lock);
> fs_reclaim_release(GFP_KERNEL);
>
> - ret = drm_sched_init(sched, &args);
> - if (ret) {
> - XDNA_ERR(xdna, "Failed to init DRM scheduler. ret %d", ret);
> - goto free_cmd_bufs;
> - }
> -
> - ret = drm_sched_entity_init(&priv->entity, DRM_SCHED_PRIORITY_NORMAL,
> - &sched, 1, NULL);
> - if (ret) {
> - XDNA_ERR(xdna, "Failed to initial sched entiry. ret %d", ret);
> - goto free_sched;
> - }
> -
> ret = aie2_hwctx_col_list(hwctx);
> if (ret) {
> XDNA_ERR(xdna, "Create col list failed, ret %d", ret);
> - goto free_entity;
> + goto free_cmd_bufs;
> }
>
> ret = amdxdna_pm_resume_get_locked(xdna);
> @@ -791,10 +848,6 @@ int aie2_hwctx_init(struct amdxdna_hwctx *hwctx)
> amdxdna_pm_suspend_put(xdna);
> free_col_list:
> kfree(hwctx->col_list);
> -free_entity:
> - drm_sched_entity_destroy(&priv->entity);
> -free_sched:
> - drm_sched_fini(&priv->sched);
> free_cmd_bufs:
> for (i = 0; i < ARRAY_SIZE(priv->cmd_buf); i++) {
> if (!priv->cmd_buf[i])
> @@ -805,6 +858,10 @@ int aie2_hwctx_init(struct amdxdna_hwctx *hwctx)
> amdxdna_gem_unpin(heap);
> put_heap:
> drm_gem_object_put(to_gobj(heap));
> +free_entity:
> + drm_sched_entity_destroy(&priv->entity);
> +free_sched:
> + aie2_hwctx_sched_free(ndev, priv->sched);
> free_priv:
> kfree(priv);
> return ret;
> @@ -812,18 +869,20 @@ int aie2_hwctx_init(struct amdxdna_hwctx *hwctx)
>
> void aie2_hwctx_fini(struct amdxdna_hwctx *hwctx)
> {
> + struct amdxdna_dev_hdl *ndev;
> struct amdxdna_dev *xdna;
> int idx;
>
> xdna = hwctx->client->xdna;
> + ndev = xdna->dev_handle;
>
> XDNA_DBG(xdna, "%s sequence number %lld", hwctx->name, hwctx->priv->seq);
> aie2_hwctx_wait_for_idle(hwctx);
>
> /* Request fw to destroy hwctx and cancel the rest pending requests */
> - drm_sched_stop(&hwctx->priv->sched, NULL);
> + drm_sched_stop(hwctx->priv->sched, NULL);
> aie2_release_resource(hwctx);
> - drm_sched_start(&hwctx->priv->sched, 0);
> + drm_sched_start(hwctx->priv->sched, 0);
>
> mutex_unlock(&xdna->dev_lock);
> drm_sched_entity_destroy(&hwctx->priv->entity);
> @@ -834,7 +893,7 @@ void aie2_hwctx_fini(struct amdxdna_hwctx *hwctx)
> atomic64_read(&hwctx->job_free_cnt));
> mutex_lock(&xdna->dev_lock);
>
> - drm_sched_fini(&hwctx->priv->sched);
> + aie2_hwctx_sched_free(ndev, hwctx->priv->sched);
> aie2_ctx_syncobj_destroy(hwctx);
>
> for (idx = 0; idx < ARRAY_SIZE(hwctx->priv->cmd_buf); idx++) {
> diff --git a/drivers/accel/amdxdna/aie2_pci.c b/drivers/accel/amdxdna/aie2_pci.c
> index daec1f6b4907..7a4314ca843b 100644
> --- a/drivers/accel/amdxdna/aie2_pci.c
> +++ b/drivers/accel/amdxdna/aie2_pci.c
> @@ -503,10 +503,16 @@ static int aie2_init(struct amdxdna_dev *xdna)
> ndev->priv = xdna->dev_info->dev_priv;
> ndev->aie.xdna = xdna;
>
> + ret = aie2_hwctx_sched_init(ndev);
> + if (ret)
> + return ret;
> +
> for (i = 0; i < ARRAY_SIZE(npu_fw); i++) {
> fw_full_path = kasprintf(GFP_KERNEL, "%s%s", ndev->priv->fw_path, npu_fw[i]);
> - if (!fw_full_path)
> - return -ENOMEM;
> + if (!fw_full_path) {
> + ret = -ENOMEM;
> + goto free_hwctx_sched;
> + }
>
> ret = firmware_request_nowarn(&fw, fw_full_path, &pdev->dev);
> kfree(fw_full_path);
> @@ -519,7 +525,7 @@ static int aie2_init(struct amdxdna_dev *xdna)
> if (ret) {
> XDNA_ERR(xdna, "failed to request_firmware %s, ret %d",
> ndev->priv->fw_path, ret);
> - return ret;
> + goto free_hwctx_sched;
> }
>
> ret = pcim_enable_device(pdev);
> @@ -623,6 +629,8 @@ static int aie2_init(struct amdxdna_dev *xdna)
> aie2_hw_stop(xdna);
> release_fw:
> release_firmware(fw);
> +free_hwctx_sched:
> + aie2_hwctx_sched_fini(ndev);
>
> return ret;
> }
> @@ -631,6 +639,7 @@ static void aie2_fini(struct amdxdna_dev *xdna)
> {
> amdxdna_pm_fini(xdna);
> aie2_hw_stop(xdna);
> + aie2_hwctx_sched_fini(xdna->dev_handle);
> }
>
> static int aie2_get_aie_status(struct amdxdna_client *client,
> diff --git a/drivers/accel/amdxdna/aie2_pci.h b/drivers/accel/amdxdna/aie2_pci.h
> index ea1dac106400..2c7019bd26b5 100644
> --- a/drivers/accel/amdxdna/aie2_pci.h
> +++ b/drivers/accel/amdxdna/aie2_pci.h
> @@ -107,7 +107,7 @@ struct amdxdna_hwctx_priv {
> struct amdxdna_gem_obj *heap;
> void *mbox_chann;
>
> - struct drm_gpu_scheduler sched;
> + struct drm_gpu_scheduler *sched;
> struct drm_sched_entity entity;
>
> struct mutex io_lock; /* protect seq and cmd order */
> @@ -152,6 +152,8 @@ struct amdxdna_dev_hdl {
> u32 total_col;
> struct amdxdna_drm_query_aie_version version;
> struct aie2_exec_msg_ops *exec_msg_ops;
> + struct drm_gpu_scheduler *hwctx_sched;
> + struct ida hwctx_sched_ida;
>
> /* power management and clock*/
> enum amdxdna_power_mode_type pw_mode;
> @@ -294,6 +296,8 @@ int aie2_update_prop_time_quota(struct amdxdna_dev_hdl *ndev, u32 us);
> /* aie2_hwctx.c */
> int aie2_hwctx_init(struct amdxdna_hwctx *hwctx);
> void aie2_hwctx_fini(struct amdxdna_hwctx *hwctx);
> +int aie2_hwctx_sched_init(struct amdxdna_dev_hdl *ndev);
> +void aie2_hwctx_sched_fini(struct amdxdna_dev_hdl *ndev);
> int aie2_hwctx_config(struct amdxdna_hwctx *hwctx, u32 type, u64 value, void *buf, u32 size);
> int aie2_hwctx_sync_debug_bo(struct amdxdna_hwctx *hwctx, u32 debug_bo_hdl);
> void aie2_hwctx_suspend(struct amdxdna_client *client);
prev parent reply other threads:[~2026-09-23 20:59 UTC|newest]
Thread overview: 2+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-23 20:05 Lizhi Hou
2026-09-23 20:58 ` Max Zhen [this message]
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=a570abe2-4da7-4a0f-8c7f-4c227ac924cd@amd.com \
--to=max.zhen@amd.com \
--cc=dri-devel@lists.freedesktop.org \
--cc=karol.wachowski@linux.intel.com \
--cc=linux-kernel@vger.kernel.org \
--cc=lizhi.hou@amd.com \
--cc=malhyuk97@gmail.com \
--cc=mario.limonciello@amd.com \
--cc=ogabbay@kernel.org \
--cc=quic_jhugo@quicinc.com \
--cc=sonal.santan@amd.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox
all inboxes | Powered by JetHome®