mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
From: David Zhang <yidong.zhang@amd.com>
To: <quic_jhugo@quicinc.com>, <karol.wachowski@linux.intel.com>,
	<max.zhen@amd.com>, <lizhi.hou@amd.com>, <ogabbay@kernel.org>,
	<dri-devel@lists.freedesktop.org>, <linux-kernel@vger.kernel.org>
Cc: David Zhang <yidong.zhang@amd.com>, <sonal.santan@amd.com>,
	<mario.limonciello@amd.com>, Wendy Liang <wendy.liang@amd.com>
Subject: [PATCH V3 12/19] accel/amdxdna: Implement AIE4 kernel queue lifecycle and memory layout
Date: Wed, 7 Oct 2026 20:23:41 -0700	[thread overview]
Message-ID: <20261008032348.2044667-13-yidong.zhang@amd.com> (raw)
In-Reply-To: <20261008032348.2044667-1-yidong.zhang@amd.com>

Initialize kernel-mode submission required buffers, workqueue, and
hardware contexts:
- Update queue definition that is being used to send requests.
- Add job workqueue for pending and running jobs.
- Add mutex protection for each io.
- Initialize queue with direct and indirect packet, format queue header.
- Add kernel-mode submission required steps in hwctx create/destroy.
- Add aie4_get_cert_comp() to safely acquire a reference to the
  per-hwctx completion tracker under io_lock in aie4_cmd_wait() before
  waiting on the queue, preventing race conditions with
  aie4_hwctx_destroy().

Co-developed-by: Max Zhen <max.zhen@amd.com>
Signed-off-by: Max Zhen <max.zhen@amd.com>
Co-developed-by: Wendy Liang <wendy.liang@amd.com>
Signed-off-by: Wendy Liang <wendy.liang@amd.com>
Signed-off-by: David Zhang <yidong.zhang@amd.com>
---
 drivers/accel/amdxdna/aie4_ctx.c        | 196 ++++++++++++++++++++----
 drivers/accel/amdxdna/aie4_host_queue.h |  65 ++++++++
 drivers/accel/amdxdna/aie4_pci.h        |  40 +++++
 3 files changed, 268 insertions(+), 33 deletions(-)

diff --git a/drivers/accel/amdxdna/aie4_ctx.c b/drivers/accel/amdxdna/aie4_ctx.c
index 5a2fc19bad20..226570367f71 100644
--- a/drivers/accel/amdxdna/aie4_ctx.c
+++ b/drivers/accel/amdxdna/aie4_ctx.c
@@ -22,6 +22,13 @@
 #include "amdxdna_mailbox_helper.h"
 #include "amdxdna_pci_drv.h"
 
+#define CTX_INVALID_ID			(~0U)
+#define CTX_INVALID_DOORBELL		AMDXDNA_INVALID_DOORBELL_OFFSET
+
+static void job_worker(struct work_struct *work)
+{
+}
+
 static struct cert_comp *aie4_lookup_cert_comp(struct amdxdna_dev_hdl *ndev, u32 msix_idx)
 {
 	struct amdxdna_dev *xdna = ndev->aie.xdna;
@@ -38,7 +45,7 @@ static struct cert_comp *aie4_lookup_cert_comp(struct amdxdna_dev_hdl *ndev, u32
 
 	cert_comp = kzalloc_obj(*cert_comp);
 	if (!cert_comp)
-		return NULL;
+		return ERR_PTR(-ENOMEM);
 
 	cert_comp->ndev = ndev;
 	cert_comp->msix_idx = msix_idx;
@@ -65,7 +72,7 @@ static struct cert_comp *aie4_lookup_cert_comp(struct amdxdna_dev_hdl *ndev, u32
 	aie4_free_notification(cert_comp);
 free_cert_comp:
 	kfree(cert_comp);
-	return NULL;
+	return ERR_PTR(ret);
 }
 
 static void cert_comp_release(struct kref *kref)
@@ -73,8 +80,6 @@ static void cert_comp_release(struct kref *kref)
 	struct cert_comp *cert_comp = container_of(kref, struct cert_comp, kref);
 	struct amdxdna_dev_hdl *ndev = cert_comp->ndev;
 
-	drm_WARN_ON(&ndev->aie.xdna->ddev, !mutex_is_locked(&ndev->cert_comp_lock));
-
 	xa_erase(&ndev->cert_comp_xa, cert_comp->msix_idx);
 	aie4_free_notification(cert_comp);
 	kfree(cert_comp);
@@ -82,20 +87,41 @@ static void cert_comp_release(struct kref *kref)
 
 static void aie4_put_cert_comp(struct cert_comp *cert_comp)
 {
-	struct amdxdna_dev_hdl *ndev;
+	struct amdxdna_dev_hdl *ndev = cert_comp->ndev;
 
-	ndev = cert_comp->ndev;
 	guard(mutex)(&ndev->cert_comp_lock);
 
 	kref_put(&cert_comp->kref, cert_comp_release);
 }
 
-static int aie4_msg_destroy_context(struct amdxdna_dev_hdl *ndev, u32 hw_context_id)
+static struct cert_comp *aie4_get_cert_comp(struct amdxdna_hwctx *hwctx)
+{
+	struct amdxdna_hwctx_priv *priv = hwctx->priv;
+	struct cert_comp *cert_comp;
+
+	/* Take io_lock to serialize against cert_comp link/unlink. */
+	guard(mutex)(&priv->io_lock);
+
+	cert_comp = READ_ONCE(priv->cert_comp);
+	if (cert_comp)
+		kref_get(&cert_comp->kref);
+
+	return cert_comp;
+}
+
+static void aie4_msg_destroy_context(struct amdxdna_dev_hdl *ndev, u32 hw_context_id)
 {
 	DECLARE_AIE_MSG(aie4_msg_destroy_hw_context, AIE4_MSG_OP_DESTROY_HW_CONTEXT);
+	struct amdxdna_dev *xdna = ndev->aie.xdna;
+	int ret;
+
+	if (hw_context_id == CTX_INVALID_ID)
+		return;
 
 	req.hw_context_id = hw_context_id;
-	return aie_send_mgmt_msg_wait(&ndev->aie, &msg);
+	ret = aie_send_mgmt_msg_wait(&ndev->aie, &msg);
+	if (ret)
+		XDNA_WARN(xdna, "destroy ctx id %d failed %d", hw_context_id, ret);
 }
 
 static u8 aie4_parse_priority_to_dev(u32 priority)
@@ -114,19 +140,20 @@ static u8 aie4_parse_priority_to_dev(u32 priority)
 	}
 }
 
-static int aie4_hwctx_create(struct amdxdna_hwctx *hwctx)
+int aie4_hwctx_create(struct amdxdna_hwctx *hwctx)
 {
 	DECLARE_AIE_MSG(aie4_msg_create_hw_context, AIE4_MSG_OP_CREATE_HW_CONTEXT);
 	struct amdxdna_client *client = hwctx->client;
 	struct amdxdna_hwctx_priv *priv = hwctx->priv;
-	struct amdxdna_dev *xdna = hwctx->client->xdna;
+	struct amdxdna_dev *xdna = client->xdna;
 	struct amdxdna_dev_hdl *ndev = xdna->dev_handle;
+	struct cert_comp *cert_comp;
 	int ret;
 
 	drm_WARN_ON(&xdna->ddev, !mutex_is_locked(&xdna->dev_lock));
 
 	if (!ndev->partition_id || !hwctx->num_tiles) {
-		XDNA_ERR(xdna, "invalid request partition_id %d, num_tiles %d",
+		XDNA_ERR(xdna, "invalid request partition_id %u, num_tiles %d",
 			 ndev->partition_id, hwctx->num_tiles);
 		return -EINVAL;
 	}
@@ -136,7 +163,6 @@ static int aie4_hwctx_create(struct amdxdna_hwctx *hwctx)
 	req.pasid = aie4_msg_pasid(client);
 	req.pasid = req.pasid == IOMMU_PASID_INVALID ? 0 : req.pasid;
 	req.priority_band = aie4_parse_priority_to_dev(hwctx->qos.priority);
-
 	req.hsa_addr_high = upper_32_bits(amdxdna_gem_dev_addr(priv->umq_bo));
 	req.hsa_addr_low = lower_32_bits(amdxdna_gem_dev_addr(priv->umq_bo));
 
@@ -150,72 +176,142 @@ static int aie4_hwctx_create(struct amdxdna_hwctx *hwctx)
 	}
 
 	XDNA_DBG(xdna, "resp msix: %d, ctx id: %d, doorbell: %d",
-		 resp.job_complete_msix_idx,
-		 resp.hw_context_id,
+		 resp.job_complete_msix_idx, resp.hw_context_id,
 		 resp.doorbell_offset);
 
 	/* setup interrupt completion per msix index */
-	priv->cert_comp = aie4_lookup_cert_comp(ndev, resp.job_complete_msix_idx);
-	if (!priv->cert_comp) {
+	cert_comp = aie4_lookup_cert_comp(ndev, resp.job_complete_msix_idx);
+	if (IS_ERR(cert_comp)) {
 		aie4_msg_destroy_context(ndev, resp.hw_context_id);
-		return -EINVAL;
+		return PTR_ERR(cert_comp);
 	}
 
 	priv->hw_ctx_id = resp.hw_context_id;
-	hwctx->doorbell_offset = AMDXDNA_INVALID_DOORBELL_OFFSET;
+
+	hwctx->fw_ctx_id = resp.hw_context_id;
+	hwctx->start_col = 0;
+	hwctx->num_col = ndev->total_col;
+
+	/* Set up driver doorbell and return invalid offset to userspace. */
+	mutex_lock(&priv->io_lock);
+	ret = aie4_doorbell_setup(hwctx, &resp);
+	if (ret) {
+		mutex_unlock(&priv->io_lock);
+		aie4_msg_destroy_context(ndev, resp.hw_context_id);
+		aie4_put_cert_comp(cert_comp);
+		priv->hw_ctx_id = CTX_INVALID_ID;
+		hwctx->fw_ctx_id = -1;
+		return ret;
+	}
+	WRITE_ONCE(priv->cert_comp, cert_comp);
+	mutex_unlock(&priv->io_lock);
+	hwctx->doorbell_offset = CTX_INVALID_DOORBELL;
+	wake_up_all(&priv->job_list_wq);
 
 	return 0;
 }
 
-static void aie4_hwctx_destroy(struct amdxdna_hwctx *hwctx)
+void aie4_hwctx_destroy(struct amdxdna_hwctx *hwctx, enum aie4_hwctx_flags flags)
 {
 	struct amdxdna_client *client = hwctx->client;
 	struct amdxdna_hwctx_priv *priv = hwctx->priv;
 	struct amdxdna_dev *xdna = client->xdna;
 	struct amdxdna_dev_hdl *ndev = xdna->dev_handle;
+	struct cert_comp *cert_comp;
 
 	drm_WARN_ON(&xdna->ddev, !mutex_is_locked(&xdna->dev_lock));
 
-	aie4_msg_destroy_context(ndev, priv->hw_ctx_id);
-	aie4_put_cert_comp(priv->cert_comp);
+	mutex_lock(&priv->io_lock);
+	cert_comp = priv->cert_comp;
+	WRITE_ONCE(priv->cert_comp, NULL);
+	mutex_unlock(&priv->io_lock);
+
+	if (cert_comp)
+		wake_up_all(&cert_comp->waitq);
+
+	if (flags != AIE4_HWCTX_DISCONNECT)
+		aie4_msg_destroy_context(ndev, priv->hw_ctx_id);
+
+	if (cert_comp)
+		aie4_put_cert_comp(cert_comp);
+
+	priv->hw_ctx_id = CTX_INVALID_ID;
+	hwctx->fw_ctx_id = -1;
+	hwctx->doorbell_offset = CTX_INVALID_DOORBELL;
+
+	cancel_work_sync(&priv->job_work);
 }
 
 static void aie4_hwctx_umq_fini(struct amdxdna_hwctx *hwctx)
 {
 	if (hwctx->priv && hwctx->priv->umq_bo)
-		amdxdna_gem_put_obj(hwctx->priv->umq_bo);
+		drm_gem_object_put(to_gobj(hwctx->priv->umq_bo));
 }
 
 static int aie4_hwctx_umq_init(struct amdxdna_hwctx *hwctx)
 {
+	const size_t indir_pkts_sz = CTX_MAX_CMDS * HSA_MAX_LEVEL1_INDIRECT_ENTRIES *
+				     sizeof(struct host_indirect_packet_data);
+	const size_t pkts_sz = CTX_MAX_CMDS * sizeof(struct host_queue_packet);
 	struct amdxdna_hwctx_priv *priv = hwctx->priv;
 	struct amdxdna_dev *xdna = hwctx->client->xdna;
 	struct amdxdna_gem_obj *umq_bo;
 	struct host_queue_header *qhdr;
+	u64 data_dev_addr;
+	void *umq_va;
 	int ret;
+	int i;
 
+	/* Queue lives in user-allocated BO so device can access it under PASID. */
 	umq_bo = amdxdna_gem_get_obj(hwctx->client, hwctx->umq_bo_hdl, AMDXDNA_BO_SHARE);
 	if (!umq_bo) {
 		XDNA_ERR(xdna, "cannot find umq_bo handle %d", hwctx->umq_bo_hdl);
 		return -ENOENT;
 	}
-	if (umq_bo->mem.size < sizeof(*qhdr)) {
-		XDNA_ERR(xdna, "umq_bo size is too small");
+
+	/* Ensure user BO can hold queue header and direct/indirect packet arrays. */
+	if (umq_bo->mem.size < sizeof(*qhdr) ||
+	    (umq_bo->mem.size < sizeof(*qhdr) + pkts_sz + indir_pkts_sz)) {
+		XDNA_ERR(xdna, "umq_bo size %zu is too small",
+			 (size_t)umq_bo->mem.size);
 		ret = -EINVAL;
 		goto put_umq_bo;
 	}
 
-	/* get kva address for host queue read index and write index */
-	qhdr = amdxdna_gem_vmap(umq_bo);
-	if (!qhdr) {
+	umq_va = amdxdna_gem_vmap(umq_bo);
+	if (!umq_va) {
 		ret = -ENOMEM;
 		goto put_umq_bo;
 	}
+	qhdr = umq_va;
 
 	priv->umq_bo = umq_bo;
 	priv->umq_read_index = &qhdr->read_index;
 	priv->umq_write_index = &qhdr->write_index;
 
+	/* Lay out direct packets after header, followed by indirect packets. */
+	data_dev_addr = amdxdna_gem_dev_addr(umq_bo) + sizeof(*qhdr);
+	priv->umq_pkts = umq_va + sizeof(*qhdr);
+	priv->umq_indirect_pkts = umq_va + sizeof(*qhdr) + pkts_sz;
+	priv->umq_indirect_pkts_dev_addr = data_dev_addr + pkts_sz;
+
+	/* Clear only the driver-managed header and packet regions. */
+	memset(umq_va, 0, sizeof(*qhdr) + pkts_sz + indir_pkts_sz);
+	priv->write_index = QUEUE_INDEX_START;
+	qhdr->read_index = QUEUE_INDEX_START;
+	qhdr->write_index = QUEUE_INDEX_START;
+	qhdr->version.major = HOST_QUEUE_MAJOR_VERSION;
+	qhdr->version.minor = HOST_QUEUE_MINOR_VERSION;
+	qhdr->capacity = CTX_MAX_CMDS;
+	qhdr->data_address = data_dev_addr;
+	for (i = 0; i < CTX_MAX_CMDS; i++)
+		priv->umq_pkts[i].pkt_header.common_header.opcode = OPCODE_EXEC_BUF;
+	for (i = 0; i < CTX_MAX_CMDS * HSA_MAX_LEVEL1_INDIRECT_ENTRIES; i++) {
+		priv->umq_indirect_pkts[i].header.opcode = OPCODE_EXEC_BUF;
+		priv->umq_indirect_pkts[i].header.count = sizeof(struct exec_buf);
+		priv->umq_indirect_pkts[i].header.distribute = 1;
+	}
+
 	return 0;
 
 put_umq_bo:
@@ -227,28 +323,52 @@ int aie4_hwctx_init(struct amdxdna_hwctx *hwctx)
 {
 	struct amdxdna_client *client = hwctx->client;
 	struct amdxdna_dev *xdna = client->xdna;
+	struct amdxdna_dev_hdl *ndev = xdna->dev_handle;
 	struct amdxdna_hwctx_priv *priv;
 	int ret;
 
+	if (!AIE_FEATURE_ON(&ndev->aie, AIE4_HSA_COMMAND))
+		return -EOPNOTSUPP;
+
 	priv = kzalloc_obj(*priv);
 	if (!priv)
 		return -ENOMEM;
 	hwctx->priv = priv;
+	priv->hwctx = hwctx;
+
+	/* Initialize io_lock guarding cert_comp binding. */
+	mutex_init(&priv->io_lock);
+
+	INIT_LIST_HEAD(&priv->pending_job_list);
+	INIT_LIST_HEAD(&priv->running_job_list);
+	init_waitqueue_head(&priv->job_list_wq);
+	INIT_WORK(&priv->job_work, job_worker);
 
 	ret = aie4_hwctx_umq_init(hwctx);
 	if (ret)
-		goto free_priv;
+		goto destroy_lock;
 
 	ret = aie4_hwctx_create(hwctx);
 	if (ret)
 		goto umq_fini;
 
-	XDNA_DBG(xdna, "hwctx %s init completed", hwctx->name);
+	priv->job_work_q = alloc_ordered_workqueue("aie4_job_%d_%d", 0,
+						   client->pid, hwctx->fw_ctx_id);
+	if (!priv->job_work_q) {
+		XDNA_ERR(xdna, "Create job_work_q failed");
+		ret = -ENOMEM;
+		goto destroy_ctx;
+	}
+
+	XDNA_DBG(xdna, "hwctx %d.%d init completed", client->pid, hwctx->fw_ctx_id);
 	return 0;
 
+destroy_ctx:
+	aie4_hwctx_destroy(hwctx, AIE4_HWCTX_NORMAL);
 umq_fini:
 	aie4_hwctx_umq_fini(hwctx);
-free_priv:
+destroy_lock:
+	mutex_destroy(&priv->io_lock);
 	kfree(priv);
 	hwctx->priv = NULL;
 	return ret;
@@ -256,8 +376,14 @@ int aie4_hwctx_init(struct amdxdna_hwctx *hwctx)
 
 void aie4_hwctx_fini(struct amdxdna_hwctx *hwctx)
 {
-	aie4_hwctx_destroy(hwctx);
+	struct amdxdna_hwctx_priv *priv = hwctx->priv;
+
+	aie4_hwctx_destroy(hwctx, AIE4_HWCTX_ERROR);
+	cancel_work_sync(&priv->job_work);
+	if (priv->job_work_q)
+		destroy_workqueue(priv->job_work_q);
 	aie4_hwctx_umq_fini(hwctx);
+	mutex_destroy(&priv->io_lock);
 	kfree(hwctx->priv);
 }
 
@@ -301,10 +427,12 @@ static inline bool check_cmd_done(struct amdxdna_hwctx *hwctx, u64 seq)
 int aie4_cmd_wait(struct amdxdna_hwctx *hwctx, u64 seq, u32 timeout)
 {
 	unsigned long wait_jifs = MAX_SCHEDULE_TIMEOUT;
-	struct amdxdna_hwctx_priv *priv = hwctx->priv;
-	struct cert_comp *cert_comp = priv->cert_comp;
+	struct cert_comp *cert_comp = aie4_get_cert_comp(hwctx);
 	long ret;
 
+	if (!cert_comp)
+		return -EAGAIN;
+
 	if (timeout)
 		wait_jifs = msecs_to_jiffies(timeout);
 
@@ -315,5 +443,7 @@ int aie4_cmd_wait(struct amdxdna_hwctx *hwctx, u64 seq, u32 timeout)
 	if (!ret)
 		ret = -ETIME;
 
+	aie4_put_cert_comp(cert_comp);
+
 	return ret <= 0 ? ret : 0;
 }
diff --git a/drivers/accel/amdxdna/aie4_host_queue.h b/drivers/accel/amdxdna/aie4_host_queue.h
index 95ebbb714c6f..d2ed2fd65fb1 100644
--- a/drivers/accel/amdxdna/aie4_host_queue.h
+++ b/drivers/accel/amdxdna/aie4_host_queue.h
@@ -6,9 +6,14 @@
 #ifndef _AIE4_HOST_QUEUE_H_
 #define _AIE4_HOST_QUEUE_H_
 
+#include <linux/bits.h>
 #include <linux/types.h>
 
 #define CTX_MAX_CMDS                    32
+#define HSA_MAX_LEVEL1_INDIRECT_ENTRIES	6
+#define QUEUE_INDEX_START		0
+#define HOST_QUEUE_MAJOR_VERSION	1
+#define HOST_QUEUE_MINOR_VERSION	0
 
 /* Host queue header layout. */
 struct host_queue_header {
@@ -24,4 +29,64 @@ struct host_queue_header {
 	__u64 data_address; /* The xdna dev addr for payload. */
 } __packed;
 
+/* Payload for an OPCODE_EXEC_BUF host-queue packet (single command). */
+struct exec_buf {
+	u32 dtrace_buf_host_addr_low;
+	u32 dpu_control_code_host_addr_low;
+	u32 dpu_control_code_host_addr_high;
+	u16 args_len;
+	u16 dtrace_buf_host_addr_high;
+	u32 args_host_addr_low;
+	u32 args_host_addr_high;
+} __packed;
+
+#define OPCODE_EXEC_BUF		1
+#define CHAIN_FLG_LAST_CMD	0
+#define CHAIN_FLG_NOT_LAST_CMD	1
+struct common_header {
+	u16 reserved; /* MBZ. */
+	u8 opcode;
+	u8 chain_flag;
+	u16 count;
+	u8 distribute;
+	u8 indirect;
+} __packed;
+
+struct host_queue_packet_header {
+	struct common_header common_header;
+	u64 completion_signal;
+} __packed;
+
+struct host_queue_packet {
+	struct host_queue_packet_header pkt_header;
+	u32 data[12]; /* total 64-byte packet */
+} __packed;
+
+struct host_indirect_packet_entry {
+	u32 host_addr_low;
+	u32 host_addr_high_uc_index;
+} __packed;
+
+#define HIPE_HOST_ADDR_HIGH_SHIFT	0
+#define HIPE_HOST_ADDR_HIGH_MASK	GENMASK(24, 0)
+#define HIPE_UC_INDEX_SHIFT		25
+#define HIPE_UC_INDEX_MASK		GENMASK(31, 25)
+
+static inline void hipe_set_host_addr_high(u32 *val, u32 addr_hi)
+{
+	*val &= ~HIPE_HOST_ADDR_HIGH_MASK;
+	*val |= (addr_hi << HIPE_HOST_ADDR_HIGH_SHIFT) & HIPE_HOST_ADDR_HIGH_MASK;
+}
+
+static inline void hipe_set_uc_index(u32 *val, u32 uc_idx)
+{
+	*val &= ~HIPE_UC_INDEX_MASK;
+	*val |= (uc_idx << HIPE_UC_INDEX_SHIFT) & HIPE_UC_INDEX_MASK;
+}
+
+struct host_indirect_packet_data {
+	struct common_header header;
+	struct exec_buf payload;
+} __packed;
+
 #endif /* _AIE4_HOST_QUEUE_H_ */
diff --git a/drivers/accel/amdxdna/aie4_pci.h b/drivers/accel/amdxdna/aie4_pci.h
index 9fcdfcc5a15f..3d3301ce11ae 100644
--- a/drivers/accel/amdxdna/aie4_pci.h
+++ b/drivers/accel/amdxdna/aie4_pci.h
@@ -8,7 +8,10 @@
 
 #include <linux/device.h>
 #include <linux/iopoll.h>
+#include <linux/list.h>
 #include <linux/pci.h>
+#include <linux/wait.h>
+#include <linux/workqueue.h>
 
 #include "aie.h"
 #include "aie4_msg_priv.h"
@@ -25,7 +28,20 @@ struct cert_comp {
 	wait_queue_head_t               waitq;
 };
 
+/*
+ * aie4 kernel-submission job states (stored in amdxdna_sched_job priv.aie4.state).
+ * Anonymous enum - the aie4_job_state identifier is already a field-access macro.
+ */
+enum {
+	AIE4_JOB_STATE_INIT,
+	AIE4_JOB_STATE_PENDING,
+	AIE4_JOB_STATE_SUBMITTING,
+	AIE4_JOB_STATE_SUBMITTED,
+	AIE4_JOB_STATE_DONE,
+};
+
 struct amdxdna_hwctx_priv {
+	struct amdxdna_hwctx		*hwctx;
 	struct amdxdna_gem_obj          *umq_bo;
 	u64                             *umq_read_index;
 	u64                             *umq_write_index;
@@ -33,7 +49,22 @@ struct amdxdna_hwctx_priv {
 	struct cert_comp                *cert_comp;
 	u32                             hw_ctx_id;
 
+	/* Direct and indirect packet storage aliasing umq_bo. */
+	u64                             write_index;
+	struct host_queue_packet        *umq_pkts;
+	struct host_indirect_packet_data *umq_indirect_pkts;
+	u64                             umq_indirect_pkts_dev_addr;
+	/* Transport doorbell target set by aie4_doorbell_setup(). */
 	void                    __iomem *doorbell_addr;
+
+	struct mutex                    io_lock; /* serialize submit, protect job lists */
+	struct list_head                pending_job_list;
+	/* Head of pending_job_list, read locklessly by wait conditions. */
+	struct amdxdna_sched_job        *pending_head;
+	struct list_head                running_job_list;
+	wait_queue_head_t               job_list_wq;
+	struct work_struct              job_work;
+	struct workqueue_struct         *job_work_q;
 };
 
 struct amdxdna_dev_priv {
@@ -107,9 +138,18 @@ int aie4_set_ctx_hysteresis(struct amdxdna_dev_hdl *ndev, u32 timeout_us);
 u32 aie4_msg_pasid(struct amdxdna_client *client);
 
 /* aie4_ctx.c */
+enum aie4_hwctx_flags {
+	AIE4_HWCTX_NORMAL = 0,
+	AIE4_HWCTX_GRACEFUL,
+	AIE4_HWCTX_DISCONNECT, /* sets has_error, do not destroy context */
+	AIE4_HWCTX_ERROR, /* sets has_error, destroy context */
+};
+
 int aie4_hwctx_init(struct amdxdna_hwctx *hwctx);
 void aie4_hwctx_fini(struct amdxdna_hwctx *hwctx);
 int aie4_cmd_wait(struct amdxdna_hwctx *hwctx, u64 seq, u32 timeout);
+int aie4_hwctx_create(struct amdxdna_hwctx *hwctx);
+void aie4_hwctx_destroy(struct amdxdna_hwctx *hwctx, enum aie4_hwctx_flags);
 
 /* aie4_pci.c */
 int aie4_restore_power_mode(struct amdxdna_dev_hdl *ndev);
-- 
2.34.1


  parent reply	other threads:[~2026-10-08  3:24 UTC|newest]

Thread overview: 24+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-10-08  3:23 [PATCH V3 00/19] accel/amdxdna: Kernel submission and PM for AIE4 David Zhang
2026-10-08  3:23 ` [PATCH V3 01/19] accel/amdxdna: Rename NPU3 firmware files David Zhang
2026-10-08 16:23   ` Lizhi Hou
2026-10-08  3:23 ` [PATCH V3 02/19] accel/amdxdna: Remove mmap for doorbell David Zhang
2026-10-08  3:23 ` [PATCH V3 03/19] accel/amdxdna: Add CERT firmware version support David Zhang
2026-10-08 17:12   ` Lizhi Hou
2026-10-08 17:26     ` Zhang, Yidong (David)
2026-10-08  3:23 ` [PATCH V3 04/19] accel/amdxdna: Upgrade firmware version to 6.0 David Zhang
2026-10-08 17:15   ` Lizhi Hou
2026-10-08  3:23 ` [PATCH V3 05/19] accel/amdxdna: Add NPU3 classic device support David Zhang
2026-10-08  3:23 ` [PATCH V3 06/19] accel/amdxdna: Add AIE version query to aie4_get_info David Zhang
2026-10-08  3:23 ` [PATCH V3 07/19] accel/amdxdna: Add get and set power_mode for AIE4 David Zhang
2026-10-08  3:23 ` [PATCH V3 08/19] accel/amdxdna: Add clock, DPM frequency, and resource info queries " David Zhang
2026-10-08  3:23 ` [PATCH V3 09/19] accel/amdxdna: Add context switch hysteresis with debugfs control David Zhang
2026-10-08  3:23 ` [PATCH V3 10/19] accel/amdxdna: Refactor AIE4 hardware initialization sequence David Zhang
2026-10-08  3:23 ` [PATCH V3 11/19] accel/amdxdna: Decouple AIE4 doorbell and MSI-X notify transport hooks David Zhang
2026-10-08  3:23 ` David Zhang [this message]
2026-10-08  3:23 ` [PATCH V3 13/19] accel/amdxdna: Prepare for AIE4 command submission David Zhang
2026-10-08  3:23 ` [PATCH V3 14/19] accel/amdxdna: Move HMM invalidate wait into common GEM code David Zhang
2026-10-08  3:23 ` [PATCH V3 15/19] accel/amdxdna: Make populate_range common for AIE2 and AIE4 David Zhang
2026-10-08  3:23 ` [PATCH V3 16/19] accel/amdxdna: Implement AIE4 command packet building and submission David Zhang
2026-10-08  3:23 ` [PATCH V3 17/19] accel/amdxdna: Enable AIE4 firmware logging to DRAM David Zhang
2026-10-08  3:23 ` [PATCH V3 18/19] accel/amdxdna: Implement AIE4 suspend and resume David Zhang
2026-10-08  3:23 ` [PATCH V3 19/19] accel/amdxdna: Implement runtime suspend and resume support David Zhang

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20261008032348.2044667-13-yidong.zhang@amd.com \
    --to=yidong.zhang@amd.com \
    --cc=dri-devel@lists.freedesktop.org \
    --cc=karol.wachowski@linux.intel.com \
    --cc=linux-kernel@vger.kernel.org \
    --cc=lizhi.hou@amd.com \
    --cc=mario.limonciello@amd.com \
    --cc=max.zhen@amd.com \
    --cc=ogabbay@kernel.org \
    --cc=quic_jhugo@quicinc.com \
    --cc=sonal.santan@amd.com \
    --cc=wendy.liang@amd.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®