[PATCH V3 13/19] accel/amdxdna: Prepare for AIE4 command submission

From: David Zhang

Date: Wed Oct 07 2026 - 23:26:05 EST


Prepare the data structures and completion wait helpers required for
AIE4 command submission:
- Define struct amdxdna_cmd_start_dpu in amdxdna_ctx.h for the
ERT_START_DPU payload.
- Extend union amdxdna_job_priv with an aie4 member for queue list
linkage and job state tracking.
- Implement smp_rmb() ordering and non-sleeping retry in
get_read_index(), returning zero (no completion) when the sample
is still invalid after the retry.
- Update check_cmd_done() and aie4_cmd_wait() to detect asynchronous
device disconnect and reset via check_cert_comp_linked().

get_read_index() returns 0 for an invalid sample. Since 0 <= any
sequence number, callers never treat it as a completion. If this
happens while the context is disconnected, aie4_cmd_wait() returns
-EAGAIN so the caller retries the wait until a valid read_index is
observed; the command state and fence are not affected.

Co-developed-by: Max Zhen <max.zhen@xxxxxxx>
Signed-off-by: Max Zhen <max.zhen@xxxxxxx>
Co-developed-by: Wendy Liang <wendy.liang@xxxxxxx>
Signed-off-by: Wendy Liang <wendy.liang@xxxxxxx>
Signed-off-by: David Zhang <yidong.zhang@xxxxxxx>
---
drivers/accel/amdxdna/aie4_ctx.c | 58 ++++++++++++++++++++---------
drivers/accel/amdxdna/amdxdna_ctx.h | 20 ++++++++++
2 files changed, 60 insertions(+), 18 deletions(-)

diff --git a/drivers/accel/amdxdna/aie4_ctx.c b/drivers/accel/amdxdna/aie4_ctx.c
index 226570367f71..09c92b1b5134 100644
--- a/drivers/accel/amdxdna/aie4_ctx.c
+++ b/drivers/accel/amdxdna/aie4_ctx.c
@@ -394,34 +394,46 @@ static inline bool valid_queue_index(u64 read, u64 write, u32 capacity)

static u64 get_read_index(struct amdxdna_hwctx *hwctx)
{
- u64 wi = READ_ONCE(*hwctx->priv->umq_write_index);
- u64 ri = READ_ONCE(*hwctx->priv->umq_read_index);
+ struct amdxdna_hwctx_priv *priv = hwctx->priv;
struct amdxdna_dev *xdna = hwctx->client->xdna;
+ u64 ri, wi;
+
+ /* Sample read_index before write_index to guarantee wi >= ri. */
+ ri = READ_ONCE(*priv->umq_read_index);
+ /* Order the read_index sample before the write_index sample. */
+ smp_rmb();
+ wi = READ_ONCE(priv->write_index);

- /*
- * CERT cannot update read index as uint64 atomically. Driver may read
- * half-updated read index when it has bits in high 32bit. In case read
- * index is not valid, wait for some time and retry once. It should
- * allow CERT to complete the read index update.
- */
+ /* Non-atomic 64-bit counter update by CERT; re-sample once if invalid. */
if (!valid_queue_index(ri, wi, CTX_MAX_CMDS)) {
- XDNA_WARN(xdna, "Invalid index, ri %llu, wi %llu", ri, wi);
- usleep_range(100, 200);
- ri = READ_ONCE(*hwctx->priv->umq_read_index);
+ ri = READ_ONCE(*priv->umq_read_index);
+ /* Order the read_index sample before the write_index sample. */
+ smp_rmb();
+ wi = READ_ONCE(priv->write_index);
if (!valid_queue_index(ri, wi, CTX_MAX_CMDS)) {
- XDNA_ERR(xdna, "Invalid index after retry, ri %llu, wi %llu", ri, wi);
- ri = 0;
+ /* 0 <= any seq: callers always treat it as "not done" and retry. */
+ XDNA_DBG(xdna, "Invalid index, ri %llu, wi %llu", ri, wi);
+ return 0;
}
}

return ri;
}

-static inline bool check_cmd_done(struct amdxdna_hwctx *hwctx, u64 seq)
+/* Verify cert_comp remains linked to detect disconnect or reset. */
+static bool check_cert_comp_linked(struct amdxdna_hwctx *hwctx, struct cert_comp *comp)
+{
+ /* READ_ONCE pairs with the link/unlink WRITE_ONCE. */
+ return comp == READ_ONCE(hwctx->priv->cert_comp);
+}
+
+static inline bool check_cmd_done(struct amdxdna_hwctx *hwctx, u64 seq, struct cert_comp *comp)
{
- u64 read_idx = get_read_index(hwctx);
+ /* Lockless check for wait_event condition; detects completion or disconnect. */
+ if (!check_cert_comp_linked(hwctx, comp))
+ return true;

- return read_idx > seq;
+ return get_read_index(hwctx) > seq;
}

int aie4_cmd_wait(struct amdxdna_hwctx *hwctx, u64 seq, u32 timeout)
@@ -437,11 +449,21 @@ int aie4_cmd_wait(struct amdxdna_hwctx *hwctx, u64 seq, u32 timeout)
wait_jifs = msecs_to_jiffies(timeout);

ret = wait_event_interruptible_timeout(cert_comp->waitq,
- (check_cmd_done(hwctx, seq)),
+ check_cmd_done(hwctx, seq, cert_comp),
wait_jifs);

- if (!ret)
+ if (!ret) {
ret = -ETIME;
+ } else if (ret > 0 && !check_cert_comp_linked(hwctx, cert_comp) &&
+ get_read_index(hwctx) <= seq) {
+ /*
+ * Disconnected, or read_index invalid (0), before completion was
+ * seen. -EAGAIN is by design: the caller retries the wait until a
+ * valid read_index is observed. Command state and fence are not
+ * affected.
+ */
+ ret = -EAGAIN;
+ }

aie4_put_cert_comp(cert_comp);

diff --git a/drivers/accel/amdxdna/amdxdna_ctx.h b/drivers/accel/amdxdna/amdxdna_ctx.h
index 9bbc3db4ebde..b3677851d1c5 100644
--- a/drivers/accel/amdxdna/amdxdna_ctx.h
+++ b/drivers/accel/amdxdna/amdxdna_ctx.h
@@ -48,6 +48,18 @@ struct amdxdna_cmd_start_npu {
u32 prop_args[]; /* properties and regular kernel arguments */
};

+/*
+ * struct amdxdna_cmd_start_dpu - interpretation of data payload for
+ * ERT_START_DPU in amdxdna_cmd.
+ */
+struct amdxdna_cmd_start_dpu {
+ u64 dtrace_buffer; /* dtrace buffer address 2 words */
+ u64 instruction_buffer; /* buffer address 2 words */
+ u32 instruction_buffer_size; /* size of buffer in bytes */
+ u16 uc_index; /* microblaze controller index */
+ u16 chained; /* number of following amdxdna_cmd_start_dpu elements */
+};
+
/*
* Interpretation of the beginning of data payload for ERT_CMD_CHAIN in
* amdxdna_cmd. The rest of the payload in amdxdna_cmd is cmd BO handles.
@@ -138,8 +150,14 @@ struct amdxdna_drv_cmd {
};

struct app_health_report;
+
union amdxdna_job_priv {
struct app_health_report *aie2_health;
+ /* aie4 kernel submission: queue linkage + job state */
+ struct {
+ struct list_head list;
+ u32 state;
+ } aie4;
};

struct amdxdna_sched_job {
@@ -162,6 +180,8 @@ struct amdxdna_sched_job {
};

#define aie2_job_health priv.aie2_health
+#define aie4_job_list priv.aie4.list
+#define aie4_job_state priv.aie4.state

static inline u32
amdxdna_cmd_get_op(struct amdxdna_gem_obj *abo)
--
2.34.1