[PATCH V0 18/21] accel/amdxdna: Implement AIE4 suspend and resume

From: David Zhang

Date: Fri Sep 25 2026 - 21:40:17 EST


Implement suspend and resume callbacks for AIE4 Physical Function (PF),
Virtual Function (VF), and Classic device types:
- Add .suspend and .resume hooks in amdxdna_dev_ops for aie4_pf_ops,
aie4_vf_ops, and aie4_classic_ops.
- Implement aie4_hwctx_suspend_all() to destroy or drain contexts across
all registered clients and wait for in-flight jobs.
- Implement aie4_hwctx_resume_all() to recreate firmware contexts and
kick doorbells via aie4_hwctx_resume_jobs() to resume hardware queue
consumption.
- Guard firmware destroy message in aie4_hwctx_destroy() when context
ID is invalid.
- Restore SR-IOV virtual functions on PF resume via aie4_restore_sriov().

Signed-off-by: David Zhang <yidong.zhang@xxxxxxx>
---
drivers/accel/amdxdna/aie4_ctx.c | 17 +-
drivers/accel/amdxdna/aie4_pci.c | 243 ++++++++++++++++++++++++
drivers/accel/amdxdna/aie4_pci.h | 9 +
drivers/accel/amdxdna/aie4_sriov.c | 4 +-
drivers/accel/amdxdna/amdxdna_pci_drv.h | 3 +
5 files changed, 274 insertions(+), 2 deletions(-)

diff --git a/drivers/accel/amdxdna/aie4_ctx.c b/drivers/accel/amdxdna/aie4_ctx.c
index ec00ea0bbc55..016baf5b02c7 100644
--- a/drivers/accel/amdxdna/aie4_ctx.c
+++ b/drivers/accel/amdxdna/aie4_ctx.c
@@ -260,7 +260,7 @@ void aie4_hwctx_destroy(struct amdxdna_hwctx *hwctx, enum aie4_hwctx_flags flags
if (has_reset)
wake_up_all(&priv->job_list_wq);

- if (flags != AIE4_HWCTX_DISCONNECT)
+ if (flags != AIE4_HWCTX_DISCONNECT && priv->hw_ctx_id != CTX_INVALID_ID)
aie4_msg_destroy_context(ndev, priv->hw_ctx_id);

priv->hw_ctx_id = CTX_INVALID_ID;
@@ -391,6 +391,7 @@ int aie4_hwctx_init(struct amdxdna_hwctx *hwctx)
return -ENOMEM;
hwctx->priv = priv;
priv->hwctx = hwctx;
+ priv->hw_ctx_id = CTX_INVALID_ID;

/*
* io_lock guards the per-hwctx cert_comp binding (the connected sentinel)
@@ -972,6 +973,20 @@ int aie4_hwctx_wait_for_running(struct amdxdna_hwctx *hwctx)
return ret;
}

+void aie4_hwctx_resume_jobs(struct amdxdna_hwctx *hwctx)
+{
+ struct amdxdna_hwctx_priv *priv = hwctx->priv;
+
+ mutex_lock(&priv->io_lock);
+ if (list_empty(&priv->running_job_list)) {
+ mutex_unlock(&priv->io_lock);
+ return;
+ }
+ aie4_doorbell_ring(hwctx);
+ mutex_unlock(&priv->io_lock);
+
+ queue_work(priv->job_work_q, &priv->job_work);
+}
/*
* Submit the command(s) carried by @job into the host queue. Called with
* io_lock held. A single ERT_START_DPU maps to one queue entry; an
diff --git a/drivers/accel/amdxdna/aie4_pci.c b/drivers/accel/amdxdna/aie4_pci.c
index e5266237977c..3c190003865a 100644
--- a/drivers/accel/amdxdna/aie4_pci.c
+++ b/drivers/accel/amdxdna/aie4_pci.c
@@ -1095,11 +1095,250 @@ static void aie4_debugfs_init(struct amdxdna_dev *xdna)
&aie4_ctx_hysteresis_fops);
}

+void aie4_hwctx_suspend_all(struct amdxdna_dev_hdl *ndev, int clean_jobs)
+{
+ struct amdxdna_dev *xdna = ndev->aie.xdna;
+ struct amdxdna_client *client;
+ struct amdxdna_hwctx *hwctx;
+ unsigned long hwctx_id;
+ int idx;
+
+ drm_WARN_ON(&xdna->ddev, !mutex_is_locked(&xdna->dev_lock));
+
+ amdxdna_for_each_client(xdna, client) {
+ idx = srcu_read_lock(&client->hwctx_srcu);
+ amdxdna_for_each_hwctx(client, hwctx_id, hwctx) {
+ /* clean up workers and drain running jobs */
+ if (clean_jobs) {
+ int ret;
+
+ aie4_hwctx_destroy(hwctx, AIE4_HWCTX_ERROR);
+ ret = aie4_hwctx_wait_for_running(hwctx);
+ if (ret)
+ XDNA_WARN(xdna, "hwctx %s wait for running failed %d",
+ hwctx->name, ret);
+ } else {
+ aie4_hwctx_destroy(hwctx, AIE4_HWCTX_NORMAL);
+ }
+ }
+ srcu_read_unlock(&client->hwctx_srcu, idx);
+ }
+
+ XDNA_DBG(xdna, "Finished hwctx suspend");
+}
+
+int aie4_hwctx_resume_all(struct amdxdna_dev_hdl *ndev)
+{
+ struct amdxdna_dev *xdna = ndev->aie.xdna;
+ struct amdxdna_client *client;
+ struct amdxdna_hwctx *hwctx;
+ unsigned long hwctx_id;
+ int ret, idx;
+
+ drm_WARN_ON(&xdna->ddev, !mutex_is_locked(&xdna->dev_lock));
+
+ amdxdna_for_each_client(xdna, client) {
+ idx = srcu_read_lock(&client->hwctx_srcu);
+ amdxdna_for_each_hwctx(client, hwctx_id, hwctx) {
+ ret = aie4_hwctx_create(hwctx);
+ if (ret)
+ goto error;
+ aie4_hwctx_resume_jobs(hwctx);
+ }
+ srcu_read_unlock(&client->hwctx_srcu, idx);
+ }
+
+ XDNA_DBG(xdna, "Finished hwctx resume");
+ return 0;
+error:
+ srcu_read_unlock(&client->hwctx_srcu, idx);
+ XDNA_DBG(xdna, "Failed hwctx resume");
+ return ret;
+}
+
+static int aie4_restore_sriov(struct amdxdna_dev_hdl *ndev)
+{
+ struct amdxdna_dev *xdna = ndev->aie.xdna;
+ struct pci_dev *pdev = to_pci_dev(xdna->ddev.dev);
+ int ret;
+
+ if (ndev->num_vfs) {
+ if (pci_num_vf(pdev) != ndev->num_vfs) {
+ XDNA_ERR(xdna, "inconsistent vf number");
+ return -EINVAL;
+ }
+ ret = aie4_create_vfs(ndev, ndev->num_vfs);
+ if (ret) {
+ XDNA_ERR(xdna, "create vfs failed, %d", ret);
+ return ret;
+ }
+ XDNA_DBG(xdna, "restored num_vfs %d", ndev->num_vfs);
+ }
+
+ return 0;
+}
+
+static int aie4_pf_suspend(struct amdxdna_dev *xdna)
+{
+ struct amdxdna_dev_hdl *ndev = xdna->dev_handle;
+ struct pci_dev *pdev = to_pci_dev(xdna->ddev.dev);
+
+ drm_WARN_ON(&xdna->ddev, !mutex_is_locked(&xdna->dev_lock));
+ aie4_pf_hw_stop(ndev);
+ pci_disable_device(pdev);
+
+ XDNA_DBG(xdna, "pf suspend done");
+ return 0;
+}
+
+static int aie4_pf_resume(struct amdxdna_dev *xdna)
+{
+ struct amdxdna_dev_hdl *ndev = xdna->dev_handle;
+ struct pci_dev *pdev = to_pci_dev(xdna->ddev.dev);
+ int ret;
+
+ drm_WARN_ON(&xdna->ddev, !mutex_is_locked(&xdna->dev_lock));
+
+ ret = pci_enable_device(pdev);
+ if (ret) {
+ XDNA_ERR(xdna, "enable pci device failed %d", ret);
+ return ret;
+ }
+ pci_set_master(pdev);
+
+ ret = aie4_pf_hw_start(ndev);
+ if (ret) {
+ XDNA_ERR(xdna, "hw_start failed %d", ret);
+ goto pci_disable;
+ }
+
+ ret = aie4_restore_sriov(ndev);
+ if (ret)
+ goto hw_stop;
+
+ XDNA_DBG(xdna, "pf resume done");
+ return 0;
+hw_stop:
+ aie4_pf_hw_stop(ndev);
+pci_disable:
+ pci_disable_device(pdev);
+ return ret;
+}
+
+static int aie4_vf_suspend(struct amdxdna_dev *xdna)
+{
+ struct amdxdna_dev_hdl *ndev = xdna->dev_handle;
+ struct pci_dev *pdev = to_pci_dev(xdna->ddev.dev);
+
+ drm_WARN_ON(&xdna->ddev, !mutex_is_locked(&xdna->dev_lock));
+ aie4_hwctx_suspend_all(ndev, false);
+ /*
+ * partition_fini and mailbox messages should not be called here
+ * because PF suspend will do the cleanup for all VFs.
+ */
+ aie4_mailbox_fini(ndev);
+ pci_disable_device(pdev);
+
+ XDNA_DBG(xdna, "vf suspend done");
+ return 0;
+}
+
+static int aie4_vf_resume(struct amdxdna_dev *xdna)
+{
+ struct amdxdna_dev_hdl *ndev = xdna->dev_handle;
+ struct pci_dev *pdev = to_pci_dev(xdna->ddev.dev);
+ int ret;
+
+ drm_WARN_ON(&xdna->ddev, !mutex_is_locked(&xdna->dev_lock));
+
+ ret = pci_enable_device(pdev);
+ if (ret) {
+ XDNA_ERR(xdna, "enable pci device failed %d", ret);
+ return ret;
+ }
+ pci_set_master(pdev);
+
+ ret = aie4_vf_hw_start(ndev);
+ if (ret) {
+ XDNA_ERR(xdna, "hw_start failed %d", ret);
+ goto pci_disable;
+ }
+
+ ret = aie4_hwctx_resume_all(ndev);
+ if (ret) {
+ XDNA_ERR(xdna, "hwctx_resume failed %d", ret);
+ goto hw_clear;
+ }
+
+ XDNA_DBG(xdna, "vf resume done");
+ return 0;
+
+hw_clear:
+ aie4_hwctx_suspend_all(ndev, true);
+ aie4_vf_hw_stop(ndev);
+pci_disable:
+ pci_disable_device(pdev);
+ return ret;
+}
+
+static int aie4_classic_suspend(struct amdxdna_dev *xdna)
+{
+ struct amdxdna_dev_hdl *ndev = xdna->dev_handle;
+ struct pci_dev *pdev = to_pci_dev(xdna->ddev.dev);
+
+ drm_WARN_ON(&xdna->ddev, !mutex_is_locked(&xdna->dev_lock));
+ aie4_hwctx_suspend_all(ndev, false);
+ aie4_classic_hw_stop(ndev);
+ pci_disable_device(pdev);
+
+ XDNA_DBG(xdna, "classic suspend done");
+ return 0;
+}
+
+static int aie4_classic_resume(struct amdxdna_dev *xdna)
+{
+ struct amdxdna_dev_hdl *ndev = xdna->dev_handle;
+ struct pci_dev *pdev = to_pci_dev(xdna->ddev.dev);
+ int ret;
+
+ drm_WARN_ON(&xdna->ddev, !mutex_is_locked(&xdna->dev_lock));
+
+ ret = pci_enable_device(pdev);
+ if (ret) {
+ XDNA_ERR(xdna, "enable pci device failed %d", ret);
+ return ret;
+ }
+ pci_set_master(pdev);
+
+ ret = aie4_classic_hw_start(ndev);
+ if (ret) {
+ XDNA_ERR(xdna, "hw_start failed %d", ret);
+ goto pci_disable;
+ }
+
+ ret = aie4_hwctx_resume_all(ndev);
+ if (ret) {
+ XDNA_ERR(xdna, "hwctx_resume failed %d", ret);
+ goto hw_clear;
+ }
+
+ XDNA_DBG(xdna, "classic resume done");
+ return 0;
+hw_clear:
+ aie4_hwctx_suspend_all(ndev, true);
+ aie4_classic_hw_stop(ndev);
+pci_disable:
+ pci_disable_device(pdev);
+ return ret;
+}
+
const struct amdxdna_dev_ops aie4_pf_ops = {
.init = aie4_pf_init,
.fini = aie4_pf_fini,
.debugfs_init = aie4_debugfs_init,
.sriov_configure = aie4_sriov_configure,
+ .resume = aie4_pf_resume,
+ .suspend = aie4_pf_suspend,
};

const struct amdxdna_dev_ops aie4_vf_ops = {
@@ -1112,6 +1351,8 @@ const struct amdxdna_dev_ops aie4_vf_ops = {
.cmd_wait = aie4_cmd_wait,
.get_aie_info = aie4_get_info,
.set_aie_state = aie4_set_state,
+ .resume = aie4_vf_resume,
+ .suspend = aie4_vf_suspend,
};

const struct amdxdna_dev_ops aie4_classic_ops = {
@@ -1124,4 +1365,6 @@ const struct amdxdna_dev_ops aie4_classic_ops = {
.cmd_wait = aie4_cmd_wait,
.get_aie_info = aie4_get_info,
.set_aie_state = aie4_set_state,
+ .resume = aie4_classic_resume,
+ .suspend = aie4_classic_suspend,
};
diff --git a/drivers/accel/amdxdna/aie4_pci.h b/drivers/accel/amdxdna/aie4_pci.h
index 6b67c9da560e..df15c63317b1 100644
--- a/drivers/accel/amdxdna/aie4_pci.h
+++ b/drivers/accel/amdxdna/aie4_pci.h
@@ -109,6 +109,7 @@ struct amdxdna_dev_hdl {
u32 total_col;
u32 max_aieclk_level;
u32 max_npuhclk_level;
+ u32 num_vfs;

struct dpm_clk_freq dpm_clk_tbl[AIE4_MAX_DPM_LEVEL_COUNT];

@@ -170,8 +171,11 @@ int aie4_cmd_submit(struct amdxdna_hwctx *hwctx, struct amdxdna_sched_job *job,
int aie4_hwctx_create(struct amdxdna_hwctx *hwctx);
void aie4_hwctx_destroy(struct amdxdna_hwctx *hwctx, enum aie4_hwctx_flags);
int aie4_hwctx_wait_for_running(struct amdxdna_hwctx *hwctx);
+void aie4_hwctx_resume_jobs(struct amdxdna_hwctx *hwctx);

/* aie4_pci.c */
+void aie4_hwctx_suspend_all(struct amdxdna_dev_hdl *ndev, int clean_jobs);
+int aie4_hwctx_resume_all(struct amdxdna_dev_hdl *ndev);
int aie4_restore_power_mode(struct amdxdna_dev_hdl *ndev);

/*
@@ -191,9 +195,14 @@ void aie4_free_notification(struct cert_comp *comp);

/* aie4_sriov.c */
#if IS_ENABLED(CONFIG_PCI_IOV)
+int aie4_create_vfs(struct amdxdna_dev_hdl *ndev, int num_vfs);
int aie4_sriov_configure(struct amdxdna_dev *xdna, int num_vfs);
int aie4_sriov_stop(struct amdxdna_dev_hdl *ndev);
#else
+static inline int aie4_create_vfs(struct amdxdna_dev_hdl *ndev, int num_vfs)
+{
+ return 0;
+}
#define aie4_sriov_configure NULL
static inline int aie4_sriov_stop(struct amdxdna_dev_hdl *ndev)
{
diff --git a/drivers/accel/amdxdna/aie4_sriov.c b/drivers/accel/amdxdna/aie4_sriov.c
index e1ce633768a5..0eea28f62676 100644
--- a/drivers/accel/amdxdna/aie4_sriov.c
+++ b/drivers/accel/amdxdna/aie4_sriov.c
@@ -26,7 +26,7 @@ static int aie4_destroy_vfs(struct amdxdna_dev_hdl *ndev)
return ret;
}

-static int aie4_create_vfs(struct amdxdna_dev_hdl *ndev, int num_vfs)
+int aie4_create_vfs(struct amdxdna_dev_hdl *ndev, int num_vfs)
{
DECLARE_AIE_MSG(aie4_msg_create_vfs, AIE4_MSG_OP_CREATE_VFS);
int ret;
@@ -55,6 +55,7 @@ int aie4_sriov_stop(struct amdxdna_dev_hdl *ndev)
}

pci_disable_sriov(pdev);
+ ndev->num_vfs = 0;
return aie4_destroy_vfs(ndev);
}

@@ -75,6 +76,7 @@ static int aie4_sriov_start(struct amdxdna_dev_hdl *ndev, int num_vfs)
return ret;
}

+ ndev->num_vfs = num_vfs;
return num_vfs;
}

diff --git a/drivers/accel/amdxdna/amdxdna_pci_drv.h b/drivers/accel/amdxdna/amdxdna_pci_drv.h
index 11f46ec738d7..45170c945d71 100644
--- a/drivers/accel/amdxdna/amdxdna_pci_drv.h
+++ b/drivers/accel/amdxdna/amdxdna_pci_drv.h
@@ -168,6 +168,9 @@ struct amdxdna_client {
#define amdxdna_for_each_hwctx(client, hwctx_id, entry) \
xa_for_each(&(client)->hwctx_xa, hwctx_id, entry)

+#define amdxdna_for_each_client(xdna, client) \
+ list_for_each_entry(client, &(xdna)->client_list, node)
+
/* Add device info below */
extern const struct amdxdna_dev_info dev_npu1_info;
extern const struct amdxdna_dev_info dev_npu3_classic_info;
--
2.34.1