[PATCH v2 1/3] drm/msm/a2xx: Add hangcheck progress detection
From: Dmitry Baryshkov
Date: Wed Sep 30 2026 - 11:23:08 EST
A2xx has no progress callback, so any submit still running when the
hangcheck timer expires is declared hung, no matter whether the CP is
advancing. A false lockup is particularly expensive here: a2xx recovery
is unreliable, and a second reset shortly after the first one leaves
a2xx_hw_init() failing for good, so everything submitted afterwards
fails as well.
Compare the CP IB1/IB2 base and remaining size between timer
expirations, like a6xx does.
Assisted-by: LLM
Signed-off-by: Dmitry Baryshkov <dmitry.baryshkov@xxxxxxxxxxxxxxxx>
---
drivers/gpu/drm/msm/adreno/a2xx_gpu.c | 18 ++++++++++++++++++
1 file changed, 18 insertions(+)
diff --git a/drivers/gpu/drm/msm/adreno/a2xx_gpu.c b/drivers/gpu/drm/msm/adreno/a2xx_gpu.c
index df4cded9143f..59d2f1684e6a 100644
--- a/drivers/gpu/drm/msm/adreno/a2xx_gpu.c
+++ b/drivers/gpu/drm/msm/adreno/a2xx_gpu.c
@@ -489,6 +489,23 @@ static u32 a2xx_get_rptr(struct msm_gpu *gpu, struct msm_ringbuffer *ring)
return ring->memptrs->rptr;
}
+static bool a2xx_progress(struct msm_gpu *gpu, struct msm_ringbuffer *ring)
+{
+ struct msm_cp_state cp_state = {
+ .ib1_base = gpu_read(gpu, REG_AXXX_CP_IB1_BASE),
+ .ib2_base = gpu_read(gpu, REG_AXXX_CP_IB2_BASE),
+ .ib1_rem = gpu_read(gpu, REG_AXXX_CP_IB1_BUFSZ),
+ .ib2_rem = gpu_read(gpu, REG_AXXX_CP_IB2_BUFSZ),
+ };
+ bool progress;
+
+ progress = !!memcmp(&cp_state, &ring->last_cp_state, sizeof(cp_state));
+
+ ring->last_cp_state = cp_state;
+
+ return progress;
+}
+
static struct msm_gpu *a2xx_gpu_init(struct drm_device *dev)
{
struct a2xx_gpu *a2xx_gpu = NULL;
@@ -553,6 +570,7 @@ const struct adreno_gpu_funcs a2xx_gpu_funcs = {
.gpu_state_put = adreno_gpu_state_put,
.create_vm = a2xx_create_vm,
.get_rptr = a2xx_get_rptr,
+ .progress = a2xx_progress,
},
.init = a2xx_gpu_init,
};
--
2.47.3