Re: [PATCH v7 07/17] drm/panfrost: Stop all jobs before commencing device teardown
From: Boris Brezillon
Date: Tue Sep 01 2026 - 10:08:18 EST
On Fri, 28 Aug 2026 21:56:47 +0100
Adrián Larumbe <adrian.larumbe@xxxxxxxxxxxxx> wrote:
> We should make sure there are no inflight jobs at the time we finish driver
> structures and synchronise their IRQs, and even more so by the time clocks
> and power domains are disabled.
>
> To that end, stop submissions from existing scheduler entities and force HW
> to stop all running jobs.
>
> Signed-off-by: Adrián Larumbe <adrian.larumbe@xxxxxxxxxxxxx>
> ---
> drivers/gpu/drm/panfrost/panfrost_device.c | 2 +
> drivers/gpu/drm/panfrost/panfrost_job.c | 71 ++++++++++++++++++------------
> drivers/gpu/drm/panfrost/panfrost_job.h | 1 +
> 3 files changed, 46 insertions(+), 28 deletions(-)
>
> diff --git a/drivers/gpu/drm/panfrost/panfrost_device.c b/drivers/gpu/drm/panfrost/panfrost_device.c
> index 9549ddb7854a..9e02fb5f73c8 100644
> --- a/drivers/gpu/drm/panfrost/panfrost_device.c
> +++ b/drivers/gpu/drm/panfrost/panfrost_device.c
> @@ -341,6 +341,8 @@ void panfrost_device_fini(struct panfrost_device *pfdev)
> pm_runtime_get_sync(pfdev->base.dev);
> pm_runtime_disable(pfdev->base.dev);
>
> + panfrost_jm_stop_sched_jobs(pfdev);
Do we really want to do the SOFT_STOP+HARD_STOP dance in that case?
Can't we just SOFT_RESET the GPU and collect the pieces in the
panfrost_<component>_fini() functions?
> +
> panfrost_gem_fini(pfdev);
> panfrost_perfcnt_fini(pfdev);
> panfrost_jm_fini(pfdev);
> diff --git a/drivers/gpu/drm/panfrost/panfrost_job.c b/drivers/gpu/drm/panfrost/panfrost_job.c
> index 5016d2d53da2..630298b7ea8a 100644
> --- a/drivers/gpu/drm/panfrost/panfrost_job.c
> +++ b/drivers/gpu/drm/panfrost/panfrost_job.c
> @@ -655,39 +655,12 @@ static u32 panfrost_active_slots(struct panfrost_device *pfdev,
> return js_state & *js_state_mask;
> }
>
> -static void
> -panfrost_reset(struct panfrost_device *pfdev,
> - struct drm_sched_job *bad)
> +static void panfrost_stop_jobs(struct panfrost_device *pfdev)
> {
> u32 js_state, js_state_mask = 0xffffffff;
> unsigned int i, j;
> - bool cookie;
> int ret;
>
> - if (!atomic_read(&pfdev->reset.pending))
> - return;
> -
> - /* Stop the schedulers.
> - *
> - * FIXME: We temporarily get out of the dma_fence_signalling section
> - * because the cleanup path generate lockdep splats when taking locks
> - * to release job resources. We should rework the code to follow this
> - * pattern:
> - *
> - * try_lock
> - * if (locked)
> - * release
> - * else
> - * schedule_work_to_release_later
> - */
> - for (i = 0; i < NUM_JOB_SLOTS; i++)
> - drm_sched_stop(&pfdev->js->queue[i].sched, bad);
> -
> - cookie = dma_fence_begin_signalling();
> -
> - if (bad)
> - drm_sched_increase_karma(bad);
> -
> /* Mask job interrupts and synchronize to make sure we won't be
> * interrupted during our reset.
> */
> @@ -730,6 +703,48 @@ panfrost_reset(struct panfrost_device *pfdev,
> }
> memset(pfdev->jobs, 0, sizeof(pfdev->jobs));
> spin_unlock(&pfdev->js->job_lock);
> +}
> +
> +void panfrost_jm_stop_sched_jobs(struct panfrost_device *pfdev)
> +{
> + for (u32 i = 0; i < NUM_JOB_SLOTS; i++)
> + drm_sched_wqueue_stop(&pfdev->js->queue[i].sched);
Do we have something cleaning up the pending list if this is stopped
while jobs are in-flight? If not, we probably need a ::cancel_job()
implementation.
> +
> + panfrost_stop_jobs(pfdev);
> +}
> +
> +static void
> +panfrost_reset(struct panfrost_device *pfdev,
> + struct drm_sched_job *bad)
> +{
> + unsigned int i;
> + bool cookie;
> +
> + if (!atomic_read(&pfdev->reset.pending))
> + return;
> +
> + /* Stop the schedulers.
> + *
> + * FIXME: We temporarily get out of the dma_fence_signalling section
> + * because the cleanup path generate lockdep splats when taking locks
> + * to release job resources. We should rework the code to follow this
> + * pattern:
> + *
> + * try_lock
> + * if (locked)
> + * release
> + * else
> + * schedule_work_to_release_later
> + */
> + for (i = 0; i < NUM_JOB_SLOTS; i++)
> + drm_sched_stop(&pfdev->js->queue[i].sched, bad);
> +
> + cookie = dma_fence_begin_signalling();
> +
> + if (bad)
> + drm_sched_increase_karma(bad);
> +
> + panfrost_stop_jobs(pfdev);
>
> /* Proceed with reset now. */
> panfrost_device_reset(pfdev, false);
> diff --git a/drivers/gpu/drm/panfrost/panfrost_job.h b/drivers/gpu/drm/panfrost/panfrost_job.h
> index 3e3b9717838f..dc9d3c9dc67b 100644
> --- a/drivers/gpu/drm/panfrost/panfrost_job.h
> +++ b/drivers/gpu/drm/panfrost/panfrost_job.h
> @@ -73,6 +73,7 @@ int panfrost_jm_is_idle(struct panfrost_device *pfdev);
> int panfrost_job_get_slot(struct panfrost_job *job);
> int panfrost_job_push(struct panfrost_job *job);
> void panfrost_job_put(struct panfrost_job *job);
> +void panfrost_jm_stop_sched_jobs(struct panfrost_device *pfdev);
>
> #ifdef CONFIG_DEBUG_FS
> void panfrost_sched_debugfs_init(struct drm_minor *minor);
>