]> git.ipfire.org Git - thirdparty/kernel/linux.git/commitdiff
drm/v3d: Serialize the scheduler timeout handlers
authorMaíra Canal <mcanal@igalia.com>
Wed, 29 Jul 2026 02:09:22 +0000 (23:09 -0300)
committerMaíra Canal <mcanal@igalia.com>
Sun, 2 Aug 2026 21:39:20 +0000 (18:39 -0300)
V3D exposes several independent hardware queues (BIN, RENDER, TFU and
CSD) but has only a single, global reset. A timeout on any one queue
therefore has to stop, reset and restart the schedulers of every other
queue as well. That makes concurrent timeout handlers unsafe.

`reset_lock` was never able to make them safe, as a driver-side lock can
only cover the driver's &drm_sched_backend_ops.timedout_job callback.
The scheduler handles the timed out job and its pending list around that
callback, outside of the driver's control, so a global reset triggered
by one queue can still interfere with another queue that is in the
middle of handling a timeout of its own.

Consequently, if a reset happens in the CSD queue while a CL-intensive
application is running, the global reset stops and restarts the CL
queue's scheduler while that queue is handling a timeout of its own. As
drm_sched_stop() and drm_sched_start() subtract and add the credits of
every job sitting on the pending list of the scheduler they are called
on, and as the CL queue's handler concurrently takes its job off that
same list and puts it back, the stop and the start no longer see the
same set of jobs. The CL queue is left with more credits in flight than
its limit:

[  327.302739] ------------[ cut here ]------------
[  327.302744] WARNING: CPU: 2 PID: 43 at drivers/gpu/drm/scheduler/sched_main.c:102 drm_sched_run_job_work+0x238/0x4d0 [gpu_sched]
[  327.302884] CPU: 2 UID: 0 PID: 43 Comm: kworker/u16:1 Not tainted 6.18.39-v8-16k+ #3 PREEMPT
[  327.302889] Hardware name: Raspberry Pi 5 Model B Rev 1.0 (DT)
[  327.302893] Workqueue: v3d_bin drm_sched_run_job_work [gpu_sched]
[  327.302984] Call trace:
[  327.302987]  drm_sched_run_job_work+0x238/0x4d0 [gpu_sched] (P)
[  327.302997]  process_scheduled_works+0x180/0x3d0
[  327.303010]  worker_thread+0x268/0x3e8
[  327.303016]  kthread+0x140/0x250
[  327.303022]  ret_from_fork+0x10/0x20
[  327.303031] ---[ end trace 0000000000000000 ]---

From that point on, the credit count of the CL queue is broken, causing
a complete GPU hang and UI freeze.

The DRM scheduler already provides a mechanism to serialize the timeout
handlers of different schedulers: an ordered workqueue passed as
drm_sched_init()'s @timeout_wq parameter. By default, each scheduler
queues its timeout work on the system workqueue, which runs the handlers
concurrently. Give all of the queues a shared ordered workqueue instead,
as recommended by the DRM scheduler documentation for hardware that has
distinct queues but resets globally.

Cc: stable@vger.kernel.org # 6.15
Reviewed-by: Iago Toral Quiroga <itoral@igalia.com>
Link: https://patch.msgid.link/20260728-v3d-order-global-reset-v1-1-e47be838158d@igalia.com
Signed-off-by: Maíra Canal <mcanal@igalia.com>
drivers/gpu/drm/v3d/v3d_drv.h
drivers/gpu/drm/v3d/v3d_sched.c

index 423bcfe331a0141395da6fc32292c68f0bf47a1e..8779a42c65081a7836ccf2f49341b0c420960475 100644 (file)
@@ -182,6 +182,12 @@ struct v3d_dev {
         */
        struct mutex reset_lock;
 
+       /* Ordered workqueue shared by every queue's scheduler timeout work.
+        * V3D reset is global to all queues, so the timeout handlers must not
+        * run concurrently.
+        */
+       struct workqueue_struct *reset_wq;
+
        /* Lock taken when creating and pushing the GPU scheduler
         * jobs, to keep the sched-fence seqnos in order.
         */
index c35f9f70cf8d63edd71e24442c4be00ffd941f36..63dd95b828c850a0891bc344fe838cd2c9753d04 100644 (file)
@@ -837,6 +837,7 @@ v3d_queue_sched_init(struct v3d_dev *v3d, const struct drm_sched_backend_ops *op
        struct drm_sched_init_args args = {
                .credit_limit = 1,
                .timeout = msecs_to_jiffies(500),
+               .timeout_wq = v3d->reset_wq,
                .dev = v3d->drm.dev,
        };
 
@@ -851,9 +852,13 @@ v3d_sched_init(struct v3d_dev *v3d)
 {
        int ret;
 
+       v3d->reset_wq = alloc_ordered_workqueue("v3d_reset", 0);
+       if (!v3d->reset_wq)
+               return -ENOMEM;
+
        ret = v3d_queue_sched_init(v3d, &v3d_bin_sched_ops, V3D_BIN, "v3d_bin");
        if (ret)
-               return ret;
+               goto fail;
 
        ret = v3d_queue_sched_init(v3d, &v3d_render_sched_ops, V3D_RENDER,
                                   "v3d_render");
@@ -896,4 +901,6 @@ v3d_sched_fini(struct v3d_dev *v3d)
                if (v3d->queue[q].sched.ready)
                        drm_sched_fini(&v3d->queue[q].sched);
        }
+
+       destroy_workqueue(v3d->reset_wq);
 }