]> git.ipfire.org Git - thirdparty/linux.git/commitdiff
blk-cgroup: don't nest queue_lock under blkcg->lock in blkcg_destroy_blkgs()
authorYu Kuai <yukuai@fygo.io>
Mon, 8 Jun 2026 03:42:47 +0000 (11:42 +0800)
committerJens Axboe <axboe@kernel.dk>
Wed, 24 Jun 2026 12:42:31 +0000 (06:42 -0600)
The correct lock order is q->queue_lock before blkcg->lock, and in order
to prevent deadlock from blkcg_destroy_blkgs(), trylock is used for
q->queue_lock while blkcg->lock is already held, this is hacky.

Refactor blkcg_destroy_blkgs() to hold blkcg->lock only long enough to
get the first blkg and then release it. Then take q->queue_lock and
blkcg->lock in the correct order to destroy the blkg. This is a very cold
path, so the extra lock/unlock cycles are acceptable.

Also prepare to convert protecting blkcg with blkcg_mutex instead of
queue_lock.

Signed-off-by: Yu Kuai <yukuai@fygo.io>
Link: https://patch.msgid.link/00b03cf74a9937cb4d6dd67a189ddc00a3de0451.1780621988.git.yukuai@fygo.io
Signed-off-by: Jens Axboe <axboe@kernel.dk>
block/blk-cgroup.c

index e1bde48852ae13f1dcfff18139e61b178c7bd0b6..d2a1f5903f248a7a6b6b938ffc050de9cc00afdb 100644 (file)
@@ -1239,6 +1239,21 @@ struct list_head *blkcg_get_cgwb_list(struct cgroup_subsys_state *css)
  *    This finally frees the blkcg.
  */
 
+static struct blkcg_gq *blkcg_get_first_blkg(struct blkcg *blkcg)
+{
+       struct blkcg_gq *blkg = NULL;
+
+       spin_lock_irq(&blkcg->lock);
+       if (!hlist_empty(&blkcg->blkg_list)) {
+               blkg = hlist_entry(blkcg->blkg_list.first, struct blkcg_gq,
+                                  blkcg_node);
+               blkg_get(blkg);
+       }
+       spin_unlock_irq(&blkcg->lock);
+
+       return blkg;
+}
+
 /**
  * blkcg_destroy_blkgs - responsible for shooting down blkgs
  * @blkcg: blkcg of interest
@@ -1252,32 +1267,24 @@ struct list_head *blkcg_get_cgwb_list(struct cgroup_subsys_state *css)
  */
 static void blkcg_destroy_blkgs(struct blkcg *blkcg)
 {
-       might_sleep();
+       struct blkcg_gq *blkg;
 
-       spin_lock_irq(&blkcg->lock);
+       might_sleep();
 
-       while (!hlist_empty(&blkcg->blkg_list)) {
-               struct blkcg_gq *blkg = hlist_entry(blkcg->blkg_list.first,
-                                               struct blkcg_gq, blkcg_node);
+       while ((blkg = blkcg_get_first_blkg(blkcg))) {
                struct request_queue *q = blkg->q;
 
-               if (need_resched() || !spin_trylock(&q->queue_lock)) {
-                       /*
-                        * Given that the system can accumulate a huge number
-                        * of blkgs in pathological cases, check to see if we
-                        * need to rescheduling to avoid softlockup.
-                        */
-                       spin_unlock_irq(&blkcg->lock);
-                       cond_resched();
-                       spin_lock_irq(&blkcg->lock);
-                       continue;
-               }
+               spin_lock_irq(&q->queue_lock);
+               spin_lock(&blkcg->lock);
 
                blkg_destroy(blkg);
-               spin_unlock(&q->queue_lock);
-       }
 
-       spin_unlock_irq(&blkcg->lock);
+               spin_unlock(&blkcg->lock);
+               spin_unlock_irq(&q->queue_lock);
+
+               blkg_put(blkg);
+               cond_resched();
+       }
 }
 
 /**