On 9/15/26 9:24 PM, yu kuai wrote:
Hi,

在 2026/9/13 21:29, Nilay Shroff 写道:
On 9/13/26 12:24 PM, Yu Kuai wrote:
From: Yu Kuai <[email protected]>

blkcg_punt_bio_submit() currently queues punted bios on
blkg->async_bios,
so it has to call bio_blkg() to find or create a queue-local blkg.  Bios
now carry and pin the blkcg css, so punted bio lifetime no longer
needs to
be anchored by a blkg.

Keeping the punt state in blkg can instantiate a blkg even when no blkcg
policy is enabled, just to bounce submission from a shared kthread.
Move
async_bio_lock, async_bios and async_bio_work to struct blkcg, and queue
punted bios on bio_blkcg() for non-root cgroups.  Root or
unassociated bios
are submitted directly.

This preserves the priority-inversion avoidance while preventing
blkcg_punt_bio_submit() from creating blkgs that are not needed by any
policy.

Signed-off-by: Yu Kuai <[email protected]>
---
   block/blk-cgroup.c | 52 ++++++++++++++++++++++++++--------------------
   block/blk-cgroup.h | 14 ++++++-------
   2 files changed, 35 insertions(+), 31 deletions(-)

diff --git a/block/blk-cgroup.c b/block/blk-cgroup.c
index 59ccfefe16a8..aa3cee107ebe 100644
--- a/block/blk-cgroup.c
+++ b/block/blk-cgroup.c
@@ -180,14 +180,10 @@ static void blkg_free(struct blkcg_gq *blkg)
     static void __blkg_release(struct rcu_head *rcu)
   {
       struct blkcg_gq *blkg = container_of(rcu, struct blkcg_gq,
rcu_head);
   -#ifdef CONFIG_BLK_CGROUP_PUNT_BIO
-    WARN_ON(!bio_list_empty(&blkg->async_bios));
-#endif
-
       blkg_free(blkg);
   }
     /*
    * A group is RCU protected, but having an rcu lock does not mean
that one
@@ -226,23 +222,23 @@ static void blkg_release(struct percpu_ref *ref)
   }
     #ifdef CONFIG_BLK_CGROUP_PUNT_BIO
   static struct workqueue_struct *blkcg_punt_bio_wq;
   -static void blkg_async_bio_workfn(struct work_struct *work)
+static void blkcg_async_bio_workfn(struct work_struct *work)
   {
-    struct blkcg_gq *blkg = container_of(work, struct blkcg_gq,
-                         async_bio_work);
+    struct blkcg *blkcg = container_of(work, struct blkcg,
async_bio_work);
       struct bio_list bios = BIO_EMPTY_LIST;
       struct bio *bio;
       struct blk_plug plug;
       bool need_plug = false;
   -    /* as long as there are pending bios, @blkg can't go away */
-    spin_lock(&blkg->async_bio_lock);
-    bio_list_merge_init(&bios, &blkg->async_bios);
-    spin_unlock(&blkg->async_bio_lock);
+    /* as long as there are pending bios, @blkcg can't go away */
+    {
+        guard(spinlock)(&blkcg->async_bio_lock);
+        bio_list_merge_init(&bios, &blkcg->async_bios);
+    }

Instead of using guard(spinlock)(...) here, I think we could use the
simpler spin_lock()/spin_unlock() helpers. IMO, they are easier
to read and reason about for these short critical sections.
Ok.

       /* start plug only when bio_list contains at least 2 bios */
       if (bios.head && bios.head->bi_next) {
           need_plug = true;
           blk_start_plug(&plug);
@@ -259,19 +255,20 @@ static void blkg_async_bio_workfn(struct
work_struct *work)
    * cgroup.  Use this helper instead of submit_bio to punt the
actual issuing to
    * a dedicated per-blkcg work item to avoid such priority inversions.
    */
   void blkcg_punt_bio_submit(struct bio *bio)
   {
-    struct blkcg_gq *blkg = bio_blkg(bio);
+    struct blkcg *blkcg = bio_blkcg(bio);
   -    if (blkg && blkg->parent) {
-        spin_lock(&blkg->async_bio_lock);
-        bio_list_add(&blkg->async_bios, bio);
-        spin_unlock(&blkg->async_bio_lock);
-        queue_work(blkcg_punt_bio_wq, &blkg->async_bio_work);
+    if (blkcg && cgroup_parent(blkcg->css.cgroup)) {
+        {
+            guard(spinlock)(&blkcg->async_bio_lock);
+            bio_list_add(&blkcg->async_bios, bio);
+        }
+        queue_work(blkcg_punt_bio_wq, &blkcg->async_bio_work);

Again same here, replace guard() with spin_lock() and spin_unlock()
helpers.

       } else {
-        /* Never bounce if there is no non-root blkg to queue on. */
+        /* Never bounce if there is no non-root blkcg to queue on. */
           submit_bio(bio);
       }
   }
   EXPORT_SYMBOL_GPL(blkcg_punt_bio_submit);
   @@ -350,15 +347,10 @@ static struct blkcg_gq *blkg_alloc(struct
blkcg *blkcg, struct gendisk *disk,
       blkg->q = disk->queue;
       INIT_LIST_HEAD(&blkg->q_node);
       blkg->blkcg = blkcg;
       blkg->blkcg_id = blkcg->css.id;
       blkg->iostat.blkg = blkg;
-#ifdef CONFIG_BLK_CGROUP_PUNT_BIO
-    spin_lock_init(&blkg->async_bio_lock);
-    bio_list_init(&blkg->async_bios);
-    INIT_WORK(&blkg->async_bio_work, blkg_async_bio_workfn);
-#endif
         u64_stats_init(&blkg->iostat.sync);
       for_each_possible_cpu(cpu) {
           u64_stats_init(&per_cpu_ptr(blkg->iostat_cpu, cpu)->sync);
           per_cpu_ptr(blkg->iostat_cpu, cpu)->blkg = blkg;
@@ -1399,10 +1391,16 @@ static void blkcg_css_free(struct
cgroup_subsys_state *css)
           if (blkcg->cpd[i])
               blkcg_policy[i]->cpd_free_fn(blkcg->cpd[i]);
         mutex_unlock(&blkcg_pol_mutex);
   +#ifdef CONFIG_BLK_CGROUP_PUNT_BIO
+    {
+        guard(spinlock)(&blkcg->async_bio_lock);
+        WARN_ON(!bio_list_empty(&blkcg->async_bios));
+    }
+#endif

This is a slightly different case. At this point blkcg_css_free() is
freeing the blkcg object after its final reference has gone away, so
there should be no concurrent context accessing blkcg->async_bios.
Therefore, I don't think we need to acquire async_bio_lock here just
to perform the WARN_ON() check.

Perhaps is it better just to remove the check? blkcg will be pinned
by any bio inside the list, so I believe this is safe.

Ideally yes we would not enter into blkcg_css_free() until all
references to blkcg are dropped. So the WARN_ON() appears to be
used just as a paranoia check. I'm okay either to drop it or
if you want to keep it then I suggest replacing bio_list_empty()
with bio_list_empty_careful(), so that the check explicitly allows
lockless inspection during teardown.

Thanks,
--Nilay


Reply via email to