[PATCH v5 3/3] writeback: switch a replaced cgwb's inodes to its successor

From: Liz Fong-Jones

Date: Thu Oct 01 2026 - 19:50:56 EST


When a live memcg's blkcg association changes, cgwb_create() replaces
its killed wb in bdi->cgwb_tree. Foreign flushes for the memcg then
find the new wb, but the inodes, dirty or not, are still attached to the
old one, and they only move over as they are written back (see
wbc_attach_and_unlock_inode()). Until then, other memcgs appending to
those inodes stall in balance_dirty_pages().

After the takeover, queue a work item that switches all of the old wb's
inodes to the wb foreign flushes now find, like cleanup_offline_cgwb()
does for b_attached and b_dirty_time, but including b_dirty, b_io and
b_more_io. The switch carries the dirty and writeback page counts, so
nothing needs to be written back first. cgwb_create() can run with
interrupts disabled (folio_account_dirtied() -> inode_attach_wb()), so
it only queues the work, after dropping cgwb_lock. Inodes already being
switched when the work runs are left to the existing per-inode
switching, so this is best effort. If the successor is already gone,
fall back like cleanup_offline_cgwb() does, to the nearest live
ancestor's wb or the root wb.

Test: a live cgroup appends to 5000 files on XFS at 250MiB/s, io is
disabled on its parent so its wb is killed, it creates a file (the
takeover), and a sibling keeps appending to its files under a 4G parent
limit. Worst lag of the sibling behind schedule over 30s, in three runs:

without this patch 9.1s 4.0s 1.0s
with this patch 1.2s 1.0s 1.0s

Without this patch, moving the old wb's inodes over took up to 11s,
first switch to last; with it, all 5001 were switched in a 30-60ms
window, with up to 553MiB of the old cgroup's pages dirty.

Suggested-by: Tejun Heo <tj@xxxxxxxxxx>
Assisted-by: Claude:claude-opus-5-5 checkpatch sparse
Assisted-by: Claude:claude-fable-5-1
Signed-off-by: Liz Fong-Jones <lizf@xxxxxxxxxxxx>
---
fs/fs-writeback.c | 59 ++++++++++++++++++++++++++++++++++++++++
include/linux/backing-dev-defs.h | 1 +
include/linux/writeback.h | 1 +
mm/backing-dev.c | 28 ++++++++++++++++++-
4 files changed, 88 insertions(+), 1 deletion(-)

diff --git a/fs/fs-writeback.c b/fs/fs-writeback.c
index 050d507aa554..9be641800934 100644
--- a/fs/fs-writeback.c
+++ b/fs/fs-writeback.c
@@ -809,6 +809,65 @@ bool cleanup_offline_cgwb(struct bdi_writeback *wb)
return restart;
}

+/**
+ * switch_replaced_cgwb - switch a replaced wb's inodes to its successor
+ * @wb: target wb, replaced in bdi->cgwb_tree by another wb of its memcg
+ *
+ * Switch all inodes attached to @wb, dirty or not, to the wb that foreign
+ * flushes now find for @wb's memcg. If that one is already gone, fall back
+ * like cleanup_offline_cgwb() does, to the nearest live ancestor's wb or the
+ * root wb. The switch carries the dirty and writeback page counts, so
+ * nothing needs to be written back first. Returns %true if not all inodes
+ * were switched and the function has to be restarted.
+ */
+bool switch_replaced_cgwb(struct bdi_writeback *wb)
+{
+ struct cgroup_subsys_state *memcg_css;
+ struct inode_switch_wbs_context *isw;
+ struct bdi_writeback *new_wb;
+ bool restart;
+ int nr = 0;
+
+ new_wb = wb_get_lookup(wb->bdi, wb->memcg_css);
+ for (memcg_css = wb->memcg_css->parent; !new_wb && memcg_css;
+ memcg_css = memcg_css->parent)
+ new_wb = wb_get_create(wb->bdi, memcg_css, GFP_KERNEL);
+ if (!new_wb)
+ new_wb = &wb->bdi->wb; /* wb_get() is noop for bdi's wb */
+ if (WARN_ON_ONCE(new_wb == wb)) {
+ wb_put(new_wb);
+ return false;
+ }
+
+ isw = kzalloc_flex(*isw, inodes, WB_MAX_INODES_PER_ISW);
+ if (!isw) {
+ wb_put(new_wb);
+ return false;
+ }
+
+ atomic_inc(&isw_nr_in_flight);
+
+ spin_lock(&wb->list_lock);
+ restart = isw_prepare_wbs_switch(new_wb, isw, &wb->b_attached, &nr) ||
+ isw_prepare_wbs_switch(new_wb, isw, &wb->b_dirty, &nr) ||
+ isw_prepare_wbs_switch(new_wb, isw, &wb->b_io, &nr) ||
+ isw_prepare_wbs_switch(new_wb, isw, &wb->b_more_io, &nr) ||
+ isw_prepare_wbs_switch(new_wb, isw, &wb->b_dirty_time, &nr);
+ spin_unlock(&wb->list_lock);
+
+ if (nr == 0) {
+ atomic_dec(&isw_nr_in_flight);
+ wb_put(new_wb);
+ kfree(isw);
+ return restart;
+ }
+
+ trace_inode_switch_wbs_queue(wb, new_wb, nr);
+ wb_queue_isw(new_wb, isw);
+
+ return restart;
+}
+
/**
* wbc_attach_and_unlock_inode - associate wbc with target inode and unlock it
* @wbc: writeback_control of interest
diff --git a/include/linux/backing-dev-defs.h b/include/linux/backing-dev-defs.h
index 035d25f5b81b..0d9a113f6fd6 100644
--- a/include/linux/backing-dev-defs.h
+++ b/include/linux/backing-dev-defs.h
@@ -160,6 +160,7 @@ struct bdi_writeback {
* to this wb */
struct llist_head switch_wbs_ctxs; /* queued contexts for
* writeback switching */
+ struct work_struct replaced_work; /* see switch_replaced_cgwb() */

union {
struct work_struct release_work;
diff --git a/include/linux/writeback.h b/include/linux/writeback.h
index b749a9a5a5ee..88e4b5049698 100644
--- a/include/linux/writeback.h
+++ b/include/linux/writeback.h
@@ -209,6 +209,7 @@ int cgroup_writeback_by_id(u64 bdi_id, int memcg_id,
enum wb_reason reason, struct wb_completion *done);
void cgroup_writeback_umount(struct super_block *sb);
bool cleanup_offline_cgwb(struct bdi_writeback *wb);
+bool switch_replaced_cgwb(struct bdi_writeback *wb);

/**
* inode_attach_wb - associate an inode with its wb
diff --git a/mm/backing-dev.c b/mm/backing-dev.c
index 839a3c476c5b..49b7b86a7221 100644
--- a/mm/backing-dev.c
+++ b/mm/backing-dev.c
@@ -637,9 +637,21 @@ static void cgwb_release_workfn(struct work_struct *work)
bdi_put(bdi);
WARN_ON_ONCE(!list_empty(&wb->b_attached));
WARN_ON_ONCE(work_pending(&wb->switch_work));
+ WARN_ON_ONCE(work_pending(&wb->replaced_work));
call_rcu(&wb->rcu, cgwb_free_rcu);
}

+static void cgwb_replaced_workfn(struct work_struct *work)
+{
+ struct bdi_writeback *wb = container_of(work, struct bdi_writeback,
+ replaced_work);
+
+ do {
+ cond_resched_tasks_rcu_qs();
+ } while (switch_replaced_cgwb(wb));
+ wb_put(wb);
+}
+
static void cgwb_release(struct percpu_ref *refcnt)
{
struct bdi_writeback *wb = container_of(refcnt, struct bdi_writeback,
@@ -676,7 +688,7 @@ static int cgwb_create(struct backing_dev_info *bdi,
struct mem_cgroup *memcg;
struct cgroup_subsys_state *blkcg_css;
struct list_head *memcg_cgwb_list, *blkcg_cgwb_list;
- struct bdi_writeback *wb, *old_wb;
+ struct bdi_writeback *wb, *old_wb, *replaced_wb = NULL;
void __rcu **slot;
unsigned long flags;
int ret = 0;
@@ -724,6 +736,7 @@ static int cgwb_create(struct backing_dev_info *bdi,
INIT_WORK(&wb->switch_work, inode_switch_wbs_work_fn);
init_llist_head(&wb->switch_wbs_ctxs);
INIT_WORK(&wb->release_work, cgwb_release_workfn);
+ INIT_WORK(&wb->replaced_work, cgwb_replaced_workfn);
set_bit(WB_registered, &wb->state);
bdi_get(bdi);

@@ -748,6 +761,8 @@ static int cgwb_create(struct backing_dev_info *bdi,
old_wb = radix_tree_deref_slot_protected(slot, &cgwb_lock);
if (wb_dying(old_wb)) {
radix_tree_replace_slot(&bdi->cgwb_tree, slot, wb);
+ if (wb_tryget(old_wb))
+ replaced_wb = old_wb;
ret = 0;
} else {
ret = -EEXIST;
@@ -763,6 +778,17 @@ static int cgwb_create(struct backing_dev_info *bdi,
}
}
spin_unlock_irqrestore(&cgwb_lock, flags);
+
+ /*
+ * The replaced wb is out of foreign flushes' reach but may still have
+ * inodes attached, dirty or not. Switch them over to @wb. We may be
+ * running with interrupts disabled, so use a work item. A replaced wb
+ * never returns to the tree, so its work can't already be pending.
+ */
+ if (replaced_wb &&
+ !queue_work(system_dfl_wq, &replaced_wb->replaced_work))
+ wb_put(replaced_wb);
+
if (ret) {
if (ret == -EEXIST)
ret = 0;

--
2.53.0