[RFC PATCH 3/4] mm/zswap: allow writeback to be throttled by the cgroup IO controllers

From: Alexandre Ghiti

Date: Mon Sep 28 2026 - 04:28:01 EST


bio_issue_as_root_blkg() exempts REQ_SWAP bios from the cgroup IO
controllers. The exemption avoids priority inversion: reclaim issues
swap writes on behalf of the whole system, so making them wait on one
cgroup's IO budget stalls memory reclaim for everyone.

For zswap writeback the exemption does harm: the write is charged to the
cgroup's IO budget anyway, afterwards, but is issued ahead of the
synchronous swapin and page cache reads that the workload blocks on.

Let zswap writeback opt out of it with REQ_BACKGROUND.

Signed-off-by: Alexandre Ghiti <alex@xxxxxxxx>
---
include/linux/swap_ops.h | 1 +
mm/page_io.c | 7 +++++--
mm/zswap.c | 20 ++++++++++++++++----
3 files changed, 22 insertions(+), 6 deletions(-)

diff --git a/include/linux/swap_ops.h b/include/linux/swap_ops.h
index 57ac6c703f68..152598ce6c4f 100644
--- a/include/linux/swap_ops.h
+++ b/include/linux/swap_ops.h
@@ -17,6 +17,7 @@ struct swap_iocb {
struct swap_io_ctx {
struct swap_iocb *sio;
struct swap_info_struct *sis;
+ bool throttled;
};

/*
diff --git a/mm/page_io.c b/mm/page_io.c
index 88962571cb93..14dd54ae6d65 100644
--- a/mm/page_io.c
+++ b/mm/page_io.c
@@ -591,9 +591,12 @@ static void swap_bdev_submit_write(struct swap_io_ctx *ctx)
{
struct swap_iocb *sio = ctx->sio;
struct bio *bio = &sio->bio;
+ blk_opf_t opf = REQ_OP_WRITE | REQ_SWAP;

- bio_init(bio, ctx->sis->bdev, sio->bvecs, ARRAY_SIZE(sio->bvecs),
- REQ_OP_WRITE | REQ_SWAP);
+ if (ctx->throttled)
+ opf |= REQ_BACKGROUND;
+
+ bio_init(bio, ctx->sis->bdev, sio->bvecs, ARRAY_SIZE(sio->bvecs), opf);
bio->bi_iter.bi_size = sio->len;
bio->bi_iter.bi_sector = swap_folio_sector(bio_first_folio_all(bio));
bio_associate_blkg_from_page(bio, bio_first_folio_all(bio));
diff --git a/mm/zswap.c b/mm/zswap.c
index 8925b6100402..3b902f29ed4c 100644
--- a/mm/zswap.c
+++ b/mm/zswap.c
@@ -983,16 +983,21 @@ static bool zswap_decompress(struct zswap_entry *entry, struct folio *folio)
* in the first place. After the folio has been decompressed into
* the swap cache, the compressed version stored by zswap can be
* freed.
+ *
+ * @throttled lets the cgroup IO controllers throttle the write rather than
+ * issue it as root.
*/
static int zswap_writeback_entry(struct zswap_entry *entry,
- swp_entry_t swpentry)
+ swp_entry_t swpentry, bool throttled)
{
struct xarray *tree;
pgoff_t offset = swp_offset(swpentry);
struct folio *folio;
struct mempolicy *mpol;
struct swap_info_struct *si;
- struct swap_io_ctx ctx = {};
+ struct swap_io_ctx ctx = {
+ .throttled = throttled,
+ };
int ret = 0;

/* try to allocate swap cache folio */
@@ -1070,8 +1075,14 @@ static int zswap_writeback_entry(struct zswap_entry *entry,
*
* ZSWAP_SHRINK_SWAPCACHE (out) is set when the walk stopped at an entry whose
* folio is already in the swap cache.
+ *
+ * ZSWAP_SHRINK_THROTTLED (in) lets the cgroup IO controllers throttle the
+ * writeback rather than issue it as root. Only the memory pressure shrinker
+ * sets it. The pool limit path must not: once the pool is full zswap_store()
+ * rejects everything until this writeback drains it.
*/
#define ZSWAP_SHRINK_SWAPCACHE BIT(0)
+#define ZSWAP_SHRINK_THROTTLED BIT(1)

/*
* The dynamic shrinker is modulated by the following factors:
@@ -1154,7 +1165,8 @@ static enum lru_status shrink_memcg_cb(struct list_head *item, struct list_lru_o
*/
spin_unlock(&l->lock);

- writeback_result = zswap_writeback_entry(entry, swpentry);
+ writeback_result = zswap_writeback_entry(entry, swpentry, flags &&
+ (*flags & ZSWAP_SHRINK_THROTTLED));

if (writeback_result) {
zswap_reject_reclaim_fail++;
@@ -1180,7 +1192,7 @@ static unsigned long zswap_shrinker_scan(struct shrinker *shrinker,
struct shrink_control *sc)
{
unsigned long shrink_ret;
- unsigned int flags = 0;
+ unsigned int flags = ZSWAP_SHRINK_THROTTLED;

if (!zswap_shrinker_enabled ||
!mem_cgroup_zswap_writeback_enabled(sc->memcg)) {
--
2.53.0-Meta