[PATCH 1/2] mm/zswap: batch the writeback IO of consecutive entries
From: Alexandre Ghiti
Date: Wed Oct 07 2026 - 12:00:03 EST
Commit 8f29aa226f82 ("mm/swap: introduce struct swap_io_ctx") introduced a
simple way to batch the swap writes of folios with consecutive slots into a
single bio. zswap writeback does not use it: zswap_writeback_entry()
submits a context of its own for every entry, so every entry written back
becomes its own IO.
So let the two functions that walk the zswap LRU, zswap_shrinker_scan()
and shrink_memcg(), own the context and submit it once the walk is done.
Kernel build (defconfig, -j4) in a 600M memory.max cgroup with the zswap
shrinker enabled, swapping to an NVMe partition with iocost enabled. It
runs alone, and next to a sibling fio random writer with 10 times its
io.weight so that iocost throttles it. Mean +- stddev, counters from the
build's cgroup:
Alone:
unbatched batched
build time (s) 1066 +- 9 1061 +- 2 -0.5%
write requests (k) 748 +- 47 596 +- 26 -20.4%
pages per write request 1.07 +- 0.01 1.30 +- 0.01 +21.2%
iocost debt (s) 15.0 +- 2.7 12.1 +- 1.2 -19.5%
iocost wait (s) 7.5 +- 1.2 7.2 +- 1.1 -3.6%
IO full pressure (s) 19.0 +- 1.8 18.2 +- 0.9 -4.2%
Next to the writer:
unbatched batched
build time (s) 2269 +- 50 2203 +- 62 -2.9%
write requests (k) 714 +- 31 605 +- 24 -15.2%
pages per write request 1.08 +- 0.00 1.30 +- 0.02 +20.3%
iocost debt (s) 685 +- 25 646 +- 26 -5.7%
iocost wait (s) 655 +- 26 632 +- 23 -3.6%
IO full pressure (s) 655 +- 31 624 +- 30 -4.7%
Suggested-by: Nhat Pham <nphamcs@xxxxxxxxx>
Signed-off-by: Alexandre Ghiti <alex@xxxxxxxx>
---
mm/zswap.c | 47 ++++++++++++++++++++++++++++++++++++-----------
1 file changed, 36 insertions(+), 11 deletions(-)
diff --git a/mm/zswap.c b/mm/zswap.c
index ae19e301fced..698fb74c5d47 100644
--- a/mm/zswap.c
+++ b/mm/zswap.c
@@ -990,6 +990,21 @@ static bool zswap_decompress(struct zswap_entry *entry, struct folio *folio)
/*********************************
* writeback code
**********************************/
+/* State shared by all the entries written back by one LRU walk. */
+struct zswap_shrink_ctl {
+ /*
+ * Collects the writeback of consecutive entries into as few IOs as
+ * possible. The owner of the walk must submit what is left in it
+ * once it is done.
+ */
+ struct swap_io_ctx io_ctx;
+ /*
+ * If not NULL, stop the walk when writeback finds a folio already in
+ * the swap cache, and set it to true.
+ */
+ bool *encountered_page_in_swapcache;
+};
+
/*
* Attempts to free an entry by adding a folio to the swap cache,
* decompressing the entry data into the folio, and issuing a
@@ -1001,8 +1016,12 @@ static bool zswap_decompress(struct zswap_entry *entry, struct folio *folio)
* in the first place. After the folio has been decompressed into
* the swap cache, the compressed version stored by zswap can be
* freed.
+ *
+ * The folio is only added to @ctx here, it is up to the owner of @ctx to
+ * submit the IO.
*/
-static int zswap_writeback_entry(struct zswap_entry *entry,
+static int zswap_writeback_entry(struct swap_io_ctx *ctx,
+ struct zswap_entry *entry,
swp_entry_t swpentry)
{
struct xarray *tree;
@@ -1010,7 +1029,6 @@ static int zswap_writeback_entry(struct zswap_entry *entry,
struct folio *folio;
struct mempolicy *mpol;
struct swap_info_struct *si;
- struct swap_io_ctx ctx = {};
int ret = 0;
/* try to allocate swap cache folio */
@@ -1082,8 +1100,7 @@ static int zswap_writeback_entry(struct zswap_entry *entry,
folio_put(folio);
/* start writeback */
- __swap_writeout(&ctx, folio);
- swap_write_submit(&ctx);
+ __swap_writeout(ctx, folio);
return 0;
@@ -1123,7 +1140,7 @@ static enum lru_status shrink_memcg_cb(struct list_head *item, struct list_lru_o
void *arg)
{
struct zswap_entry *entry = container_of(item, struct zswap_entry, lru);
- bool *encountered_page_in_swapcache = (bool *)arg;
+ struct zswap_shrink_ctl *ctl = arg;
swp_entry_t swpentry;
enum lru_status ret = LRU_REMOVED_RETRY;
int writeback_result;
@@ -1178,7 +1195,7 @@ static enum lru_status shrink_memcg_cb(struct list_head *item, struct list_lru_o
*/
spin_unlock(&l->lock);
- writeback_result = zswap_writeback_entry(entry, swpentry);
+ writeback_result = zswap_writeback_entry(&ctl->io_ctx, entry, swpentry);
if (writeback_result) {
zswap_reject_reclaim_fail++;
@@ -1189,9 +1206,9 @@ static enum lru_status shrink_memcg_cb(struct list_head *item, struct list_lru_o
* into the warmer region. We should terminate shrinking (if we're in the dynamic
* shrinker context).
*/
- if (writeback_result == -EEXIST && encountered_page_in_swapcache) {
+ if (writeback_result == -EEXIST && ctl->encountered_page_in_swapcache) {
ret = LRU_STOP;
- *encountered_page_in_swapcache = true;
+ *ctl->encountered_page_in_swapcache = true;
}
} else {
zswap_written_back_pages++;
@@ -1203,8 +1220,11 @@ static enum lru_status shrink_memcg_cb(struct list_head *item, struct list_lru_o
static unsigned long zswap_shrinker_scan(struct shrinker *shrinker,
struct shrink_control *sc)
{
- unsigned long shrink_ret;
bool encountered_page_in_swapcache = false;
+ struct zswap_shrink_ctl ctl = {
+ .encountered_page_in_swapcache = &encountered_page_in_swapcache,
+ };
+ unsigned long shrink_ret;
if (!zswap_shrinker_enabled ||
!mem_cgroup_zswap_writeback_enabled(sc->memcg)) {
@@ -1213,7 +1233,9 @@ static unsigned long zswap_shrinker_scan(struct shrinker *shrinker,
}
shrink_ret = list_lru_shrink_walk(&zswap_list_lru, sc, &shrink_memcg_cb,
- &encountered_page_in_swapcache);
+ &ctl);
+
+ swap_write_submit(&ctl.io_ctx);
if (encountered_page_in_swapcache)
return SHRINK_STOP;
@@ -1319,6 +1341,7 @@ static struct shrinker *zswap_alloc_shrinker(void)
*/
static int shrink_memcg(struct mem_cgroup *memcg)
{
+ struct zswap_shrink_ctl ctl = {};
int nid, shrunk = 0, scanned = 0;
if (!mem_cgroup_zswap_writeback_enabled(memcg))
@@ -1335,10 +1358,12 @@ static int shrink_memcg(struct mem_cgroup *memcg)
unsigned long nr_to_walk = SWAP_CLUSTER_MAX;
shrunk += list_lru_walk_one(&zswap_list_lru, nid, memcg,
- &shrink_memcg_cb, NULL, &nr_to_walk);
+ &shrink_memcg_cb, &ctl, &nr_to_walk);
scanned += SWAP_CLUSTER_MAX - nr_to_walk;
}
+ swap_write_submit(&ctl.io_ctx);
+
/* Nothing was scanned: every LRU under @memcg was empty. */
if (!scanned)
return -ENOENT;
--
2.53.0-Meta