[PATCH v4 01/14] mm: xswap support for zswap

From: Baoquan He

Date: Fri Oct 02 2026 - 20:32:28 EST


From: Chris Li <chrisl@xxxxxxxxxx>

Introduce extendable swap device support - xswap.

An xswap device has no backing storage and no swap data section, so
it wastes no disk space. Creation is via a sysfs interface added in a
later patch.

Zswap writeback is gated on whether a real (non-xswap) swap device is
active. nr_real_swapfiles counts such devices and is maintained at
swapon/swapoff only, so the gate reflects "a device exists to write
back to" rather than "a device currently has free slots". This keeps
writeback working even when the real swap device is full, and avoids
a double decrement when a full device is swapped off.

An xswap entry is not added to the zswap writeback LRU: with no backing
store there is nothing to write it back to. zswap_lru_del() tolerates
that, since an entry that was never added is on no list.

Co-developed-by: Baoquan He <hebaoquan@xxxxxxxxxx>
Signed-off-by: Baoquan He <hebaoquan@xxxxxxxxxx>
Signed-off-by: Chris Li <chrisl@xxxxxxxxxx>
---
include/linux/swap.h | 2 ++
mm/page_io.c | 19 +++++++++++++++++++
mm/swap_state.c | 9 +++++++++
mm/swapfile.c | 28 ++++++++++++++++++++++++----
mm/zswap.c | 11 +++++++++--
5 files changed, 63 insertions(+), 6 deletions(-)

diff --git a/include/linux/swap.h b/include/linux/swap.h
index 4f686709d71f..9073d29377d8 100644
--- a/include/linux/swap.h
+++ b/include/linux/swap.h
@@ -201,6 +201,7 @@ enum {
SWP_STABLE_WRITES = (1 << 11), /* no overwrite PG_writeback pages */
SWP_SYNCHRONOUS_IO = (1 << 12), /* synchronous IO is efficient */
SWP_HIBERNATION = (1 << 13), /* pinned for hibernation */
+ SWP_XSWAP = (1 << 14), /* extendable swap device */
/* add others here before... */
};

@@ -378,6 +379,7 @@ void free_folio_and_swap_cache(struct folio *folio);
void free_pages_and_swap_cache(struct encoded_page **, int);
/* linux/mm/swapfile.c */
extern atomic_long_t nr_swap_pages;
+extern atomic_t nr_real_swapfiles;
extern long total_swap_pages;
extern atomic_t nr_rotate_swap;

diff --git a/mm/page_io.c b/mm/page_io.c
index c6824fcd483e..25fa9b82ed46 100644
--- a/mm/page_io.c
+++ b/mm/page_io.c
@@ -248,6 +248,15 @@ int swap_writeout(struct swap_io_ctx *ctx, struct folio *folio)
}
rcu_read_unlock();

+ /*
+ * xswap has no backing store: keep the folio. ctx->sis is not set
+ * yet, so look the device up from the entry.
+ */
+ if (unlikely(__swap_entry_to_info(folio->swap)->flags & SWP_XSWAP)) {
+ folio_mark_dirty(folio);
+ return AOP_WRITEPAGE_ACTIVATE;
+ }
+
__swap_writeout(ctx, folio);
return 0;
out_unlock:
@@ -486,6 +495,16 @@ void swap_read_folio(struct swap_io_ctx *ctx, struct folio *folio)
if (zswap_load(folio) != -ENOENT)
goto finish;

+ if (unlikely(sis->flags & SWP_XSWAP)) {
+ /*
+ * An xswap entry only ever lives in zswap, so zswap_load()
+ * must have found it. Unlock and let the caller retry.
+ */
+ WARN_ON_ONCE(1);
+ folio_unlock(folio);
+ goto finish;
+ }
+
/* We have to read from slower devices. Increase zswap protection. */
zswap_folio_swapin(folio);
swap_add_folio(ctx, folio, READ);
diff --git a/mm/swap_state.c b/mm/swap_state.c
index ebef568cd62e..3a4e9e447b0b 100644
--- a/mm/swap_state.c
+++ b/mm/swap_state.c
@@ -884,6 +884,10 @@ struct folio *swap_cluster_readahead(swp_entry_t entry, gfp_t gfp_mask,
struct blk_plug plug;
swp_entry_t ra_entry;

+ /* xswap entries live only in zswap; readahead does not help. */
+ if (si->flags & SWP_XSWAP)
+ goto skip;
+
mask = swapin_nr_pages(offset) - 1;
if (!mask)
goto skip;
@@ -969,6 +973,7 @@ static int swap_vma_ra_win(struct vm_fault *vmf, unsigned long *start,
static struct folio *swap_vma_readahead(swp_entry_t targ_entry, gfp_t gfp_mask,
struct mempolicy *mpol, pgoff_t targ_ilx, struct vm_fault *vmf)
{
+ struct swap_info_struct *si = __swap_entry_to_info(targ_entry);
struct swap_io_ctx ctx = {};
struct blk_plug plug;
struct folio *folio;
@@ -977,6 +982,10 @@ static struct folio *swap_vma_readahead(swp_entry_t targ_entry, gfp_t gfp_mask,
unsigned long start, end, addr;
pgoff_t ilx = targ_ilx;

+ /* xswap entries live only in zswap; readahead does not help. */
+ if (si->flags & SWP_XSWAP)
+ goto skip;
+
win = swap_vma_ra_win(vmf, &start, &end);
if (win == 1)
goto skip;
diff --git a/mm/swapfile.c b/mm/swapfile.c
index c3288910b3e3..32c3133e1211 100644
--- a/mm/swapfile.c
+++ b/mm/swapfile.c
@@ -66,6 +66,7 @@ static void move_cluster(struct swap_info_struct *si,
static DEFINE_SPINLOCK(swap_lock);
static unsigned int nr_swapfiles;
atomic_long_t nr_swap_pages;
+atomic_t nr_real_swapfiles;
/*
* Some modules use swappable objects and may try to swap them out under
* memory pressure (via the shrinker). Before doing so, they may wish to
@@ -733,7 +734,8 @@ static void free_cluster(struct swap_info_struct *si, struct swap_cluster_info *
/*
* If the swap is discardable, prepare discard the cluster
* instead of free it immediately. The cluster will be freed
- * after discard.
+ * after discard. xswap has no bdev and never sets
+ * SWP_PAGE_DISCARD, so it always takes the free path below.
*/
if ((si->flags & (SWP_WRITEOK | SWP_PAGE_DISCARD)) ==
(SWP_WRITEOK | SWP_PAGE_DISCARD)) {
@@ -1216,6 +1218,9 @@ static void del_from_avail_list(struct swap_info_struct *si, bool swapoff)
*/
lockdep_assert_held(&si->lock);
si->flags &= ~SWP_WRITEOK;
+ /* Count active devices, not merely those on the avail list. */
+ if (!(si->flags & SWP_XSWAP))
+ atomic_sub(1, &nr_real_swapfiles);
atomic_long_or(SWAP_USAGE_OFFLIST_BIT, &si->inuse_pages);
} else {
/*
@@ -1273,6 +1278,8 @@ static void add_to_avail_list(struct swap_info_struct *si, bool swapon)
}

plist_add(&si->avail_list, &swap_avail_head);
+ if (swapon && !(si->flags & SWP_XSWAP))
+ atomic_add(1, &nr_real_swapfiles);

skip:
spin_unlock(&swap_avail_lock);
@@ -3268,7 +3275,8 @@ SYSCALL_DEFINE1(swapoff, const char __user *, specialfile)

destroy_swap_extents(p, p->swap_file);

- if (!(p->flags & SWP_SOLIDSTATE))
+ if (!(p->flags & SWP_XSWAP) &&
+ !(p->flags & SWP_SOLIDSTATE))
atomic_dec(&nr_rotate_swap);

mutex_lock(&swapon_mutex);
@@ -3378,6 +3386,19 @@ static void swap_stop(struct seq_file *swap, void *v)
mutex_unlock(&swapon_mutex);
}

+static const char *swap_type_str(struct swap_info_struct *si)
+{
+ struct file *file = si->swap_file;
+
+ if (si->flags & SWP_XSWAP)
+ return "xswap\t";
+
+ if (S_ISBLK(file_inode(file)->i_mode))
+ return "partition";
+
+ return "file\t";
+}
+
static int swap_show(struct seq_file *swap, void *v)
{
struct swap_info_struct *si = v;
@@ -3397,8 +3418,7 @@ static int swap_show(struct seq_file *swap, void *v)
len = seq_file_path(swap, file, " \t\n\\");
seq_printf(swap, "%*s%s\t%lu\t%s%lu\t%s%d\n",
len < 40 ? 40 - len : 1, " ",
- S_ISBLK(file_inode(file)->i_mode) ?
- "partition" : "file\t",
+ swap_type_str(si),
bytes, bytes < 10000000 ? "\t" : "",
inuse, inuse < 10000000 ? "\t" : "",
si->prio);
diff --git a/mm/zswap.c b/mm/zswap.c
index ae19e301fced..7fe25f0b157c 100644
--- a/mm/zswap.c
+++ b/mm/zswap.c
@@ -1018,6 +1018,11 @@ static int zswap_writeback_entry(struct zswap_entry *entry,
if (IS_ERR_OR_NULL(si))
return -ENOENT;

+ if (si->flags & SWP_XSWAP) {
+ put_swap_device(si);
+ return -EINVAL;
+ }
+
mpol = get_task_policy(current);
folio = __swap_cache_alloc_folio(swpentry, GFP_KERNEL, BIT(0), NULL, mpol,
NO_INTERLEAVE_INDEX);
@@ -1511,7 +1516,9 @@ static bool zswap_store_page(struct folio *folio, long index,
entry->referenced = true;
if (entry->length) {
INIT_LIST_HEAD(&entry->lru);
- zswap_lru_add(entry);
+ /* No backing store: nothing to write these back to. */
+ if (!(__swap_entry_to_info(page_swpentry)->flags & SWP_XSWAP))
+ zswap_lru_add(entry);
}

return true;
@@ -1581,7 +1588,7 @@ bool zswap_store(struct folio *folio)
zswap_pool_put(pool);
put_objcg:
obj_cgroup_put(objcg);
- if (!ret && zswap_pool_reached_full)
+ if (!ret && zswap_pool_reached_full && atomic_read(&nr_real_swapfiles))
queue_work(shrink_wq, &zswap_shrink_work);
check_old:
/*
--
2.54.0