Re: [PATCH v4 04/14] mm, swap: add sysfs create interface for xswap
From: KunWu Chan
Date: Wed Oct 07 2026 - 02:32:34 EST
On Sat, Oct 3, 2026 at 8:32 AM Baoquan He <hebaoquan@xxxxxxxxxx> wrote:
>
> xswap devices have no backing storage, so there is no file to swapon.
> Add a sysfs interface to create them directly:
>
> /sys/kernel/mm/xswap/create write an optional priority to create
> a device, empty for the default
>
> A new device is created with si->max and nr_clusters_max both equal to
> twice RAM: the cluster_info array is a sparse VM_SPARSE area mapped
> lazily, so an idle device costs nothing. The device shows up in
> /proc/swaps as "xswap<N>".
>
> It takes the highest priority by default. Swapout picks the
> highest-priority device that has room, so a real device ahead of an xswap
> one would take the page directly: zswap would still cache it, but the
> slot on the real device is taken either way, and not taking it is the
> point of xswap. A lower priority can be asked for, not a higher one.
>
> Hibernation device discovery skips xswap devices: they have no bdev
> to carry a resume image.
>
> Signed-off-by: Baoquan He <hebaoquan@xxxxxxxxxx>
> ---
> mm/swapfile.c | 169 ++++++++++++++++++++++++++++++++++++++++++++++++--
> 1 file changed, 164 insertions(+), 5 deletions(-)
>
> diff --git a/mm/swapfile.c b/mm/swapfile.c
> index 767d3f46877f..9c204e455ef9 100644
> --- a/mm/swapfile.c
> +++ b/mm/swapfile.c
> @@ -16,6 +16,8 @@
> #include <linux/kernel_stat.h>
> #include <linux/swap.h>
> #include <linux/vmalloc.h>
> +#include <linux/kobject.h>
> +#include <linux/sysfs.h>
> #include <linux/pagemap.h>
> #include <linux/namei.h>
> #include <linux/shmem_fs.h>
> @@ -48,6 +50,13 @@
> #include "swap_table.h"
> #include "internal.h"
> #include "swap.h"
> +#define DEF_SWAP_PRIO -1
> +
> +/*
> + * Above every real swap device: one ahead of xswap would take the page
> + * directly, and taking that slot is what xswap exists to avoid.
> + */
> +#define DEF_XSWAP_PRIO SWAP_FLAG_PRIO_MASK
>
> #ifdef CONFIG_XSWAP
> /*
> @@ -66,7 +75,65 @@ static int xswap_map_clusters(struct swap_info_struct *si,
> static void xswap_unmap_clusters(struct swap_info_struct *si,
> unsigned long start_idx, unsigned long nr);
> static int xswap_mapped_end(pte_t *pte, unsigned long addr, void *data);
> -#endif
> +
> +static int xswap_create(int prio);
> +
> +static ssize_t xswap_create_store(struct kobject *kobj,
> + struct kobj_attribute *attr,
> + const char *buf, size_t count)
> +{
> + int prio = DEF_XSWAP_PRIO;
> + int err;
> +
> + if (!capable(CAP_SYS_ADMIN))
> + return -EPERM;
> +
> + /* "[<prio>]" is optional; a lower one is allowed, not a higher. */
> + if (*skip_spaces(buf)) {
> + err = kstrtoint(buf, 10, &prio);
> + if (err)
> + return err;
> + }
> +
> + err = xswap_create(prio);
> + if (err < 0)
> + return err;
> +
> + return count;
> +}
> +
> +static struct kobj_attribute xswap_create_attr = __ATTR(create, 0200, NULL,
> + xswap_create_store);
> +
> +static struct attribute *xswap_attrs[] = {
> + &xswap_create_attr.attr,
> + NULL,
> +};
> +
> +static const struct attribute_group xswap_attr_group = {
> + .attrs = xswap_attrs,
> +};
> +
> +static struct kobject *xswap_kobj;
> +
> +static void xswap_sysfs_init(void)
> +{
> + xswap_kobj = kobject_create_and_add("xswap", mm_kobj);
> + if (!xswap_kobj) {
> + pr_err("xswap: failed to create sysfs kobject\n");
> + return;
> + }
> + if (sysfs_create_group(xswap_kobj, &xswap_attr_group)) {
> + pr_err("xswap: failed to create sysfs group\n");
> + kobject_put(xswap_kobj);
> + xswap_kobj = NULL;
> + }
> +}
> +#else /* !CONFIG_XSWAP */
> +static inline void xswap_sysfs_init(void)
> +{
> +}
> +#endif /* CONFIG_XSWAP */
>
> static void swap_range_alloc(struct swap_info_struct *si,
> unsigned int nr_entries);
> @@ -94,7 +161,6 @@ atomic_t nr_real_swapfiles;
> EXPORT_SYMBOL_GPL(nr_swap_pages);
> /* protected with swap_lock. reading in vm_swap_full() doesn't need lock */
> long total_swap_pages;
> -#define DEF_SWAP_PRIO -1
> unsigned long swapfile_maximum_size;
> #ifdef CONFIG_MIGRATION
> bool swap_migration_ad_supported;
> @@ -2275,6 +2341,9 @@ static int __find_hibernation_swap_type(dev_t device, sector_t offset)
>
> if (!(sis->flags & SWP_WRITEOK))
> continue;
> + /* xswap has no bdev to match a resume device */
> + if (sis->flags & SWP_XSWAP)
> + continue;
>
> if (device == sis->bdev->bd_dev) {
> struct swap_extent *se = first_se(sis);
> @@ -2462,6 +2531,8 @@ int find_first_swap(dev_t *device)
>
> if (!(sis->flags & SWP_WRITEOK))
> continue;
> + if (sis->flags & SWP_XSWAP)
> + continue;
> *device = sis->bdev->bd_dev;
> spin_unlock(&swap_lock);
> return type;
> @@ -3425,7 +3496,8 @@ static void *swap_start(struct seq_file *swap, loff_t *pos)
> return SEQ_START_TOKEN;
>
> for (type = 0; (si = swap_type_to_info(type)); type++) {
> - if (!(si->swap_file))
> + if (!si->swap_file &&
> + !((si->flags & SWP_XSWAP) && (si->flags & SWP_WRITEOK)))
> continue;
> if (!--l)
> return si;
> @@ -3446,7 +3518,8 @@ static void *swap_next(struct seq_file *swap, void *v, loff_t *pos)
>
> ++(*pos);
> for (; (si = swap_type_to_info(type)); type++) {
> - if (!(si->swap_file))
> + if (!si->swap_file &&
> + !((si->flags & SWP_XSWAP) && (si->flags & SWP_WRITEOK)))
> continue;
> return si;
> }
> @@ -3488,7 +3561,14 @@ static int swap_show(struct seq_file *swap, void *v)
> inuse = K(swap_usage_in_pages(si));
>
> file = si->swap_file;
> - len = seq_file_path(swap, file, " \t\n\\");
> + if (file)
> + len = seq_file_path(swap, file, " \t\n\\");
> + else {
> + char name[16];
> +
> + len = scnprintf(name, sizeof(name), "xswap%d", si->type);
> + seq_puts(swap, name);
> + }
> seq_printf(swap, "%*s%s\t%lu\t%s%lu\t%s%d\n",
> len < 40 ? 40 - len : 1, " ",
> swap_type_str(si),
> @@ -4011,6 +4091,83 @@ static int setup_swap_clusters_info(struct swap_info_struct *si,
> return err;
> }
>
> +#ifdef CONFIG_XSWAP
> +/* Create a file-less xswap device, sized at twice RAM. */
> +static int xswap_create(int prio)
> +{
> + struct swap_info_struct *si;
> + unsigned long ram, maxpages;
> + int error;
> +
> + if (prio != DEF_XSWAP_PRIO &&
> + (prio < 0 || prio > SWAP_FLAG_PRIO_MASK))
> + return -EINVAL;
> +
> + si = alloc_swap_info();
> + if (IS_ERR(si))
> + return PTR_ERR(si);
> +
> + INIT_WORK(&si->discard_work, swap_discard_work);
> + INIT_WORK(&si->reclaim_work, swap_reclaim_work);
> +
> + ram = totalram_pages();
> + maxpages = min_t(unsigned long, ram * 2, swapfile_maximum_size);
> + /* si->max is an unsigned int: don't overflow it. */
> + if (maxpages > UINT_MAX)
> + maxpages = UINT_MAX;
> + /* Cluster-aligned, so no cluster holds a slot past si->max. */
> + if (maxpages > SWAPFILE_CLUSTER)
> + maxpages = rounddown(maxpages, SWAPFILE_CLUSTER);
> + if (maxpages < 2)
> + maxpages = 2;
> +
> + si->bdev = NULL;
> + si->flags |= SWP_XSWAP | SWP_SOLIDSTATE;
> + si->max = maxpages;
> + si->pages = maxpages - 1;
> + /*
> + * No backing file: setup_swap_extents() is only reachable from the
> + * file-backed swapon() path, so set ops here. Only ops->flags is
> + * used, by may_enter_fs(); the IO methods are never called because
> + * swap_writeout()/swap_read_folio() short circuit xswap.
> + */
> + si->ops = &swap_bdev_ops;
> +
> + error = setup_swap_clusters_info(si, NULL, maxpages);
> + if (error)
> + goto bad_swap;
> +
> + error = zswap_swapon(si->type, si->max);
> + if (error)
> + goto bad_swap;
> +
> + mutex_lock(&swapon_mutex);
> + si->prio = prio;
> + si->list.prio = -si->prio;
> + si->avail_list.prio = -si->prio;
> + /* si->swap_file stays NULL: this is a file-less device */
> + enable_swap_info(si);
> + pr_info("xswap: adding extendable swap type %d (prio %d, %u pages, max %lu)\n",
> + si->type, prio, si->pages, maxpages);
> + mutex_unlock(&swapon_mutex);
> + atomic_inc(&proc_poll_event);
> + wake_up_interruptible(&proc_poll_wait);
> +
> + return si->type;
> +
> +bad_swap:
> + kfree(si->global_cluster);
> + si->global_cluster = NULL;
> + destroy_swap_extents(si, NULL); /* safe: xswap never sets SWP_ACTIVATED */
> + free_swap_cluster_info(si);
> + si->cluster_info = NULL;
> + spin_lock(&swap_lock);
> + si->flags = 0;
> + spin_unlock(&swap_lock);
> + return error;
> +}
> +#endif /* CONFIG_XSWAP */
> +
> SYSCALL_DEFINE2(swapon, const char __user *, specialfile, int, swap_flags)
> {
> struct swap_info_struct *si;
> @@ -4357,6 +4514,8 @@ static int __init swapfile_init(void)
> swap_migration_ad_supported = true;
> #endif /* CONFIG_MIGRATION */
>
> + xswap_sysfs_init();
> +
> return 0;
> }
> subsys_initcall(swapfile_init);
> --
> 2.54.0
>
The create path and priority handling look consistent with the
existing swap priority rules. In particular, an xswap device has no
`swap_file`, and the changes to `swap_show()` / `swap_start()` /
`swap_next()` account for that when enumerating and displaying swap
devices.
The `VM_SPARSE` mapping avoids allocating the full cluster metadata
backing up front, while allowing the mapped range to grow later.
The overcommit interaction (`si->pages` -> `total_swap_pages` ->
`__vm_enough_memory()`) is an open design item already called out in
the cover letter. I don't see it as a reason to block this phase, but
it should remain explicitly tracked.
Acked-by: Kunwu Chan <chentao@xxxxxxxxxx>
Thanks,
Kunwu