Re: [RFC v3 05/15] mm, swap: add xswap cluster grow via VM_SPARSE vmalloc

From: Klara Modin

Date: Fri Aug 14 2026 - 19:09:35 EST


Hi,

On 2026-08-13 18:48:44 +0800, Baoquan He wrote:
> Implement dynamic cluster_info array growth for xswap devices using a
> VM_SPARSE vmalloc area:
>
> 1. xswap_map_clusters(): Allocate physical pages and map them into
> the pre-reserved VM_SPARSE KVA region via vm_area_map_pages().
>
> 2. xswap_unmap_clusters(): Unmap pages from the VM_SPARSE area via
> vm_area_unmap_pages() (used by the error/teardown paths, shrink
> comes later).
>
> 3. setup_swap_clusters_info() xswap path: Use get_vm_area(VM_SPARSE)
> for the cluster_info array, lazily mapping only the initial chunk.
>
> 4. free_swap_cluster_info() xswap path: Unmap all clusters and
> free_vm_area(). Built on the refactoring in the previous patch.
>
> 5. wait_for_allocation() xswap guard: Skip shrinker-unmapped clusters
> beyond nr_clusters_mapped.
>
> The grow path avoids emergency reserves via __GFP_HIGH|__GFP_NOMEMALLOC
> and wraps allocations with memalloc_noreclaim_save(). A per-device
> mutex (xswap_lock) serializes concurrent map/unmap page table
> modifications.
>
> Signed-off-by: Baoquan He <hebaoquan@xxxxxxxxxx>
> ---
> include/linux/swap.h | 1 +
> mm/swapfile.c | 257 ++++++++++++++++++++++++++++++++++++++++++-
> 2 files changed, 256 insertions(+), 2 deletions(-)
>
> diff --git a/include/linux/swap.h b/include/linux/swap.h
> index 7ffc62a3b2d7..c824848c6cfc 100644
> --- a/include/linux/swap.h
> +++ b/include/linux/swap.h
> @@ -252,6 +252,7 @@ struct swap_info_struct {
> struct vm_struct *cluster_vm; /* VM_SPARSE area for xswap dynamic cluster_info */
> unsigned long nr_clusters; /* total cluster count for xswap */
> unsigned long nr_clusters_mapped; /* currently mapped cluster count */
> + struct mutex xswap_lock; /* serialize map/unmap operations */
> #endif
> struct list_head free_clusters; /* free clusters list */
> struct list_head full_clusters; /* full clusters list */
> diff --git a/mm/swapfile.c b/mm/swapfile.c
> index 4ce30e9ecdf6..178c3b798f8e 100644
> --- a/mm/swapfile.c
> +++ b/mm/swapfile.c
> @@ -49,6 +49,25 @@
> #include "internal.h"
> #include "swap.h"
>
> +#ifdef CONFIG_XSWAP
> +/*
> + * xswap: dynamically grow the cluster_info array via a VM_SPARSE area.
> + *
> + * XSWAP_GROW_CLUSTERS is the number of clusters to map in one grow
> + * operation. It is set to the number of cluster_info structs that
> + * fit in a single page (at least 16), so that the vmalloc page table
> + * overhead is proportional to the number of clusters mapped.
> + */
> +#define XSWAP_GROW_CLUSTERS \
> + max_t(unsigned long, PAGE_SIZE / sizeof(struct swap_cluster_info), 16)
> +
> +static int xswap_map_clusters(struct swap_info_struct *si,
> + unsigned long start_idx, unsigned long nr);
> +static void xswap_unmap_clusters(struct swap_info_struct *si,
> + unsigned long start_idx, unsigned long nr);
> +static int xswap_check_mapped(pte_t *pte, unsigned long addr, void *data);
> +#endif
> +
> static void swap_range_alloc(struct swap_info_struct *si,
> unsigned int nr_entries);
> static bool folio_swapcache_freeable(struct folio *folio);
> @@ -2708,15 +2727,27 @@ static unsigned int find_next_to_unuse(struct swap_info_struct *si,
> unsigned int prev)
> {
> unsigned int i;
> + unsigned int end = si->max;
> unsigned long swp_tb;
>
> +#ifdef CONFIG_XSWAP
> + /* xswap may have shrunk and unmapped the cluster_info tail. */
> + if (si->flags & SWP_XSWAP) {
> + unsigned long mapped_end;
> +
> + mapped_end = READ_ONCE(si->nr_clusters_mapped) * SWAPFILE_CLUSTER;
> + if (mapped_end < end)
> + end = mapped_end;
> + }
> +#endif
> +
> /*
> * No need for swap_lock here: we're just looking
> * for whether an entry is in use, not modifying it; false
> * hits are okay, and sys_swapoff() has already prevented new
> * allocations from this area (while holding swap_lock).
> */
> - for (i = prev + 1; i < si->max; i++) {
> + for (i = prev + 1; i < end; i++) {
> swp_tb = swap_table_get(__swap_offset_to_cluster(si, i),
> i % SWAPFILE_CLUSTER);
> if (!swp_tb_is_null(swp_tb) && !swp_tb_is_bad(swp_tb))
> @@ -2725,7 +2756,7 @@ static unsigned int find_next_to_unuse(struct swap_info_struct *si,
> cond_resched();
> }
>
> - if (i == si->max)
> + if (i == end)
> i = 0;
>
> return i;
> @@ -3041,6 +3072,13 @@ static void wait_for_allocation(struct swap_info_struct *si)
>
> BUG_ON(si->flags & SWP_WRITEOK);
>
> +#ifdef CONFIG_XSWAP
> + /* Skip shrinker-unmapped cluster tail. */
> + if (si->flags & SWP_XSWAP)
> + end = min(end, READ_ONCE(si->nr_clusters_mapped) *
> + SWAPFILE_CLUSTER);
> +#endif
> +
> for (offset = 0; offset < end; offset += SWAPFILE_CLUSTER) {
> ci = swap_cluster_lock(si, offset);
> swap_cluster_unlock(ci);
> @@ -3057,6 +3095,19 @@ static void free_swap_cluster_info(struct swap_info_struct *si)
> if (!cluster_info)
> return;
>
> +#ifdef CONFIG_XSWAP
> + if (si->flags & SWP_XSWAP) {
> + /* Unmap all mapped clusters and free the VM_SPARSE area */
> + if (si->nr_clusters_mapped > 0)
> + xswap_unmap_clusters(si, 0, si->nr_clusters_mapped);
> + free_vm_area(si->cluster_vm);
> + si->cluster_vm = NULL;
> + si->nr_clusters = 0;
> + si->nr_clusters_mapped = 0;
> + return;
> + }
> +#endif
> +
> nr_clusters = DIV_ROUND_UP(maxpages, SWAPFILE_CLUSTER);
> for (i = 0; i < nr_clusters; i++) {
> ci = cluster_info + i;
> @@ -3553,6 +3604,150 @@ static unsigned long read_swap_header(struct swap_info_struct *si,
> return maxpages;
> }
>
> +#ifdef CONFIG_XSWAP
> +static int xswap_map_clusters(struct swap_info_struct *si,
> + unsigned long start_idx, unsigned long nr)
> +{
> + unsigned long start_addr = (unsigned long)si->cluster_info +
> + (size_t)start_idx * sizeof(struct swap_cluster_info);
> + unsigned long end_addr = start_addr + (size_t)nr * sizeof(struct swap_cluster_info);
> + /* Round to page boundaries for vm_area_map_pages(). */
> + unsigned long vm_start = PAGE_ALIGN(start_addr);
> + unsigned long vm_end = PAGE_ALIGN(end_addr);
> + unsigned int noreclaim_flags;
> + unsigned long npages;
> + struct page **pages;
> + unsigned long i;
> + int err;
> +
> + mutex_lock(&si->xswap_lock);
> +
> + if (vm_start >= vm_end) {
> + /* All requested clusters fall within already-mapped pages. */
> + for (i = start_idx; i < start_idx + nr; i++)
> + spin_lock_init(&si->cluster_info[i].lock);
> + WRITE_ONCE(si->nr_clusters_mapped, start_idx + nr);
> + mutex_unlock(&si->xswap_lock);
> + return 0;
> + }
> +
> + npages = (vm_end - vm_start) >> PAGE_SHIFT;
> +
> + /* Prevent recursive reclaim during vmap page table allocation. */
> + noreclaim_flags = memalloc_noreclaim_save();
> +
> + pages = kmalloc_array(npages, sizeof(*pages),
> + __GFP_HIGH | __GFP_NOMEMALLOC | GFP_KERNEL);
> + if (!pages) {
> + memalloc_noreclaim_restore(noreclaim_flags);
> + mutex_unlock(&si->xswap_lock);
> + return -ENOMEM;
> + }
> +
> + for (i = 0; i < npages; i++) {
> + /* __GFP_ZERO: cluster_info pointer fields must start NULL. */
> + pages[i] = alloc_page(__GFP_HIGH | __GFP_NOMEMALLOC |
> + GFP_KERNEL | __GFP_ZERO);
> + if (!pages[i])
> + goto fail;
> + }
> +
> + /* Detect racing grower that already mapped these pages. */
> + if (apply_to_existing_page_range(&init_mm, vm_start,
> + vm_end - vm_start,
> + xswap_check_mapped, NULL)) {
> + i = npages;
> + goto fail_nounmap;
> + }
> +
> + err = vm_area_map_pages(si->cluster_vm, vm_start, vm_end, pages);
> + if (err) {
> + /* -EBUSY: defensive, the page was already mapped. */
> + if (err == -EBUSY) {
> + i = npages;
> + goto fail_nounmap;
> + }
> + i = npages;
> + goto fail;
> + }
> +
> + kfree(pages);
> + memalloc_noreclaim_restore(noreclaim_flags);
> +
> + /* Initialize spinlocks for newly mapped clusters */
> + for (i = start_idx; i < start_idx + nr; i++)
> + spin_lock_init(&si->cluster_info[i].lock);
> +
> + /*
> + * Pairs with READ_ONCE() in shrink/grow paths.
> + */
> + WRITE_ONCE(si->nr_clusters_mapped, start_idx + nr);
> + mutex_unlock(&si->xswap_lock);
> + return 0;
> +
> +fail_nounmap:
> + /*
> + * The concurrent grower already mapped the range, initialized the
> + * cluster spinlocks and advanced nr_clusters_mapped. It may still
> + * be holding those locks while adding clusters to the free list, so
> + * do not touch them here; just free our unused pages.
> + */
> + while (i > 0) {
> + i--;
> + if (pages[i])
> + __free_page(pages[i]);
> + }
> + kfree(pages);
> + memalloc_noreclaim_restore(noreclaim_flags);
> + mutex_unlock(&si->xswap_lock);
> + return 0;
> +
> +fail:
> + while (i > 0) {
> + i--;
> + if (pages[i])
> + __free_page(pages[i]);
> + }
> + memalloc_noreclaim_restore(noreclaim_flags);
> + kfree(pages);
> + mutex_unlock(&si->xswap_lock);
> + return -ENOMEM;
> +}
> +
> +static void xswap_unmap_clusters(struct swap_info_struct *si,
> + unsigned long start_idx, unsigned long nr)
> +{
> + unsigned long start_addr = (unsigned long)si->cluster_info +
> + (size_t)start_idx * sizeof(struct swap_cluster_info);
> + unsigned long end_addr = start_addr + (size_t)nr * sizeof(struct swap_cluster_info);
> + /* Round to page boundaries for vm_area_unmap_pages(). */
> + unsigned long vm_start = PAGE_ALIGN(start_addr);
> + unsigned long vm_end = PAGE_ALIGN(end_addr);
> +
> + mutex_lock(&si->xswap_lock);
> +
> + if (vm_start >= vm_end) {
> + WRITE_ONCE(si->nr_clusters_mapped, start_idx);
> + mutex_unlock(&si->xswap_lock);
> + return;
> + }
> +
> + vm_area_unmap_pages(si->cluster_vm, vm_start, vm_end);
> + /* vm_area_unmap_pages() clears PTEs but does not free pages. */
> + /* TODO: free backing pages via page table walk or tracking bitmap */
> +
> + /* Pairs with READ_ONCE() in shrink/grow paths. */
> + WRITE_ONCE(si->nr_clusters_mapped, start_idx);
> + mutex_unlock(&si->xswap_lock);
> +}
> +
> +/* Return 1 at first present PTE to signal range is already mapped. */
> +static int xswap_check_mapped(pte_t *pte, unsigned long addr, void *data)
> +{
> + return 1;
> +}
> +#endif /* CONFIG_XSWAP */
> +
> static int setup_swap_clusters_info(struct swap_info_struct *si,
> union swap_header *swap_header,
> unsigned long maxpages)
> @@ -3562,6 +3757,64 @@ static int setup_swap_clusters_info(struct swap_info_struct *si,
> int err = -ENOMEM;
> unsigned long i;
>
> +#ifdef CONFIG_XSWAP
> + if (si->flags & SWP_XSWAP) {
> + unsigned long size = PAGE_ALIGN(nr_clusters * sizeof(*cluster_info));
> + struct vm_struct *vm;
> +
> + vm = get_vm_area(size, VM_SPARSE);
> + if (!vm)
> + goto err;
> +
> + cluster_info = vm->addr;
> + si->cluster_vm = vm;
> + si->nr_clusters = nr_clusters;
> + si->cluster_info = cluster_info;

Should probably initialise the mutex here instead since
xswap_map_clusters() uses it?

> +
> + /* Map the initial chunk (at least cluster 0) */
> + if (xswap_map_clusters(si, 0, min_t(unsigned long,
> + XSWAP_GROW_CLUSTERS, nr_clusters)))
> + goto err_free_vm;

> +
> + /* xswap: only cluster 0 slot 0 is bad */
> + err = swap_cluster_setup_bad_slot(si, cluster_info, 0, false);
> + if (err)
> + goto err_unmap;
> +
> + INIT_LIST_HEAD(&si->free_clusters);
> + INIT_LIST_HEAD(&si->full_clusters);
> + INIT_LIST_HEAD(&si->discard_clusters);
> + for (i = 0; i < SWAP_NR_ORDERS; i++) {
> + INIT_LIST_HEAD(&si->nonfull_clusters[i]);
> + INIT_LIST_HEAD(&si->frag_clusters[i]);
> + }
> +
> + /* Mark mapped clusters: cluster 0 has 1 bad slot, rest free */
> + for (i = 0; i < si->nr_clusters_mapped; i++) {
> + struct swap_cluster_info *ci = &cluster_info[i];
> +
> + if (i == 0) {
> + ci->flags = CLUSTER_FLAG_NONFULL;
> + list_add_tail(&ci->list, &si->nonfull_clusters[0]);
> + } else {
> + ci->flags = CLUSTER_FLAG_FREE;
> + list_add_tail(&ci->list, &si->free_clusters);
> + }
> + }
> +
> + mutex_init(&si->xswap_lock);
> + return 0;
> +
> +err_unmap:
> + xswap_unmap_clusters(si, 0, si->nr_clusters_mapped);
> +err_free_vm:
> + free_vm_area(si->cluster_vm);
> + si->cluster_vm = NULL;
> + si->cluster_info = NULL;
> + return err;
> + }
> +#endif /* CONFIG_XSWAP */
> +
> cluster_info = kvzalloc_objs(*cluster_info, nr_clusters);
> if (!cluster_info)
> goto err;
> --
> 2.54.0
>

Regards,
Klara Modin