Re: [PATCH v4 05/14] mm, swap: add xswap grow trigger on cluster allocation
From: KunWu Chan
Date: Wed Oct 07 2026 - 02:54:47 EST
On Sat, Oct 3, 2026 at 8:32 AM Baoquan He <hebaoquan@xxxxxxxxxx> wrote:
>
> Grow the mapped range before it runs out rather than after, from
> cluster_alloc_swap_entry().
>
> Growing maps pages into the VM_SPARSE area and can sleep. Doing it from
> the allocation that would otherwise have failed puts that sleep on the
> swap-out path with nothing left to fall back on. Growing at 85% in use
> instead leaves the allocation that pays for it a reserve to land in, and
> leaves one grow's worth of room in front of the next.
>
> The target is 65%: past the point where the shrink starts looking (50%)
> and short of the point that triggers the next grow (85%), so a grow
> cannot leave the range in a state the shrink would undo.
>
> Since xswap is always SWP_SOLIDSTATE, global_cluster_lock is never
> held on this path.
>
> This makes the xswap cluster space grow transparently as swap usage
> increases, without any userspace intervention.
>
> The caller holds local_lock(&percpu_swap_cluster.lock) across the whole
> slow path, so drop it around xswap_map_clusters() and take it again
> afterwards; it only protects the per-cpu cluster cache, which this path
> does not touch.
>
> Signed-off-by: Baoquan He <hebaoquan@xxxxxxxxxx>
> ---
> mm/swapfile.c | 93 +++++++++++++++++++++++++++++++++++++++++++++++++++
> 1 file changed, 93 insertions(+)
>
> diff --git a/mm/swapfile.c b/mm/swapfile.c
> index 9c204e455ef9..45050f8a2ed8 100644
> --- a/mm/swapfile.c
> +++ b/mm/swapfile.c
> @@ -70,6 +70,19 @@
> #define XSWAP_GROW_CLUSTERS \
> max_t(unsigned long, PAGE_SIZE / sizeof(struct swap_cluster_info), 16)
>
> +/*
> + * Grow and shrink thresholds, as a percentage of the mapped range in use.
> + * Each one lands inside (SHRINK_WHEN, GROW_WHEN), so neither can leave the
> + * range in a state that wakes the other up.
> + */
> +#define XSWAP_SHRINK_WHEN 50
> +#define XSWAP_SHRINK_UNTIL 70
> +#define XSWAP_GROW_UNTIL 65
> +#define XSWAP_GROW_WHEN 85
> +#define XSWAP_GROW_CHUNKS 4
> +
> +static bool xswap_should_grow(struct swap_info_struct *si);
> +static unsigned long xswap_grow(struct swap_info_struct *si);
> static int xswap_map_clusters(struct swap_info_struct *si,
> unsigned long start_idx, unsigned long nr);
> static void xswap_unmap_clusters(struct swap_info_struct *si,
> @@ -1204,6 +1217,17 @@ static unsigned long cluster_alloc_swap_entry(struct swap_info_struct *si,
> if (order && !(si->flags & SWP_BLKDEV))
> return 0;
>
> +#ifdef CONFIG_XSWAP
> + /*
> + * Top the range up early. Every path below leaves through `done`,
> + * so this has to come first; mapping pages can sleep, and doing it
> + * from the allocation that would otherwise fail leaves nothing to
> + * fall back on.
> + */
> + if ((si->flags & SWP_XSWAP) && xswap_should_grow(si))
> + xswap_grow(si);
> +#endif
> +
> if (!(si->flags & SWP_SOLIDSTATE)) {
> /* Serialize HDD SWAP allocation for each device. */
> spin_lock(&si->global_cluster_lock);
> @@ -1280,6 +1304,12 @@ static unsigned long cluster_alloc_swap_entry(struct swap_info_struct *si,
> if (found)
> goto done;
> }
> +
> +#ifdef CONFIG_XSWAP
> + /* A concurrent free or grow may have added clusters; retry once. */
> + if (!found && (si->flags & SWP_XSWAP))
> + found = alloc_swap_scan_list(si, &si->free_clusters, folio, false);
> +#endif
> done:
> if (!(si->flags & SWP_SOLIDSTATE))
> spin_unlock(&si->global_cluster_lock);
> @@ -3909,6 +3939,69 @@ static int xswap_map_clusters(struct swap_info_struct *si,
> return -ENOMEM;
> }
>
> +static bool xswap_should_grow(struct swap_info_struct *si)
> +{
> + unsigned long mapped = READ_ONCE(si->nr_clusters_mapped);
> +
> + if (mapped >= READ_ONCE(si->nr_clusters_max))
> + return false;
> +
> + return swap_usage_in_pages(si) * 100 >=
> + mapped * SWAPFILE_CLUSTER * XSWAP_GROW_WHEN;
> +}
> +
> +/*
> + * Map more of the cluster_info array, up to GROW_UNTIL in use. The caller
> + * holds percpu_swap_cluster.lock, which this drops while it sleeps.
> + */
> +static unsigned long xswap_grow(struct swap_info_struct *si)
> +{
> + unsigned long mapped = READ_ONCE(si->nr_clusters_mapped);
> + unsigned long nr_new, want, i;
> + int ret;
> +
> + /* The unmap below is paired with the caller's lock. */
> + lockdep_assert_held(&percpu_swap_cluster.lock);
> +
> + want = DIV_ROUND_UP(swap_usage_in_pages(si) * 100,
> + XSWAP_GROW_UNTIL * SWAPFILE_CLUSTER);
> + if (want <= mapped)
> + return 0;
> +
> + nr_new = rounddown(want - mapped, XSWAP_GROW_CLUSTERS);
> + if (!nr_new)
> + nr_new = XSWAP_GROW_CLUSTERS;
> + if (nr_new > XSWAP_GROW_CHUNKS * XSWAP_GROW_CLUSTERS)
> + nr_new = XSWAP_GROW_CHUNKS * XSWAP_GROW_CLUSTERS;
> + nr_new = min(nr_new, READ_ONCE(si->nr_clusters_max) - mapped);
> +
> + /*
> + * Mapping pages can sleep. The lock only guards the per-cpu
> + * cluster cache, which this path does not touch.
> + */
> + local_unlock(&percpu_swap_cluster.lock);
> + ret = xswap_map_clusters(si, mapped, nr_new);
> + local_lock(&percpu_swap_cluster.lock);
> + if (ret)
> + return 0;
> +
> + for (i = mapped; i < mapped + nr_new; i++) {
> + struct swap_cluster_info *ci = &si->cluster_info[i];
> +
> + /*
> + * A concurrent grower may have taken these already;
> + * only add the off-list ones.
> + */
> + spin_lock(&ci->lock);
> + if (ci->flags == CLUSTER_FLAG_NONE)
> + move_cluster(si, ci, &si->free_clusters,
> + CLUSTER_FLAG_FREE);
> + spin_unlock(&ci->lock);
> + }
> +
> + return nr_new;
> +}
> +
> static void xswap_unmap_clusters(struct swap_info_struct *si,
> unsigned long start_idx, unsigned long nr)
> {
> --
> 2.54.0
>
Hi Baoquan,
The 85% grow threshold gives the allocation path headroom before
the currently mapped range is exhausted.
I also checked the lock scope in `xswap_grow()`. Dropping the
per-CPU `local_lock` around `xswap_map_clusters()` is appropriate
because the mapping operation may sleep, while this path does not
access the per-CPU cluster cache.
`xswap_should_grow()` uses `READ_ONCE()` for
`nr_clusters_mapped`, so a concurrent grow can make an allocation
miss the grow check for one round. This is benign because a
subsequent allocation will observe the updated mapped range.
Reviewed-by: Kunwu Chan <chentao@xxxxxxxxxx>
Thanks,
Kunwu