Re: [PATCH v5 26/36] mm: add NODE_PRIVATE_CAP_RECLAIM for opted-in private node reclaim
From: Richard Cheng
Date: Wed Jul 22 2026 - 09:48:30 EST
On Mon, Jul 20, 2026 at 03:34:20PM +0800, Gregory Price wrote:
> Provide a mechanism to opt private nodes into the reclaim process.
>
> Reclaim as a "singular service" is actually made up of:
>
> - kswapd reclaim
> - direct reclaim
> - kcompactd compaction
> - direct compaction
> - both mglru / lru paths
> - madvise reclaim hints
> - damon reclaim operations
>
> CAP_RECLAIM gates whether the kernel may do inter-node placement
> (compaction) or swap for the folios on that private node.
>
> node_allows_reclaim() encapsulates the policy as a whole. Ordinary
> nodes are always reclaimable, private nodes only when opted in.
>
> With the exception of madvise and DAMON, reclaim operations are
> highly integrated with one another, so they are opted in/out of
> together - otherwise reclaim becomes unpredictable.
>
> To prevent bisect debugging failures, this stays in a single commit.
>
> For example: reclaim without compaction will OOM on high-order
> allocation failure despite a large amount of free memory. Normally
> compaction would (potentially) resolve this issue.
>
> If a private node opts into reclaim, we create normal watermarks for
> that node - otherwise pgdat_balanced() is always true and reclaim
> thinks there is no work to do.
>
> Cross-node operations (demotion, promotion, khugepaged) are NOT
> included in CAP_RECLAIM because some workflows may desire different
> behaviors for this kind of operation:
>
> - prefer direct-to-swap, do not demote
> - remain resident and OOM
> - __GFP_THISNODE: fail and let the driver augment reclaim
>
> Normal swap-out is allowed because there are clear userland controls
> (not registering swap, cgroup.swap, etc) to control that per-workload.
>
> Signed-off-by: Gregory Price <gourry@xxxxxxxxxx>
> ---
> include/linux/node_private.h | 33 +++++++++++++++++++++++++
> mm/compaction.c | 9 ++++---
> mm/damon/paddr.c | 4 +--
> mm/huge_memory.c | 2 +-
> mm/internal.h | 13 ++++++++++
> mm/madvise.c | 6 ++---
> mm/memory_hotplug.c | 3 +--
> mm/page_alloc.c | 2 +-
> mm/vmscan.c | 48 ++++++++++++++++++++++++++++++------
> 9 files changed, 100 insertions(+), 20 deletions(-)
>
> diff --git a/include/linux/node_private.h b/include/linux/node_private.h
> index 475496c84249f..f7cbae1309904 100644
> --- a/include/linux/node_private.h
> +++ b/include/linux/node_private.h
> @@ -7,6 +7,13 @@
>
> struct page;
>
> +/*
> + * Per-node service opt-ins (node_private.caps). A private node is isolated
> + * from all general mm services by default; the registering driver sets these
> + * to let specific services operate on its node.
> + */
> +#define NODE_PRIVATE_CAP_RECLAIM (1UL << 0) /* allow mm reclaim */
> +
> /**
> * struct node_private - Per-node container for N_MEMORY_PRIVATE nodes
> *
> @@ -41,6 +48,27 @@ static inline bool node_is_private(int nid)
> return node_state(nid, N_MEMORY_PRIVATE);
> }
>
> +/**
> + * node_allows_reclaim - may the mm reclaim from this node?
> + * @nid: the node to test
> + *
> + * Only a private node is ever excluded. Every other node can safely
> + * be operated on by reclaim.
> + */
> +static inline bool node_allows_reclaim(int nid)
> +{
> + struct node_private *np;
> + bool ret;
> +
> + if (!node_state(nid, N_MEMORY_PRIVATE))
> + return true;
> + rcu_read_lock();
> + np = rcu_dereference(NODE_DATA(nid)->node_private);
> + ret = np && (np->caps & NODE_PRIVATE_CAP_RECLAIM);
> + rcu_read_unlock();
> + return ret;
> +}
> +
> #else /* !CONFIG_NUMA */
>
> static inline bool folio_is_private_node(struct folio *folio)
> @@ -58,6 +86,11 @@ static inline bool node_is_private(int nid)
> return false;
> }
>
> +static inline bool node_allows_reclaim(int nid)
> +{
> + return true;
> +}
> +
> #endif /* CONFIG_NUMA */
>
> #if defined(CONFIG_NUMA) && defined(CONFIG_MEMORY_HOTPLUG)
> diff --git a/mm/compaction.c b/mm/compaction.c
> index 8c1351cce7bcc..b9c25c599732f 100644
> --- a/mm/compaction.c
> +++ b/mm/compaction.c
> @@ -25,6 +25,7 @@
> #include <linux/psi.h>
> #include <linux/cpuset.h>
> #include "page_alloc.h"
> +#include <linux/node_private.h>
> #include "internal.h"
>
> #ifdef CONFIG_COMPACTION
> @@ -2462,7 +2463,7 @@ bool compaction_zonelist_suitable(struct alloc_context *ac, int order,
> !__cpuset_zone_allowed(zone, gfp_mask))
> continue;
>
> - if (node_is_private(zone_to_nid(zone)))
> + if (!node_allows_reclaim(zone_to_nid(zone)))
> continue;
>
> /*
> @@ -2856,7 +2857,7 @@ enum compact_result try_to_compact_pages(gfp_t gfp_mask, unsigned int order,
> !__cpuset_zone_allowed(zone, gfp_mask))
> continue;
>
> - if (node_is_private(zone_to_nid(zone)))
> + if (!node_allows_reclaim(zone_to_nid(zone)))
> continue;
>
> if (prio > MIN_COMPACT_PRIORITY
> @@ -2928,7 +2929,7 @@ static int compact_node(pg_data_t *pgdat, bool proactive)
> .proactive_compaction = proactive,
> };
>
> - if (node_is_private(pgdat->node_id))
> + if (!node_allows_reclaim(pgdat->node_id))
> return 0;
>
> for (zoneid = 0; zoneid < MAX_NR_ZONES; zoneid++) {
> @@ -3026,7 +3027,7 @@ static ssize_t compact_store(struct device *dev,
> {
> int nid = dev->id;
>
> - if (node_is_private(nid))
> + if (!node_allows_reclaim(nid))
> return -EINVAL;
>
> if (nid >= 0 && nid < nr_node_ids && node_online(nid)) {
> diff --git a/mm/damon/paddr.c b/mm/damon/paddr.c
> index c741a94319750..b668cf55d1b67 100644
> --- a/mm/damon/paddr.c
> +++ b/mm/damon/paddr.c
> @@ -251,8 +251,8 @@ static unsigned long damon_pa_pageout(struct damon_region *r,
> continue;
> }
>
> - /* private node memory is not reclaimable by default */
> - if (folio_is_private_node(folio))
> + /* DAMOS pageout is reclaim; gate a private node on CAP_RECLAIM */
> + if (!node_allows_reclaim(folio_nid(folio)))
> goto put_folio;
>
> if (damos_pa_filter_out(s, folio))
> diff --git a/mm/huge_memory.c b/mm/huge_memory.c
> index 1df91b4e5c2bc..cebc89eb0541f 100644
> --- a/mm/huge_memory.c
> +++ b/mm/huge_memory.c
> @@ -2340,7 +2340,7 @@ bool madvise_free_huge_pmd(struct mmu_gather *tlb, struct vm_area_struct *vma,
>
> folio = pmd_folio(orig_pmd);
>
> - if (folio_is_private_node(folio))
> + if (!node_allows_reclaim(folio_nid(folio)))
> goto out;
>
> /*
> diff --git a/mm/internal.h b/mm/internal.h
> index 85c460296cea1..8329034ae561f 100644
> --- a/mm/internal.h
> +++ b/mm/internal.h
> @@ -110,6 +110,19 @@ static inline bool page_is_private_managed(struct page *page)
> return folio_is_private_managed(page_folio(page));
> }
>
> +/*
> + * folio_allows_madvise() - may madvise reclaim hints act on this folio?
> + *
> + * madvise reclaim hints (COLD/PAGEOUT/FREE) are userland-driven reclaim, so
> + * they follow reclaim opt-in: false for ZONE_DEVICE and for N_MEMORY_PRIVATE
> + * nodes without CAP_RECLAIM, true for all other normal folios.
> + */
> +static inline bool folio_allows_madvise(struct folio *folio)
> +{
> + return !folio_is_zone_device(folio) &&
> + node_allows_reclaim(folio_nid(folio));
> +}
> +
> /*
> * folio_allows_longterm_pin() - may this folio be long-term GUP-pinned?
> *
> diff --git a/mm/madvise.c b/mm/madvise.c
> index 29f35a23919a0..56ca974542707 100644
> --- a/mm/madvise.c
> +++ b/mm/madvise.c
> @@ -396,7 +396,7 @@ static int madvise_cold_or_pageout_pte_range(pmd_t *pmd,
>
> folio = pmd_folio(orig_pmd);
>
> - if (folio_is_private_node(folio))
> + if (!node_allows_reclaim(folio_nid(folio)))
> goto huge_unlock;
>
> /* Do not interfere with other mappings of this folio */
> @@ -478,7 +478,7 @@ static int madvise_cold_or_pageout_pte_range(pmd_t *pmd,
> continue;
>
> folio = vm_normal_folio(vma, addr, ptent);
> - if (!folio || folio_is_private_managed(folio))
> + if (!folio || !folio_allows_madvise(folio))
> continue;
>
> /*
> @@ -707,7 +707,7 @@ static int madvise_free_pte_range(pmd_t *pmd, unsigned long addr,
> }
>
> folio = vm_normal_folio(vma, addr, ptent);
> - if (!folio || folio_is_private_managed(folio))
> + if (!folio || !folio_allows_madvise(folio))
> continue;
>
> /*
> diff --git a/mm/memory_hotplug.c b/mm/memory_hotplug.c
> index be230ac9efe5a..1f42ed303366c 100644
> --- a/mm/memory_hotplug.c
> +++ b/mm/memory_hotplug.c
> @@ -1237,8 +1237,7 @@ int online_pages(unsigned long pfn, unsigned long nr_pages,
> /* reinitialise watermarks and update pcp limits */
> init_per_zone_wmark_min();
>
> - /* Private nodes opt-out of reclaim/compaction by default */
> - if (!node_is_private(nid)) {
> + if (node_allows_reclaim(nid)) {
> kswapd_run(nid);
> kcompactd_run(nid);
> }
> diff --git a/mm/page_alloc.c b/mm/page_alloc.c
> index 2b08bea2379a9..2667a4564b7ac 100644
> --- a/mm/page_alloc.c
> +++ b/mm/page_alloc.c
> @@ -6667,7 +6667,7 @@ static void __setup_per_zone_wmarks(void)
> u64 tmp;
>
> spin_lock_irqsave(&zone->lock, flags);
> - if (node_is_private(zone_to_nid(zone))) {
> + if (!node_allows_reclaim(zone_to_nid(zone))) {
> zone->_watermark[WMARK_MIN] = 0;
> zone->_watermark[WMARK_LOW] = 0;
> zone->_watermark[WMARK_HIGH] = 0;
> diff --git a/mm/vmscan.c b/mm/vmscan.c
> index 86b2334c23b98..f1722693ac2db 100644
> --- a/mm/vmscan.c
> +++ b/mm/vmscan.c
> @@ -5396,6 +5396,21 @@ static const struct attribute_group lru_gen_attr_group = {
> * debugfs interface
> ******************************************************************************/
>
> +/*
> + * Nodes the lru_gen debugfs interface lists: ordinary memory nodes plus any
> + * N_MEMORY_PRIVATE nodes opted into reclaim. run_cmd() already accepts the
> + * latter, so keep the listing in sync with what it accepts.
> + */
> +static void lru_gen_seq_nodes(nodemask_t *nodes)
> +{
> + int nid;
> +
> + *nodes = node_states[N_MEMORY];
> + for_each_node_state(nid, N_MEMORY_PRIVATE)
> + if (node_allows_reclaim(nid))
> + node_set(nid, *nodes);
> +}
> +
> static void *lru_gen_seq_start(struct seq_file *m, loff_t *pos)
> {
> struct mem_cgroup *memcg;
> @@ -5407,9 +5422,11 @@ static void *lru_gen_seq_start(struct seq_file *m, loff_t *pos)
>
> memcg = mem_cgroup_iter(NULL, NULL, NULL);
> do {
> + nodemask_t nodes;
> int nid;
>
> - for_each_node_state(nid, N_MEMORY) {
> + lru_gen_seq_nodes(&nodes);
> + for_each_node_mask(nid, nodes) {
> if (!nr_to_skip--)
> return get_lruvec(memcg, nid);
> }
> @@ -5431,16 +5448,18 @@ static void *lru_gen_seq_next(struct seq_file *m, void *v, loff_t *pos)
> {
> int nid = lruvec_pgdat(v)->node_id;
> struct mem_cgroup *memcg = lruvec_memcg(v);
> + nodemask_t nodes;
>
> ++*pos;
>
> - nid = next_memory_node(nid);
> + lru_gen_seq_nodes(&nodes);
> + nid = next_node(nid, nodes);
> if (nid == MAX_NUMNODES) {
> memcg = mem_cgroup_iter(NULL, memcg, NULL);
> if (!memcg)
> return NULL;
>
> - nid = first_memory_node;
> + nid = first_node(nodes);
> }
>
> return get_lruvec(memcg, nid);
> @@ -5509,10 +5528,12 @@ static int lru_gen_seq_show(struct seq_file *m, void *v)
> struct lru_gen_folio *lrugen = &lruvec->lrugen;
> int nid = lruvec_pgdat(lruvec)->node_id;
> struct mem_cgroup *memcg = lruvec_memcg(lruvec);
> + nodemask_t nodes;
> DEFINE_MAX_SEQ(lruvec);
> DEFINE_MIN_SEQ(lruvec);
>
> - if (nid == first_memory_node) {
> + lru_gen_seq_nodes(&nodes);
> + if (nid == first_node(nodes)) {
> const char *path = memcg ? m->private : "";
>
> #ifdef CONFIG_MEMCG
> @@ -5612,7 +5633,9 @@ static int run_cmd(char cmd, u64 memcg_id, int nid, unsigned long seq,
> int err = -EINVAL;
> struct mem_cgroup *memcg = NULL;
>
> - if (nid < 0 || nid >= MAX_NUMNODES || !node_state(nid, N_MEMORY))
> + if (nid < 0 || nid >= MAX_NUMNODES ||
> + !(node_state(nid, N_MEMORY) ||
> + (node_is_private(nid) && node_allows_reclaim(nid))))
> return -EINVAL;
>
> if (!mem_cgroup_disabled()) {
> @@ -6145,7 +6168,7 @@ static void shrink_node(pg_data_t *pgdat, struct scan_control *sc)
> * Private nodes do not support reclaim by default, filtering here
> * captures all normal reclaim paths that may attempt eviction.
> */
> - if (node_is_private(pgdat->node_id))
> + if (!node_allows_reclaim(pgdat->node_id))
> return;
>
> if ((lru_gen_enabled() || lru_gen_switching()) && root_reclaim(sc)) {
> @@ -6758,6 +6781,16 @@ unsigned long mem_cgroup_shrink_node(struct mem_cgroup *memcg,
> return sc.nr_reclaimed;
> }
>
> +static struct zonelist *memcg_reclaim_zonelist(int nid, gfp_t gfp_mask)
> +{
> + unsigned int aflags = ALLOC_DEFAULT;
> +
> + if (unlikely(!nodes_empty(node_states[N_MEMORY_PRIVATE])))
> + aflags = ALLOC_ZONELIST_PRIVATE;
> +
> + return select_zonelist(nid, gfp_mask, aflags);
> +}
I applied this code change and notice in mem_cgroup_shrink_node(), it calls
shrink_lruvec() directly and never passes through shrink_node().
This patch makes private nodes reachable through memcg reclaim zonelist, but
what prevents the direct per-node memcg path from reclaiming a private node
that has not set NODE_PRIVATE_CAP_RECLAIM ?
--Richard
> +
> unsigned long try_to_free_mem_cgroup_pages(struct mem_cgroup *memcg,
> unsigned long nr_pages,
> gfp_t gfp_mask,
> @@ -6784,7 +6817,8 @@ unsigned long try_to_free_mem_cgroup_pages(struct mem_cgroup *memcg,
> * equal pressure on all the nodes. This is based on the assumption that
> * the reclaim does not bail out early.
> */
> - struct zonelist *zonelist = node_zonelist(numa_node_id(), sc.gfp_mask);
> + struct zonelist *zonelist = memcg_reclaim_zonelist(numa_node_id(),
> + sc.gfp_mask);
>
> set_task_reclaim_state(current, &sc.reclaim_state);
> trace_mm_vmscan_memcg_reclaim_begin(sc.gfp_mask, 0, memcg);
> --
> 2.53.0-Meta
>
>