[PATCH 06/25] mm/fbatch: fbatch_drain_lazyfree(onstack fbatch) before ptl unlock

From: Hugh Dickins

Date: Mon Aug 24 2026 - 10:08:15 EST


Re-enable lazyfree batching for MADV_FREE. But it's not safe now to leave
potentially stale (then reused) folios in a per-cpu fbatch for lazyfree.
Instead, madvise_free_pte_range() keep an fbatch on its stack, and drain
it each time before dropping pagetable lock, while the folios are secure.

Ignore folio_may_be_lru_cached() and lru_cache_disabled(): limitations
irrelevant to this fbatch drained under spinlock (even if RT); though
in practice madvise_free_huge_pmd() does have to drain every time.

Signed-off-by: Hugh Dickins <hughd@xxxxxxxxxx>
---
include/linux/huge_mm.h | 6 ++++--
mm/folio.c | 34 +++++++++++++++++++++-------------
mm/huge_memory.c | 6 ++++--
mm/internal.h | 3 ++-
mm/madvise.c | 9 +++++++--
5 files changed, 38 insertions(+), 20 deletions(-)

diff --git a/include/linux/huge_mm.h b/include/linux/huge_mm.h
index c745f7ad2298..d50906327d1d 100644
--- a/include/linux/huge_mm.h
+++ b/include/linux/huge_mm.h
@@ -24,9 +24,11 @@ static inline void huge_pud_set_accessed(struct vm_fault *vmf, pud_t orig_pud)
}
#endif

-vm_fault_t do_huge_pmd_wp_page(struct vm_fault *vmf);
+struct folio_batch;
bool madvise_free_huge_pmd(struct mmu_gather *tlb, struct vm_area_struct *vma,
- pmd_t *pmd, unsigned long addr, unsigned long next);
+ pmd_t *pmd, unsigned long addr, unsigned long next,
+ struct folio_batch *fbatch);
+vm_fault_t do_huge_pmd_wp_page(struct vm_fault *vmf);
bool zap_huge_pmd(struct mmu_gather *tlb, struct vm_area_struct *vma, pmd_t *pmd,
unsigned long addr);
int zap_huge_pud(struct mmu_gather *tlb, struct vm_area_struct *vma, pud_t *pud,
diff --git a/mm/folio.c b/mm/folio.c
index 88e3ebd7e652..e76868c95acc 100644
--- a/mm/folio.c
+++ b/mm/folio.c
@@ -50,7 +50,6 @@ struct cpu_fbatches {
struct folio_batch lru_activate;
struct folio_batch lru_deactivate_file;
struct folio_batch lru_deactivate;
- struct folio_batch lru_lazyfree;
/* Protecting the following batches which require disabling interrupts */
local_lock_t lock_irq;
struct folio_batch lru_move_tail;
@@ -193,8 +192,6 @@ static void __folio_batch_add_and_move(struct folio_batch __percpu *fbatch,
local_lock(&cpu_fbatches.lock);

if (!folio_batch_add(this_cpu_ptr(fbatch), folio) ||
- /* XXX Temporarily disable lazyfree batching */
- fbatch == &cpu_fbatches.lru_lazyfree ||
!folio_may_be_lru_cached(folio) || lru_cache_disabled())
folio_batch_move_lru(this_cpu_ptr(fbatch), move_fn);

@@ -651,10 +648,6 @@ void lru_add_drain_cpu(int cpu)
fbatch = &fbatches->lru_deactivate;
if (folio_batch_count(fbatch))
folio_batch_move_lru(fbatch, lru_deactivate);
-
- fbatch = &fbatches->lru_lazyfree;
- if (folio_batch_count(fbatch))
- folio_batch_move_lru(fbatch, lru_lazyfree);
}

/**
@@ -700,19 +693,35 @@ void folio_deactivate(struct folio *folio)

/**
* folio_mark_lazyfree - make an anon folio lazyfree
- * @folio: folio to deactivate
+ * @fbatch: batch to which folio will be added
+ * @folio: folio to be lazily freed
*
- * folio_mark_lazyfree() moves @folio to the inactive file list.
- * This is done to accelerate the reclaim of @folio.
+ * folio_mark_lazyfree() moves @folio to the inactive file list
+ * via @fbatch. This is done to accelerate the reclaim of @folio.
*/
-void folio_mark_lazyfree(struct folio *folio)
+void folio_mark_lazyfree(struct folio_batch *fbatch, struct folio *folio)
{
if (!folio_test_anon(folio) || !folio_test_swapbacked(folio) ||
!folio_test_lru(folio) ||
folio_test_swapcache(folio) || folio_test_unevictable(folio))
return;

- folio_batch_add_and_move(folio, lru_lazyfree);
+ if (!folio_batch_add(fbatch, folio))
+ folio_batch_move_lru(fbatch, lru_lazyfree);
+}
+
+/**
+ * fbatch_drain_lazyfree - drain the caller's folio batch
+ * @fbatch: batch of folios to be lazily freed
+ *
+ * Must be called before caller drops the page table lock: that is,
+ * before dropping the last certain reference to the folios in @fbatch.
+ * It would be very bad to lazyfree a folio after it was freed and reused.
+ */
+void fbatch_drain_lazyfree(struct folio_batch *fbatch)
+{
+ if (folio_batch_count(fbatch))
+ folio_batch_move_lru(fbatch, lru_lazyfree);
}

void lru_add_drain(void)
@@ -766,7 +775,6 @@ static bool cpu_needs_drain(unsigned int cpu)
folio_batch_count(&fbatches->lru_move_tail) ||
folio_batch_count(&fbatches->lru_deactivate_file) ||
folio_batch_count(&fbatches->lru_deactivate) ||
- folio_batch_count(&fbatches->lru_lazyfree) ||
need_mlock_drain(cpu)) ||
has_bh_in_lru(cpu, NULL);
}
diff --git a/mm/huge_memory.c b/mm/huge_memory.c
index 98b1d0ea50f0..b1f315400111 100644
--- a/mm/huge_memory.c
+++ b/mm/huge_memory.c
@@ -2356,7 +2356,8 @@ vm_fault_t do_huge_pmd_numa_page(struct vm_fault *vmf)
* Otherwise, return false.
*/
bool madvise_free_huge_pmd(struct mmu_gather *tlb, struct vm_area_struct *vma,
- pmd_t *pmd, unsigned long addr, unsigned long next)
+ pmd_t *pmd, unsigned long addr, unsigned long next,
+ struct folio_batch *fbatch)
{
spinlock_t *ptl;
pmd_t orig_pmd;
@@ -2417,7 +2418,8 @@ bool madvise_free_huge_pmd(struct mmu_gather *tlb, struct vm_area_struct *vma,
tlb_remove_pmd_tlb_entry(tlb, pmd, addr);
}

- folio_mark_lazyfree(folio);
+ folio_mark_lazyfree(fbatch, folio);
+ fbatch_drain_lazyfree(fbatch);
ret = true;
out:
spin_unlock(ptl);
diff --git a/mm/internal.h b/mm/internal.h
index 68db5abd0a4c..ababee1a8872 100644
--- a/mm/internal.h
+++ b/mm/internal.h
@@ -66,7 +66,8 @@ void lru_add_drain(void);
void lru_add_drain_cpu(int cpu);
void lru_add_drain_cpu_zone(struct zone *zone);
void folio_deactivate(struct folio *folio);
-void folio_mark_lazyfree(struct folio *folio);
+void folio_mark_lazyfree(struct folio_batch *fbatch, struct folio *folio);
+void fbatch_drain_lazyfree(struct folio_batch *fbatch);

/* mm/vmscan.c */
unsigned long zone_reclaimable_pages(struct zone *zone);
diff --git a/mm/madvise.c b/mm/madvise.c
index 240d9161ee74..6ef1f489123c 100644
--- a/mm/madvise.c
+++ b/mm/madvise.c
@@ -27,6 +27,7 @@
#include <linux/file.h>
#include <linux/blk_plug.h>
#include <linux/backing-dev.h>
+#include <linux/folio_batch.h>
#include <linux/pagewalk.h>
#include <linux/swap.h>
#include <linux/leafops.h>
@@ -657,6 +658,7 @@ static int madvise_free_pte_range(pmd_t *pmd, unsigned long addr,
struct mmu_gather *tlb = walk->private;
struct mm_struct *mm = tlb->mm;
struct vm_area_struct *vma = walk->vma;
+ struct folio_batch fbatch;
spinlock_t *ptl;
pte_t *start_pte, *pte, ptent;
struct folio *folio;
@@ -664,9 +666,10 @@ static int madvise_free_pte_range(pmd_t *pmd, unsigned long addr,
unsigned long next;
int nr, max_nr;

+ folio_batch_init(&fbatch);
next = pmd_addr_end(addr, end);
if (pmd_trans_huge(*pmd))
- if (madvise_free_huge_pmd(tlb, vma, pmd, addr, next))
+ if (madvise_free_huge_pmd(tlb, vma, pmd, addr, next, &fbatch))
return 0;

tlb_change_page_size(tlb, PAGE_SIZE);
@@ -724,6 +727,7 @@ static int madvise_free_pte_range(pmd_t *pmd, unsigned long addr,
continue;
folio_get(folio);
lazy_mmu_mode_disable();
+ fbatch_drain_lazyfree(&fbatch);
pte_unmap_unlock(start_pte, ptl);
start_pte = NULL;
err = split_folio(folio);
@@ -768,13 +772,14 @@ static int madvise_free_pte_range(pmd_t *pmd, unsigned long addr,
clear_young_dirty_ptes(vma, addr, pte, nr, cydp_flags);
tlb_remove_tlb_entries(tlb, pte, nr, addr);
}
- folio_mark_lazyfree(folio);
+ folio_mark_lazyfree(&fbatch, folio);
}

if (nr_swap)
add_mm_counter(mm, MM_SWAPENTS, nr_swap);
if (start_pte) {
lazy_mmu_mode_disable();
+ fbatch_drain_lazyfree(&fbatch);
pte_unmap_unlock(start_pte, ptl);
}
cond_resched();
--
2.51.0