[PATCH v4 09/13] mm/collapse: separate scanning a PTE table from collapsing it
From: Kiryl Shutsemau
Date: Mon Sep 28 2026 - 06:11:36 EST
From: "Kiryl Shutsemau (Meta)" <kas@xxxxxxxxxx>
A collapse is two jobs. One reads a PTE table under mmap_lock and decides
whether the range is worth collapsing. The other allocates, isolates,
copies and flushes, and wants the lock given up first.
collapse_single_pmd() did both, so the boundary between them was somewhere
in the middle of a function.
Give each half its own function:
- collapse_scan_pmd() scans one table. The anonymous scan that used to
carry that name keeps its body as collapse_scan_anon_pmd(), and
collapse_scan_pmd() is now the entry that picks the anonymous or the
file side.
- collapse_run_pmd() does the collapse the scan asked for, and is
handed what the scan returned. SCAN_SUCCEED means there is
something to collapse. SCAN_PTE_MAPPED_HUGEPAGE means the page
cache already holds the PMD folio and only the PTE table is left to
retract. Both are work for the run; anything else is why there is
nothing to do.
collapse_single_pmd() is now the two of them with the mmap_lock drop in
between, so its callers see what they saw before. collapse_control_init()
sets a control up before its first scan.
What the scan found and the run needs travels in collapse_control. For
an anonymous table that is the orders and the referenced and swapped-out
counts, which mthp_collapse() and collapse_huge_page() now read from
there instead of taking as arguments. For a file it is the file itself
and the offset in it: a file collapse works on the page cache and never
sees a VMA, so the scan takes the reference while it still has one and
the run gives it back.
The file scan moves under mmap_lock with the anonymous one, where before
the lock was given up first. The lock is now held over the page cache
walk, an RCU walk over one table's worth of slots with no PTL, and taken
fewer times. collapse_scan_mm_slot() ends its walk whenever the lock was
dropped, so a refused file table used to cost khugepaged an unlock, a
trip back through khugepaged_do_scan(), a relock and a VMA lookup. Now
only a table that goes on to be collapsed does.
Assisted-by: LLM
Signed-off-by: Kiryl Shutsemau (Meta) <kas@xxxxxxxxxx>
---
mm/collapse.h | 14 +++++
mm/khugepaged.c | 141 ++++++++++++++++++++++++++++++++----------------
2 files changed, 108 insertions(+), 47 deletions(-)
diff --git a/mm/collapse.h b/mm/collapse.h
index dcd117071955..ca7b367c89cb 100644
--- a/mm/collapse.h
+++ b/mm/collapse.h
@@ -98,6 +98,20 @@ struct collapse_control {
/* Each bit marks a PTE the scan accepted as a collapse source */
DECLARE_BITMAP(eligible_ptes, MAX_PTRS_PER_PTE);
+
+ /*
+ * What a scan found and the run after it needs. Live only between the
+ * two, and read by nobody else.
+ *
+ * The file side takes a reference while it still has the VMA, since a
+ * file collapse works on the page cache and never sees one; the run is
+ * what gives it back.
+ */
+ unsigned long scan_orders;
+ int scan_referenced;
+ int scan_unmapped;
+ struct file *scan_file;
+ pgoff_t scan_pgoff;
};
#endif /* __MM_COLLAPSE_H */
diff --git a/mm/khugepaged.c b/mm/khugepaged.c
index 89a4c3f5a91c..2b044e63d8c9 100644
--- a/mm/khugepaged.c
+++ b/mm/khugepaged.c
@@ -1235,8 +1235,8 @@ static enum scan_result alloc_charge_folio(struct folio **foliop, struct mm_stru
* while allocating a THP, as that could trigger direct reclaim/compaction.
* Note that the VMA must be rechecked after grabbing the mmap_lock again.
*/
-static enum scan_result collapse_huge_page(struct mm_struct *mm, unsigned long start_addr,
- int referenced, int unmapped, struct collapse_control *cc,
+static enum scan_result collapse_huge_page(struct mm_struct *mm,
+ unsigned long start_addr, struct collapse_control *cc,
unsigned int order)
{
const unsigned long pmd_addr = start_addr & HPAGE_PMD_MASK;
@@ -1275,14 +1275,14 @@ static enum scan_result collapse_huge_page(struct mm_struct *mm, unsigned long s
goto out_nolock;
}
- if (unmapped) {
+ if (cc->scan_unmapped) {
/*
* __collapse_huge_page_swapin() will return with mmap_lock
* released when it fails. So we jump out_nolock directly in
* that case. Continuing to collapse causes inconsistency.
*/
result = __collapse_huge_page_swapin(mm, vma, start_addr, pmd,
- referenced, order);
+ cc->scan_referenced, order);
if (result != SCAN_SUCCEED)
goto out_nolock;
}
@@ -1446,9 +1446,8 @@ static unsigned int max_order_from_offset(unsigned int offset)
* If a collapse is permitted, we attempt to collapse the PTE range into a
* mTHP.
*/
-static enum scan_result mthp_collapse(struct mm_struct *mm,
- unsigned long address, int referenced, int unmapped,
- struct collapse_control *cc, unsigned long enabled_orders)
+static enum scan_result mthp_collapse(struct mm_struct *mm, unsigned long address,
+ struct collapse_control *cc)
{
unsigned int nr_eligible_ptes, nr_ptes, max_ptes_none;
enum scan_result last_result = SCAN_FAIL;
@@ -1461,7 +1460,7 @@ static enum scan_result mthp_collapse(struct mm_struct *mm,
while (offset < HPAGE_PMD_NR) {
nr_ptes = 1UL << order;
- if (!test_bit(order, &enabled_orders))
+ if (!test_bit(order, &cc->scan_orders))
goto next_order;
max_ptes_none = collapse_max_ptes_none(cc, NULL, order);
@@ -1469,19 +1468,18 @@ static enum scan_result mthp_collapse(struct mm_struct *mm,
offset + nr_ptes);
/*
- * Swap PTEs accepted during the scan are counted in @unmapped,
- * not in cc->eligible_ptes. Account them for the PMD-order
- * candidate.
+ * Swap PTEs accepted during the scan are counted in
+ * cc->scan_unmapped, not in cc->eligible_ptes. Account them for
+ * the PMD-order candidate.
*/
if (is_pmd_order(order))
- nr_eligible_ptes += unmapped;
+ nr_eligible_ptes += cc->scan_unmapped;
if (nr_eligible_ptes >= nr_ptes - max_ptes_none) {
enum scan_result ret;
collapse_address = address + offset * PAGE_SIZE;
- ret = collapse_huge_page(mm, collapse_address, referenced,
- unmapped, cc, order);
+ ret = collapse_huge_page(mm, collapse_address, cc, order);
switch (ret) {
/* Cases where we continue to next collapse candidate */
@@ -1523,7 +1521,7 @@ static enum scan_result mthp_collapse(struct mm_struct *mm,
* we must always move to the next offset.
*/
if (order > COLLAPSE_MIN_MTHP_ORDER &&
- (enabled_orders & GENMASK(order - 1, 0))) {
+ (cc->scan_orders & GENMASK(order - 1, 0))) {
order--;
continue;
}
@@ -1548,14 +1546,14 @@ static enum scan_result mthp_collapse(struct mm_struct *mm,
return last_result;
}
-static enum scan_result collapse_scan_pmd(struct mm_struct *mm,
- struct vm_area_struct *vma, unsigned long start_addr,
- bool *lock_dropped, struct collapse_control *cc)
+static enum scan_result collapse_scan_anon_pmd(struct vm_area_struct *vma,
+ unsigned long start_addr, struct collapse_control *cc)
{
const unsigned int max_ptes_shared = collapse_max_ptes_shared(cc, HPAGE_PMD_ORDER);
const unsigned int max_ptes_swap = collapse_max_ptes_swap(cc, HPAGE_PMD_ORDER);
unsigned int max_ptes_none = collapse_max_ptes_none(cc, vma, HPAGE_PMD_ORDER);
enum tva_type tva_flags = cc->policy.tva_type;
+ struct mm_struct *mm = vma->vm_mm;
pmd_t *pmd;
pte_t *pte, *_pte, pteval;
int i;
@@ -1735,12 +1733,9 @@ static enum scan_result collapse_scan_pmd(struct mm_struct *mm,
out_unmap:
pte_unmap_unlock(pte, ptl);
if (result == SCAN_SUCCEED) {
- /* collapse_huge_page() expects the lock to be dropped before calling */
- mmap_read_unlock(mm);
- result = mthp_collapse(mm, start_addr, referenced,
- unmapped, cc, enabled_orders);
- /* mmap_lock was released above, set lock_dropped */
- *lock_dropped = true;
+ cc->scan_orders = enabled_orders;
+ cc->scan_referenced = referenced;
+ cc->scan_unmapped = unmapped;
}
out:
trace_mm_khugepaged_scan_pmd(mm, failed_pfn, referenced,
@@ -2742,41 +2737,67 @@ static enum scan_result collapse_scan_file(struct mm_struct *mm,
count_vm_event(THP_SCAN_EXCEED_NONE_PTE);
}
- trace_mm_khugepaged_scan_file(mm, failed_pfn, file, present, swap, result);
+ trace_mm_khugepaged_scan_file(mm, failed_pfn, file, present, swap,
+ result);
return result;
}
-/*
- * Try to collapse a single PMD starting at a PMD aligned addr, and return
- * the results.
- */
-static enum scan_result collapse_single_pmd(unsigned long addr,
- struct vm_area_struct *vma, bool *lock_dropped,
- struct collapse_control *cc)
+static void collapse_control_init(struct collapse_control *cc)
+{
+ cc->progress = 0;
+ cc->scan_file = NULL;
+}
+
+static enum scan_result collapse_scan_pmd(struct vm_area_struct *vma,
+ unsigned long addr, struct collapse_control *cc)
{
- struct mm_struct *mm = vma->vm_mm;
- bool triggered_wb = false;
enum scan_result result;
- struct file *file;
pgoff_t pgoff;
- mmap_assert_locked(mm);
+ mmap_assert_locked(vma->vm_mm);
+ /* Whatever the last scan found has to have been run by now */
+ if (WARN_ON_ONCE(cc->scan_file)) {
+ fput(cc->scan_file);
+ cc->scan_file = NULL;
+ }
if (vma_is_anonymous(vma))
- return collapse_scan_pmd(mm, vma, addr, lock_dropped, cc);
+ return collapse_scan_anon_pmd(vma, addr, cc);
- file = get_file(vma->vm_file);
pgoff = linear_page_index(vma, addr);
-
- mmap_read_unlock(mm);
- *lock_dropped = true;
-
+ result = collapse_scan_file(vma->vm_mm, addr, vma->vm_file, pgoff, cc);
/*
* SCAN_PTE_MAPPED_HUGEPAGE is work too: the page cache already holds
- * the PMD folio, and only the PTE table is left to retract.
+ * the PMD folio, and retracting the PTE table is the run's job.
*/
- result = collapse_scan_file(mm, addr, file, pgoff, cc);
- if (result != SCAN_SUCCEED)
+ if (result != SCAN_SUCCEED && result != SCAN_PTE_MAPPED_HUGEPAGE)
+ return result;
+
+ /*
+ * A file collapse works on the page cache and never sees a VMA, so take
+ * what it needs from this one while it is still here.
+ */
+ cc->scan_file = get_file(vma->vm_file);
+ cc->scan_pgoff = pgoff;
+ return result;
+}
+
+static enum scan_result collapse_run_pmd(struct mm_struct *mm,
+ unsigned long addr, enum scan_result result,
+ struct collapse_control *cc)
+{
+ struct file *file = cc->scan_file;
+ bool triggered_wb = false;
+ pgoff_t pgoff;
+
+ if (!file)
+ return mthp_collapse(mm, addr, cc);
+
+ cc->scan_file = NULL;
+ pgoff = cc->scan_pgoff;
+
+ /* The scan found the PMD folio in place: nothing to collapse */
+ if (result == SCAN_PTE_MAPPED_HUGEPAGE)
goto put;
retry:
result = collapse_file(mm, addr, file, pgoff, cc);
@@ -2794,6 +2815,10 @@ static enum scan_result collapse_single_pmd(unsigned long addr,
put:
fput(file);
+ /*
+ * A PMD folio is in the page cache, whether the collapse just put it
+ * there or found it: retract the PTE table, and map the PMD if asked.
+ */
if (result == SCAN_PTE_MAPPED_HUGEPAGE) {
mmap_read_lock(mm);
if (collapse_test_exit_or_disable(mm))
@@ -2808,6 +2833,28 @@ static enum scan_result collapse_single_pmd(unsigned long addr,
return result;
}
+/*
+ * Try to collapse a single PMD starting at a PMD aligned addr, and return
+ * the results.
+ */
+static enum scan_result collapse_single_pmd(unsigned long addr,
+ struct vm_area_struct *vma, bool *lock_dropped,
+ struct collapse_control *cc)
+{
+ struct mm_struct *mm = vma->vm_mm;
+ enum scan_result result;
+
+ result = collapse_scan_pmd(vma, addr, cc);
+ if (result != SCAN_SUCCEED && result != SCAN_PTE_MAPPED_HUGEPAGE)
+ return result;
+
+ /* The collapse takes its own locks, so give this up */
+ mmap_read_unlock(mm);
+ *lock_dropped = true;
+
+ return collapse_run_pmd(mm, addr, result, cc);
+}
+
static void collapse_scan_mm_slot(unsigned int progress_max,
enum scan_result *result, struct collapse_control *cc)
__releases(&khugepaged_mm_lock)
@@ -2950,9 +2997,9 @@ static void khugepaged_do_scan(struct collapse_control *cc)
lru_add_drain_all();
+ collapse_control_init(cc);
collapse_policy_khugepaged(&cc->policy);
- cc->progress = 0;
while (true) {
cond_resched();
@@ -3179,8 +3226,8 @@ int madvise_collapse(struct vm_area_struct *vma, unsigned long start,
cc = kmalloc_obj(*cc);
if (!cc)
return -ENOMEM;
+ collapse_control_init(cc);
collapse_policy_madvise(&cc->policy);
- cc->progress = 0;
lru_add_drain_all();
--
2.54.0