Re: [PATCH 08/12] mm/collapse: separate scanning a PTE table from collapsing it
From: Kiryl Shutsemau
Date: Mon Sep 07 2026 - 07:49:39 EST
On Sat, Sep 05, 2026 at 10:30:16PM -0400, Zi Yan wrote:
> On Fri Sep 4, 2026 at 11:10 AM EDT, Kiryl Shutsemau wrote:
> > From: "Kiryl Shutsemau (Meta)" <kas@xxxxxxxxxx>
> >
> > A collapse is two jobs. One reads a PTE table under mmap_lock and decides
> > whether the range is worth collapsing. The other allocates, isolates,
> > copies and flushes, and wants the lock given up first.
> >
> > collapse_single_pmd() did both, so the boundary between them was somewhere
> > in the middle of a function.
> >
> > Give each half its own function:
> >
> > - collapse_scan_pmd() scans one table and only reads. The anonymous
> > scan that used to carry that name keeps its body as
> > collapse_scan_anon_pmd(), and collapse_scan_pmd() is now the entry
> > that picks the anonymous or the file side.
> >
> > - collapse_run_pmd() does the collapse the scan asked for.
> > SCAN_SUCCEED from the scan means there is something to run; anything
> > else is why there is not.
> >
> > collapse_single_pmd() is now the two of them with the mmap_lock drop in
> > between, so its callers see what they saw before.
> >
> > Scan results (beyond SCAN_SUCCEED) communicated via collapse_control
> > structure: the orders, the referenced and swapped-out counts, and for a
> > file the file itself and the offset in it.
> >
> > A file collapse works on the page cache and never sees a VMA. The scan
> > takes the file reference while it still has VMA and the run unpins it
> > when it is done.
> >
> > Tracing changes with it. mm_khugepaged_scan_pmd now fires before
> > mm_collapse_huge_page instead of after it. Its status field already reads
> > SCAN_SUCCEED for an accepted table, so what the collapse then made of that
> > table is mm_collapse_huge_page's to report, per order.
> >
> > The two calls to that tracepoint become one. They differed in what the
> > collapse between them changed; with the collapse no longer here, both
> > carry the same arguments. failed_pfn is set only where a PTE was refused,
> > so it is -1 exactly when the result is SCAN_SUCCEED.
> >
> > Assisted-by: Claude-Code:claude-opus-5
> > Signed-off-by: Kiryl Shutsemau (Meta) <kas@xxxxxxxxxx>
> > ---
> > mm/collapse.h | 14 ++++++
> > mm/khugepaged.c | 121 ++++++++++++++++++++++++++++++++++--------------
> > 2 files changed, 100 insertions(+), 35 deletions(-)
> >
>
> <snip>
>
> >
> > - mmap_read_unlock(mm);
> > - *lock_dropped = true;
> > +static enum scan_result collapse_run_pmd(struct mm_struct *mm,
> > + unsigned long addr, struct collapse_control *cc)
> > +{
> > + struct file *file = cc->scan_file;
> > + bool triggered_wb = false;
> > + enum scan_result result;
> > + pgoff_t pgoff;
> > +
> > + if (!file)
> > + return mthp_collapse(mm, addr, cc->scan_referenced,
> > + cc->scan_unmapped, cc, cc->scan_orders);
> > +
> > + cc->scan_file = NULL;
> > + pgoff = cc->scan_pgoff;
> > retry:
> > result = collapse_scan_file(mm, addr, file, pgoff, cc);
>
> In the commit message, collapse_run_pmd() is said to do the collapse
> work, but collapse_scan_file() is scanning, right?
>
> It seems that the code only separate anonymous scan and collapse.
> Why cannot pagecache code be separated in a similar way?
Will fold the patch below into v2:
diff --git a/mm/collapse.h b/mm/collapse.h
index 4baf2228d2c4..1ebbbf63fb25 100644
--- a/mm/collapse.h
+++ b/mm/collapse.h
@@ -95,13 +95,15 @@ struct collapse_control {
*
* The file side takes a reference while it still has the VMA, since a
* file collapse works on the page cache and never sees one; the run is
- * what gives it back.
+ * what gives it back. A scan that found the PMD folio already in the
+ * cache leaves only the PTE table to retract.
*/
unsigned long scan_orders;
int scan_referenced;
int scan_unmapped;
struct file *scan_file;
pgoff_t scan_pgoff;
+ bool scan_retract_only;
};
/* Which orders a VMA may collapse to, zero when it may not collapse at all */
diff --git a/mm/khugepaged.c b/mm/khugepaged.c
index 4ae292a6392e..403e5fee942d 100644
--- a/mm/khugepaged.c
+++ b/mm/khugepaged.c
@@ -2722,20 +2722,13 @@ static enum scan_result collapse_scan_file(struct mm_struct *mm,
else
cc->progress += HPAGE_PMD_NR;
- if (result == SCAN_SUCCEED) {
- if (present < HPAGE_PMD_NR - max_ptes_none) {
- result = SCAN_EXCEED_NONE_PTE;
- count_vm_event(THP_SCAN_EXCEED_NONE_PTE);
- } else {
- result = collapse_file(mm, addr, file, start, cc);
- }
- trace_mm_khugepaged_scan_file(mm, -1, file, present, swap,
- SCAN_SUCCEED);
- } else {
- trace_mm_khugepaged_scan_file(mm, failed_pfn, file, present,
- swap, result);
+ if (result == SCAN_SUCCEED && present < HPAGE_PMD_NR - max_ptes_none) {
+ result = SCAN_EXCEED_NONE_PTE;
+ count_vm_event(THP_SCAN_EXCEED_NONE_PTE);
}
+ trace_mm_khugepaged_scan_file(mm, failed_pfn, file, present, swap,
+ result);
return result;
}
@@ -2758,6 +2751,9 @@ enum scan_result collapse_scan_pmd(struct vm_area_struct *vma,
unsigned long addr, struct collapse_control *cc,
unsigned long orders)
{
+ enum scan_result result;
+ pgoff_t pgoff;
+
mmap_assert_locked(vma->vm_mm);
/* Whatever the last scan found has to have been run by now */
if (WARN_ON_ONCE(cc->scan_file)) {
@@ -2768,14 +2764,31 @@ enum scan_result collapse_scan_pmd(struct vm_area_struct *vma,
if (vma_is_anonymous(vma))
return collapse_scan_anon_pmd(vma, addr, cc, orders);
+ pgoff = linear_page_index(vma, addr);
+ result = collapse_scan_file(vma->vm_mm, addr, vma->vm_file, pgoff, cc);
+ switch (result) {
+ case SCAN_SUCCEED:
+ cc->scan_retract_only = false;
+ break;
+ case SCAN_PTE_MAPPED_HUGEPAGE:
+ /*
+ * The page cache already holds the PMD folio; what is left is
+ * to retract the PTE table, which is the run's job.
+ */
+ cc->scan_retract_only = true;
+ result = SCAN_SUCCEED;
+ break;
+ default:
+ return result;
+ }
+
/*
* A file collapse works on the page cache and never sees a VMA, so take
- * what it needs from this one while it is still here. Judging the
- * range needs the page cache and no lock, so it happens in the run.
+ * what it needs from this one while it is still here.
*/
cc->scan_file = get_file(vma->vm_file);
- cc->scan_pgoff = linear_page_index(vma, addr);
- return SCAN_SUCCEED;
+ cc->scan_pgoff = pgoff;
+ return result;
}
enum scan_result collapse_run_pmd(struct mm_struct *mm, unsigned long addr,
@@ -2792,8 +2805,13 @@ enum scan_result collapse_run_pmd(struct mm_struct *mm, unsigned long addr,
cc->scan_file = NULL;
pgoff = cc->scan_pgoff;
+
+ if (cc->scan_retract_only) {
+ result = SCAN_PTE_MAPPED_HUGEPAGE;
+ goto retract;
+ }
retry:
- result = collapse_scan_file(mm, addr, file, pgoff, cc);
+ result = collapse_file(mm, addr, file, pgoff, cc);
/* Dirty pages are worth a writeback and one more try, if asked for */
if (cc->policy.writeback_dirty && result == SCAN_PAGE_DIRTY_OR_WRITEBACK &&
@@ -2805,8 +2823,13 @@ enum scan_result collapse_run_pmd(struct mm_struct *mm, unsigned long addr,
triggered_wb = true;
goto retry;
}
+retract:
fput(file);
+ /*
+ * A PMD folio is in the page cache, whether the collapse just put it
+ * there or found it: retract the PTE table, and map the PMD if asked.
+ */
if (result == SCAN_PTE_MAPPED_HUGEPAGE) {
mmap_read_lock(mm);
if (collapse_test_exit_or_disable(mm))
--
Kiryl Shutsemau / Kirill A. Shutemov