[PATCH RFC 8/9] mm/huge_memory: split present and non-present huge PMDs separately

From: Yin Tirui

Date: Fri Aug 28 2026 - 14:43:45 EST


Move the shared PTE rebuild into split_huge_pmd_to_ptes(), then add
split_present_huge_pmd() and split_non_present_huge_pmd() on top of it to
separate the present and non-present cases in __split_huge_pmd_locked().

No functional change intended.

Suggested-by: David Hildenbrand <david@xxxxxxxxxx>
Link: https://lore.kernel.org/linux-mm/67a655e3-fa23-4d2a-9685-14e6221d5d26@xxxxxxxxxx/
Signed-off-by: Yin Tirui <yintirui@xxxxxxxxx>
---
mm/huge_memory.c | 334 +++++++++++++++++++++++++----------------------
1 file changed, 180 insertions(+), 154 deletions(-)

diff --git a/mm/huge_memory.c b/mm/huge_memory.c
index 72e2cd1d7672..fdb751a1e525 100644
--- a/mm/huge_memory.c
+++ b/mm/huge_memory.c
@@ -3245,143 +3245,31 @@ static bool split_huge_pmd_anon_rmap(const struct split_pmd_state *state,
return false;
}

-static void __split_huge_pmd_locked(struct vm_area_struct *vma, pmd_t *pmd,
- unsigned long haddr, bool freeze)
+/*
+ * Replace an anonymous huge PMD entry with a page table mapping the same
+ * folio at PTE granularity.
+ */
+static void split_huge_pmd_to_ptes(struct vm_area_struct *vma,
+ unsigned long haddr, pmd_t *pmd, struct split_pmd_state *state)
{
- const pmd_t old_pmd = *pmd;
- const bool is_present = pmd_present(old_pmd);
+ /* Present mappings and device private entries hold a PMD-level rmap. */
+ const bool rmapped = state->is_present || state->is_device_private;
struct mm_struct *mm = vma->vm_mm;
- struct split_pmd_state state = {
- .is_present = is_present,
- .freeze = freeze,
- };
- struct folio *folio;
+ struct page *page = state->page;
unsigned long addr;
pgtable_t pgtable;
pmd_t _pmd;
pte_t *pte;
int i;

- VM_BUG_ON(haddr & ~HPAGE_PMD_MASK);
- VM_BUG_ON_VMA(vma->vm_start > haddr, vma);
- VM_BUG_ON_VMA(vma->vm_end < haddr + HPAGE_PMD_SIZE, vma);
-
- VM_WARN_ON_ONCE(!pmd_is_valid_softleaf(old_pmd) &&
- !pmd_trans_huge(old_pmd));
-
- count_vm_event(THP_SPLIT_PMD);
+ if (rmapped)
+ state->freeze = split_huge_pmd_anon_rmap(state, vma, haddr);

/*
- * FIXME: Do we want to invalidate secondary mmu by calling
- * mmu_notifier_arch_invalidate_secondary_tlbs() see comments below
- * inside __split_huge_pmd() ?
- *
- * We are going from a zero huge page write protected to zero small
- * page also write protected so it does not seems useful to invalidate
- * secondary mmu at this time.
- */
- if (huge_zero_pmd_can_split(vma, old_pmd)) {
- __split_huge_zero_page_pmd(vma, haddr, pmd);
- return;
- }
-
- folio = normal_or_softleaf_folio_pmd(vma, haddr, old_pmd, is_present);
-
- /*
- * A non-present entry which is neither a migration nor a device
- * private entry is corrupt, and pmd_to_softleaf_folio() has already
- * warned about it. Leave it alone rather than act on a PFN which
- * means nothing.
- */
- if (unlikely(!is_present && !folio))
- return;
-
- if (!folio || !folio_test_anon(folio)) {
- unmap_huge_pmd_entry(vma, haddr, pmd, folio, is_present);
- return;
- }
-
- state.folio = folio;
-
- if (pmd_is_migration_entry(old_pmd)) {
- const softleaf_t entry = softleaf_from_pmd(old_pmd);
-
- state.page = softleaf_to_page(entry);
-
- state.soft_dirty = pmd_swp_soft_dirty(old_pmd);
- state.uffd = pmd_swp_uffd(old_pmd);
-
- state.write = softleaf_is_migration_write(entry);
- state.anon_exclusive =
- softleaf_is_migration_read_exclusive(entry);
- state.young = softleaf_is_migration_young(entry);
- state.dirty = softleaf_is_migration_dirty(entry);
- } else if (pmd_is_device_private_entry(old_pmd)) {
- const softleaf_t entry = softleaf_from_pmd(old_pmd);
-
- state.is_device_private = true;
- state.page = softleaf_to_page(entry);
-
- state.soft_dirty = pmd_swp_soft_dirty(old_pmd);
- state.uffd = pmd_swp_uffd(old_pmd);
-
- state.write = softleaf_is_device_private_write(entry);
- state.anon_exclusive = PageAnonExclusive(state.page);
-
- /*
- * Device private folios are treated the same as regular folios
- * w.r.t. anon exclusive handling, see
- * split_huge_pmd_anon_rmap().
- */
- state.freeze = split_huge_pmd_anon_rmap(&state, vma, haddr);
- } else {
- /*
- * Up to this point the pmd is present and huge and userland has
- * the whole access to the hugepage during the split (which
- * happens in place). If we overwrite the pmd with the not-huge
- * version pointing to the pte here (which of course we could if
- * all CPUs were bug free), userland could trigger a small page
- * size TLB miss on the small sized TLB while the hugepage TLB
- * entry is still established in the huge TLB. Some CPU doesn't
- * like that. See
- * http://support.amd.com/TechDocs/41322_10h_Rev_Gd.pdf, Erratum
- * 383 on page 105. Intel should be safe but is also warns that
- * it's only safe if the permission and cache attributes of the
- * two entries loaded in the two TLB is identical (which should
- * be the case here). But it is generally safer to never allow
- * small and huge TLB entries for the same virtual address to be
- * loaded simultaneously. So instead of doing "pmd_populate();
- * flush_pmd_tlb_range();" we first mark the current pmd
- * notpresent (atomically because here the pmd_trans_huge must
- * remain set at all times on the pmd until the split is
- * complete for this pmd), then we flush the SMP TLB and finally
- * we write the non-huge version of the pmd entry with
- * pmd_populate.
- *
- * This must also happen before PageAnonExclusive() is read
- * below, see folio_try_share_anon_rmap_pmd().
- */
- const pmd_t pmdval = pmdp_invalidate(vma, haddr, pmd);
-
- state.page = pmd_page(pmdval);
- state.write = pmd_write(pmdval);
- state.young = pmd_young(pmdval);
- state.dirty = pmd_dirty(pmdval);
- state.soft_dirty = pmd_soft_dirty(pmdval);
- state.uffd = pmd_uffd(pmdval);
- state.anon_exclusive = PageAnonExclusive(state.page);
-
- if (state.dirty)
- folio_set_dirty(folio);
-
- VM_WARN_ON_FOLIO(!folio_ref_count(folio), folio);
-
- state.freeze = split_huge_pmd_anon_rmap(&state, vma, haddr);
- }
-
- /*
- * Withdraw the table only after we mark the pmd entry invalid.
- * This's critical for some architectures (Power).
+ * The caller has already invalidated a present entry, and a softleaf
+ * entry is not present to begin with. Either way the entry is out of
+ * service before we withdraw the deposited page table, which is
+ * critical for some architectures (Power).
*/
pgtable = pgtable_trans_huge_withdraw(mm, pmd);
pmd_populate(mm, &_pmd, pgtable);
@@ -3393,33 +3281,34 @@ static void __split_huge_pmd_locked(struct vm_area_struct *vma, pmd_t *pmd,
* Note that NUMA hinting access restrictions are not transferred to
* avoid any possibility of altering permissions across VMAs.
*/
- if (state.freeze || (!state.is_present && !state.is_device_private)) {
+ if (state->freeze ||
+ (!state->is_present && !state->is_device_private)) {
pte_t entry;
swp_entry_t swp_entry;

for (i = 0, addr = haddr; i < HPAGE_PMD_NR; i++, addr += PAGE_SIZE) {
- if (state.write)
+ if (state->write)
swp_entry = make_writable_migration_entry(
- page_to_pfn(state.page + i));
- else if (state.anon_exclusive)
+ page_to_pfn(page + i));
+ else if (state->anon_exclusive)
swp_entry = make_readable_exclusive_migration_entry(
- page_to_pfn(state.page + i));
+ page_to_pfn(page + i));
else
swp_entry = make_readable_migration_entry(
- page_to_pfn(state.page + i));
- if (state.young)
+ page_to_pfn(page + i));
+ if (state->young)
swp_entry = make_migration_entry_young(swp_entry);
- if (state.dirty)
+ if (state->dirty)
swp_entry = make_migration_entry_dirty(swp_entry);
entry = swp_entry_to_pte(swp_entry);
- if (state.soft_dirty)
+ if (state->soft_dirty)
entry = pte_swp_mksoft_dirty(entry);
- if (state.uffd)
+ if (state->uffd)
entry = pte_swp_mkuffd(entry);
VM_WARN_ON(!pte_none(ptep_get(pte + i)));
set_pte_at(mm, addr, pte + i, entry);
}
- } else if (state.is_device_private) {
+ } else if (state->is_device_private) {
pte_t entry;
swp_entry_t swp_entry;

@@ -3429,19 +3318,19 @@ static void __split_huge_pmd_locked(struct vm_area_struct *vma, pmd_t *pmd,
* pages corresponding to the pte entries when freeze
* is false.
*/
- if (state.write)
+ if (state->write)
swp_entry = make_writable_device_private_entry(
- page_to_pfn(state.page + i));
+ page_to_pfn(page + i));
else
swp_entry = make_readable_device_private_entry(
- page_to_pfn(state.page + i));
+ page_to_pfn(page + i));
/*
* Young and dirty bits are not progated via swp_entry
*/
entry = swp_entry_to_pte(swp_entry);
- if (state.soft_dirty)
+ if (state->soft_dirty)
entry = pte_swp_mksoft_dirty(entry);
- if (state.uffd)
+ if (state->uffd)
entry = pte_swp_mkuffd(entry);
VM_WARN_ON(!pte_none(ptep_get(pte + i)));
set_pte_at(mm, addr, pte + i, entry);
@@ -3449,21 +3338,21 @@ static void __split_huge_pmd_locked(struct vm_area_struct *vma, pmd_t *pmd,
} else {
pte_t entry;

- entry = mk_pte(state.page, READ_ONCE(vma->vm_page_prot));
- if (state.write)
+ entry = mk_pte(page, READ_ONCE(vma->vm_page_prot));
+ if (state->write)
entry = pte_mkwrite(entry, vma);
- if (!state.young)
+ if (!state->young)
entry = pte_mkold(entry);
/* NOTE: this may set soft-dirty too on some archs */
- if (state.dirty)
+ if (state->dirty)
entry = pte_mkdirty(entry);
- if (state.soft_dirty)
+ if (state->soft_dirty)
entry = pte_mksoft_dirty(entry);
- if (state.uffd)
+ if (state->uffd)
entry = pte_mkuffd(entry);

/* Restore PAGE_NONE so an RWP marker keeps trapping */
- if (userfaultfd_rwp(vma) && state.uffd)
+ if (userfaultfd_rwp(vma) && state->uffd)
entry = pte_modify(entry, PAGE_NONE);

for (i = 0; i < HPAGE_PMD_NR; i++)
@@ -3473,15 +3362,152 @@ static void __split_huge_pmd_locked(struct vm_area_struct *vma, pmd_t *pmd,
}
pte_unmap(pte);

- if (state.is_present || state.is_device_private)
- folio_remove_rmap_pmd(state.folio, state.page, vma);
- if (state.freeze)
- put_page(state.page);
+ if (rmapped)
+ folio_remove_rmap_pmd(state->folio, page, vma);
+ if (state->freeze)
+ put_page(page);

smp_wmb(); /* make pte visible before pmd */
pmd_populate(mm, pmd, pgtable);
}

+static void split_present_huge_pmd(struct vm_area_struct *vma,
+ unsigned long haddr, pmd_t *pmd, struct folio *folio,
+ bool freeze)
+{
+ struct split_pmd_state state = {
+ .folio = folio,
+ .is_present = true,
+ .freeze = freeze,
+ };
+
+ /*
+ * Up to this point the pmd is present and huge and userland has the
+ * whole access to the hugepage during the split (which happens in
+ * place). If we overwrite the pmd with the not-huge version pointing
+ * to the pte here (which of course we could if all CPUs were bug
+ * free), userland could trigger a small page size TLB miss on the
+ * small sized TLB while the hugepage TLB entry is still established in
+ * the huge TLB. Some CPU doesn't like that. See
+ * http://support.amd.com/TechDocs/41322_10h_Rev_Gd.pdf, Erratum 383 on
+ * page 105. Intel should be safe but is also warns that it's only safe
+ * if the permission and cache attributes of the two entries loaded in
+ * the two TLB is identical (which should be the case here). But it is
+ * generally safer to never allow small and huge TLB entries for the
+ * same virtual address to be loaded simultaneously. So instead of
+ * doing "pmd_populate(); flush_pmd_tlb_range();" we first mark the
+ * current pmd notpresent (atomically because here the pmd_trans_huge
+ * must remain set at all times on the pmd until the split is complete
+ * for this pmd), then we flush the SMP TLB and finally we write the
+ * non-huge version of the pmd entry with pmd_populate.
+ *
+ * This must also happen before PageAnonExclusive() is read below, see
+ * folio_try_share_anon_rmap_pmd().
+ */
+ const pmd_t pmdval = pmdp_invalidate(vma, haddr, pmd);
+
+ state.page = pmd_page(pmdval);
+ state.write = pmd_write(pmdval);
+ state.young = pmd_young(pmdval);
+ state.dirty = pmd_dirty(pmdval);
+ state.soft_dirty = pmd_soft_dirty(pmdval);
+ state.uffd = pmd_uffd(pmdval);
+ state.anon_exclusive = PageAnonExclusive(state.page);
+
+ if (state.dirty)
+ folio_set_dirty(folio);
+
+ VM_WARN_ON_FOLIO(!folio_ref_count(folio), folio);
+
+ split_huge_pmd_to_ptes(vma, haddr, pmd, &state);
+}
+
+static void split_non_present_huge_pmd(struct vm_area_struct *vma,
+ unsigned long haddr, pmd_t *pmd, pmd_t old_pmd,
+ struct folio *folio, bool freeze)
+{
+ const softleaf_t entry = softleaf_from_pmd(old_pmd);
+ struct split_pmd_state state = {
+ .folio = folio,
+ .page = softleaf_to_page(entry),
+ .is_device_private = softleaf_is_device_private(entry),
+ .freeze = freeze,
+ .soft_dirty = pmd_swp_soft_dirty(old_pmd),
+ .uffd = pmd_swp_uffd(old_pmd),
+ };
+
+ if (state.is_device_private) {
+ /*
+ * Device private folios are treated the same as regular folios
+ * w.r.t. anon exclusive handling, see
+ * split_huge_pmd_anon_rmap().
+ */
+ state.write = softleaf_is_device_private_write(entry);
+ state.anon_exclusive = PageAnonExclusive(state.page);
+ } else {
+ state.write = softleaf_is_migration_write(entry);
+ state.young = softleaf_is_migration_young(entry);
+ state.dirty = softleaf_is_migration_dirty(entry);
+ state.anon_exclusive =
+ softleaf_is_migration_read_exclusive(entry);
+ }
+
+ split_huge_pmd_to_ptes(vma, haddr, pmd, &state);
+}
+
+static void __split_huge_pmd_locked(struct vm_area_struct *vma, pmd_t *pmd,
+ unsigned long haddr, bool freeze)
+{
+ const pmd_t old_pmd = *pmd;
+ const bool is_present = pmd_present(old_pmd);
+ struct folio *folio;
+
+ VM_BUG_ON(haddr & ~HPAGE_PMD_MASK);
+ VM_BUG_ON_VMA(vma->vm_start > haddr, vma);
+ VM_BUG_ON_VMA(vma->vm_end < haddr + HPAGE_PMD_SIZE, vma);
+
+ VM_WARN_ON_ONCE(!pmd_is_valid_softleaf(old_pmd) &&
+ !pmd_trans_huge(old_pmd));
+
+ count_vm_event(THP_SPLIT_PMD);
+
+ /*
+ * FIXME: Do we want to invalidate secondary mmu by calling
+ * mmu_notifier_arch_invalidate_secondary_tlbs() see comments below
+ * inside __split_huge_pmd() ?
+ *
+ * We are going from a zero huge page write protected to zero small
+ * page also write protected so it does not seems useful to invalidate
+ * secondary mmu at this time.
+ */
+ if (huge_zero_pmd_can_split(vma, old_pmd)) {
+ __split_huge_zero_page_pmd(vma, haddr, pmd);
+ return;
+ }
+
+ folio = normal_or_softleaf_folio_pmd(vma, haddr, old_pmd, is_present);
+
+ /*
+ * A non-present entry which is neither a migration nor a device
+ * private entry is corrupt, and pmd_to_softleaf_folio() has already
+ * warned about it. Leave it alone rather than act on a PFN which
+ * means nothing.
+ */
+ if (unlikely(!is_present && !folio))
+ return;
+
+ if (!folio || !folio_test_anon(folio)) {
+ unmap_huge_pmd_entry(vma, haddr, pmd, folio, is_present);
+ return;
+ }
+
+ if (is_present)
+ split_present_huge_pmd(vma, haddr, pmd, folio, freeze);
+ else
+ split_non_present_huge_pmd(vma, haddr, pmd, old_pmd, folio,
+ freeze);
+}
+
void split_huge_pmd_locked(struct vm_area_struct *vma, unsigned long address,
pmd_t *pmd, bool freeze)
{
--
2.34.1