[PATCH v6 09/12] powerpc/mm: switch device DAX to shared tail vmemmap pages

From: Muchun Song

Date: Wed Sep 30 2026 - 11:28:53 EST


The powerpc radix compound vmemmap population path still finds a reusable
tail page by walking the vmemmap page tables.

Switch it to the common vmemmap_shared_tail_page() helper instead, so it
can use the shared vmemmap page directly to simplify the code.

This removes the powerpc-specific tail-page lookup and its fallback path
and aligns the device DAX vmemmap optimization path with HugeTLB.

Signed-off-by: Muchun Song <songmuchun@xxxxxxxxxxxxx>
Acked-by: David Hildenbrand (Arm) <david@xxxxxxxxxx>
---
v6:
- Collect Acked-by from David Hildenbrand
---
arch/powerpc/mm/book3s64/radix_pgtable.c | 80 +++---------------------
include/linux/vmemmap-optimization.h | 6 ++
mm/sparse-vmemmap.c | 6 --
3 files changed, 15 insertions(+), 77 deletions(-)

diff --git a/arch/powerpc/mm/book3s64/radix_pgtable.c b/arch/powerpc/mm/book3s64/radix_pgtable.c
index cf692b2b5f7b..ee068f24a79f 100644
--- a/arch/powerpc/mm/book3s64/radix_pgtable.c
+++ b/arch/powerpc/mm/book3s64/radix_pgtable.c
@@ -19,6 +19,7 @@
#include <linux/string_helpers.h>
#include <linux/memory.h>
#include <linux/kfence.h>
+#include <linux/vmemmap-optimization.h>

#include <asm/pgalloc.h>
#include <asm/mmu_context.h>
@@ -1250,59 +1251,6 @@ static pte_t * __meminit radix__vmemmap_populate_address(unsigned long addr, int
return pte;
}

-static pte_t * __meminit vmemmap_compound_tail_page(unsigned long addr,
- unsigned long pfn_offset, int node)
-{
- pgd_t *pgd;
- p4d_t *p4d;
- pud_t *pud;
- pmd_t *pmd;
- pte_t *pte;
- unsigned long map_addr;
-
- /* the second vmemmap page which we use for duplication */
- map_addr = addr - pfn_offset * sizeof(struct page) + PAGE_SIZE;
- pgd = pgd_offset_k(map_addr);
- p4d = p4d_offset(pgd, map_addr);
- pud = vmemmap_pud_alloc(p4d, node, map_addr);
- if (!pud)
- return NULL;
- pmd = vmemmap_pmd_alloc(pud, node, map_addr);
- if (!pmd)
- return NULL;
- if (pmd_leaf(*pmd))
- /*
- * The second page is mapped as a hugepage due to a nearby request.
- * Force our mapping to page size without deduplication
- */
- return NULL;
- pte = vmemmap_pte_alloc(pmd, node, map_addr);
- if (!pte)
- return NULL;
- /*
- * Check if there exist a mapping to the left
- */
- if (pte_none(*pte)) {
- /*
- * Populate the head page vmemmap page.
- * It can fall in different pmd, hence
- * vmemmap_populate_address()
- */
- pte = radix__vmemmap_populate_address(map_addr - PAGE_SIZE, node, NULL, NULL);
- if (!pte)
- return NULL;
- /*
- * Populate the tail pages vmemmap page
- */
- pte = radix__vmemmap_pte_populate(pmd, map_addr, node, NULL, NULL);
- if (!pte)
- return NULL;
- vmemmap_verify(pte, node, map_addr, map_addr + PAGE_SIZE);
- return pte;
- }
- return pte;
-}
-
int __meminit vmemmap_populate_compound_pages(unsigned long start_pfn,
unsigned long start,
unsigned long end, int node,
@@ -1320,6 +1268,12 @@ int __meminit vmemmap_populate_compound_pages(unsigned long start_pfn,
pud_t *pud;
pmd_t *pmd;
pte_t *pte;
+ struct page *tail_page;
+ unsigned int order = pfn_to_section_compound_order(start_pfn);
+
+ tail_page = vmemmap_shared_tail_page(order, device_zone(node));
+ if (!tail_page)
+ return -ENOMEM;

for (addr = start; addr < end; addr = next) {

@@ -1349,10 +1303,9 @@ int __meminit vmemmap_populate_compound_pages(unsigned long start_pfn,
next = addr + PAGE_SIZE;
continue;
} else {
- unsigned long nr_pages = pgmap_vmemmap_nr(pgmap);
+ unsigned long nr_pages = 1UL << order;
unsigned long addr_pfn = page_to_pfn((struct page *)addr);
unsigned long pfn_offset = addr_pfn - ALIGN_DOWN(addr_pfn, nr_pages);
- pte_t *tail_page_pte;

/*
* if the address is aligned to huge page size it is the
@@ -1377,23 +1330,8 @@ int __meminit vmemmap_populate_compound_pages(unsigned long start_pfn,
next = addr + 2 * PAGE_SIZE;
continue;
}
- /*
- * get the 2nd mapping details
- * Also create it if that doesn't exist
- */
- tail_page_pte = vmemmap_compound_tail_page(addr, pfn_offset, node);
- if (!tail_page_pte) {
-
- pte = radix__vmemmap_pte_populate(pmd, addr, node, NULL, NULL);
- if (!pte)
- return -ENOMEM;
- vmemmap_verify(pte, node, addr, addr + PAGE_SIZE);
-
- next = addr + PAGE_SIZE;
- continue;
- }

- pte = radix__vmemmap_pte_populate(pmd, addr, node, NULL, pte_page(*tail_page_pte));
+ pte = radix__vmemmap_pte_populate(pmd, addr, node, NULL, tail_page);
if (!pte)
return -ENOMEM;
vmemmap_verify(pte, node, addr, addr + PAGE_SIZE);
diff --git a/include/linux/vmemmap-optimization.h b/include/linux/vmemmap-optimization.h
index bd0974b262a4..fa9e9abd6656 100644
--- a/include/linux/vmemmap-optimization.h
+++ b/include/linux/vmemmap-optimization.h
@@ -83,6 +83,12 @@ static inline unsigned int pfn_to_section_compound_order(unsigned long pfn)
{
return 0;
}
+
+static inline struct page *vmemmap_shared_tail_page(unsigned int order,
+ struct zone *zone)
+{
+ return NULL;
+}
#endif /* CONFIG_VMEMMAP_OPTIMIZATION */

static inline bool vmemmap_optimizable_pfn(unsigned long pfn)
diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c
index d40a2f5b5fca..e1f8a03e3d49 100644
--- a/mm/sparse-vmemmap.c
+++ b/mm/sparse-vmemmap.c
@@ -240,12 +240,6 @@ struct page __ref *vmemmap_shared_tail_page(unsigned int order, struct zone *zon

return page;
}
-#else
-static inline struct page *vmemmap_shared_tail_page(unsigned int order,
- struct zone *zone)
-{
- return NULL;
-}
#endif

static __meminit void *vmemmap_alloc_pte(unsigned long pfn, int node,
--
2.54.0