Re: [PATCH 04/17] mm/mm_init: skip initializing shared vmemmap tail pages
From: Mike Rapoport
Date: Wed Jul 15 2026 - 01:09:14 EST
> memmap_init_range() initializes every struct page in the target range.
> For compound pages with vmemmap optimization, the tail struct pages are
> backed by a shared vmemmap page.
>
> Initializing those tail struct pages would overwrite the shared
> vmemmap page contents, so users such as HugeTLB have to open-code
> follow-up handling to restore the metadata afterwards.
>
> Use the section's compound page order to detect struct pages that fall
> into the shared tail vmemmap range and skip their initialization in
> memmap_init_range(). Still initialize the pageblock migratetypes for
> the skipped range so the surrounding setup remains intact.
>
> This is a preparatory change for consolidating handling across users of
> vmemmap optimization, and it also avoids redundant initialization of
> shared tail vmemmap pages during early boot.
>
> Signed-off-by: Muchun Song <songmuchun@xxxxxxxxxxxxx>
>
> diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h
> index ea884245f499..6fa6e7f0abf9 100644
> --- a/include/linux/mmzone.h
> +++ b/include/linux/mmzone.h
> @@ -2375,6 +2375,10 @@ struct mem_section;
>
> #define sparse_vmemmap_init_nid_early(_nid) do {} while (0)
> #define pfn_in_present_section pfn_valid
> +static inline struct mem_section *__pfn_to_section(unsigned long pfn)
> +{
> + return NULL;
> +}
Changelog should mention why this is a part of the patch, it's really
not obvious.
> #endif /* CONFIG_SPARSEMEM */
>
> #ifdef CONFIG_SPARSEMEM_VMEMMAP
> diff --git a/mm/internal.h b/mm/internal.h
> index 430aa72a4575..ebbab7421633 100644
> --- a/mm/internal.h
> +++ b/mm/internal.h
> @@ -1002,10 +1002,26 @@ static inline void sparse_init(void) {}
> */
> #ifdef CONFIG_SPARSEMEM_VMEMMAP
> void sparse_init_subsection_map(void);
> +
> +static inline bool page_vmemmap_optimizable(const struct page *page, unsigned int order)
> +{
> + const unsigned long pfn = page_to_pfn(page);
> + const unsigned long nr_pages = 1UL << order;
> +
Don't you want to gate it on HUGETLB_PAGE_OPTIMIZE_VMEMMAP_DEFAULT_ON?
Even if it's false for sections without order explicitly set (which is
btw not very obvious), this adds checks for every struct page that we
initialize.
We can revisit this later when the optimization would be relevant for
memory hotplug.
> + if (!is_power_of_2(sizeof(struct page)))
> + return false;
> +
> + return (pfn & (nr_pages - 1)) >= OPTIMIZED_FOLIO_VMEMMAP_NR_STRUCT_PAGES;
> +}
> #else
> static inline void sparse_init_subsection_map(void)
> {
> }
> +
> +static inline bool page_vmemmap_optimizable(const struct page *page, unsigned int order)
> +{
> + return false;
> +}
> #endif /* CONFIG_SPARSEMEM_VMEMMAP */
>
> #if defined CONFIG_COMPACTION || defined CONFIG_CMA
> diff --git a/mm/mm_init.c b/mm/mm_init.c
> index 4026b084bd4b..7ef1ac105058 100644
> --- a/mm/mm_init.c
> +++ b/mm/mm_init.c
> @@ -674,19 +674,21 @@ static inline void fixup_hashdist(void)
> static inline void fixup_hashdist(void) {}
> #endif /* CONFIG_NUMA */
>
> -#if defined(CONFIG_ZONE_DEVICE) || defined(CONFIG_DEFERRED_STRUCT_PAGE_INIT)
> static __meminit void pageblock_migratetype_init_range(unsigned long pfn,
> - unsigned long nr_pages, int migratetype, bool atomic)
> + unsigned long nr_pages, int migratetype, bool isolate, bool atomic)
> {
> const unsigned long end = pfn + nr_pages;
>
> for (pfn = pageblock_align(pfn); pfn < end; pfn += pageblock_nr_pages) {
> - init_pageblock_migratetype(pfn_to_page(pfn), migratetype, false);
> + init_pageblock_migratetype(pfn_to_page(pfn), migratetype, isolate);
> +#ifdef CONFIG_SPARSEMEM
> if (!atomic && IS_ALIGNED(pfn, PAGES_PER_SECTION))
> +#else
> + if (!atomic && IS_ALIGNED(pfn, MAX_FOLIO_NR_PAGES))
> +#endif
> cond_resched();
> }
> }
> -#endif
>
> #ifdef CONFIG_DEFERRED_STRUCT_PAGE_INIT
> static inline void pgdat_set_deferred_range(pg_data_t *pgdat)
> @@ -892,6 +894,8 @@ void __meminit memmap_init_range(unsigned long size, int nid, unsigned long zone
> #endif
>
> for (pfn = start_pfn; pfn < end_pfn; ) {
> + unsigned int order = section_order(__pfn_to_section(pfn));
> +
I'd move this into page_vmemmap_optimizable(). This way it would be
clearer that a section must have non 0 order for optimization to catch.
> /*
> * There can be holes in boot-time mem_map[]s handed to this
> * function. They do not exist on hotplugged memory.
> @@ -906,6 +910,15 @@ void __meminit memmap_init_range(unsigned long size, int nid, unsigned long zone
> }
>
> page = pfn_to_page(pfn);
> + if (page_vmemmap_optimizable(page, order)) {
You can pass pfn instead of page to page_vmemmap_optimizable(), will
save a conversion there.
> + const unsigned long start = pfn;
> +
> + pfn = min(ALIGN(start, 1UL << order), end_pfn);
> + pageblock_migratetype_init_range(start, pfn - start, migratetype,
> + isolate_pageblock, false);
I wonder if we'll see any measurable difference in the memmap initialization
if we pull pageblock initialization out unconditionally.
If we don't we can just call pageblock_migratetype_init_range() for the
entire range and kill that
if (pageblock_aligned(pfn)) {
init_pageblock_migratetype(page, migratetype,
isolate_pageblock);
cond_resched();
}
--
Sincerely yours,
Mike.