[PATCH v3 03/17] mm/mm_init: skip initializing shared vmemmap tail pages
Muchun Song <[email protected]> Tue, 4 Aug 2026 11:55:21 +0800
| Newsgroups | org.kvack.linux-mm,org.kernel.vger.linux-kernel |
|---|---|
| Message-ID | <[email protected]> |
memmap_init_range() initializes every struct page in the target range. For compound pages with vmemmap optimization, the tail struct pages are backed by a shared vmemmap page. Initializing those tail struct pages would overwrite the shared vmemmap page contents, requiring users such as HugeTLB to restore the metadata afterwards. Track the compound order for HVO-backed sections and use that metadata to detect struct pages that fall into the shared tail vmemmap range. Skip those shared tail pages in memmap_init_range(), then initialize pageblock migratetypes for the processed range with a helper after the per-page initialization loop. Keep direct mem_section access inside sparse helpers by exposing pfn_to_section_order() to users that only need the order associated with a PFN. This lets memmap_init_range() skip shared tail vmemmap pages without exposing __pfn_to_section() to !SPARSEMEM builds. This is a preparatory change for consolidating handling across users of vmemmap optimization, and it also avoids redundant initialization of shared tail vmemmap pages during early boot. Signed-off-by: Muchun Song <[email protected]> --- v3: - Replace the !SPARSEMEM __pfn_to_section() stub with pfn_to_section_order() (suggested by Mike Rapoport) v2: - Fold section order tracking into the first user instead of keeping a standalone API-only patch (suggested by Mike Rapoport) - Rename page_vmemmap_optimizable() to pfn_vmemmap_optimizable() and pass a PFN directly (suggested by Mike Rapoport) - Initialize pageblock migratetypes from a helper after the per-page loop (suggested by Mike Rapoport) - Use a 1G PFN chunk for cond_resched() in the pageblock helper (suggested by Mike Rapoport) - Guard section_order() with CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP so it returns 0 when HVO is disabled and lets the compiler optimize the code as much as possible (suggested by Mike Rapoport) - Explain why the !SPARSEMEM __pfn_to_section() stub belongs here (suggested by Mike Rapoport) --- include/linux/mmzone.h | 8 ++++++++ mm/mm_init.c | 34 +++++++++++++++++----------------- mm/sparse.h | 33 +++++++++++++++++++++++++++++++++ 3 files changed, 58 insertions(+), 17 deletions(-) diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index f96a45ac558f..81e16d71e1f0 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -2023,6 +2023,14 @@ struct mem_section { unsigned long section_mem_map; struct mem_section_usage *usage; +#ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP + /* + * Normally, sections hold regular (order-0) pages. However, for + * sections with HVO enabled, this tracks the compound page order + * to enable deduplication of redundant vmemmap pages. + */ + unsigned int order; +#endif #ifdef CONFIG_PAGE_EXTENSION /* * If SPARSEMEM, pgdat doesn't have page_ext pointer. We use diff --git a/mm/mm_init.c b/mm/mm_init.c index e5aa20a9b898..1823381a69bb 100644 --- a/mm/mm_init.c +++ b/mm/mm_init.c @@ -29,6 +29,7 @@ #include <linux/cma.h> #include <linux/crash_dump.h> #include <linux/execmem.h> +#include <linux/sizes.h> #include <linux/vmstat.h> #include <linux/kexec_handover.h> #include <linux/hugetlb.h> @@ -677,21 +678,19 @@ static inline void fixup_hashdist(void) static inline void fixup_hashdist(void) {} #endif /* CONFIG_NUMA */ -#if defined(CONFIG_ZONE_DEVICE) || defined(CONFIG_DEFERRED_STRUCT_PAGE_INIT) static __meminit void pageblock_migratetype_init_range(unsigned long pfn, - unsigned long nr_pages, int migratetype, bool atomic) + unsigned long nr_pages, int migratetype, bool isolate, bool atomic) { const unsigned long end = pfn + nr_pages; for (pfn = pageblock_align(pfn); pfn < end; pfn += pageblock_nr_pages) { enum migratetype mt = kho_scratch_migratetype(pfn, migratetype); - init_pageblock_migratetype(pfn_to_page(pfn), mt, false); - if (!atomic && IS_ALIGNED(pfn, PAGES_PER_SECTION)) + init_pageblock_migratetype(pfn_to_page(pfn), mt, isolate); + if (!atomic && IS_ALIGNED(pfn, PFN_DOWN(SZ_1G))) cond_resched(); } } -#endif #ifdef CONFIG_DEFERRED_STRUCT_PAGE_INIT static inline void pgdat_set_deferred_range(pg_data_t *pgdat) @@ -886,6 +885,13 @@ void __meminit memmap_init_range(unsigned long size, int nid, unsigned long zone } } + if (pfn_vmemmap_optimizable(pfn)) { + unsigned int order = pfn_to_section_order(pfn); + + pfn = min(ALIGN(pfn, 1UL << order), end_pfn); + continue; + } + page = pfn_to_page(pfn); __init_single_page(page, pfn, zone, nid); if (context == MEMINIT_HOTPLUG) { @@ -897,19 +903,13 @@ void __meminit memmap_init_range(unsigned long size, int nid, unsigned long zone __SetPageOffline(page); } - /* - * Usually, we want to mark the pageblock MIGRATE_MOVABLE, - * such that unmovable allocations won't be scattered all - * over the place during system boot. - */ - if (pageblock_aligned(pfn)) { - enum migratetype mt = kho_scratch_migratetype(pfn, migratetype); - - init_pageblock_migratetype(page, mt, isolate_pageblock); + if (pageblock_aligned(pfn)) cond_resched(); - } pfn++; } + + pageblock_migratetype_init_range(start_pfn, pfn - start_pfn, migratetype, + isolate_pageblock, false); } static void __init memmap_init_zone_range(struct zone *zone, @@ -1112,7 +1112,7 @@ void __ref memmap_init_zone_device(struct zone *zone, compound_nr_pages(pfn, altmap, pgmap)); } - pageblock_migratetype_init_range(start_pfn, nr_pages, MIGRATE_MOVABLE, false); + pageblock_migratetype_init_range(start_pfn, nr_pages, MIGRATE_MOVABLE, false, false); pr_debug("%s initialised %lu pages in %ums\n", __func__, nr_pages, jiffies_to_msecs(jiffies - start)); @@ -1922,7 +1922,7 @@ static void __init deferred_free_pages(unsigned long pfn, if (!nr_pages) return; - pageblock_migratetype_init_range(pfn, nr_pages, MIGRATE_MOVABLE, true); + pageblock_migratetype_init_range(pfn, nr_pages, MIGRATE_MOVABLE, false, true); page = pfn_to_page(pfn); diff --git a/mm/sparse.h b/mm/sparse.h index 95aa031213f2..b9b6b47e85ce 100644 --- a/mm/sparse.h +++ b/mm/sparse.h @@ -10,6 +10,39 @@ #include <linux/mmzone.h> +#ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP +static inline unsigned int section_order(const struct mem_section *section) +{ + return section->order; +} + +static inline unsigned int pfn_to_section_order(unsigned long pfn) +{ + return section_order(__pfn_to_section(pfn)); +} +#else +static inline unsigned int section_order(const struct mem_section *section) +{ + return 0; +} + +static inline unsigned int pfn_to_section_order(unsigned long pfn) +{ + return 0; +} +#endif + +static inline bool pfn_vmemmap_optimizable(unsigned long pfn) +{ + const unsigned int order = pfn_to_section_order(pfn); + const unsigned long nr_pages = 1UL << order; + + if (!is_power_of_2(sizeof(struct page))) + return false; + + return (pfn & (nr_pages - 1)) >= VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES; +} + /* * mm/sparse.c */ -- 2.54.0