[PATCH v3 10/17] mm/hugetlb: switch HugeTLB to section-based vmemmap optimization
Muchun Song <[email protected]> Tue, 4 Aug 2026 11:55:28 +0800
| Newsgroups | org.kvack.linux-mm,org.kernel.vger.linux-kernel |
|---|---|
| Message-ID | <[email protected]> |
HugeTLB bootmem vmemmap optimization still carries its own early setup path, including pre-populating optimized mappings before the generic sparse-vmemmap code runs. Now that the section-based vmemmap optimization can derive HugeTLB vmemmap deduplication from section metadata, HugeTLB only needs to mark the bootmem huge page range with the appropriate order. The generic sparse-vmemmap population path can then allocate and map the shared tail vmemmap pages without any HugeTLB-specific early population code. Do that by setting the section order when a bootmem huge page is allocated and dropping the dedicated pre-HVO helpers and related special-casing. This removes duplicate early setup logic and switches HugeTLB to the section-based vmemmap optimization path. Signed-off-by: Muchun Song <[email protected]> Acked-by: Mike Rapoport (Microsoft) <[email protected]> --- v3: - Use the order-based helper for the bootmem vmemmap-optimized check v2: - Collect Acked-by from Mike Rapoport --- include/linux/hugetlb.h | 1 - include/linux/mm.h | 3 -- mm/hugetlb.c | 30 ++------------ mm/hugetlb_vmemmap.c | 90 +++-------------------------------------- mm/hugetlb_vmemmap.h | 14 +++---- mm/sparse-vmemmap.c | 31 -------------- mm/sparse.h | 27 +++++++++++++ 7 files changed, 42 insertions(+), 154 deletions(-) diff --git a/include/linux/hugetlb.h b/include/linux/hugetlb.h index 16c4c4caa126..fe28f98e1b22 100644 --- a/include/linux/hugetlb.h +++ b/include/linux/hugetlb.h @@ -171,7 +171,6 @@ struct address_space *hugetlb_folio_mapping_lock_write(struct folio *folio); extern int movable_gigantic_pages __read_mostly; extern int sysctl_hugetlb_shm_group __read_mostly; -extern struct list_head huge_boot_pages[MAX_NUMNODES]; void hugetlb_bootmem_struct_page_init(void); void hugetlb_bootmem_alloc(void); diff --git a/include/linux/mm.h b/include/linux/mm.h index 7fabe6c66b4b..fd7dc85f58f3 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -5097,9 +5097,6 @@ int vmemmap_populate_hugepages(unsigned long start, unsigned long end, int node, struct vmem_altmap *altmap); int vmemmap_populate(unsigned long start, unsigned long end, int node, struct vmem_altmap *altmap); -int vmemmap_populate_hvo(unsigned long start, unsigned long end, - unsigned int order, struct zone *zone, - unsigned long headsize); void vmemmap_wrprotect_hvo(unsigned long start, unsigned long end, int node, unsigned long headsize); void vmemmap_populate_print_last(void); diff --git a/mm/hugetlb.c b/mm/hugetlb.c index d212bbff4c83..7d8507aa4c5b 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -52,6 +52,7 @@ #include "hugetlb_cma.h" #include "hugetlb_internal.h" #include "mm_init.h" +#include "sparse.h" #include <linux/page-isolation.h> int hugetlb_max_hstate __read_mostly; @@ -59,7 +60,7 @@ unsigned int default_hstate_idx; struct hstate hstates[HUGE_MAX_HSTATE]; __initdata nodemask_t hugetlb_bootmem_nodes; -__initdata struct list_head huge_boot_pages[MAX_NUMNODES]; +static struct list_head huge_boot_pages[MAX_NUMNODES] __initdata; /* * Due to ordering constraints across the init code for various @@ -3137,6 +3138,7 @@ static bool __init alloc_bootmem_huge_page(struct hstate *h, int nid) } else { list_add_tail(&m->list, &huge_boot_pages[nid]); m->flags |= HUGE_BOOTMEM_ZONES_VALID; + hugetlb_vmemmap_optimize_bootmem_page(m); /* * Only initialize the head struct page in memmap_init_reserved_pages, * rest of the struct pages will be initialized by the HugeTLB @@ -3297,6 +3299,7 @@ static void __init gather_bootmem_prealloc_node(unsigned long nid) * this folio. */ folio_set_hugetlb_vmemmap_optimized(folio); + section_set_order_range(folio_pfn(folio), folio_nr_pages(folio), 0); if (hugetlb_bootmem_page_earlycma(m)) folio_set_hugetlb_cma(folio); @@ -3340,31 +3343,6 @@ void __init hugetlb_bootmem_struct_page_init(void) .max_threads = num_node_state(N_MEMORY), .numa_aware = true, }; -#ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP - struct zone *zone; - - for_each_zone(zone) { - for (int i = 0; i < VMEMMAP_OPTIMIZATION_NR_ORDERS; i++) { - struct page *tail, *p; - unsigned int order; - - tail = zone->vmemmap_tails[i]; - if (!tail) - continue; - - order = i + VMEMMAP_OPTIMIZATION_MIN_ORDER; - p = page_to_virt(tail); - /* - * prep_and_add_bootmem_folios() can access pageblock - * flags on bootmem HugeTLB pages, so initialize the - * shared tail struct pages here before bootmem folios - * start using them. - */ - for (int j = 0; j < PAGE_SIZE / sizeof(struct page); j++) - init_compound_tail(p + j, NULL, order, zone); - } - } -#endif padata_do_multithreaded(&job); } diff --git a/mm/hugetlb_vmemmap.c b/mm/hugetlb_vmemmap.c index c48fcea076a5..7293706b532f 100644 --- a/mm/hugetlb_vmemmap.c +++ b/mm/hugetlb_vmemmap.c @@ -18,8 +18,7 @@ #include <asm/tlbflush.h> #include "hugetlb_vmemmap.h" -#include "internal.h" -#include "mm_init.h" +#include "sparse.h" /** * struct vmemmap_remap_walk - walk vmemmap page table @@ -706,95 +705,18 @@ void hugetlb_vmemmap_optimize_bootmem_folios(struct hstate *h, struct list_head __hugetlb_vmemmap_optimize_folios(h, folio_list, true); } -#ifdef CONFIG_SPARSEMEM_VMEMMAP_PREINIT - -/* Return true of a bootmem allocated HugeTLB page should be pre-HVO-ed */ -static bool vmemmap_should_optimize_bootmem_page(struct huge_bootmem_page *m) +void __init hugetlb_vmemmap_optimize_bootmem_page(struct huge_bootmem_page *m) { - unsigned long section_size, psize, pmd_vmemmap_size; - phys_addr_t paddr; - - if (!READ_ONCE(vmemmap_optimize_enabled)) - return false; - - if (!hugetlb_vmemmap_optimizable(m->hstate)) - return false; - - psize = huge_page_size(m->hstate); - paddr = virt_to_phys(m); - - /* - * Pre-HVO only works if the bootmem huge page - * is aligned to the section size. - */ - section_size = (1UL << PA_SECTION_SHIFT); - if (!IS_ALIGNED(paddr, section_size) || - !IS_ALIGNED(psize, section_size)) - return false; - - /* - * The pre-HVO code does not deal with splitting PMDS, - * so the bootmem page must be aligned to the number - * of base pages that can be mapped with one vmemmap PMD. - */ - pmd_vmemmap_size = (PMD_SIZE / (sizeof(struct page))) << PAGE_SHIFT; - if (!IS_ALIGNED(paddr, pmd_vmemmap_size) || - !IS_ALIGNED(psize, pmd_vmemmap_size)) - return false; - - return true; -} - -/* - * Initialize memmap section for a gigantic page, HVO-style. - */ -void __init hugetlb_vmemmap_init_early(int nid) -{ - unsigned long psize, paddr, section_size; - unsigned long ns, i, pnum, pfn, nr_pages; - unsigned long start, end; - struct huge_bootmem_page *m = NULL; - void *map; + struct hstate *h = m->hstate; + unsigned long pfn = PHYS_PFN(__pa(m)); if (!READ_ONCE(vmemmap_optimize_enabled)) return; - section_size = (1UL << PA_SECTION_SHIFT); - - list_for_each_entry(m, &huge_boot_pages[nid], list) { - struct zone *zone; - - if (!vmemmap_should_optimize_bootmem_page(m)) - continue; - - nr_pages = pages_per_huge_page(m->hstate); - psize = nr_pages << PAGE_SHIFT; - paddr = virt_to_phys(m); - pfn = PHYS_PFN(paddr); - map = pfn_to_page(pfn); - start = (unsigned long)map; - end = start + hugetlb_vmemmap_size(m->hstate); - zone = pfn_to_zone(pfn, nid); - - if (vmemmap_populate_hvo(start, end, huge_page_order(m->hstate), - zone, HUGETLB_VMEMMAP_RESERVE_SIZE)) - panic("Failed to allocate memmap for HugeTLB page\n"); - memmap_boot_pages_add(DIV_ROUND_UP(HUGETLB_VMEMMAP_RESERVE_SIZE, PAGE_SIZE)); - - pnum = pfn_to_section_nr(pfn); - ns = psize / section_size; - - for (i = 0; i < ns; i++) { - sparse_init_early_section(nid, map, pnum, - SECTION_IS_VMEMMAP_PREINIT); - map += section_map_size(); - pnum++; - } - + section_set_order_range(pfn, pages_per_huge_page(h), huge_page_order(h)); + if (vmemmap_optimizable_order(pfn_to_section_order(pfn))) m->flags |= HUGE_BOOTMEM_HVO; - } } -#endif static const struct ctl_table hugetlb_vmemmap_sysctls[] = { { diff --git a/mm/hugetlb_vmemmap.h b/mm/hugetlb_vmemmap.h index 7ac49c52457d..20eb03df542a 100644 --- a/mm/hugetlb_vmemmap.h +++ b/mm/hugetlb_vmemmap.h @@ -9,8 +9,7 @@ #ifndef _LINUX_HUGETLB_VMEMMAP_H #define _LINUX_HUGETLB_VMEMMAP_H #include <linux/hugetlb.h> -#include <linux/io.h> -#include <linux/memblock.h> +#include "internal.h" /* * Reserve one vmemmap page, all vmemmap addresses are mapped to it. See @@ -27,10 +26,7 @@ long hugetlb_vmemmap_restore_folios(const struct hstate *h, void hugetlb_vmemmap_optimize_folio(const struct hstate *h, struct folio *folio); void hugetlb_vmemmap_optimize_folios(struct hstate *h, struct list_head *folio_list); void hugetlb_vmemmap_optimize_bootmem_folios(struct hstate *h, struct list_head *folio_list); -#ifdef CONFIG_SPARSEMEM_VMEMMAP_PREINIT -void hugetlb_vmemmap_init_early(int nid); -#endif - +void hugetlb_vmemmap_optimize_bootmem_page(struct huge_bootmem_page *m); static inline unsigned int hugetlb_vmemmap_size(const struct hstate *h) { @@ -76,13 +72,13 @@ static inline void hugetlb_vmemmap_optimize_bootmem_folios(struct hstate *h, { } -static inline void hugetlb_vmemmap_init_early(int nid) +static inline unsigned int hugetlb_vmemmap_optimizable_size(const struct hstate *h) { + return 0; } -static inline unsigned int hugetlb_vmemmap_optimizable_size(const struct hstate *h) +static inline void hugetlb_vmemmap_optimize_bootmem_page(struct huge_bootmem_page *m) { - return 0; } #endif /* CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP */ diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index b69a7af76858..7759f9de748c 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -32,8 +32,6 @@ #include <asm/dma.h> #include <asm/tlbflush.h> -#include "hugetlb_vmemmap.h" - /* * Flags for vmemmap_populate_range and friends. */ @@ -369,34 +367,6 @@ void vmemmap_wrprotect_hvo(unsigned long addr, unsigned long end, } } -#ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP -int __meminit vmemmap_populate_hvo(unsigned long addr, unsigned long end, - unsigned int order, struct zone *zone, - unsigned long headsize) -{ - unsigned long maddr; - struct page *tail; - pte_t *pte; - int node = zone_to_nid(zone); - - tail = vmemmap_get_tail(order, zone); - if (!tail) - return -ENOMEM; - - for (maddr = addr; maddr < addr + headsize; maddr += PAGE_SIZE) { - pte = vmemmap_populate_address(maddr, node, NULL, -1, 0); - if (!pte) - return -ENOMEM; - } - - /* - * Reuse the last page struct page mapped above for the rest. - */ - return vmemmap_populate_range(maddr, end, node, NULL, - page_to_pfn(tail), 0); -} -#endif - void __weak __meminit vmemmap_set_pmd(pmd_t *pmd, void *p, int node, unsigned long addr, unsigned long next) { @@ -599,7 +569,6 @@ struct page * __meminit __populate_section_memmap(unsigned long pfn, */ void __init sparse_vmemmap_init_nid_early(int nid) { - hugetlb_vmemmap_init_early(nid); } #endif diff --git a/mm/sparse.h b/mm/sparse.h index 16b9bd3070aa..bc4c58ef24a9 100644 --- a/mm/sparse.h +++ b/mm/sparse.h @@ -16,6 +16,24 @@ static inline unsigned int section_order(const struct mem_section *section) return section->order; } +static inline void section_set_order(struct mem_section *section, unsigned int order) +{ + VM_WARN_ON(section_order(section) && order && section_order(section) != order); + section->order = order; +} + +static inline void section_set_order_range(unsigned long pfn, unsigned long nr_pages, + unsigned int order) +{ + unsigned long section_nr = pfn_to_section_nr(pfn); + + if (!IS_ALIGNED(pfn | nr_pages, PAGES_PER_SECTION)) + return; + + for (unsigned long i = 0; i < nr_pages / PAGES_PER_SECTION; i++) + section_set_order(__nr_to_section(section_nr + i), order); +} + static inline unsigned int pfn_to_section_order(unsigned long pfn) { return section_order(__pfn_to_section(pfn)); @@ -26,6 +44,15 @@ static inline unsigned int section_order(const struct mem_section *section) return 0; } +static inline void section_set_order(struct mem_section *section, unsigned int order) +{ +} + +static inline void section_set_order_range(unsigned long pfn, unsigned long nr_pages, + unsigned int order) +{ +} + static inline unsigned int pfn_to_section_order(unsigned long pfn) { return 0; -- 2.54.0