The Linux Kernel Mailing List
 help / color / mirror / Atom feed
From: Muchun Song <muchun.song@linux.dev>
To: Muchun Song <songmuchun@bytedance.com>
Cc: Mike Rapoport <rppt@kernel.org>,
	Vlastimil Babka <vbabka@kernel.org>,
	Lorenzo Stoakes <ljs@kernel.org>, Michal Hocko <mhocko@suse.com>,
	David Laight <david.laight.linux@gmail.com>,
	linux-mm@kvack.org, linux-kernel@vger.kernel.org,
	Andrew Morton <akpm@linux-foundation.org>,
	Oscar Salvador <osalvador@suse.de>,
	David Hildenbrand <david@kernel.org>
Subject: Re: [PATCH v3 10/17] mm/hugetlb: switch HugeTLB to section-based vmemmap optimization
Date: Tue, 11 Aug 2026 11:58:59 +0800	[thread overview]
Message-ID: <47182b7a-b388-4aff-bb22-1da35d9993cd@linux.dev> (raw)
In-Reply-To: <20260804035535.2846016-11-songmuchun@bytedance.com>



On 2026/8/4 11:55, Muchun Song wrote:
> HugeTLB bootmem vmemmap optimization still carries its own early setup
> path, including pre-populating optimized mappings before the generic
> sparse-vmemmap code runs.
>
> Now that the section-based vmemmap optimization can derive HugeTLB
> vmemmap deduplication from section metadata, HugeTLB only needs to mark
> the bootmem huge page range with the appropriate order. The generic
> sparse-vmemmap population path can then allocate and map the shared tail
> vmemmap pages without any HugeTLB-specific early population code.
>
> Do that by setting the section order when a bootmem huge page is
> allocated and dropping the dedicated pre-HVO helpers and related
> special-casing.
>
> This removes duplicate early setup logic and switches HugeTLB to the
> section-based vmemmap optimization path.
>
> Signed-off-by: Muchun Song <songmuchun@bytedance.com>
> Acked-by: Mike Rapoport (Microsoft) <rppt@kernel.org>
> ---
> v3:
> - Use the order-based helper for the bootmem vmemmap-optimized check
>
> v2:
> - Collect Acked-by from Mike Rapoport
> ---
>   include/linux/hugetlb.h |  1 -
>   include/linux/mm.h      |  3 --
>   mm/hugetlb.c            | 30 ++------------
>   mm/hugetlb_vmemmap.c    | 90 +++--------------------------------------
>   mm/hugetlb_vmemmap.h    | 14 +++----
>   mm/sparse-vmemmap.c     | 31 --------------
>   mm/sparse.h             | 27 +++++++++++++
>   7 files changed, 42 insertions(+), 154 deletions(-)
>
> diff --git a/include/linux/hugetlb.h b/include/linux/hugetlb.h
> index 16c4c4caa126..fe28f98e1b22 100644
> --- a/include/linux/hugetlb.h
> +++ b/include/linux/hugetlb.h
> @@ -171,7 +171,6 @@ struct address_space *hugetlb_folio_mapping_lock_write(struct folio *folio);
>   
>   extern int movable_gigantic_pages __read_mostly;
>   extern int sysctl_hugetlb_shm_group __read_mostly;
> -extern struct list_head huge_boot_pages[MAX_NUMNODES];
>   
>   void hugetlb_bootmem_struct_page_init(void);
>   void hugetlb_bootmem_alloc(void);
> diff --git a/include/linux/mm.h b/include/linux/mm.h
> index 7fabe6c66b4b..fd7dc85f58f3 100644
> --- a/include/linux/mm.h
> +++ b/include/linux/mm.h
> @@ -5097,9 +5097,6 @@ int vmemmap_populate_hugepages(unsigned long start, unsigned long end,
>   			       int node, struct vmem_altmap *altmap);
>   int vmemmap_populate(unsigned long start, unsigned long end, int node,
>   		struct vmem_altmap *altmap);
> -int vmemmap_populate_hvo(unsigned long start, unsigned long end,
> -			 unsigned int order, struct zone *zone,
> -			 unsigned long headsize);
>   void vmemmap_wrprotect_hvo(unsigned long start, unsigned long end, int node,
>   			  unsigned long headsize);
>   void vmemmap_populate_print_last(void);
> diff --git a/mm/hugetlb.c b/mm/hugetlb.c
> index d212bbff4c83..7d8507aa4c5b 100644
> --- a/mm/hugetlb.c
> +++ b/mm/hugetlb.c
> @@ -52,6 +52,7 @@
>   #include "hugetlb_cma.h"
>   #include "hugetlb_internal.h"
>   #include "mm_init.h"
> +#include "sparse.h"
>   #include <linux/page-isolation.h>
>   
>   int hugetlb_max_hstate __read_mostly;
> @@ -59,7 +60,7 @@ unsigned int default_hstate_idx;
>   struct hstate hstates[HUGE_MAX_HSTATE];
>   
>   __initdata nodemask_t hugetlb_bootmem_nodes;
> -__initdata struct list_head huge_boot_pages[MAX_NUMNODES];
> +static struct list_head huge_boot_pages[MAX_NUMNODES] __initdata;
>   
>   /*
>    * Due to ordering constraints across the init code for various
> @@ -3137,6 +3138,7 @@ static bool __init alloc_bootmem_huge_page(struct hstate *h, int nid)
>   	} else {
>   		list_add_tail(&m->list, &huge_boot_pages[nid]);
>   		m->flags |= HUGE_BOOTMEM_ZONES_VALID;
> +		hugetlb_vmemmap_optimize_bootmem_page(m);
>   		/*
>   		 * Only initialize the head struct page in memmap_init_reserved_pages,
>   		 * rest of the struct pages will be initialized by the HugeTLB
> @@ -3297,6 +3299,7 @@ static void __init gather_bootmem_prealloc_node(unsigned long nid)
>   			 * this folio.
>   			 */
>   			folio_set_hugetlb_vmemmap_optimized(folio);
> +		section_set_order_range(folio_pfn(folio), folio_nr_pages(folio), 0);
>   
>   		if (hugetlb_bootmem_page_earlycma(m))
>   			folio_set_hugetlb_cma(folio);
> @@ -3340,31 +3343,6 @@ void __init hugetlb_bootmem_struct_page_init(void)
>   		.max_threads	= num_node_state(N_MEMORY),
>   		.numa_aware	= true,
>   	};
> -#ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP
> -	struct zone *zone;
> -
> -	for_each_zone(zone) {
> -		for (int i = 0; i < VMEMMAP_OPTIMIZATION_NR_ORDERS; i++) {
> -			struct page *tail, *p;
> -			unsigned int order;
> -
> -			tail = zone->vmemmap_tails[i];
> -			if (!tail)
> -				continue;
> -
> -			order = i + VMEMMAP_OPTIMIZATION_MIN_ORDER;
> -			p = page_to_virt(tail);
> -			/*
> -			 * prep_and_add_bootmem_folios() can access pageblock
> -			 * flags on bootmem HugeTLB pages, so initialize the
> -			 * shared tail struct pages here before bootmem folios
> -			 * start using them.
> -			 */
> -			for (int j = 0; j < PAGE_SIZE / sizeof(struct page); j++)
> -				init_compound_tail(p + j, NULL, order, zone);
> -		}
> -	}
> -#endif
>   
>   	padata_do_multithreaded(&job);
>   }
> diff --git a/mm/hugetlb_vmemmap.c b/mm/hugetlb_vmemmap.c
> index c48fcea076a5..7293706b532f 100644
> --- a/mm/hugetlb_vmemmap.c
> +++ b/mm/hugetlb_vmemmap.c
> @@ -18,8 +18,7 @@
>   
>   #include <asm/tlbflush.h>
>   #include "hugetlb_vmemmap.h"
> -#include "internal.h"
> -#include "mm_init.h"
> +#include "sparse.h"
>   
>   /**
>    * struct vmemmap_remap_walk - walk vmemmap page table
> @@ -706,95 +705,18 @@ void hugetlb_vmemmap_optimize_bootmem_folios(struct hstate *h, struct list_head
>   	__hugetlb_vmemmap_optimize_folios(h, folio_list, true);
>   }
>   
> -#ifdef CONFIG_SPARSEMEM_VMEMMAP_PREINIT
> -
> -/* Return true of a bootmem allocated HugeTLB page should be pre-HVO-ed */
> -static bool vmemmap_should_optimize_bootmem_page(struct huge_bootmem_page *m)
> +void __init hugetlb_vmemmap_optimize_bootmem_page(struct huge_bootmem_page *m)
>   {
> -	unsigned long section_size, psize, pmd_vmemmap_size;
> -	phys_addr_t paddr;
> -
> -	if (!READ_ONCE(vmemmap_optimize_enabled))
> -		return false;
> -
> -	if (!hugetlb_vmemmap_optimizable(m->hstate))
> -		return false;
> -
> -	psize = huge_page_size(m->hstate);
> -	paddr = virt_to_phys(m);
> -
> -	/*
> -	 * Pre-HVO only works if the bootmem huge page
> -	 * is aligned to the section size.
> -	 */
> -	section_size = (1UL << PA_SECTION_SHIFT);
> -	if (!IS_ALIGNED(paddr, section_size) ||
> -	    !IS_ALIGNED(psize, section_size))
> -		return false;
> -
> -	/*
> -	 * The pre-HVO code does not deal with splitting PMDS,
> -	 * so the bootmem page must be aligned to the number
> -	 * of base pages that can be mapped with one vmemmap PMD.
> -	 */
> -	pmd_vmemmap_size = (PMD_SIZE / (sizeof(struct page))) << PAGE_SHIFT;
> -	if (!IS_ALIGNED(paddr, pmd_vmemmap_size) ||
> -	    !IS_ALIGNED(psize, pmd_vmemmap_size))
> -		return false;
> -
> -	return true;
> -}
> -
> -/*
> - * Initialize memmap section for a gigantic page, HVO-style.
> - */
> -void __init hugetlb_vmemmap_init_early(int nid)
> -{
> -	unsigned long psize, paddr, section_size;
> -	unsigned long ns, i, pnum, pfn, nr_pages;
> -	unsigned long start, end;
> -	struct huge_bootmem_page *m = NULL;
> -	void *map;
> +	struct hstate *h = m->hstate;
> +	unsigned long pfn = PHYS_PFN(__pa(m));
>   
>   	if (!READ_ONCE(vmemmap_optimize_enabled))
>   		return;
>   
> -	section_size = (1UL << PA_SECTION_SHIFT);
> -
> -	list_for_each_entry(m, &huge_boot_pages[nid], list) {
> -		struct zone *zone;
> -
> -		if (!vmemmap_should_optimize_bootmem_page(m))
> -			continue;
> -
> -		nr_pages = pages_per_huge_page(m->hstate);
> -		psize = nr_pages << PAGE_SHIFT;
> -		paddr = virt_to_phys(m);
> -		pfn = PHYS_PFN(paddr);
> -		map = pfn_to_page(pfn);
> -		start = (unsigned long)map;
> -		end = start + hugetlb_vmemmap_size(m->hstate);
> -		zone = pfn_to_zone(pfn, nid);
> -
> -		if (vmemmap_populate_hvo(start, end, huge_page_order(m->hstate),
> -					 zone, HUGETLB_VMEMMAP_RESERVE_SIZE))
> -			panic("Failed to allocate memmap for HugeTLB page\n");
> -		memmap_boot_pages_add(DIV_ROUND_UP(HUGETLB_VMEMMAP_RESERVE_SIZE, PAGE_SIZE));
> -
> -		pnum = pfn_to_section_nr(pfn);
> -		ns = psize / section_size;
> -
> -		for (i = 0; i < ns; i++) {
> -			sparse_init_early_section(nid, map, pnum,
> -					SECTION_IS_VMEMMAP_PREINIT);
> -			map += section_map_size();
> -			pnum++;
> -		}
> -
> +	section_set_order_range(pfn, pages_per_huge_page(h), huge_page_order(h));
> +	if (vmemmap_optimizable_order(pfn_to_section_order(pfn)))
>   		m->flags |= HUGE_BOOTMEM_HVO;
> -	}

Sashiko suggested that removing the pmd_vmemmap_size alignment
check might cause a kernel panic on architectures where a vmemmap
PMD covers multiple memory sections (e.g., LoongArch with 64KB pages).
However, this is a false positive, because LoongArch does not support
gigantic HugeTLB pages and therefore does not rely on the section-based
vmemmap optimization infrastructure at all.

Muchun,
Thanks.

>   }
> -#endif
>   
>   static const struct ctl_table hugetlb_vmemmap_sysctls[] = {
>   	{
> diff --git a/mm/hugetlb_vmemmap.h b/mm/hugetlb_vmemmap.h
> index 7ac49c52457d..20eb03df542a 100644
> --- a/mm/hugetlb_vmemmap.h
> +++ b/mm/hugetlb_vmemmap.h
> @@ -9,8 +9,7 @@
>   #ifndef _LINUX_HUGETLB_VMEMMAP_H
>   #define _LINUX_HUGETLB_VMEMMAP_H
>   #include <linux/hugetlb.h>
> -#include <linux/io.h>
> -#include <linux/memblock.h>
> +#include "internal.h"
>   
>   /*
>    * Reserve one vmemmap page, all vmemmap addresses are mapped to it. See
> @@ -27,10 +26,7 @@ long hugetlb_vmemmap_restore_folios(const struct hstate *h,
>   void hugetlb_vmemmap_optimize_folio(const struct hstate *h, struct folio *folio);
>   void hugetlb_vmemmap_optimize_folios(struct hstate *h, struct list_head *folio_list);
>   void hugetlb_vmemmap_optimize_bootmem_folios(struct hstate *h, struct list_head *folio_list);
> -#ifdef CONFIG_SPARSEMEM_VMEMMAP_PREINIT
> -void hugetlb_vmemmap_init_early(int nid);
> -#endif
> -
> +void hugetlb_vmemmap_optimize_bootmem_page(struct huge_bootmem_page *m);
>   
>   static inline unsigned int hugetlb_vmemmap_size(const struct hstate *h)
>   {
> @@ -76,13 +72,13 @@ static inline void hugetlb_vmemmap_optimize_bootmem_folios(struct hstate *h,
>   {
>   }
>   
> -static inline void hugetlb_vmemmap_init_early(int nid)
> +static inline unsigned int hugetlb_vmemmap_optimizable_size(const struct hstate *h)
>   {
> +	return 0;
>   }
>   
> -static inline unsigned int hugetlb_vmemmap_optimizable_size(const struct hstate *h)
> +static inline void hugetlb_vmemmap_optimize_bootmem_page(struct huge_bootmem_page *m)
>   {
> -	return 0;
>   }
>   #endif /* CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP */
>   
> diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c
> index b69a7af76858..7759f9de748c 100644
> --- a/mm/sparse-vmemmap.c
> +++ b/mm/sparse-vmemmap.c
> @@ -32,8 +32,6 @@
>   #include <asm/dma.h>
>   #include <asm/tlbflush.h>
>   
> -#include "hugetlb_vmemmap.h"
> -
>   /*
>    * Flags for vmemmap_populate_range and friends.
>    */
> @@ -369,34 +367,6 @@ void vmemmap_wrprotect_hvo(unsigned long addr, unsigned long end,
>   	}
>   }
>   
> -#ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP
> -int __meminit vmemmap_populate_hvo(unsigned long addr, unsigned long end,
> -				       unsigned int order, struct zone *zone,
> -				       unsigned long headsize)
> -{
> -	unsigned long maddr;
> -	struct page *tail;
> -	pte_t *pte;
> -	int node = zone_to_nid(zone);
> -
> -	tail = vmemmap_get_tail(order, zone);
> -	if (!tail)
> -		return -ENOMEM;
> -
> -	for (maddr = addr; maddr < addr + headsize; maddr += PAGE_SIZE) {
> -		pte = vmemmap_populate_address(maddr, node, NULL, -1, 0);
> -		if (!pte)
> -			return -ENOMEM;
> -	}
> -
> -	/*
> -	 * Reuse the last page struct page mapped above for the rest.
> -	 */
> -	return vmemmap_populate_range(maddr, end, node, NULL,
> -				      page_to_pfn(tail), 0);
> -}
> -#endif
> -
>   void __weak __meminit vmemmap_set_pmd(pmd_t *pmd, void *p, int node,
>   				      unsigned long addr, unsigned long next)
>   {
> @@ -599,7 +569,6 @@ struct page * __meminit __populate_section_memmap(unsigned long pfn,
>    */
>   void __init sparse_vmemmap_init_nid_early(int nid)
>   {
> -	hugetlb_vmemmap_init_early(nid);
>   }
>   #endif
>   
> diff --git a/mm/sparse.h b/mm/sparse.h
> index 16b9bd3070aa..bc4c58ef24a9 100644
> --- a/mm/sparse.h
> +++ b/mm/sparse.h
> @@ -16,6 +16,24 @@ static inline unsigned int section_order(const struct mem_section *section)
>   	return section->order;
>   }
>   
> +static inline void section_set_order(struct mem_section *section, unsigned int order)
> +{
> +	VM_WARN_ON(section_order(section) && order && section_order(section) != order);
> +	section->order = order;
> +}
> +
> +static inline void section_set_order_range(unsigned long pfn, unsigned long nr_pages,
> +					   unsigned int order)
> +{
> +	unsigned long section_nr = pfn_to_section_nr(pfn);
> +
> +	if (!IS_ALIGNED(pfn | nr_pages, PAGES_PER_SECTION))
> +		return;
> +
> +	for (unsigned long i = 0; i < nr_pages / PAGES_PER_SECTION; i++)
> +		section_set_order(__nr_to_section(section_nr + i), order);
> +}
> +
>   static inline unsigned int pfn_to_section_order(unsigned long pfn)
>   {
>   	return section_order(__pfn_to_section(pfn));
> @@ -26,6 +44,15 @@ static inline unsigned int section_order(const struct mem_section *section)
>   	return 0;
>   }
>   
> +static inline void section_set_order(struct mem_section *section, unsigned int order)
> +{
> +}
> +
> +static inline void section_set_order_range(unsigned long pfn, unsigned long nr_pages,
> +					   unsigned int order)
> +{
> +}
> +
>   static inline unsigned int pfn_to_section_order(unsigned long pfn)
>   {
>   	return 0;


  reply	other threads:[~2026-08-11  3:59 UTC|newest]

Thread overview: 23+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-08-04  3:55 [PATCH v3 00/17] mm: Introduce section-based vmemmap optimization for HugeTLB Muchun Song
2026-08-04  3:55 ` [PATCH v3 01/17] mm/sparse: relax struct mem_section size constraints Muchun Song
2026-08-04  3:55 ` [PATCH v3 02/17] mm/sparse-vmemmap: rename HVO order macros Muchun Song
2026-08-04  3:55 ` [PATCH v3 03/17] mm/mm_init: skip initializing shared vmemmap tail pages Muchun Song
2026-08-11  3:40   ` Muchun Song
2026-08-04  3:55 ` [PATCH v3 04/17] mm/sparse-vmemmap: initialize shared tail vmemmap pages on allocation Muchun Song
2026-08-04  3:55 ` [PATCH v3 05/17] mm/sparse-vmemmap: support section-based vmemmap accounting Muchun Song
2026-08-11  3:45   ` Muchun Song
2026-08-04  3:55 ` [PATCH v3 06/17] mm/mm_init: factor out pfn_to_zone() Muchun Song
2026-08-04  3:55 ` [PATCH v3 07/17] mm/sparse-vmemmap: move vmemmap_get_tail() before PTE population Muchun Song
2026-08-04  3:55 ` [PATCH v3 08/17] mm/sparse-vmemmap: support section-based vmemmap optimization Muchun Song
2026-08-06  3:04   ` Muchun Song
2026-08-11  3:54   ` Muchun Song
2026-08-04  3:55 ` [PATCH v3 09/17] mm/sparse: initialize memory sections earlier Muchun Song
2026-08-04  3:55 ` [PATCH v3 10/17] mm/hugetlb: switch HugeTLB to section-based vmemmap optimization Muchun Song
2026-08-11  3:58   ` Muchun Song [this message]
2026-08-04  3:55 ` [PATCH v3 11/17] mm/sparse-vmemmap: remove SPARSEMEM_VMEMMAP_PREINIT support Muchun Song
2026-08-04  3:55 ` [PATCH v3 12/17] mm/sparse: inline usemap allocation into sparse_init_nid() Muchun Song
2026-08-04  3:55 ` [PATCH v3 13/17] mm/sparse: remove section_map_size() Muchun Song
2026-08-04  3:55 ` [PATCH v3 14/17] mm/hugetlb: remove HUGE_BOOTMEM_HVO Muchun Song
2026-08-04  3:55 ` [PATCH v3 15/17] mm/hugetlb: remove HUGE_BOOTMEM_CMA Muchun Song
2026-08-04  3:55 ` [PATCH v3 16/17] mm/hugetlb: localize struct huge_bootmem_page Muchun Song
2026-08-04  3:55 ` [PATCH v3 17/17] mm/hugetlb: localize HUGE_BOOTMEM_ZONES_VALID Muchun Song

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=47182b7a-b388-4aff-bb22-1da35d9993cd@linux.dev \
    --to=muchun.song@linux.dev \
    --cc=akpm@linux-foundation.org \
    --cc=david.laight.linux@gmail.com \
    --cc=david@kernel.org \
    --cc=linux-kernel@vger.kernel.org \
    --cc=linux-mm@kvack.org \
    --cc=ljs@kernel.org \
    --cc=mhocko@suse.com \
    --cc=osalvador@suse.de \
    --cc=rppt@kernel.org \
    --cc=songmuchun@bytedance.com \
    --cc=vbabka@kernel.org \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox