All of lore.kernel.org
 help / color / mirror / Atom feed
From: Qi Zheng <qi.zheng@linux.dev>
To: Muchun Song <songmuchun@bytedance.com>,
	Andrew Morton <akpm@linux-foundation.org>,
	Oscar Salvador <osalvador@suse.de>,
	David Hildenbrand <david@kernel.org>
Cc: Mike Rapoport <rppt@kernel.org>,
	Vlastimil Babka <vbabka@kernel.org>,
	Lorenzo Stoakes <ljs@kernel.org>, Michal Hocko <mhocko@suse.com>,
	David Laight <david.laight.linux@gmail.com>,
	"Liam R . Howlett" <liam@infradead.org>,
	Suren Baghdasaryan <surenb@google.com>,
	linux-mm@kvack.org, linux-kernel@vger.kernel.org,
	Muchun Song <muchun.song@linux.dev>
Subject: Re: [PATCH v5 10/17] mm/hugetlb: switch HugeTLB to section-based vmemmap optimization
Date: Tue, 25 Aug 2026 21:12:34 +0800	[thread overview]
Message-ID: <83d8d1f2-6ca2-4c72-a494-23ce07bb4181@linux.dev> (raw)
In-Reply-To: <20260825084608.47437-11-songmuchun@bytedance.com>



On 8/25/26 4:46 PM, Muchun Song wrote:
> HugeTLB bootmem vmemmap optimization still carries its own early setup
> path, including pre-populating optimized mappings before the generic
> sparse-vmemmap code runs.
> 
> Now that the section-based vmemmap optimization can derive HugeTLB
> vmemmap deduplication from section metadata, HugeTLB only needs to mark
> the bootmem huge page range with the appropriate order. The generic
> sparse-vmemmap population path can then allocate and map the shared tail
> vmemmap pages without any HugeTLB-specific early population code.
> 
> Do that by setting the section order when a bootmem huge page is
> allocated and dropping the dedicated pre-HVO helpers and related
> special-casing.
> 
> This removes duplicate early setup logic and switches HugeTLB to the
> section-based vmemmap optimization path.
> 
> Signed-off-by: Muchun Song <songmuchun@bytedance.com>
> Acked-by: Mike Rapoport (Microsoft) <rppt@kernel.org>
> ---
> v3:
> - Use the order-based helper for the bootmem vmemmap-optimized check
> 
> v2:
> - Collect Acked-by from Mike Rapoport
> ---
>   include/linux/hugetlb.h |  1 -
>   include/linux/mm.h      |  3 --
>   mm/hugetlb.c            | 30 ++------------
>   mm/hugetlb_vmemmap.c    | 90 +++--------------------------------------
>   mm/hugetlb_vmemmap.h    | 14 +++----
>   mm/sparse-vmemmap.c     | 31 --------------
>   mm/sparse.h             | 27 +++++++++++++
>   7 files changed, 42 insertions(+), 154 deletions(-)
> 
> diff --git a/include/linux/hugetlb.h b/include/linux/hugetlb.h
> index 16c4c4caa126..fe28f98e1b22 100644
> --- a/include/linux/hugetlb.h
> +++ b/include/linux/hugetlb.h
> @@ -171,7 +171,6 @@ struct address_space *hugetlb_folio_mapping_lock_write(struct folio *folio);
>   
>   extern int movable_gigantic_pages __read_mostly;
>   extern int sysctl_hugetlb_shm_group __read_mostly;
> -extern struct list_head huge_boot_pages[MAX_NUMNODES];
>   
>   void hugetlb_bootmem_struct_page_init(void);
>   void hugetlb_bootmem_alloc(void);
> diff --git a/include/linux/mm.h b/include/linux/mm.h
> index dd09c438fa23..441bd39eab73 100644
> --- a/include/linux/mm.h
> +++ b/include/linux/mm.h
> @@ -5159,9 +5159,6 @@ int vmemmap_populate_hugepages(unsigned long start, unsigned long end,
>   			       int node, struct vmem_altmap *altmap);
>   int vmemmap_populate(unsigned long start, unsigned long end, int node,
>   		struct vmem_altmap *altmap);
> -int vmemmap_populate_hvo(unsigned long start, unsigned long end,
> -			 unsigned int order, struct zone *zone,
> -			 unsigned long headsize);
>   void vmemmap_wrprotect_hvo(unsigned long start, unsigned long end, int node,
>   			  unsigned long headsize);
>   void vmemmap_populate_print_last(void);
> diff --git a/mm/hugetlb.c b/mm/hugetlb.c
> index 04e6c4244cd6..fbb0c83bea79 100644
> --- a/mm/hugetlb.c
> +++ b/mm/hugetlb.c
> @@ -52,6 +52,7 @@
>   #include "hugetlb_cma.h"
>   #include "hugetlb_internal.h"
>   #include "mm_init.h"
> +#include "sparse.h"
>   #include <linux/page-isolation.h>
>   
>   int hugetlb_max_hstate __read_mostly;
> @@ -59,7 +60,7 @@ unsigned int default_hstate_idx;
>   struct hstate hstates[HUGE_MAX_HSTATE];
>   
>   __initdata nodemask_t hugetlb_bootmem_nodes;
> -__initdata struct list_head huge_boot_pages[MAX_NUMNODES];
> +static struct list_head huge_boot_pages[MAX_NUMNODES] __initdata;
>   
>   /*
>    * Due to ordering constraints across the init code for various
> @@ -3139,6 +3140,7 @@ static bool __init alloc_bootmem_huge_page(struct hstate *h, int nid)
>   	} else {
>   		list_add_tail(&m->list, &huge_boot_pages[nid]);
>   		m->flags |= HUGE_BOOTMEM_ZONES_VALID;
> +		hugetlb_vmemmap_optimize_bootmem_page(m);
>   		/*
>   		 * Only initialize the head struct page in memmap_init_reserved_pages,
>   		 * rest of the struct pages will be initialized by the HugeTLB
> @@ -3299,6 +3301,7 @@ static void __init gather_bootmem_prealloc_node(unsigned long nid)
>   			 * this folio.
>   			 */
>   			folio_set_hugetlb_vmemmap_optimized(folio);
> +		section_set_order_range(folio_pfn(folio), folio_nr_pages(folio), 0);

So section->order is only used during initialization. Is it ever
accessed later at runtime? If not, can we just skip zeroing it out?

>   
>   		if (hugetlb_bootmem_page_earlycma(m))
>   			folio_set_hugetlb_cma(folio);
> @@ -3342,31 +3345,6 @@ void __init hugetlb_bootmem_struct_page_init(void)
>   		.max_threads	= num_node_state(N_MEMORY),
>   		.numa_aware	= true,
>   	};
> -#ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP
> -	struct zone *zone;
> -
> -	for_each_zone(zone) {
> -		for (int i = 0; i < VMEMMAP_OPTIMIZATION_NR_ORDERS; i++) {
> -			struct page *tail, *p;
> -			unsigned int order;
> -
> -			tail = zone->vmemmap_tails[i];
> -			if (!tail)
> -				continue;
> -
> -			order = i + VMEMMAP_OPTIMIZATION_MIN_ORDER;
> -			p = page_to_virt(tail);
> -			/*
> -			 * prep_and_add_bootmem_folios() can access pageblock
> -			 * flags on bootmem HugeTLB pages, so initialize the
> -			 * shared tail struct pages here before bootmem folios
> -			 * start using them.
> -			 */
> -			for (int j = 0; j < PAGE_SIZE / sizeof(struct page); j++)
> -				init_compound_tail(p + j, NULL, order, zone);
> -		}
> -	}
> -#endif
>   
>   	padata_do_multithreaded(&job);
>   }
> diff --git a/mm/hugetlb_vmemmap.c b/mm/hugetlb_vmemmap.c
> index c48fcea076a5..7293706b532f 100644
> --- a/mm/hugetlb_vmemmap.c
> +++ b/mm/hugetlb_vmemmap.c
> @@ -18,8 +18,7 @@
>   
>   #include <asm/tlbflush.h>
>   #include "hugetlb_vmemmap.h"
> -#include "internal.h"
> -#include "mm_init.h"
> +#include "sparse.h"
>   
>   /**
>    * struct vmemmap_remap_walk - walk vmemmap page table
> @@ -706,95 +705,18 @@ void hugetlb_vmemmap_optimize_bootmem_folios(struct hstate *h, struct list_head
>   	__hugetlb_vmemmap_optimize_folios(h, folio_list, true);
>   }
>   
> -#ifdef CONFIG_SPARSEMEM_VMEMMAP_PREINIT
> -
> -/* Return true of a bootmem allocated HugeTLB page should be pre-HVO-ed */
> -static bool vmemmap_should_optimize_bootmem_page(struct huge_bootmem_page *m)
> +void __init hugetlb_vmemmap_optimize_bootmem_page(struct huge_bootmem_page *m)
>   {
> -	unsigned long section_size, psize, pmd_vmemmap_size;
> -	phys_addr_t paddr;
> -
> -	if (!READ_ONCE(vmemmap_optimize_enabled))
> -		return false;
> -
> -	if (!hugetlb_vmemmap_optimizable(m->hstate))
> -		return false;
> -
> -	psize = huge_page_size(m->hstate);
> -	paddr = virt_to_phys(m);
> -
> -	/*
> -	 * Pre-HVO only works if the bootmem huge page
> -	 * is aligned to the section size.
> -	 */
> -	section_size = (1UL << PA_SECTION_SHIFT);
> -	if (!IS_ALIGNED(paddr, section_size) ||
> -	    !IS_ALIGNED(psize, section_size))
> -		return false;
> -
> -	/*
> -	 * The pre-HVO code does not deal with splitting PMDS,
> -	 * so the bootmem page must be aligned to the number
> -	 * of base pages that can be mapped with one vmemmap PMD.
> -	 */
> -	pmd_vmemmap_size = (PMD_SIZE / (sizeof(struct page))) << PAGE_SHIFT;
> -	if (!IS_ALIGNED(paddr, pmd_vmemmap_size) ||
> -	    !IS_ALIGNED(psize, pmd_vmemmap_size))
> -		return false;
> -
> -	return true;
> -}
> -
> -/*
> - * Initialize memmap section for a gigantic page, HVO-style.
> - */
> -void __init hugetlb_vmemmap_init_early(int nid)
> -{
> -	unsigned long psize, paddr, section_size;
> -	unsigned long ns, i, pnum, pfn, nr_pages;
> -	unsigned long start, end;
> -	struct huge_bootmem_page *m = NULL;
> -	void *map;
> +	struct hstate *h = m->hstate;
> +	unsigned long pfn = PHYS_PFN(__pa(m));
>   
>   	if (!READ_ONCE(vmemmap_optimize_enabled))
>   		return;
>   
> -	section_size = (1UL << PA_SECTION_SHIFT);
> -
> -	list_for_each_entry(m, &huge_boot_pages[nid], list) {
> -		struct zone *zone;
> -
> -		if (!vmemmap_should_optimize_bootmem_page(m))
> -			continue;
> -
> -		nr_pages = pages_per_huge_page(m->hstate);
> -		psize = nr_pages << PAGE_SHIFT;
> -		paddr = virt_to_phys(m);
> -		pfn = PHYS_PFN(paddr);
> -		map = pfn_to_page(pfn);
> -		start = (unsigned long)map;
> -		end = start + hugetlb_vmemmap_size(m->hstate);
> -		zone = pfn_to_zone(pfn, nid);
> -
> -		if (vmemmap_populate_hvo(start, end, huge_page_order(m->hstate),
> -					 zone, HUGETLB_VMEMMAP_RESERVE_SIZE))
> -			panic("Failed to allocate memmap for HugeTLB page\n");
> -		memmap_boot_pages_add(DIV_ROUND_UP(HUGETLB_VMEMMAP_RESERVE_SIZE, PAGE_SIZE));
> -
> -		pnum = pfn_to_section_nr(pfn);
> -		ns = psize / section_size;
> -
> -		for (i = 0; i < ns; i++) {
> -			sparse_init_early_section(nid, map, pnum,
> -					SECTION_IS_VMEMMAP_PREINIT);
> -			map += section_map_size();
> -			pnum++;
> -		}
> -
> +	section_set_order_range(pfn, pages_per_huge_page(h), huge_page_order(h));
> +	if (vmemmap_optimizable_order(pfn_to_section_order(pfn)))
>   		m->flags |= HUGE_BOOTMEM_HVO;
> -	}
>   }
> -#endif
>   
>   static const struct ctl_table hugetlb_vmemmap_sysctls[] = {
>   	{
> diff --git a/mm/hugetlb_vmemmap.h b/mm/hugetlb_vmemmap.h
> index 7ac49c52457d..20eb03df542a 100644
> --- a/mm/hugetlb_vmemmap.h
> +++ b/mm/hugetlb_vmemmap.h
> @@ -9,8 +9,7 @@
>   #ifndef _LINUX_HUGETLB_VMEMMAP_H
>   #define _LINUX_HUGETLB_VMEMMAP_H
>   #include <linux/hugetlb.h>
> -#include <linux/io.h>
> -#include <linux/memblock.h>
> +#include "internal.h"
>   
>   /*
>    * Reserve one vmemmap page, all vmemmap addresses are mapped to it. See
> @@ -27,10 +26,7 @@ long hugetlb_vmemmap_restore_folios(const struct hstate *h,
>   void hugetlb_vmemmap_optimize_folio(const struct hstate *h, struct folio *folio);
>   void hugetlb_vmemmap_optimize_folios(struct hstate *h, struct list_head *folio_list);
>   void hugetlb_vmemmap_optimize_bootmem_folios(struct hstate *h, struct list_head *folio_list);
> -#ifdef CONFIG_SPARSEMEM_VMEMMAP_PREINIT
> -void hugetlb_vmemmap_init_early(int nid);
> -#endif
> -
> +void hugetlb_vmemmap_optimize_bootmem_page(struct huge_bootmem_page *m);
>   
>   static inline unsigned int hugetlb_vmemmap_size(const struct hstate *h)
>   {
> @@ -76,13 +72,13 @@ static inline void hugetlb_vmemmap_optimize_bootmem_folios(struct hstate *h,
>   {
>   }
>   
> -static inline void hugetlb_vmemmap_init_early(int nid)
> +static inline unsigned int hugetlb_vmemmap_optimizable_size(const struct hstate *h)
>   {
> +	return 0;
>   }
>   
> -static inline unsigned int hugetlb_vmemmap_optimizable_size(const struct hstate *h)
> +static inline void hugetlb_vmemmap_optimize_bootmem_page(struct huge_bootmem_page *m)
>   {
> -	return 0;
>   }
>   #endif /* CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP */
>   
> diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c
> index 8d39196f6d93..e48758805a8f 100644
> --- a/mm/sparse-vmemmap.c
> +++ b/mm/sparse-vmemmap.c
> @@ -32,8 +32,6 @@
>   #include <asm/dma.h>
>   #include <asm/tlbflush.h>
>   
> -#include "hugetlb_vmemmap.h"
> -
>   /*
>    * Flags for vmemmap_populate_range and friends.
>    */
> @@ -404,34 +402,6 @@ void vmemmap_wrprotect_hvo(unsigned long addr, unsigned long end,
>   	}
>   }
>   
> -#ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP
> -int __meminit vmemmap_populate_hvo(unsigned long addr, unsigned long end,
> -				       unsigned int order, struct zone *zone,
> -				       unsigned long headsize)
> -{
> -	unsigned long maddr;
> -	struct page *tail;
> -	pte_t *pte;
> -	int node = zone_to_nid(zone);
> -
> -	tail = vmemmap_get_tail(order, zone);
> -	if (!tail)
> -		return -ENOMEM;
> -
> -	for (maddr = addr; maddr < addr + headsize; maddr += PAGE_SIZE) {
> -		pte = vmemmap_populate_address(maddr, node, NULL, -1, 0);
> -		if (!pte)
> -			return -ENOMEM;
> -	}
> -
> -	/*
> -	 * Reuse the last page struct page mapped above for the rest.
> -	 */
> -	return vmemmap_populate_range(maddr, end, node, NULL,
> -				      page_to_pfn(tail), 0);
> -}
> -#endif
> -
>   void __weak __meminit vmemmap_set_pmd(pmd_t *pmd, void *p, int node,
>   				      unsigned long addr, unsigned long next)
>   {
> @@ -634,7 +604,6 @@ struct page * __meminit __populate_section_memmap(unsigned long pfn,
>    */
>   void __init sparse_vmemmap_init_nid_early(int nid)
>   {
> -	hugetlb_vmemmap_init_early(nid);
>   }
>   #endif

This turns into a no-op function and can be removed right away. I see
you already drop it in patch #11, anyway.

Besides that, LGTM, so:

Acked-by: Qi Zheng <qi.zheng@linux.dev>

Thanks,
Qi



  reply	other threads:[~2026-08-25 13:12 UTC|newest]

Thread overview: 54+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-08-25  8:45 [PATCH v5 00/17] mm: Introduce section-based vmemmap optimization for HugeTLB Muchun Song
2026-08-25  8:45 ` [PATCH v5 01/17] mm/sparse: relax struct mem_section size constraints Muchun Song
2026-09-09  9:33   ` David Hildenbrand (Arm)
2026-08-25  8:45 ` [PATCH v5 02/17] mm/sparse-vmemmap: rename HVO order macros Muchun Song
2026-08-25 10:05   ` Qi Zheng
2026-08-25  8:45 ` [PATCH v5 03/17] mm/mm_init: skip initializing shared vmemmap tail pages Muchun Song
2026-08-25  9:30   ` Mike Rapoport
2026-08-25 10:15   ` Qi Zheng
2026-09-09  9:36   ` David Hildenbrand (Arm)
2026-09-09 11:30     ` Muchun Song
2026-08-25  8:45 ` [PATCH v5 04/17] mm/sparse-vmemmap: initialize shared tail vmemmap pages on allocation Muchun Song
2026-08-25  9:30   ` Mike Rapoport
2026-08-25 10:40   ` Qi Zheng
2026-09-09  9:39   ` David Hildenbrand (Arm)
2026-09-09 12:49     ` Muchun Song
2026-08-25  8:45 ` [PATCH v5 05/17] mm/sparse-vmemmap: support section-based vmemmap accounting Muchun Song
2026-08-25 10:47   ` Qi Zheng
2026-08-25  8:45 ` [PATCH v5 06/17] mm/mm_init: factor out pfn_to_zone() Muchun Song
2026-08-25 10:51   ` Qi Zheng
2026-08-25  8:45 ` [PATCH v5 07/17] mm/sparse-vmemmap: move helpers ahead of future callers Muchun Song
2026-08-25 10:54   ` Qi Zheng
2026-08-25  8:45 ` [PATCH v5 08/17] mm/sparse-vmemmap: support section-based vmemmap optimization Muchun Song
2026-08-25  9:30   ` Mike Rapoport
2026-08-25 12:38   ` Qi Zheng
2026-08-25  8:46 ` [PATCH v5 09/17] mm/sparse: initialize memory sections earlier Muchun Song
2026-08-25 12:49   ` Qi Zheng
2026-09-09  9:56   ` David Hildenbrand (Arm)
2026-09-09 12:29     ` Muchun Song
2026-09-09 12:48       ` David Hildenbrand (Arm)
2026-09-09 12:58         ` Muchun Song
2026-08-25  8:46 ` [PATCH v5 10/17] mm/hugetlb: switch HugeTLB to section-based vmemmap optimization Muchun Song
2026-08-25 13:12   ` Qi Zheng [this message]
2026-08-26  2:38     ` Muchun Song
2026-08-26  3:24       ` Qi Zheng
2026-09-09  9:58   ` David Hildenbrand (Arm)
2026-09-09 13:08     ` Muchun Song
2026-09-09 13:39       ` David Hildenbrand (Arm)
2026-08-25  8:46 ` [PATCH v5 11/17] mm/sparse-vmemmap: remove SPARSEMEM_VMEMMAP_PREINIT support Muchun Song
2026-08-25 13:16   ` Qi Zheng
2026-09-09  9:59   ` David Hildenbrand (Arm)
2026-08-25  8:46 ` [PATCH v5 12/17] mm/sparse: inline usemap allocation into sparse_init_nid() Muchun Song
2026-09-09 10:01   ` David Hildenbrand (Arm)
2026-08-25  8:46 ` [PATCH v5 13/17] mm/sparse: remove section_map_size() Muchun Song
2026-09-09 10:02   ` David Hildenbrand (Arm)
2026-08-25  8:46 ` [PATCH v5 14/17] mm/hugetlb: remove HUGE_BOOTMEM_HVO Muchun Song
2026-08-25 13:28   ` Qi Zheng
2026-08-25  8:46 ` [PATCH v5 15/17] mm/hugetlb: remove HUGE_BOOTMEM_CMA Muchun Song
2026-08-25 13:42   ` Qi Zheng
2026-08-25  8:46 ` [PATCH v5 16/17] mm/hugetlb: localize struct huge_bootmem_page Muchun Song
2026-08-25 13:38   ` Qi Zheng
2026-08-25  8:46 ` [PATCH v5 17/17] mm/hugetlb: localize HUGE_BOOTMEM_ZONES_VALID Muchun Song
2026-08-25 13:40   ` Qi Zheng
2026-08-28  4:20 ` [PATCH v5 00/17] mm: Introduce section-based vmemmap optimization for HugeTLB Andrew Morton
2026-09-09  9:31 ` David Hildenbrand (Arm)

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=83d8d1f2-6ca2-4c72-a494-23ce07bb4181@linux.dev \
    --to=qi.zheng@linux.dev \
    --cc=akpm@linux-foundation.org \
    --cc=david.laight.linux@gmail.com \
    --cc=david@kernel.org \
    --cc=liam@infradead.org \
    --cc=linux-kernel@vger.kernel.org \
    --cc=linux-mm@kvack.org \
    --cc=ljs@kernel.org \
    --cc=mhocko@suse.com \
    --cc=muchun.song@linux.dev \
    --cc=osalvador@suse.de \
    --cc=rppt@kernel.org \
    --cc=songmuchun@bytedance.com \
    --cc=surenb@google.com \
    --cc=vbabka@kernel.org \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.