From: Usama Arif <usama.arif@linux.dev>
To: Sourav Panda <souravpanda@google.com>
Cc: Usama Arif <usama.arif@linux.dev>,
muchun.song@linux.dev, osalvador@suse.de,
akpm@linux-foundation.org, david@kernel.org, surenb@google.com,
fvdl@google.com, gthelen@google.com, rientjes@google.com,
linux-mm@kvack.org, linux-kernel@vger.kernel.org
Subject: Re: [PATCH v3] mm/hugetlb_cma: support percentage-based hugetlb_cma reservation
Date: Thu, 30 Jul 2026 05:49:37 -0700 [thread overview]
Message-ID: <20260730124938.2619968-1-usama.arif@linux.dev> (raw)
In-Reply-To: <20260729163148.3271755-1-souravpanda@google.com>
On Wed, 29 Jul 2026 16:31:48 +0000 Sourav Panda <souravpanda@google.com> wrote:
> Currently, hugetlb_cma reservation only supports absolute sizes (e.g.,
> hugetlb_cma=2G or hugetlb_cma=0:1G,1:1G). This can be restrictive in
> heterogeneous environments or when deploying common kernel command lines
> across machines with different memory capacities.
>
> Add support for percentage-based hugetlb_cma reservation (e.g.,
> hugetlb_cma=20% or hugetlb_cma=0:20%,1:10%).
>
> The percentage is calculated against the total memory (for global
> settings) or against the node-specific memory (for node-specific
> settings) using memblock APIs during early boot.
>
> Signed-off-by: Sourav Panda <souravpanda@google.com>
> ---
> Link: https://lore.kernel.org/linux-mm/20260628190155.3655895-1-souravpanda@google.com/ [v2]
> Link: https://lore.kernel.org/linux-mm/20260625215900.2151690-1-souravpanda@google.com/ [v1]
>
> v3:
> - Enhance pr_info logs to explicitly show percentage-derived sizes per Andrew Morton's suggestions.
> - Explicitly align percentage-derived sizes down to the gigantic page size to prevent allocation failures (Usama Arif).
>
> .../admin-guide/kernel-parameters.txt | 7 +-
> mm/hugetlb_cma.c | 126 ++++++++++++++++--
> 2 files changed, 123 insertions(+), 10 deletions(-)
>
> diff --git a/Documentation/admin-guide/kernel-parameters.txt b/Documentation/admin-guide/kernel-parameters.txt
> index 5a05b48d1684..846940ce2858 100644
> --- a/Documentation/admin-guide/kernel-parameters.txt
> +++ b/Documentation/admin-guide/kernel-parameters.txt
> @@ -2064,8 +2064,11 @@ Kernel parameters
> hugetlb_cma= [HW,CMA,EARLY] The size of a CMA area used for allocation
> of gigantic hugepages. Or using node format, the size
> of a CMA area per node can be specified.
> - Format: nn[KMGTPE] or (node format)
> - <node>:nn[KMGTPE][,<node>:nn[KMGTPE]]
> + The size can be an absolute value (e.g., 2G) or a
> + percentage of the total memory or node memory (e.g., 20%).
> + Format: nn[KMGTPE] or nn% or (node format)
> + <node>:nn[KMGTPE][,<node>:nn[KMGTPE]] or
> + <node>:nn%[,<node>:nn%]
>
> Reserve a CMA area of given size and allocate gigantic
> hugepages using the CMA allocator. If enabled, the
> diff --git a/mm/hugetlb_cma.c b/mm/hugetlb_cma.c
> index 7693ccefd0c6..99d05eb82abc 100644
> --- a/mm/hugetlb_cma.c
> +++ b/mm/hugetlb_cma.c
> @@ -9,6 +9,9 @@
> #include <asm/setup.h>
>
> #include <linux/hugetlb.h>
> +#include <linux/memblock.h>
> +#include <linux/math.h>
> +#include <linux/math64.h>
> #include "internal.h"
> #include "hugetlb_cma.h"
>
> @@ -18,6 +21,28 @@ static unsigned long hugetlb_cma_size_in_node[MAX_NUMNODES] __initdata;
> static bool hugetlb_cma_only __ro_after_init;
> static unsigned long hugetlb_cma_size __ro_after_init;
>
> +static unsigned int hugetlb_cma_percent __initdata;
> +static unsigned int hugetlb_cma_percent_in_node[MAX_NUMNODES] __initdata;
> +
> +#ifdef CONFIG_NUMA
> +static phys_addr_t __init memblock_node_memory_size(int nid)
> +{
> + struct memblock_region *reg;
> + phys_addr_t size = 0;
> +
> + for_each_mem_region(reg) {
> + if (reg->nid == nid)
> + size += reg->size;
> + }
> + return size;
> +}
> +#else
> +static phys_addr_t __init memblock_node_memory_size(int nid)
> +{
> + return memblock_phys_mem_size();
> +}
> +#endif
> +
> void hugetlb_cma_free_frozen_folio(struct folio *folio)
> {
> WARN_ON_ONCE(!cma_release_frozen(hugetlb_cma[folio_nid(folio)],
> @@ -100,14 +125,28 @@ static int __init cmdline_parse_hugetlb_cma(char *p)
> break;
>
> if (s[count] == ':') {
> + char *next;
> +
> if (tmp >= MAX_NUMNODES)
> break;
> nid = array_index_nospec(tmp, MAX_NUMNODES);
>
> s += count + 1;
> - tmp = memparse(s, &s);
> - hugetlb_cma_size_in_node[nid] = tmp;
> - hugetlb_cma_size += tmp;
> + tmp = memparse(s, &next);
> + if (*next == '%') {
> + if (tmp > 100) {
> + pr_warn("hugetlb_cma: invalid percentage %lu for node %d\n",
> + tmp, nid);
> + break;
> + }
> + hugetlb_cma_percent_in_node[nid] = tmp;
> + hugetlb_cma_size_in_node[nid] = 0;
> + s = next + 1;
> + } else {
> + hugetlb_cma_size_in_node[nid] = tmp;
> + hugetlb_cma_percent_in_node[nid] = 0;
> + s = next;
> + }
>
> /*
> * Skip the separator if have one, otherwise
> @@ -118,7 +157,28 @@ static int __init cmdline_parse_hugetlb_cma(char *p)
> else
> break;
> } else {
> - hugetlb_cma_size = memparse(p, &p);
> + char *next;
> +
> + tmp = memparse(p, &next);
> + if (*next == '%') {
> + if (tmp > 100) {
> + pr_warn("hugetlb_cma: invalid percentage %lu\n", tmp);
> + } else {
> + hugetlb_cma_percent = tmp;
> + hugetlb_cma_size = 0;
> + for (nid = 0; nid < MAX_NUMNODES; nid++) {
> + hugetlb_cma_size_in_node[nid] = 0;
> + hugetlb_cma_percent_in_node[nid] = 0;
> + }
> + }
> + } else {
> + hugetlb_cma_size = tmp;
> + hugetlb_cma_percent = 0;
> + for (nid = 0; nid < MAX_NUMNODES; nid++) {
> + hugetlb_cma_size_in_node[nid] = 0;
> + hugetlb_cma_percent_in_node[nid] = 0;
> + }
> + }
> break;
> }
> }
> @@ -144,8 +204,36 @@ void __init hugetlb_cma_reserve(void)
> {
> unsigned long size, reserved, per_node, order;
> bool node_specific_cma_alloc = false;
> + bool has_node_specific_param = false;
> int nid;
>
> + for (nid = 0; nid < MAX_NUMNODES; nid++) {
> + if (hugetlb_cma_size_in_node[nid] || hugetlb_cma_percent_in_node[nid]) {
> + has_node_specific_param = true;
> + break;
> + }
> + }
> +
> + if (has_node_specific_param) {
> + hugetlb_cma_size = 0;
> + for (nid = 0; nid < MAX_NUMNODES; nid++) {
> + if (hugetlb_cma_percent_in_node[nid]) {
> + phys_addr_t node_gfp_mem = memblock_node_memory_size(nid);
> + u64 s;
> +
> + s = mul_u64_u32_div((u64)node_gfp_mem,
> + hugetlb_cma_percent_in_node[nid],
> + 100);
> +
> + hugetlb_cma_size_in_node[nid] = s;
> + }
> + hugetlb_cma_size += hugetlb_cma_size_in_node[nid];
> + }
> + } else if (hugetlb_cma_percent) {
> + hugetlb_cma_size = mul_u64_u32_div((u64)memblock_phys_mem_size(),
> + hugetlb_cma_percent, 100);
> + }
> +
> if (!hugetlb_cma_size)
> return;
>
> @@ -163,6 +251,20 @@ void __init hugetlb_cma_reserve(void)
> */
> VM_WARN_ON(order <= MAX_PAGE_ORDER);
>
> + if (hugetlb_cma_percent) {
> + hugetlb_cma_size = ALIGN_DOWN(hugetlb_cma_size, PAGE_SIZE << order);
> + } else if (has_node_specific_param) {
Hello,
What will be done here if you have the following in kernel commandline?
hugetlb_cma=20% hugetlb_cma=0:20%,1:20%
It looks like the later node form doesn't clear hugetlb_cma_percent,
so node specific sizes will be ignored, which I think is not the right
behaviour.
> + hugetlb_cma_size = 0;
> + for (nid = 0; nid < MAX_NUMNODES; nid++) {
> + if (hugetlb_cma_percent_in_node[nid]) {
> + hugetlb_cma_size_in_node[nid] =
> + ALIGN_DOWN(hugetlb_cma_size_in_node[nid],
> + PAGE_SIZE << order);
Maybe print a warning here if you round down to 0?
For example, 5% of a 16G machine will round down to 0.
> + }
> + hugetlb_cma_size += hugetlb_cma_size_in_node[nid];
> + }
> + }
> +
> hugetlb_bootmem_set_nodes();
>
> for (nid = 0; nid < MAX_NUMNODES; nid++) {
> @@ -205,8 +307,12 @@ void __init hugetlb_cma_reserve(void)
> per_node = DIV_ROUND_UP(hugetlb_cma_size,
> nodes_weight(hugetlb_bootmem_nodes));
> per_node = round_up(per_node, PAGE_SIZE << order);
> - pr_info("hugetlb_cma: reserve %lu MiB, up to %lu MiB per node\n",
> - hugetlb_cma_size / SZ_1M, per_node / SZ_1M);
> + if (hugetlb_cma_percent)
> + pr_info("hugetlb_cma: reserve %lu MiB (%u%%), up to %lu MiB per node\n",
> + hugetlb_cma_size / SZ_1M, hugetlb_cma_percent, per_node / SZ_1M);
> + else
> + pr_info("hugetlb_cma: reserve %lu MiB, up to %lu MiB per node\n",
> + hugetlb_cma_size / SZ_1M, per_node / SZ_1M);
> }
>
> reserved = 0;
> @@ -241,8 +347,12 @@ void __init hugetlb_cma_reserve(void)
> }
>
> reserved += size;
> - pr_info("hugetlb_cma: reserved %lu MiB on node %d\n",
> - size / SZ_1M, nid);
> + if (hugetlb_cma_percent_in_node[nid])
> + pr_info("hugetlb_cma: reserved %lu MiB (%u%%) on node %d\n",
> + size / SZ_1M, hugetlb_cma_percent_in_node[nid], nid);
> + else
> + pr_info("hugetlb_cma: reserved %lu MiB on node %d\n",
> + size / SZ_1M, nid);
>
> if (reserved >= hugetlb_cma_size)
> break;
> --
> 2.55.0.rc0.799.gd6f94ed593-goog
>
>
prev parent reply other threads:[~2026-07-30 12:49 UTC|newest]
Thread overview: 2+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-07-29 16:31 [PATCH v3] mm/hugetlb_cma: support percentage-based hugetlb_cma reservation Sourav Panda
2026-07-30 12:49 ` Usama Arif [this message]
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260730124938.2619968-1-usama.arif@linux.dev \
--to=usama.arif@linux.dev \
--cc=akpm@linux-foundation.org \
--cc=david@kernel.org \
--cc=fvdl@google.com \
--cc=gthelen@google.com \
--cc=linux-kernel@vger.kernel.org \
--cc=linux-mm@kvack.org \
--cc=muchun.song@linux.dev \
--cc=osalvador@suse.de \
--cc=rientjes@google.com \
--cc=souravpanda@google.com \
--cc=surenb@google.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.