From: Ackerley Tng via B4 Relay <devnull+ackerleytng.google.com@kernel.org>
To: Alistair Popple <apopple@nvidia.com>,
Andrew Morton <akpm@linux-foundation.org>,
Byungchul Park <byungchul@sk.com>,
David Hildenbrand <david@kernel.org>,
Gregory Price <gourry@gourry.net>,
Joshua Hahn <joshua.hahnjy@gmail.com>,
Matthew Brost <matthew.brost@intel.com>,
Muchun Song <muchun.song@linux.dev>,
Oscar Salvador <osalvador@suse.de>, Rakie Kim <rakie.kim@sk.com>,
Ying Huang <ying.huang@linux.alibaba.com>,
Zi Yan <ziy@nvidia.com>,
erdemaktas@google.com, fvdl@google.com, jiaqiyan@google.com,
jthoughton@google.com, mhocko@kernel.org, michael.roth@amd.com,
pasha.tatashin@soleen.com, pbonzini@redhat.com,
peterx@redhat.com, pratyush@kernel.org,
rick.p.edgecombe@intel.com, rientjes@google.com,
roman.gushchin@linux.dev, seanjc@google.com,
shakeel.butt@linux.dev, shivankg@amd.com, vannapurve@google.com,
yan.y.zhao@intel.com, Jason Gunthorpe <jgg@ziepe.ca>
Cc: linux-kernel@vger.kernel.org, linux-mm@kvack.org,
Ackerley Tng <ackerleytng@google.com>
Subject: [PATCH v5 2/3] mm: hugetlb: Move mpol interpretation out of alloc_buddy_hugetlb_folio_with_mpol()
Date: Mon, 03 Aug 2026 06:37:59 -0700 [thread overview]
Message-ID: <20260803-hugetlb-mpol-interpretation-v5-2-af2b7089f8a9@google.com> (raw)
In-Reply-To: <20260803-hugetlb-mpol-interpretation-v5-0-af2b7089f8a9@google.com>
From: Ackerley Tng <ackerleytng@google.com>
Move memory policy interpretation out of
alloc_buddy_hugetlb_folio_with_mpol() and into alloc_hugetlb_folio() to
separate reading and interpretation of memory policy from actual
allocation.
This will later allow memory policy to be interpreted outside of the
process of allocating a hugetlb folio entirely. This opens doors for other
callers of the HugeTLB folio allocation function, such as guest_memfd,
where memory may not always be mapped and hence may not have an associated
vma.
Introduce struct mempolicy_interpreted to hold all the components of an
interpreted memory policy.
Rename alloc_buddy_hugetlb_folio_with_mpol() to alloc_buddy_hugetlb_folio()
since the function no longer interprets memory policy.
No functional change intended.
Reviewed-by: James Houghton <jthoughton@google.com>
Acked-by: Oscar Salvador <osalvador@suse.de>
Signed-off-by: Ackerley Tng <ackerleytng@google.com>
---
include/uapi/linux/mempolicy.h | 2 +-
mm/hugetlb.c | 54 ++++++++++++++++++++++++++++--------------
2 files changed, 37 insertions(+), 19 deletions(-)
diff --git a/include/uapi/linux/mempolicy.h b/include/uapi/linux/mempolicy.h
index 6c962d866e864..7f6fc9599693b 100644
--- a/include/uapi/linux/mempolicy.h
+++ b/include/uapi/linux/mempolicy.h
@@ -16,7 +16,7 @@
*/
/* Policies */
-enum {
+enum mempolicy_mode {
MPOL_DEFAULT,
MPOL_PREFERRED,
MPOL_BIND,
diff --git a/mm/hugetlb.c b/mm/hugetlb.c
index 7985cfd21a03c..3a159de08a0b6 100644
--- a/mm/hugetlb.c
+++ b/mm/hugetlb.c
@@ -1317,6 +1317,12 @@ static unsigned long available_huge_pages(struct hstate *h)
return h->free_huge_pages - h->resv_huge_pages;
}
+struct mempolicy_interpreted {
+ int nid;
+ nodemask_t *nodemask;
+ enum mempolicy_mode mode;
+};
+
static struct folio *dequeue_hugetlb_folio_vma(struct hstate *h,
struct vm_area_struct *vma,
unsigned long address)
@@ -2138,32 +2144,28 @@ static struct folio *alloc_migrate_hugetlb_folio(struct hstate *h, gfp_t gfp_mas
return folio;
}
-/*
- * Use the VMA's mpolicy to allocate a huge page from the buddy.
- */
static
-struct folio *alloc_buddy_hugetlb_folio_with_mpol(struct hstate *h,
- struct vm_area_struct *vma, unsigned long addr)
+struct folio *alloc_buddy_hugetlb_folio(struct hstate *h,
+ gfp_t gfp_mask, struct mempolicy_interpreted *mpoli)
{
struct folio *folio = NULL;
- struct mempolicy *mpol;
- gfp_t gfp_mask = htlb_alloc_mask(h);
- int nid;
- nodemask_t *nodemask;
+ nodemask_t *nodemask = mpoli->nodemask;
- nid = huge_node(vma, addr, gfp_mask, &mpol, &nodemask);
- if (mpol_is_preferred_many(mpol)) {
+ if (mpoli->mode == MPOL_PREFERRED_MANY) {
gfp_t gfp = gfp_mask & ~(__GFP_DIRECT_RECLAIM | __GFP_NOFAIL);
- folio = alloc_surplus_hugetlb_folio(h, gfp, nid, nodemask);
+ folio = alloc_surplus_hugetlb_folio(h, gfp, mpoli->nid,
+ nodemask);
/* Fallback to all nodes if page==NULL */
nodemask = NULL;
}
- if (!folio)
- folio = alloc_surplus_hugetlb_folio(h, gfp_mask, nid, nodemask);
- mpol_cond_put(mpol);
+ if (!folio) {
+ folio = alloc_surplus_hugetlb_folio(h, gfp_mask, mpoli->nid,
+ nodemask);
+ }
+
return folio;
}
@@ -2853,7 +2855,7 @@ struct folio *alloc_hugetlb_folio(struct vm_area_struct *vma,
int ret, idx;
struct hugetlb_cgroup *h_cg = NULL;
struct hugetlb_cgroup *h_cg_rsvd = NULL;
- gfp_t gfp = htlb_alloc_mask(h) | __GFP_RETRY_MAYFAIL;
+ gfp_t gfp = htlb_alloc_mask(h);
idx = hstate_index(h);
@@ -2926,8 +2928,24 @@ struct folio *alloc_hugetlb_folio(struct vm_area_struct *vma,
folio = dequeue_hugetlb_folio_vma(h, vma, addr);
if (!folio) {
+ struct mempolicy_interpreted mpoli;
+ struct mempolicy *mpol;
+ nodemask_t *nodemask;
+ int nid;
+
spin_unlock_irq(&hugetlb_lock);
- folio = alloc_buddy_hugetlb_folio_with_mpol(h, vma, addr);
+ nid = huge_node(vma, addr, gfp, &mpol, &nodemask);
+ mpoli = (struct mempolicy_interpreted){
+ .nid = nid,
+#ifdef CONFIG_NUMA
+ .mode = mpol ? mpol->mode : MPOL_DEFAULT,
+#else
+ .mode = MPOL_DEFAULT,
+#endif
+ .nodemask = nodemask,
+ };
+ folio = alloc_buddy_hugetlb_folio(h, gfp, &mpoli);
+ mpol_cond_put(mpol);
if (!folio)
goto out_uncharge_cgroup;
spin_lock_irq(&hugetlb_lock);
@@ -2983,7 +3001,7 @@ struct folio *alloc_hugetlb_folio(struct vm_area_struct *vma,
}
}
- ret = mem_cgroup_charge_hugetlb(folio, gfp);
+ ret = mem_cgroup_charge_hugetlb(folio, gfp | __GFP_RETRY_MAYFAIL);
/*
* Unconditionally increment NR_HUGETLB here. If it turns out that
* mem_cgroup_charge_hugetlb failed, then immediately free the page and
--
2.55.0.508.g3f0d502094-goog
WARNING: multiple messages have this Message-ID (diff)
From: Ackerley Tng <ackerleytng@google.com>
To: Alistair Popple <apopple@nvidia.com>,
Andrew Morton <akpm@linux-foundation.org>,
Byungchul Park <byungchul@sk.com>,
David Hildenbrand <david@kernel.org>,
Gregory Price <gourry@gourry.net>,
Joshua Hahn <joshua.hahnjy@gmail.com>,
Matthew Brost <matthew.brost@intel.com>,
Muchun Song <muchun.song@linux.dev>,
Oscar Salvador <osalvador@suse.de>, Rakie Kim <rakie.kim@sk.com>,
Ying Huang <ying.huang@linux.alibaba.com>,
Zi Yan <ziy@nvidia.com>,
erdemaktas@google.com, fvdl@google.com, jiaqiyan@google.com,
jthoughton@google.com, mhocko@kernel.org, michael.roth@amd.com,
pasha.tatashin@soleen.com, pbonzini@redhat.com,
peterx@redhat.com, pratyush@kernel.org,
rick.p.edgecombe@intel.com, rientjes@google.com,
roman.gushchin@linux.dev, seanjc@google.com,
shakeel.butt@linux.dev, shivankg@amd.com, vannapurve@google.com,
yan.y.zhao@intel.com, Jason Gunthorpe <jgg@ziepe.ca>
Cc: linux-kernel@vger.kernel.org, linux-mm@kvack.org,
Ackerley Tng <ackerleytng@google.com>
Subject: [PATCH v5 2/3] mm: hugetlb: Move mpol interpretation out of alloc_buddy_hugetlb_folio_with_mpol()
Date: Mon, 03 Aug 2026 06:37:59 -0700 [thread overview]
Message-ID: <20260803-hugetlb-mpol-interpretation-v5-2-af2b7089f8a9@google.com> (raw)
In-Reply-To: <20260803-hugetlb-mpol-interpretation-v5-0-af2b7089f8a9@google.com>
Move memory policy interpretation out of
alloc_buddy_hugetlb_folio_with_mpol() and into alloc_hugetlb_folio() to
separate reading and interpretation of memory policy from actual
allocation.
This will later allow memory policy to be interpreted outside of the
process of allocating a hugetlb folio entirely. This opens doors for other
callers of the HugeTLB folio allocation function, such as guest_memfd,
where memory may not always be mapped and hence may not have an associated
vma.
Introduce struct mempolicy_interpreted to hold all the components of an
interpreted memory policy.
Rename alloc_buddy_hugetlb_folio_with_mpol() to alloc_buddy_hugetlb_folio()
since the function no longer interprets memory policy.
No functional change intended.
Reviewed-by: James Houghton <jthoughton@google.com>
Acked-by: Oscar Salvador <osalvador@suse.de>
Signed-off-by: Ackerley Tng <ackerleytng@google.com>
---
include/uapi/linux/mempolicy.h | 2 +-
mm/hugetlb.c | 54 ++++++++++++++++++++++++++++--------------
2 files changed, 37 insertions(+), 19 deletions(-)
diff --git a/include/uapi/linux/mempolicy.h b/include/uapi/linux/mempolicy.h
index 6c962d866e864..7f6fc9599693b 100644
--- a/include/uapi/linux/mempolicy.h
+++ b/include/uapi/linux/mempolicy.h
@@ -16,7 +16,7 @@
*/
/* Policies */
-enum {
+enum mempolicy_mode {
MPOL_DEFAULT,
MPOL_PREFERRED,
MPOL_BIND,
diff --git a/mm/hugetlb.c b/mm/hugetlb.c
index 7985cfd21a03c..3a159de08a0b6 100644
--- a/mm/hugetlb.c
+++ b/mm/hugetlb.c
@@ -1317,6 +1317,12 @@ static unsigned long available_huge_pages(struct hstate *h)
return h->free_huge_pages - h->resv_huge_pages;
}
+struct mempolicy_interpreted {
+ int nid;
+ nodemask_t *nodemask;
+ enum mempolicy_mode mode;
+};
+
static struct folio *dequeue_hugetlb_folio_vma(struct hstate *h,
struct vm_area_struct *vma,
unsigned long address)
@@ -2138,32 +2144,28 @@ static struct folio *alloc_migrate_hugetlb_folio(struct hstate *h, gfp_t gfp_mas
return folio;
}
-/*
- * Use the VMA's mpolicy to allocate a huge page from the buddy.
- */
static
-struct folio *alloc_buddy_hugetlb_folio_with_mpol(struct hstate *h,
- struct vm_area_struct *vma, unsigned long addr)
+struct folio *alloc_buddy_hugetlb_folio(struct hstate *h,
+ gfp_t gfp_mask, struct mempolicy_interpreted *mpoli)
{
struct folio *folio = NULL;
- struct mempolicy *mpol;
- gfp_t gfp_mask = htlb_alloc_mask(h);
- int nid;
- nodemask_t *nodemask;
+ nodemask_t *nodemask = mpoli->nodemask;
- nid = huge_node(vma, addr, gfp_mask, &mpol, &nodemask);
- if (mpol_is_preferred_many(mpol)) {
+ if (mpoli->mode == MPOL_PREFERRED_MANY) {
gfp_t gfp = gfp_mask & ~(__GFP_DIRECT_RECLAIM | __GFP_NOFAIL);
- folio = alloc_surplus_hugetlb_folio(h, gfp, nid, nodemask);
+ folio = alloc_surplus_hugetlb_folio(h, gfp, mpoli->nid,
+ nodemask);
/* Fallback to all nodes if page==NULL */
nodemask = NULL;
}
- if (!folio)
- folio = alloc_surplus_hugetlb_folio(h, gfp_mask, nid, nodemask);
- mpol_cond_put(mpol);
+ if (!folio) {
+ folio = alloc_surplus_hugetlb_folio(h, gfp_mask, mpoli->nid,
+ nodemask);
+ }
+
return folio;
}
@@ -2853,7 +2855,7 @@ struct folio *alloc_hugetlb_folio(struct vm_area_struct *vma,
int ret, idx;
struct hugetlb_cgroup *h_cg = NULL;
struct hugetlb_cgroup *h_cg_rsvd = NULL;
- gfp_t gfp = htlb_alloc_mask(h) | __GFP_RETRY_MAYFAIL;
+ gfp_t gfp = htlb_alloc_mask(h);
idx = hstate_index(h);
@@ -2926,8 +2928,24 @@ struct folio *alloc_hugetlb_folio(struct vm_area_struct *vma,
folio = dequeue_hugetlb_folio_vma(h, vma, addr);
if (!folio) {
+ struct mempolicy_interpreted mpoli;
+ struct mempolicy *mpol;
+ nodemask_t *nodemask;
+ int nid;
+
spin_unlock_irq(&hugetlb_lock);
- folio = alloc_buddy_hugetlb_folio_with_mpol(h, vma, addr);
+ nid = huge_node(vma, addr, gfp, &mpol, &nodemask);
+ mpoli = (struct mempolicy_interpreted){
+ .nid = nid,
+#ifdef CONFIG_NUMA
+ .mode = mpol ? mpol->mode : MPOL_DEFAULT,
+#else
+ .mode = MPOL_DEFAULT,
+#endif
+ .nodemask = nodemask,
+ };
+ folio = alloc_buddy_hugetlb_folio(h, gfp, &mpoli);
+ mpol_cond_put(mpol);
if (!folio)
goto out_uncharge_cgroup;
spin_lock_irq(&hugetlb_lock);
@@ -2983,7 +3001,7 @@ struct folio *alloc_hugetlb_folio(struct vm_area_struct *vma,
}
}
- ret = mem_cgroup_charge_hugetlb(folio, gfp);
+ ret = mem_cgroup_charge_hugetlb(folio, gfp | __GFP_RETRY_MAYFAIL);
/*
* Unconditionally increment NR_HUGETLB here. If it turns out that
* mem_cgroup_charge_hugetlb failed, then immediately free the page and
--
2.55.0.508.g3f0d502094-goog
next prev parent reply other threads:[~2026-08-03 13:38 UTC|newest]
Thread overview: 11+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-08-03 13:37 [PATCH v5 0/3] Consolidate memory policy interpretation in alloc_hugetlb_folio() Ackerley Tng via B4 Relay
2026-08-03 13:37 ` Ackerley Tng
2026-08-03 13:37 ` [PATCH v5 1/3] mm: hugetlb: Consolidate interpretation of gbl_chg within alloc_hugetlb_folio() Ackerley Tng via B4 Relay
2026-08-03 13:37 ` Ackerley Tng
2026-08-03 14:49 ` Gregory Price
2026-08-03 13:37 ` Ackerley Tng via B4 Relay [this message]
2026-08-03 13:37 ` [PATCH v5 2/3] mm: hugetlb: Move mpol interpretation out of alloc_buddy_hugetlb_folio_with_mpol() Ackerley Tng
2026-08-03 14:55 ` Gregory Price
2026-08-03 13:38 ` [PATCH v5 3/3] mm: hugetlb: Move mpol interpretation out of dequeue_hugetlb_folio_vma() Ackerley Tng via B4 Relay
2026-08-03 13:38 ` Ackerley Tng
2026-08-03 14:59 ` Gregory Price
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260803-hugetlb-mpol-interpretation-v5-2-af2b7089f8a9@google.com \
--to=devnull+ackerleytng.google.com@kernel.org \
--cc=ackerleytng@google.com \
--cc=akpm@linux-foundation.org \
--cc=apopple@nvidia.com \
--cc=byungchul@sk.com \
--cc=david@kernel.org \
--cc=erdemaktas@google.com \
--cc=fvdl@google.com \
--cc=gourry@gourry.net \
--cc=jgg@ziepe.ca \
--cc=jiaqiyan@google.com \
--cc=joshua.hahnjy@gmail.com \
--cc=jthoughton@google.com \
--cc=linux-kernel@vger.kernel.org \
--cc=linux-mm@kvack.org \
--cc=matthew.brost@intel.com \
--cc=mhocko@kernel.org \
--cc=michael.roth@amd.com \
--cc=muchun.song@linux.dev \
--cc=osalvador@suse.de \
--cc=pasha.tatashin@soleen.com \
--cc=pbonzini@redhat.com \
--cc=peterx@redhat.com \
--cc=pratyush@kernel.org \
--cc=rakie.kim@sk.com \
--cc=rick.p.edgecombe@intel.com \
--cc=rientjes@google.com \
--cc=roman.gushchin@linux.dev \
--cc=seanjc@google.com \
--cc=shakeel.butt@linux.dev \
--cc=shivankg@amd.com \
--cc=vannapurve@google.com \
--cc=yan.y.zhao@intel.com \
--cc=ying.huang@linux.alibaba.com \
--cc=ziy@nvidia.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.