From: "David Hildenbrand (Arm)" <david@kernel.org>
To: "Lorenzo Stoakes (ARM)" <ljs@kernel.org>
Cc: Li Zhe <lizhe.67@bytedance.com>,
"Jose A. Perez de Azpillaga" <azpijr@gmail.com>,
akpm@linux-foundation.org, muchun.song@linux.dev,
osalvador@suse.de, linux-mm@kvack.org,
linux-kernel@vger.kernel.org
Subject: Re: [PATCH] mm/hugetlb: avoid recursive i_mmap_rwsem in PMD sharing
Date: Fri, 9 Oct 2026 20:05:15 +0200 [thread overview]
Message-ID: <70ea8ee6-6fc5-4209-8cee-362da37f5035@kernel.org> (raw)
In-Reply-To: <asjVUa6XBZ8ti-0_@gremlin>
On 10/9/26 13:52, Lorenzo Stoakes (ARM) wrote:
> On Mon, Oct 05, 2026 at 11:25:01AM +0200, David Hildenbrand (Arm) wrote:
>>> David mentioned that Oscar was planning to send a proper fix for this
>>> issue, so I will wait for that patch instead of moving forward with this
>>> trylock fallback approach.
>> @Lorenzo, if Oscar is too busy, I guess we can paste the overall idea for the
>> fix here as well (publicly)?
>
> Yeah I don't see why not! I don't think there's anything that needs to be
> private here.
Li, see below, can you work with the below?
----8<----
From 9e1c0676ea6c9cbcabf91cd8badf81a9d88fd3ef Mon Sep 17 00:00:00 2001
From: "Lorenzo Stoakes (ARM)" <ljs@kernel.org>
Date: Wed, 29 Jul 2026 18:33:08 +0100
Subject: [PATCH] fix
---
mm/hugetlb.c | 103 +++++++++++++++++++++++++++++++++++++++------------
1 file changed, 79 insertions(+), 24 deletions(-)
diff --git a/mm/hugetlb.c b/mm/hugetlb.c
index e93c4d2456aa..88ff2c4ad483 100644
--- a/mm/hugetlb.c
+++ b/mm/hugetlb.c
@@ -5077,6 +5077,79 @@ static void move_huge_pte(struct vm_area_struct *vma, unsigned long old_addr,
spin_unlock(dst_ptl);
}
+/* Returns last address processed */
+static unsigned long prealloc_or_move_page_tables(struct vm_area_struct *vma,
+ struct vm_area_struct *new_vma, unsigned long old_addr,
+ unsigned long new_addr, unsigned long len, unsigned long sz,
+ unsigned long last_addr_mask, struct mmu_gather *tlb,
+ bool is_prealloc)
+{
+ unsigned long old_end = old_addr + len;
+ struct hstate *h = hstate_vma(vma);
+ struct mm_struct *mm = vma->vm_mm;
+ pte_t *src_pte, *dst_pte;
+
+ hugetlb_vma_assert_locked(vma);
+
+ for (; old_addr < old_end; old_addr += sz, new_addr += sz) {
+ src_pte = hugetlb_walk(vma, old_addr, sz);
+ if (!src_pte) {
+ old_addr |= last_addr_mask;
+ new_addr |= last_addr_mask;
+ continue;
+ }
+ if (huge_pte_none(huge_ptep_get(mm, old_addr, src_pte)))
+ continue;
+
+ if (is_prealloc) {
+ if (!huge_pte_alloc(mm, new_vma, new_addr, sz))
+ break;
+ continue;
+ }
+
+ if (huge_pmd_unshare(tlb, vma, old_addr, src_pte)) {
+ old_addr |= last_addr_mask;
+ new_addr |= last_addr_mask;
+ continue;
+ }
+
+ dst_pte = hugetlb_walk(new_vma, new_addr, sz);
+ if (!dst_pte)
+ break;
+
+ move_huge_pte(vma, old_addr, new_addr, src_pte, dst_pte, sz);
+ tlb_remove_huge_tlb_entry(h, tlb, src_pte, old_addr);
+ }
+
+ return old_addr;
+}
+
+/* Returns true if preallocation succeeded across range, otherwise false. */
+static bool prealloc_hugetlb_page_tables(struct vm_area_struct *vma,
+ struct vm_area_struct *new_vma, unsigned long old_addr,
+ unsigned long new_addr, unsigned long len, unsigned long sz,
+ unsigned long last_addr_mask)
+{
+ unsigned long old_end = old_addr + len;
+ unsigned long addr_end;
+
+ addr_end = prealloc_or_move_page_tables(vma, new_vma, old_addr, new_addr,
+ len, sz, last_addr_mask, NULL, /*is_prealloc=*/true);
+
+ return addr_end >= old_end;
+}
+
+/* Returns last processed address. */
+static unsigned long __move_hugetlb_page_tables(struct vm_area_struct *vma,
+ struct vm_area_struct *new_vma, unsigned long old_addr,
+ unsigned long new_addr, unsigned long len, unsigned long sz,
+ unsigned long last_addr_mask, struct mmu_gather *tlb)
+{
+ i_mmap_assert_write_locked(vma->vm_file->f_mapping);
+ return prealloc_or_move_page_tables(vma, new_vma, old_addr, new_addr,
+ len, sz, last_addr_mask, tlb, /*is_prealloc=*/false);
+}
+
int move_hugetlb_page_tables(struct vm_area_struct *vma,
struct vm_area_struct *new_vma,
unsigned long old_addr, unsigned long new_addr,
@@ -5088,9 +5161,9 @@ int move_hugetlb_page_tables(struct vm_area_struct *vma,
struct mm_struct *mm = vma->vm_mm;
unsigned long old_end = old_addr + len;
unsigned long last_addr_mask;
- pte_t *src_pte, *dst_pte;
struct mmu_notifier_range range;
struct mmu_gather tlb;
+ bool preallocated;
mmu_notifier_range_init(&range, MMU_NOTIFY_CLEAR, 0, mm, old_addr,
old_end);
@@ -5106,30 +5179,12 @@ int move_hugetlb_page_tables(struct vm_area_struct *vma,
last_addr_mask = hugetlb_mask_last_page(h);
/* Prevent race with file truncation */
hugetlb_vma_lock_write(vma);
+ preallocated = prealloc_hugetlb_page_tables(vma, new_vma, old_addr,
+ new_addr, len, sz, last_addr_mask);
i_mmap_lock_write(mapping);
- for (; old_addr < old_end; old_addr += sz, new_addr += sz) {
- src_pte = hugetlb_walk(vma, old_addr, sz);
- if (!src_pte) {
- old_addr |= last_addr_mask;
- new_addr |= last_addr_mask;
- continue;
- }
- if (huge_pte_none(huge_ptep_get(mm, old_addr, src_pte)))
- continue;
-
- if (huge_pmd_unshare(&tlb, vma, old_addr, src_pte)) {
- old_addr |= last_addr_mask;
- new_addr |= last_addr_mask;
- continue;
- }
-
- dst_pte = huge_pte_alloc(mm, new_vma, new_addr, sz);
- if (!dst_pte)
- break;
-
- move_huge_pte(vma, old_addr, new_addr, src_pte, dst_pte, sz);
- tlb_remove_huge_tlb_entry(h, &tlb, src_pte, old_addr);
- }
+ if (preallocated)
+ old_addr = __move_hugetlb_page_tables(vma, new_vma, old_addr,
+ new_addr, len, sz, last_addr_mask, &tlb);
tlb_flush_mmu_tlbonly(&tlb);
huge_pmd_unshare_flush(&tlb, vma);
--
2.55.0
--
Cheers,
David
next prev parent reply other threads:[~2026-10-09 18:05 UTC|newest]
Thread overview: 7+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-30 6:43 [PATCH] mm/hugetlb: avoid recursive i_mmap_rwsem in PMD sharing Li Zhe
2026-09-30 9:31 ` Jose A. Perez de Azpillaga
2026-10-05 9:03 ` Li Zhe
2026-10-05 9:25 ` David Hildenbrand (Arm)
2026-10-09 11:52 ` Lorenzo Stoakes (ARM)
2026-10-09 18:05 ` David Hildenbrand (Arm) [this message]
2026-09-30 10:03 ` David Hildenbrand (Arm)
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=70ea8ee6-6fc5-4209-8cee-362da37f5035@kernel.org \
--to=david@kernel.org \
--cc=akpm@linux-foundation.org \
--cc=azpijr@gmail.com \
--cc=linux-kernel@vger.kernel.org \
--cc=linux-mm@kvack.org \
--cc=lizhe.67@bytedance.com \
--cc=ljs@kernel.org \
--cc=muchun.song@linux.dev \
--cc=osalvador@suse.de \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox