Linux-mm Archive on lore.kernel.org
 help / color / mirror / Atom feed
From: "David Hildenbrand (Arm)" <david@kernel.org>
To: "Lorenzo Stoakes (ARM)" <ljs@kernel.org>
Cc: Li Zhe <lizhe.67@bytedance.com>,
	"Jose A. Perez de Azpillaga" <azpijr@gmail.com>,
	akpm@linux-foundation.org, muchun.song@linux.dev,
	osalvador@suse.de, linux-mm@kvack.org,
	linux-kernel@vger.kernel.org
Subject: Re: [PATCH] mm/hugetlb: avoid recursive i_mmap_rwsem in PMD sharing
Date: Fri, 9 Oct 2026 20:05:15 +0200	[thread overview]
Message-ID: <70ea8ee6-6fc5-4209-8cee-362da37f5035@kernel.org> (raw)
In-Reply-To: <asjVUa6XBZ8ti-0_@gremlin>

On 10/9/26 13:52, Lorenzo Stoakes (ARM) wrote:
> On Mon, Oct 05, 2026 at 11:25:01AM +0200, David Hildenbrand (Arm) wrote:
>>> David mentioned that Oscar was planning to send a proper fix for this
>>> issue, so I will wait for that patch instead of moving forward with this
>>> trylock fallback approach.
>> @Lorenzo, if Oscar is too busy, I guess we can paste the overall idea for the
>> fix here as well (publicly)?
> 
> Yeah I don't see why not! I don't think there's anything that needs to be
> private here.

Li, see below, can you work with the below?


----8<----
From 9e1c0676ea6c9cbcabf91cd8badf81a9d88fd3ef Mon Sep 17 00:00:00 2001
From: "Lorenzo Stoakes (ARM)" <ljs@kernel.org>
Date: Wed, 29 Jul 2026 18:33:08 +0100
Subject: [PATCH] fix

---
 mm/hugetlb.c | 103 +++++++++++++++++++++++++++++++++++++++------------
 1 file changed, 79 insertions(+), 24 deletions(-)

diff --git a/mm/hugetlb.c b/mm/hugetlb.c
index e93c4d2456aa..88ff2c4ad483 100644
--- a/mm/hugetlb.c
+++ b/mm/hugetlb.c
@@ -5077,6 +5077,79 @@ static void move_huge_pte(struct vm_area_struct *vma, unsigned long old_addr,
 	spin_unlock(dst_ptl);
 }

+/* Returns last address processed */
+static unsigned long prealloc_or_move_page_tables(struct vm_area_struct *vma,
+		struct vm_area_struct *new_vma, unsigned long old_addr,
+		unsigned long new_addr, unsigned long len, unsigned long sz,
+		unsigned long last_addr_mask, struct mmu_gather *tlb,
+		bool is_prealloc)
+{
+	unsigned long old_end = old_addr + len;
+	struct hstate *h = hstate_vma(vma);
+	struct mm_struct *mm = vma->vm_mm;
+	pte_t *src_pte, *dst_pte;
+
+	hugetlb_vma_assert_locked(vma);
+
+	for (; old_addr < old_end; old_addr += sz, new_addr += sz) {
+		src_pte = hugetlb_walk(vma, old_addr, sz);
+		if (!src_pte) {
+			old_addr |= last_addr_mask;
+			new_addr |= last_addr_mask;
+			continue;
+		}
+		if (huge_pte_none(huge_ptep_get(mm, old_addr, src_pte)))
+			continue;
+
+		if (is_prealloc) {
+			if (!huge_pte_alloc(mm, new_vma, new_addr, sz))
+				break;
+			continue;
+		}
+
+		if (huge_pmd_unshare(tlb, vma, old_addr, src_pte)) {
+			old_addr |= last_addr_mask;
+			new_addr |= last_addr_mask;
+			continue;
+		}
+
+		dst_pte = hugetlb_walk(new_vma, new_addr, sz);
+		if (!dst_pte)
+			break;
+
+		move_huge_pte(vma, old_addr, new_addr, src_pte, dst_pte, sz);
+		tlb_remove_huge_tlb_entry(h, tlb, src_pte, old_addr);
+	}
+
+	return old_addr;
+}
+
+/* Returns true if preallocation succeeded across range, otherwise false. */
+static bool prealloc_hugetlb_page_tables(struct vm_area_struct *vma,
+		struct vm_area_struct *new_vma, unsigned long old_addr,
+		unsigned long new_addr, unsigned long len, unsigned long sz,
+		unsigned long last_addr_mask)
+{
+	unsigned long old_end = old_addr + len;
+	unsigned long addr_end;
+
+	addr_end = prealloc_or_move_page_tables(vma, new_vma, old_addr, new_addr,
+			len, sz, last_addr_mask, NULL, /*is_prealloc=*/true);
+
+	return addr_end >= old_end;
+}
+
+/* Returns last processed address. */
+static unsigned long __move_hugetlb_page_tables(struct vm_area_struct *vma,
+		struct vm_area_struct *new_vma, unsigned long old_addr,
+		unsigned long new_addr, unsigned long len, unsigned long sz,
+		unsigned long last_addr_mask, struct mmu_gather *tlb)
+{
+	i_mmap_assert_write_locked(vma->vm_file->f_mapping);
+	return prealloc_or_move_page_tables(vma, new_vma, old_addr, new_addr,
+			len, sz, last_addr_mask, tlb, /*is_prealloc=*/false);
+}
+
 int move_hugetlb_page_tables(struct vm_area_struct *vma,
 			     struct vm_area_struct *new_vma,
 			     unsigned long old_addr, unsigned long new_addr,
@@ -5088,9 +5161,9 @@ int move_hugetlb_page_tables(struct vm_area_struct *vma,
 	struct mm_struct *mm = vma->vm_mm;
 	unsigned long old_end = old_addr + len;
 	unsigned long last_addr_mask;
-	pte_t *src_pte, *dst_pte;
 	struct mmu_notifier_range range;
 	struct mmu_gather tlb;
+	bool preallocated;

 	mmu_notifier_range_init(&range, MMU_NOTIFY_CLEAR, 0, mm, old_addr,
 				old_end);
@@ -5106,30 +5179,12 @@ int move_hugetlb_page_tables(struct vm_area_struct *vma,
 	last_addr_mask = hugetlb_mask_last_page(h);
 	/* Prevent race with file truncation */
 	hugetlb_vma_lock_write(vma);
+	preallocated = prealloc_hugetlb_page_tables(vma, new_vma, old_addr,
+			new_addr, len, sz, last_addr_mask);
 	i_mmap_lock_write(mapping);
-	for (; old_addr < old_end; old_addr += sz, new_addr += sz) {
-		src_pte = hugetlb_walk(vma, old_addr, sz);
-		if (!src_pte) {
-			old_addr |= last_addr_mask;
-			new_addr |= last_addr_mask;
-			continue;
-		}
-		if (huge_pte_none(huge_ptep_get(mm, old_addr, src_pte)))
-			continue;
-
-		if (huge_pmd_unshare(&tlb, vma, old_addr, src_pte)) {
-			old_addr |= last_addr_mask;
-			new_addr |= last_addr_mask;
-			continue;
-		}
-
-		dst_pte = huge_pte_alloc(mm, new_vma, new_addr, sz);
-		if (!dst_pte)
-			break;
-
-		move_huge_pte(vma, old_addr, new_addr, src_pte, dst_pte, sz);
-		tlb_remove_huge_tlb_entry(h, &tlb, src_pte, old_addr);
-	}
+	if (preallocated)
+		old_addr = __move_hugetlb_page_tables(vma, new_vma, old_addr,
+			new_addr, len, sz, last_addr_mask, &tlb);

 	tlb_flush_mmu_tlbonly(&tlb);
 	huge_pmd_unshare_flush(&tlb, vma);
--
2.55.0

-- 
Cheers,

David


  reply	other threads:[~2026-10-09 18:05 UTC|newest]

Thread overview: 7+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-30  6:43 [PATCH] mm/hugetlb: avoid recursive i_mmap_rwsem in PMD sharing Li Zhe
2026-09-30  9:31 ` Jose A. Perez de Azpillaga
2026-10-05  9:03   ` Li Zhe
2026-10-05  9:25     ` David Hildenbrand (Arm)
2026-10-09 11:52       ` Lorenzo Stoakes (ARM)
2026-10-09 18:05         ` David Hildenbrand (Arm) [this message]
2026-09-30 10:03 ` David Hildenbrand (Arm)

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=70ea8ee6-6fc5-4209-8cee-362da37f5035@kernel.org \
    --to=david@kernel.org \
    --cc=akpm@linux-foundation.org \
    --cc=azpijr@gmail.com \
    --cc=linux-kernel@vger.kernel.org \
    --cc=linux-mm@kvack.org \
    --cc=lizhe.67@bytedance.com \
    --cc=ljs@kernel.org \
    --cc=muchun.song@linux.dev \
    --cc=osalvador@suse.de \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox