From: Vincent Donnefort <vdonnefort@google.com>
To: maz@kernel.org, oupton@kernel.org, kvmarm@lists.linux.dev,
linux-arm-kernel@lists.infradead.org
Cc: joey.gouly@arm.com, seiden@linux.ibm.com, suzuki.poulose@arm.com,
yuzenghui@huawei.com, catalin.marinas@arm.com, will@kernel.org,
kernel-team@android.com, fuad.tabba@linux.dev,
qperret@google.com, weilin.chang@arm.com,
Vincent Donnefort <vdonnefort@google.com>
Subject: [PATCH v2 22/22] KVM: arm64: Stage-2 huge mappings for protected VMs
Date: Fri, 11 Sep 2026 14:50:53 +0100 [thread overview]
Message-ID: <20260911135053.146435-23-vdonnefort@google.com> (raw)
In-Reply-To: <20260911135053.146435-1-vdonnefort@google.com>
Enable PMD-sized stage-2 block mappings for protected VMs. This is
possible whenever the stage-1 mapping allows it, that is, if it is
itself backed by THPs.
When a THP is found, an entire PMD_SIZE mapping is donated to the guest.
This mapping can only be broken down via the HVC
__pkvm_host_split_guest() which the hypervisor can request with
PKVM_HYP_REQ_SPLIT.
Signed-off-by: Vincent Donnefort <vdonnefort@google.com>
---
arch/arm64/kvm/mmu.c | 119 +++++++++++++++++++++++++++---------------
arch/arm64/kvm/pkvm.c | 21 ++++----
2 files changed, 89 insertions(+), 51 deletions(-)
diff --git a/arch/arm64/kvm/mmu.c b/arch/arm64/kvm/mmu.c
index 9ba86450fe4a..218df096c72e 100644
--- a/arch/arm64/kvm/mmu.c
+++ b/arch/arm64/kvm/mmu.c
@@ -1690,52 +1690,26 @@ static int gmem_abort(const struct kvm_s2_fault_desc *s2fd)
return ret != -EAGAIN ? ret : 0;
}
-struct kvm_s2_fault_vma_info {
- unsigned long mmu_seq;
- long vma_pagesize;
- vm_flags_t vm_flags;
- unsigned long max_map_size;
- struct page *page;
- kvm_pfn_t pfn;
- gfn_t gfn;
- bool device;
- bool mte_allowed;
- bool is_vma_cacheable;
- bool map_writable;
- bool map_non_cacheable;
-};
-
-static int pkvm_mem_abort(const struct kvm_s2_fault_desc *s2fd)
+static int pkvm_pin_user_pages(const struct kvm_s2_fault_desc *s2fd, struct page **__page,
+ unsigned long *__size, kvm_pfn_t *__pfn, gfn_t *__gfn)
{
unsigned int flags = FOLL_HWPOISON | FOLL_LONGTERM | FOLL_WRITE;
struct kvm_vcpu *vcpu = s2fd->vcpu;
- struct kvm_pgtable *pgt = vcpu->arch.hw_mmu->pgt;
struct mm_struct *mm = current->mm;
struct kvm *kvm = vcpu->kvm;
- void *hyp_memcache;
struct page *page;
- int ret;
+ kvm_pfn_t pfn;
+ gfn_t gfn;
+ long ret;
- hyp_memcache = get_mmu_memcache(vcpu);
- ret = topup_mmu_memcache(vcpu, hyp_memcache);
- if (ret)
- return -ENOMEM;
+ guard(mmap_read_lock)(mm);
- ret = account_locked_vm(mm, 1, true);
- if (ret)
- return ret;
-
- mmap_read_lock(mm);
ret = pin_user_pages(s2fd->hva, 1, flags, &page);
- mmap_read_unlock(mm);
-
if (ret == -EHWPOISON) {
kvm_send_hwpoison_signal(s2fd->hva, PAGE_SHIFT);
- ret = 0;
- goto dec_account;
+ return 0;
} else if (ret != 1) {
- ret = -EFAULT;
- goto dec_account;
+ return -EFAULT;
} else if (!folio_test_swapbacked(page_folio(page))) {
/*
* We really can't deal with page-cache pages returned by GUP
@@ -1751,29 +1725,90 @@ static int pkvm_mem_abort(const struct kvm_s2_fault_desc *s2fd)
* pages backed by swap in the knowledge that the GUP pin will
* prevent try_to_unmap() from succeeding.
*/
- ret = -EIO;
- goto unpin;
+ unpin_user_page(page);
+ return -EIO;
}
+ pfn = page_to_pfn(page);
+ gfn = gpa_to_gfn(s2fd->fault_ipa);
+
+ ret = transparent_hugepage_adjust(kvm, s2fd->memslot, s2fd->hva, &pfn, &gfn);
+ if (ret < 0) {
+ unpin_user_page(page);
+ return ret;
+ } else if (ret == PMD_SIZE && WARN_ON_ONCE(folio_size(page_folio(page)) < PMD_SIZE)) {
+ unpin_user_page(page);
+ return -EINVAL;
+ }
+
+ *__page = page;
+ *__size = ret;
+ *__pfn = pfn;
+ *__gfn = gfn;
+
+ return 0;
+}
+
+static int pkvm_mem_abort(const struct kvm_s2_fault_desc *s2fd)
+{
+ struct kvm_vcpu *vcpu = s2fd->vcpu;
+ struct kvm_pgtable *pgt = vcpu->arch.hw_mmu->pgt;
+ struct mm_struct *mm = current->mm;
+ struct kvm *kvm = vcpu->kvm;
+ unsigned long size;
+ void *hyp_memcache;
+ struct page *page;
+ kvm_pfn_t pfn;
+ gfn_t gfn;
+ int ret;
+
+ hyp_memcache = get_mmu_memcache(vcpu);
+ ret = topup_mmu_memcache(vcpu, hyp_memcache);
+ if (ret)
+ return -ENOMEM;
+
+ ret = pkvm_pin_user_pages(s2fd, &page, &size, &pfn, &gfn);
+ if (ret)
+ return ret;
+
+ ret = account_locked_vm(mm, size / PAGE_SIZE, true);
+ if (ret)
+ goto unpin;
+
write_lock(&kvm->mmu_lock);
- ret = pkvm_pgtable_stage2_map(pgt, s2fd->fault_ipa, PAGE_SIZE,
- page_to_phys(page), KVM_PGTABLE_PROT_RWX,
- hyp_memcache, 0);
+ ret = pkvm_pgtable_stage2_map(pgt, gfn_to_gpa(gfn), size, __pfn_to_phys(pfn),
+ KVM_PGTABLE_PROT_RWX, hyp_memcache, 0);
write_unlock(&kvm->mmu_lock);
if (ret) {
if (ret == -EAGAIN)
ret = 0;
+
+ account_locked_vm(mm, size / PAGE_SIZE, false);
goto unpin;
}
return 0;
+
unpin:
- unpin_user_pages(&page, 1);
-dec_account:
- account_locked_vm(mm, 1, false);
+ unpin_user_page(page);
return ret;
}
+struct kvm_s2_fault_vma_info {
+ unsigned long mmu_seq;
+ long vma_pagesize;
+ vm_flags_t vm_flags;
+ unsigned long max_map_size;
+ struct page *page;
+ kvm_pfn_t pfn;
+ gfn_t gfn;
+ bool device;
+ bool mte_allowed;
+ bool is_vma_cacheable;
+ bool map_writable;
+ bool map_non_cacheable;
+};
+
static short kvm_s2_resolve_vma_size(const struct kvm_s2_fault_desc *s2fd,
struct kvm_s2_fault_vma_info *s2vi,
struct vm_area_struct *vma)
diff --git a/arch/arm64/kvm/pkvm.c b/arch/arm64/kvm/pkvm.c
index 2840053ef2f4..6a4f35067642 100644
--- a/arch/arm64/kvm/pkvm.c
+++ b/arch/arm64/kvm/pkvm.c
@@ -446,9 +446,8 @@ static int __pkvm_pgtable_stage2_reclaim(struct kvm_pgtable *pgt, u64 start, u64
continue;
page = pfn_to_page(mapping->pfn);
- WARN_ON_ONCE(mapping->nr_pages != 1);
unpin_user_pages_dirty_lock(&page, 1, true);
- account_locked_vm(kvm->mm, 1, false);
+ account_locked_vm(kvm->mm, mapping->nr_pages, false);
pkvm_mapping_remove(mapping, &pgt->pkvm_mappings);
kfree(mapping);
}
@@ -513,17 +512,23 @@ int pkvm_pgtable_stage2_map(struct kvm_pgtable *pgt, u64 addr, u64 size,
u64 end = addr + size;
int ret;
+ if (WARN_ON_ONCE(size != PAGE_SIZE && size != PMD_SIZE))
+ return -EINVAL;
+
lockdep_assert_held_write(&kvm->mmu_lock);
mapping = pkvm_mapping_iter_first(&pgt->pkvm_mappings, addr, end - 1);
if (kvm_vm_is_protected(kvm)) {
- /* Protected VMs are mapped using RWX page-granular mappings */
- if (WARN_ON_ONCE(size != PAGE_SIZE))
- return -EINVAL;
-
if (WARN_ON_ONCE(prot != KVM_PGTABLE_PROT_RWX))
return -EINVAL;
+ /*
+ * If a huge mapping overlaps an existing PAGE_SIZE one,
+ * then the VMM has played games with the stage-1. Abort.
+ */
+ if (WARN_ON_ONCE(mapping && mapping->nr_pages == 1 && size > PAGE_SIZE))
+ return -EFAULT;
+
/*
* We either raced with another vCPU or the guest PTE
* has been poisoned by an erroneous host access.
@@ -533,10 +538,8 @@ int pkvm_pgtable_stage2_map(struct kvm_pgtable *pgt, u64 addr, u64 size,
return ret ? -EFAULT : -EAGAIN;
}
- ret = kvm_call_hyp_nvhe(__pkvm_host_donate_guest, pfn, gfn, 1);
+ ret = kvm_call_hyp_nvhe(__pkvm_host_donate_guest, pfn, gfn, size / PAGE_SIZE);
} else {
- if (WARN_ON_ONCE(size != PAGE_SIZE && size != PMD_SIZE))
- return -EINVAL;
/*
* We either raced with another vCPU or we're changing between
--
2.55.0.1007.g17ff1f9808-goog
next prev parent reply other threads:[~2026-09-11 13:51 UTC|newest]
Thread overview: 27+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-11 13:50 [PATCH v2 00/22] Huge mapping support for protected VMs Vincent Donnefort
2026-09-11 13:50 ` [PATCH v2 01/22] KVM: arm64: Prefault host stage-2 entries on block split Vincent Donnefort
2026-09-24 17:38 ` Mostafa Saleh
2026-09-25 15:02 ` Vincent Donnefort
2026-09-11 13:50 ` [PATCH v2 02/22] KVM: arm64: Propagate host stage-2 annotated " Vincent Donnefort
2026-09-11 13:50 ` [PATCH v2 03/22] KVM: arm64: Allow block-level stage-2 annotation Vincent Donnefort
2026-09-11 13:50 ` [PATCH v2 04/22] KVM: arm64: Use block-level annotations when setting up the host stage-2 Vincent Donnefort
2026-09-11 13:50 ` [PATCH v2 05/22] KVM: arm64: Make pKVM ownership selftest an HVC Vincent Donnefort
2026-09-11 14:14 ` sashiko-bot
2026-09-11 13:50 ` [PATCH v2 06/22] KVM: arm64: Add a range to __pkvm_host_share/unshare_hyp() Vincent Donnefort
2026-09-11 13:50 ` [PATCH v2 07/22] KVM: arm64: Add a range to __pkvm_host_donate_guest() Vincent Donnefort
2026-09-11 13:50 ` [PATCH v2 08/22] KVM: arm64: Add a range to hyp_poison_page() Vincent Donnefort
2026-09-11 13:50 ` [PATCH v2 09/22] KVM: arm64: Add a range to __pkvm_host_reclaim_guest() Vincent Donnefort
2026-09-11 13:50 ` [PATCH v2 10/22] KVM: arm64: Add a range to __pkvm_guest_share_host() Vincent Donnefort
2026-09-11 13:50 ` [PATCH v2 11/22] KVM: arm64: Add a range to __pkvm_guest_unshare_host() Vincent Donnefort
2026-09-11 13:50 ` [PATCH v2 12/22] KVM: arm64: Handle huge mappings in __pkvm_host_force_reclaim_page_guest() Vincent Donnefort
2026-09-11 13:50 ` [PATCH v2 13/22] KVM: arm64: Handle huge mappings in __pkvm_vcpu_in_poison_fault() Vincent Donnefort
2026-09-11 13:50 ` [PATCH v2 14/22] KVM: arm64: Add a range to pKVM ownership selftest Vincent Donnefort
2026-09-11 13:50 ` [PATCH v2 15/22] KVM: arm64: Warn on pKVM guest stage-2 block collapse Vincent Donnefort
2026-09-11 13:50 ` [PATCH v2 16/22] KVM: arm64: Add pkvm_hyp_req infrastructure Vincent Donnefort
2026-09-11 13:50 ` [PATCH v2 17/22] KVM: arm64: Introduce kvm_pgtable_stage2_table_install() Vincent Donnefort
2026-09-11 13:50 ` [PATCH v2 18/22] KVM: arm64: Add __pkvm_host_split_guest HVC Vincent Donnefort
2026-09-11 13:50 ` [PATCH v2 19/22] KVM: arm64: Extend pKVM page ownership selftests to cover guest block split Vincent Donnefort
2026-09-11 13:50 ` [PATCH v2 20/22] KVM: arm64: Add PKVM_HYP_REQ_SPLIT Vincent Donnefort
2026-09-11 13:50 ` [PATCH v2 21/22] KVM: arm64: Raise PKVM_HYP_REQ_SPLIT on guest to host sharing Vincent Donnefort
2026-09-11 13:50 ` Vincent Donnefort [this message]
2026-09-11 14:20 ` [PATCH v2 22/22] KVM: arm64: Stage-2 huge mappings for protected VMs sashiko-bot
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260911135053.146435-23-vdonnefort@google.com \
--to=vdonnefort@google.com \
--cc=catalin.marinas@arm.com \
--cc=fuad.tabba@linux.dev \
--cc=joey.gouly@arm.com \
--cc=kernel-team@android.com \
--cc=kvmarm@lists.linux.dev \
--cc=linux-arm-kernel@lists.infradead.org \
--cc=maz@kernel.org \
--cc=oupton@kernel.org \
--cc=qperret@google.com \
--cc=seiden@linux.ibm.com \
--cc=suzuki.poulose@arm.com \
--cc=weilin.chang@arm.com \
--cc=will@kernel.org \
--cc=yuzenghui@huawei.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.