From: Gavin Shan <gshan@redhat.com>
To: Suzuki K Poulose <suzuki.poulose@arm.com>,
kvm@vger.kernel.org, kvmarm@lists.linux.dev
Cc: maz@kernel.org, will@kernel.org, catalin.marinas@arm.com,
linux-kernel@vger.kernel.org,
linux-arm-kernel@lists.infradead.org, steven.price@arm.com,
aneesh.kumar@kernel.org, oupton@kernel.org, joey.gouly@arm.com,
tabba@google.com, yuzenghui@huawei.com,
linux-coco@lists.linux.dev, gankulkarni@os.amperecomputing.com,
sdonthineni@nvidia.com, alpergun@google.com,
fj0570is@fujitsu.com, WeiLin.Chang@arm.com,
lpieralisi@kernel.org, enju.kohei@fujitsu.com,
sudeep.holla@arm.com, jonathan.cameron@oss.qualcomm.com
Subject: Re: [PATCH v22 11/23] KVM: arm64: Add VM specific callback for S2 MMU operations
Date: Tue, 6 Oct 2026 13:00:55 +1000 [thread overview]
Message-ID: <b6de5d5c-e7e5-4e24-abd6-bdea9bcf8db6@redhat.com> (raw)
In-Reply-To: <20261005090754.2140522-12-suzuki.poulose@arm.com>
On 10/5/26 7:07 PM, Suzuki K Poulose wrote:
> Add VM type specific S2 MMU operation backends which can be initialized per
> VM flavor, to keep the handling cleaner.
>
> Signed-off-by: Suzuki K Poulose <suzuki.poulose@arm.com>
> ---
> Change since v21:
> - Define all vm_s2_ops call back. All calls are mandatory.
> - Define callback for each flavor, disjointing the non-protetcted pKVM and
> normal KVM (VHE & nVHE) and remove the KVM_PGT_FN() hacks.
> - Dropped Reviews due to the changes.
> - Add "no_age_gfn" and "no_stage2_unmap_range" for pKVM callbacks, no_age_*
> to be also reused by Realms later.
> - Move kvm_vm_s2_ops field to keep the structure packed
> ---
> arch/arm64/include/asm/kvm_host.h | 14 +++
> arch/arm64/kvm/mmu.c | 173 +++++++++++++++++++++++++-----
> 2 files changed, 160 insertions(+), 27 deletions(-)
>
With the following nitpicks addressed:
Reviewed-by: Gavin Shan <gshan@redhat.com>
> diff --git a/arch/arm64/include/asm/kvm_host.h b/arch/arm64/include/asm/kvm_host.h
> index 839f5e9c7c65e..0778c308ce597 100644
> --- a/arch/arm64/include/asm/kvm_host.h
> +++ b/arch/arm64/include/asm/kvm_host.h
> @@ -155,6 +155,19 @@ struct kvm_vcpu_ops {
> void (*vcpu_put)(struct kvm_vcpu *vcpu);
> };
>
> +struct kvm_gfn_range;
> +
> +struct kvm_vm_s2_ops {
> + bool (*vm_age_gfn)(struct kvm *kvm, struct kvm_gfn_range *range);
> + bool (*vm_test_age_gfn)(struct kvm *kvm, struct kvm_gfn_range *range);
> + int (*vm_flush_remote_tlbs)(struct kvm *kvm);
> + int (*vm_flush_remote_tlbs_range)(struct kvm *kvm, gfn_t gfn,
> + u64 nr_pages);
> + void (*vm_stage2_unmap_range)(struct kvm_s2_mmu *mmu,
> + phys_addr_t start, u64 size,
> + bool may_block);
> +};
> +
> struct kvm_s2_mmu {
> struct kvm_vmid vmid;
>
> @@ -321,6 +334,7 @@ enum kvm_arm_vm_flavor {
>
> struct kvm_arch {
> struct kvm_s2_mmu mmu;
> + const struct kvm_vm_s2_ops *vm_s2_ops;
>
> enum kvm_arm_vm_flavor vm_flavor;
> /* Mandated version of PSCI */
> diff --git a/arch/arm64/kvm/mmu.c b/arch/arm64/kvm/mmu.c
> index 4ae3bb6baef1a..c046c76833876 100644
> --- a/arch/arm64/kvm/mmu.c
> +++ b/arch/arm64/kvm/mmu.c
> @@ -37,6 +37,8 @@ static unsigned long __ro_after_init io_map_base;
>
> #define KVM_PGT_FN(fn) (!is_protected_kvm_enabled() ? fn : p ## fn)
>
> +static int kvm_vm_init_vm_s2_ops(struct kvm *kvm);
> +
> static phys_addr_t __stage2_range_addr_end(phys_addr_t addr, phys_addr_t end,
> phys_addr_t size)
> {
> @@ -166,6 +168,18 @@ static bool memslot_is_logging(struct kvm_memory_slot *memslot)
> return memslot->dirty_bitmap && !(memslot->flags & KVM_MEM_READONLY);
> }
>
> +static int pkvm_flush_remote_tlbs(struct kvm *kvm)
> +{
> + kvm_call_hyp_nvhe(__pkvm_tlb_flush_vmid, kvm->arch.pkvm.handle);
> + return 0;
> +}
> +
> +static int kvm_vm_flush_remote_tlbs(struct kvm *kvm)
> +{
> + kvm_call_hyp(__kvm_tlb_flush_vmid, &kvm->arch.mmu);
> + return 0;
> +}
> +
> /**
> * kvm_arch_flush_remote_tlbs() - flush all VM TLB entries for v7/8
> * @kvm: pointer to kvm structure.
> @@ -174,26 +188,31 @@ static bool memslot_is_logging(struct kvm_memory_slot *memslot)
> */
> int kvm_arch_flush_remote_tlbs(struct kvm *kvm)
> {
> - if (is_protected_kvm_enabled())
> - kvm_call_hyp_nvhe(__pkvm_tlb_flush_vmid, kvm->arch.pkvm.handle);
> - else
> - kvm_call_hyp(__kvm_tlb_flush_vmid, &kvm->arch.mmu);
> - return 0;
> + return kvm->arch.vm_s2_ops->vm_flush_remote_tlbs(kvm);
> }
>
> -int kvm_arch_flush_remote_tlbs_range(struct kvm *kvm,
> - gfn_t gfn, u64 nr_pages)
> +static int pkvm_flush_remote_tlbs_range(struct kvm *kvm,
> + gfn_t gfn, u64 nr_pages)
> +{
> + return pkvm_flush_remote_tlbs(kvm);
No need to have another function call, which causes unnecessary overhead?
kvm_call_hyp_nvhe(__pkvm_tlb_flush_vmid, kvm->arch.pkvm.handle);
return 0;
> +}
> +
> +static int kvm_vm_flush_remote_tlbs_range(struct kvm *kvm,
> + gfn_t gfn, u64 nr_pages)
> {
> u64 size = nr_pages << PAGE_SHIFT;
> u64 addr = gfn << PAGE_SHIFT;
>
> - if (is_protected_kvm_enabled())
> - kvm_call_hyp_nvhe(__pkvm_tlb_flush_vmid, kvm->arch.pkvm.handle);
> - else
> - kvm_tlb_flush_vmid_range(&kvm->arch.mmu, addr, size);
> + kvm_tlb_flush_vmid_range(&kvm->arch.mmu, addr, size);
> return 0;
> }
>
Since we're here, the local variable 'addr' and 'size' can be dropped by:
kvm_tlb_flush_vmid_range(&kvm->arch.mmu,
gfn << PAGE_SHIFT,
nr_pages << PAGE_SHIFT);
> +int kvm_arch_flush_remote_tlbs_range(struct kvm *kvm,
> + gfn_t gfn, u64 nr_pages)
> +{
> + return kvm->arch.vm_s2_ops->vm_flush_remote_tlbs_range(kvm, gfn, nr_pages);
> +}
> +
> static void *stage2_memcache_zalloc_page(void *arg)
> {
> struct kvm_mmu_memory_cache *mc = arg;
> @@ -289,6 +308,27 @@ static void invalidate_icache_guest_page(void *va, size_t size)
> __invalidate_icache_guest_page(va, size);
> }
>
> +static void kvm_vm_stage2_unmap_range(struct kvm_s2_mmu *mmu,
> + phys_addr_t start,
> + u64 size, bool may_block)
> +{
> + WARN_ON(stage2_apply_range(mmu, start, start + size,
> + kvm_pgtable_stage2_unmap, may_block));
> +}
> +
> +static void pkvm_stage2_unmap_range(struct kvm_s2_mmu *mmu,
> + phys_addr_t start,
> + u64 size, bool may_block)
> +{
> + WARN_ON(stage2_apply_range(mmu, start, start + size,
> + pkvm_pgtable_stage2_unmap, may_block));
> +}
> +
> +static void no_stage2_unmap_range(struct kvm_s2_mmu *mmu,
> + phys_addr_t start, u64 size, bool may_block)
> +{
> +}
> +
> /*
> * Unmapping vs dcache management:
> *
> @@ -329,15 +369,10 @@ void kvm_stage2_unmap_range(struct kvm_s2_mmu *mmu, phys_addr_t start,
> {
> struct kvm *kvm = kvm_s2_mmu_to_kvm(mmu);
>
> - if (kvm_vm_is_protected(kvm))
> - return;
> -
> lockdep_assert_held_write(&kvm->mmu_lock);
> WARN_ON(size & ~PAGE_MASK);
>
> - WARN_ON(stage2_apply_range(mmu, start, start + size,
> - KVM_PGT_FN(kvm_pgtable_stage2_unmap),
> - may_block));
> + kvm->arch.vm_s2_ops->vm_stage2_unmap_range(mmu, start, size, may_block);
> }
>
> void kvm_stage2_flush_range(struct kvm_s2_mmu *mmu, phys_addr_t addr, phys_addr_t end)
> @@ -977,6 +1012,12 @@ int kvm_init_stage2_mmu(struct kvm *kvm, struct kvm_s2_mmu *mmu, unsigned long t
> int cpu, err;
> struct kvm_pgtable *pgt;
>
> + /* Initialize the VM ops for the VM instance for the first time */
> + if (mmu == &kvm->arch.mmu) {
> + err = kvm_vm_init_vm_s2_ops(kvm);
> + if (err)
> + return err;
> + }
> /*
> * If we already have our page tables in place, and that the
> * MMU context is the canonical one, we have a bug somewhere,
> @@ -2441,34 +2482,68 @@ bool kvm_unmap_gfn_range(struct kvm *kvm, struct kvm_gfn_range *range)
> return false;
> }
>
> -bool kvm_age_gfn(struct kvm *kvm, struct kvm_gfn_range *range)
> +static bool kvm_vm_age_gfn(struct kvm *kvm, struct kvm_gfn_range *range)
> {
> u64 size = (range->end - range->start) << PAGE_SHIFT;
>
> - if (!kvm->arch.mmu.pgt || kvm_vm_is_protected(kvm))
> - return false;
> -
> - return KVM_PGT_FN(kvm_pgtable_stage2_test_clear_young)(kvm->arch.mmu.pgt,
> + return kvm_pgtable_stage2_test_clear_young(kvm->arch.mmu.pgt,
> range->start << PAGE_SHIFT,
> size, true);
> +}
> +
> +static bool pkvm_age_gfn(struct kvm *kvm, struct kvm_gfn_range *range)
> +{
> + u64 size = (range->end - range->start) << PAGE_SHIFT;
> +
> + return pkvm_pgtable_stage2_test_clear_young(kvm->arch.mmu.pgt,
> + range->start << PAGE_SHIFT,
> + size, true);
> +}
> +
> +static bool no_age_gfn(struct kvm *kvm, struct kvm_gfn_range *range)
> +{
> + /* The hypervisor doesn't support aging */
> + return false;
> +}
> +
> +bool kvm_age_gfn(struct kvm *kvm, struct kvm_gfn_range *range)
> +{
> + if (!kvm->arch.mmu.pgt)
> + return false;
> +
> + return kvm->arch.vm_s2_ops->vm_age_gfn(kvm, range);
> /*
> * TODO: Handle nested_mmu structures here using the reverse mapping in
> * a later version of patch series.
> */
> }
>
> -bool kvm_test_age_gfn(struct kvm *kvm, struct kvm_gfn_range *range)
> +static bool kvm_vm_test_age_gfn(struct kvm *kvm, struct kvm_gfn_range *range)
> {
> u64 size = (range->end - range->start) << PAGE_SHIFT;
>
> - if (!kvm->arch.mmu.pgt || kvm_vm_is_protected(kvm))
> - return false;
> -
> - return KVM_PGT_FN(kvm_pgtable_stage2_test_clear_young)(kvm->arch.mmu.pgt,
> + return kvm_pgtable_stage2_test_clear_young(kvm->arch.mmu.pgt,
> range->start << PAGE_SHIFT,
> size, false);
> }
>
> +static bool pkvm_test_age_gfn(struct kvm *kvm, struct kvm_gfn_range *range)
> +{
> + u64 size = (range->end - range->start) << PAGE_SHIFT;
> +
> + return pkvm_pgtable_stage2_test_clear_young(kvm->arch.mmu.pgt,
> + range->start << PAGE_SHIFT,
> + size, false);
> +}
> +
> +bool kvm_test_age_gfn(struct kvm *kvm, struct kvm_gfn_range *range)
> +{
> + if (!kvm->arch.mmu.pgt)
> + return false;
> +
> + return kvm->arch.vm_s2_ops->vm_test_age_gfn(kvm, range);
> +}
> +
> phys_addr_t kvm_mmu_get_httbr(void)
> {
> return __pa(hyp_pgtable->pgd);
> @@ -2790,3 +2865,47 @@ void kvm_toggle_cache(struct kvm_vcpu *vcpu, bool was_enabled)
>
> trace_kvm_toggle_cache(*vcpu_pc(vcpu), was_enabled, now_enabled);
> }
> +
> +static const struct kvm_vm_s2_ops protected_pkvm_vm_s2_ops = {
> + .vm_flush_remote_tlbs = pkvm_flush_remote_tlbs,
> + .vm_flush_remote_tlbs_range = pkvm_flush_remote_tlbs_range,
> + .vm_age_gfn = no_age_gfn,
> + .vm_test_age_gfn = no_age_gfn,
> + .vm_stage2_unmap_range = no_stage2_unmap_range,
> +};
> +
> +static const struct kvm_vm_s2_ops pkvm_vm_s2_ops = {
> + .vm_flush_remote_tlbs = pkvm_flush_remote_tlbs,
> + .vm_flush_remote_tlbs_range = pkvm_flush_remote_tlbs_range,
> + .vm_age_gfn = pkvm_age_gfn,
> + .vm_test_age_gfn = pkvm_test_age_gfn,
> + .vm_stage2_unmap_range = pkvm_stage2_unmap_range,
> +};
> +
> +static const struct kvm_vm_s2_ops kvm_default_vm_s2_ops = {
> + .vm_flush_remote_tlbs = kvm_vm_flush_remote_tlbs,
> + .vm_flush_remote_tlbs_range = kvm_vm_flush_remote_tlbs_range,
> + .vm_age_gfn = kvm_vm_age_gfn,
> + .vm_test_age_gfn = kvm_vm_test_age_gfn,
> + .vm_stage2_unmap_range = kvm_vm_stage2_unmap_range,
> +};
> +
> +#define KVM_VM_S2_OPS(flavor, ops) \
> + [flavor] = &(ops)
s/[flavor]/[(flavor)]
> +
> +static const struct kvm_vm_s2_ops *arm64_vm_s2_ops[] = {
> + KVM_VM_S2_OPS(VM_VHE, kvm_default_vm_s2_ops),
> + KVM_VM_S2_OPS(VM_NVHE, kvm_default_vm_s2_ops),
> + KVM_VM_S2_OPS(VM_PKVM, pkvm_vm_s2_ops),
> + KVM_VM_S2_OPS(VM_PROTECTED_PKVM, protected_pkvm_vm_s2_ops),
> +};
> +
> +static int kvm_vm_init_vm_s2_ops(struct kvm *kvm)
> +{
> + BUILD_BUG_ON(ARRAY_SIZE(arm64_vm_s2_ops) != VM_FLAVOR_MAX);
> +
> + kvm->arch.vm_s2_ops = arm64_vm_s2_ops[kvm->arch.vm_flavor];
> + if (WARN_ON(!kvm->arch.vm_s2_ops))
> + return -EINVAL;
> + return 0;
> +}
Thanks,
Gavin
next prev parent reply other threads:[~2026-10-06 3:01 UTC|newest]
Thread overview: 64+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-10-05 9:07 [PATCH v22 00/23] KVM: arm64: CCA: Add basic plumbing for Realms Suzuki K Poulose
2026-10-05 9:07 ` [PATCH v22 01/23] KVM: arm64: protected VM: Handle user writes to CNTVCT_EL0/CNTPCT_EL0 Suzuki K Poulose
2026-10-06 0:02 ` Gavin Shan
2026-10-05 9:07 ` [PATCH v22 02/23] KVM: arm64: Disable Steal time accounting for protected guests Suzuki K Poulose
2026-10-06 0:03 ` Gavin Shan
2026-10-05 9:07 ` [PATCH v22 03/23] KVM: arm64: Include kvm_emulate.h in kvm/arm_psci.h Suzuki K Poulose
2026-10-05 9:07 ` [PATCH v22 04/23] KVM: arm64: Avoid including linux/kvm_host.h in kvm_pgtable.h Suzuki K Poulose
2026-10-05 9:07 ` [PATCH v22 05/23] KVM: arm64: Track the type of VM in kvm_arch Suzuki K Poulose
2026-10-06 3:55 ` Gavin Shan
2026-10-06 8:33 ` Marc Zyngier
2026-10-06 8:49 ` Suzuki K Poulose
2026-10-05 9:07 ` [PATCH v22 06/23] KVM: arm64: Don't call vcpu_set_pauth_traps for pKVM host Suzuki K Poulose
2026-10-06 0:29 ` Gavin Shan
2026-10-05 9:07 ` [PATCH v22 07/23] KVM: arm64: Refactor the vcpu_load to allow for VM specific callbacks Suzuki K Poulose
2026-10-05 9:07 ` [PATCH v22 08/23] KVM: arm64: Add vcpu load/put call backs for flavors Suzuki K Poulose
2026-10-06 2:15 ` Gavin Shan
2026-10-05 9:07 ` [PATCH v22 09/23] KVM: arm64: Prevent unsupported vcpu features for VM types Suzuki K Poulose
2026-10-06 2:24 ` Gavin Shan
2026-10-06 2:25 ` Gavin Shan
2026-10-06 5:16 ` Suzuki K Poulose
2026-10-06 8:50 ` Marc Zyngier
2026-10-05 9:07 ` [PATCH v22 10/23] KVM: arm64: Consolidate stage2 unmap range into kvm_stage2_unmap_range Suzuki K Poulose
2026-10-06 2:37 ` Gavin Shan
2026-10-05 9:07 ` [PATCH v22 11/23] KVM: arm64: Add VM specific callback for S2 MMU operations Suzuki K Poulose
2026-10-06 3:00 ` Gavin Shan [this message]
2026-10-06 5:22 ` Suzuki K Poulose
2026-10-06 9:24 ` Marc Zyngier
2026-10-06 10:36 ` Suzuki K Poulose
2026-10-06 15:14 ` Suzuki K Poulose
2026-10-05 9:07 ` [PATCH v22 12/23] KVM: arm64: Use a local kvm pointer in kvm_handle_guest_abort() Suzuki K Poulose
2026-10-06 3:02 ` Gavin Shan
2026-10-05 9:07 ` [PATCH v22 13/23] KVM: arm64: Abstract out memory abort handling Suzuki K Poulose
2026-10-06 3:07 ` Gavin Shan
2026-10-06 5:25 ` Suzuki K Poulose
2026-10-05 9:07 ` [PATCH v22 14/23] KVM: arm64: Mandate VGIC v3 for pKVM VMs and Realms Suzuki K Poulose
2026-10-06 3:10 ` Gavin Shan
2026-10-05 9:07 ` [PATCH v22 15/23] KVM: arm64: CCA: Add a new mode for supporting Realm guests Suzuki K Poulose
2026-10-06 3:11 ` Gavin Shan
2026-10-05 9:07 ` [PATCH v22 16/23] KVM: arm64: CCA: Add VCPU load/put for Realms Suzuki K Poulose
2026-10-06 3:16 ` Gavin Shan
2026-10-06 5:09 ` Suzuki K Poulose
2026-10-05 9:07 ` [PATCH v22 17/23] KVM: arm64: CCA: Add bare minimal S2 operations for Realm Suzuki K Poulose
2026-10-06 3:18 ` Gavin Shan
2026-10-05 9:07 ` [PATCH v22 18/23] KVM: arm64: CCA: Introduce Realms Suzuki K Poulose
2026-10-06 3:42 ` Gavin Shan
2026-10-06 5:10 ` Suzuki K Poulose
2026-10-05 9:07 ` [PATCH v22 19/23] KVM: arm64: CCA: Don't expose unsupported capabilities for realm guests Suzuki K Poulose
2026-10-06 4:58 ` Gavin Shan
2026-10-05 9:07 ` [PATCH v22 20/23] KVM: arm64: CCA: WARN on injected undef exceptions Suzuki K Poulose
2026-10-06 3:49 ` Gavin Shan
2026-10-05 9:07 ` [PATCH v22 21/23] KVM: arm64: CCA: Support timers in realm RECs Suzuki K Poulose
2026-10-06 5:44 ` Gavin Shan
2026-10-05 9:07 ` [PATCH v22 22/23] KVM: arm64: CCA: Expose SVE VL register before VCPU finalization Suzuki K Poulose
2026-10-06 5:35 ` Gavin Shan
2026-10-06 5:57 ` Suzuki K Poulose
2026-10-05 9:07 ` [PATCH v22 23/23] KVM: arm64: CCA: Control user register access for Realms Suzuki K Poulose
2026-10-05 9:30 ` sashiko-bot
2026-10-05 13:08 ` Suzuki K Poulose
2026-10-06 5:47 ` Gavin Shan
2026-10-06 6:01 ` Suzuki K Poulose
2026-10-06 6:16 ` Gavin Shan
2026-10-06 12:36 ` Suzuki K Poulose
2026-10-06 22:00 ` Gavin Shan
2026-10-06 6:23 ` Gavin Shan
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=b6de5d5c-e7e5-4e24-abd6-bdea9bcf8db6@redhat.com \
--to=gshan@redhat.com \
--cc=WeiLin.Chang@arm.com \
--cc=alpergun@google.com \
--cc=aneesh.kumar@kernel.org \
--cc=catalin.marinas@arm.com \
--cc=enju.kohei@fujitsu.com \
--cc=fj0570is@fujitsu.com \
--cc=gankulkarni@os.amperecomputing.com \
--cc=joey.gouly@arm.com \
--cc=jonathan.cameron@oss.qualcomm.com \
--cc=kvm@vger.kernel.org \
--cc=kvmarm@lists.linux.dev \
--cc=linux-arm-kernel@lists.infradead.org \
--cc=linux-coco@lists.linux.dev \
--cc=linux-kernel@vger.kernel.org \
--cc=lpieralisi@kernel.org \
--cc=maz@kernel.org \
--cc=oupton@kernel.org \
--cc=sdonthineni@nvidia.com \
--cc=steven.price@arm.com \
--cc=sudeep.holla@arm.com \
--cc=suzuki.poulose@arm.com \
--cc=tabba@google.com \
--cc=will@kernel.org \
--cc=yuzenghui@huawei.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.