From: Gavin Shan <gshan@redhat.com>
To: Suzuki K Poulose <suzuki.poulose@arm.com>,
kvm@vger.kernel.org, kvmarm@lists.linux.dev
Cc: maz@kernel.org, will@kernel.org, catalin.marinas@arm.com,
linux-kernel@vger.kernel.org,
linux-arm-kernel@lists.infradead.org, steven.price@arm.com,
aneesh.kumar@kernel.org, oupton@kernel.org, joey.gouly@arm.com,
tabba@google.com, yuzenghui@huawei.com,
linux-coco@lists.linux.dev, gankulkarni@os.amperecomputing.com,
sdonthineni@nvidia.com, alpergun@google.com,
fj0570is@fujitsu.com, WeiLin.Chang@arm.com,
lpieralisi@kernel.org, enju.kohei@fujitsu.com
Subject: Re: [PATCH v19 09/20] KVM: arm64: Add VM specific callback for S2 MMU operations
Date: Mon, 28 Sep 2026 11:25:46 +1000 [thread overview]
Message-ID: <df7e9870-1176-4679-bd79-807c0d87a4f2@redhat.com> (raw)
In-Reply-To: <647ae455-4175-4070-a40e-d2d89f18971f@redhat.com>
On 9/28/26 11:09 AM, Gavin Shan wrote:
> On 9/21/26 7:28 AM, Suzuki K Poulose wrote:
>> Add VM type specific S2 MMU operation backends which can be initialized per
>> VM flavor, to keep the handling cleaner.
>>
>> Signed-off-by: Suzuki K Poulose <suzuki.poulose@arm.com>
>> ---
>> arch/arm64/include/asm/kvm_host.h | 15 ++++
>> arch/arm64/kvm/mmu.c | 137 +++++++++++++++++++++++++-----
>> 2 files changed, 131 insertions(+), 21 deletions(-)
>>
>
> Apart from the comments from Jonathan, some nitpicks and questions below.
>
>> diff --git a/arch/arm64/include/asm/kvm_host.h b/arch/arm64/include/asm/kvm_host.h
>> index 149f4582c8b6a..7664d8b8cce5a 100644
>> --- a/arch/arm64/include/asm/kvm_host.h
>> +++ b/arch/arm64/include/asm/kvm_host.h
>> @@ -155,6 +155,19 @@ struct kvm_vcpu_ops {
>> void (*vcpu_put)(struct kvm_vcpu *vcpu);
>> };
>> +struct kvm_gfn_range;
>> +
>> +struct kvm_vm_s2_ops {
>> + bool (*vm_age_gfn)(struct kvm *kvm, struct kvm_gfn_range *range);
>> + bool (*vm_test_age_gfn)(struct kvm *kvm, struct kvm_gfn_range *range);
>> + int (*vm_flush_remote_tlbs)(struct kvm *kvm);
>> + int (*vm_flush_remote_tlbs_range)(struct kvm *kvm, gfn_t gfn,
>> + u64 nr_pages);
>> + void (*vm_stage2_unmap_range)(struct kvm_s2_mmu *mmu,
>> + phys_addr_t start, u64 size,
>> + bool may_block);
>> +};
>> +
>> struct kvm_s2_mmu {
>> struct kvm_vmid vmid;
>> @@ -332,6 +345,8 @@ struct kvm_arch {
>> */
>> u64 fgu[__NR_FGT_GROUP_IDS__];
>> + const struct kvm_vm_s2_ops *vm_s2_ops;
>> +
>> /*
>> * Stage 2 paging state for VMs with nested S2 using a virtual
>> * VMID.
>> diff --git a/arch/arm64/kvm/mmu.c b/arch/arm64/kvm/mmu.c
>> index 03f2017a7404a..d97a4a1bca23f 100644
>> --- a/arch/arm64/kvm/mmu.c
>> +++ b/arch/arm64/kvm/mmu.c
>> @@ -37,6 +37,8 @@ static unsigned long __ro_after_init io_map_base;
>> #define KVM_PGT_FN(fn) (!is_protected_kvm_enabled() ? fn : p ## fn)
>> +static int kvm_vm_init_vm_s2_ops(struct kvm *kvm);
>> +
>> static phys_addr_t __stage2_range_addr_end(phys_addr_t addr, phys_addr_t end,
>> phys_addr_t size)
>> {
>> @@ -166,6 +168,18 @@ static bool memslot_is_logging(struct kvm_memory_slot *memslot)
>> return memslot->dirty_bitmap && !(memslot->flags & KVM_MEM_READONLY);
>> }
>> +static int pkvm_flush_remote_tlbs(struct kvm *kvm)
>> +{
>> + kvm_call_hyp_nvhe(__pkvm_tlb_flush_vmid, kvm->arch.pkvm.handle);
>> + return 0;
>> +}
>> +
>> +static int kvm_vm_flush_remote_tlbs(struct kvm *kvm)
>> +{
>> + kvm_call_hyp(__kvm_tlb_flush_vmid, &kvm->arch.mmu);
>> + return 0;
>> +}
>> +
>> /**
>> * kvm_arch_flush_remote_tlbs() - flush all VM TLB entries for v7/8
>> * @kvm: pointer to kvm structure.
>> @@ -174,26 +188,36 @@ static bool memslot_is_logging(struct kvm_memory_slot *memslot)
>> */
>> int kvm_arch_flush_remote_tlbs(struct kvm *kvm)
>> {
>> - if (is_protected_kvm_enabled())
>> - kvm_call_hyp_nvhe(__pkvm_tlb_flush_vmid, kvm->arch.pkvm.handle);
>> - else
>> - kvm_call_hyp(__kvm_tlb_flush_vmid, &kvm->arch.mmu);
>> - return 0;
>> + if (!kvm->arch.vm_s2_ops->vm_flush_remote_tlbs)
>> + return 1;
>
> For the return value, I'm wandering if 0 should be returned. More details
> can be found below.
>
>> + return kvm->arch.vm_s2_ops->vm_flush_remote_tlbs(kvm);
>> }
>> -int kvm_arch_flush_remote_tlbs_range(struct kvm *kvm,
>> - gfn_t gfn, u64 nr_pages)
>> +static int pkvm_flush_remote_tlbs_range(struct kvm *kvm,
>> + gfn_t gfn, u64 nr_pages)
>> +{
>> + return pkvm_flush_remote_tlbs(kvm);
>> +}
>> +
>> +static int kvm_vm_flush_remote_tlbs_range(struct kvm *kvm,
>> + gfn_t gfn, u64 nr_pages)
>> {
>> u64 size = nr_pages << PAGE_SHIFT;
>> u64 addr = gfn << PAGE_SHIFT;
>> - if (is_protected_kvm_enabled())
>> - kvm_call_hyp_nvhe(__pkvm_tlb_flush_vmid, kvm->arch.pkvm.handle);
>> - else
>> - kvm_tlb_flush_vmid_range(&kvm->arch.mmu, addr, size);
>> + kvm_tlb_flush_vmid_range(&kvm->arch.mmu, addr, size);
>> return 0;
>> }
>> +int kvm_arch_flush_remote_tlbs_range(struct kvm *kvm,
>> + gfn_t gfn, u64 nr_pages)
>> +{
>> + if (!kvm->arch.vm_s2_ops->vm_flush_remote_tlbs_range)
>> + return 1;
>> +
>
> Realm would the only case where vm_s2_ops->vm_flush_remote_{tlbs, tlbs_range)
> are NULL. On request to flush remote TLBs by kvm_flush_remote_tlbs_range(), it
> ends up with event KVM_REQ_TLB_FLUSH queued for each vCPU. How this queued event
> is linked to a remote TLB flush for realm? The problem is TLBs are owned by EL2
> realm and there are no RMI calls for the management. So I'm wandering we should
> return 0 here?
>
vm_s2_ops->vm_flush_remote_{tlbs, tlbs_range} are added in PATCH[14] where 0 is
returned for both function. So I guess needn't this check at all?
if (!kvm->arch.vm_s2_ops->vm_flush_remote_tlbs_range)
>> + return kvm->arch.vm_s2_ops->vm_flush_remote_tlbs_range(kvm, gfn, nr_pages);
>> +}
>> +
>> static void *stage2_memcache_zalloc_page(void *arg)
>> {
>> struct kvm_mmu_memory_cache *mc = arg;
>> @@ -337,13 +361,20 @@ static void __unmap_stage2_range(struct kvm_s2_mmu *mmu, phys_addr_t start, u64
>> may_block));
>> }
>> +static void kvm_vm_stage2_unmap_range(struct kvm_s2_mmu *mmu,
>> + phys_addr_t start,
>> + u64 size, bool may_block)
>> +{
>> + __unmap_stage2_range(mmu, start, size, may_block);
>> +}
>> +
>> void kvm_stage2_unmap_range(struct kvm_s2_mmu *mmu, phys_addr_t start,
>> u64 size, bool may_block)
>> {
>> - if (kvm_vm_is_protected(kvm_s2_mmu_to_kvm(mmu)))
>> - return;
>> + struct kvm *kvm = kvm_s2_mmu_to_kvm(mmu);
>> - __unmap_stage2_range(mmu, start, size, may_block);
>> + if (kvm->arch.vm_s2_ops->vm_stage2_unmap_range)
>> + kvm->arch.vm_s2_ops->vm_stage2_unmap_range(mmu, start, size, may_block);
>> }
>> void kvm_stage2_flush_range(struct kvm_s2_mmu *mmu, phys_addr_t addr, phys_addr_t end)
>> @@ -983,6 +1014,12 @@ int kvm_init_stage2_mmu(struct kvm *kvm, struct kvm_s2_mmu *mmu, unsigned long t
>> int cpu, err;
>> struct kvm_pgtable *pgt;
>> + /* Initialize the VM ops for the VM instance for the first time */
>> + if (mmu == &kvm->arch.mmu) {
>> + err = kvm_vm_init_vm_s2_ops(kvm);
>> + if (err)
>> + return err;
>> + }
>> /*
>> * If we already have our page tables in place, and that the
>> * MMU context is the canonical one, we have a bug somewhere,
>> @@ -2447,34 +2484,46 @@ bool kvm_unmap_gfn_range(struct kvm *kvm, struct kvm_gfn_range *range)
>> return false;
>> }
>> -bool kvm_age_gfn(struct kvm *kvm, struct kvm_gfn_range *range)
>> +static bool kvm_vm_age_gfn(struct kvm *kvm, struct kvm_gfn_range *range)
>> {
>> u64 size = (range->end - range->start) << PAGE_SHIFT;
>> - if (!kvm->arch.mmu.pgt || kvm_vm_is_protected(kvm))
>> - return false;
>> -
>> return KVM_PGT_FN(kvm_pgtable_stage2_test_clear_young)(kvm->arch.mmu.pgt,
>> range->start << PAGE_SHIFT,
>> size, true);
>> +}
>> +
>> +bool kvm_age_gfn(struct kvm *kvm, struct kvm_gfn_range *range)
>> +{
>> + if (!kvm->arch.mmu.pgt || !kvm->arch.vm_s2_ops->vm_age_gfn)
>> + return false;
>> +
>> + return kvm->arch.vm_s2_ops->vm_age_gfn(kvm, range);
>> /*
>> * TODO: Handle nested_mmu structures here using the reverse mapping in
>> * a later version of patch series.
>> */
>> }
>> -bool kvm_test_age_gfn(struct kvm *kvm, struct kvm_gfn_range *range)
>> +static bool kvm_vm_test_age_gfn(struct kvm *kvm, struct kvm_gfn_range *range)
>> {
>> u64 size = (range->end - range->start) << PAGE_SHIFT;
>> - if (!kvm->arch.mmu.pgt || kvm_vm_is_protected(kvm))
>> - return false;
>> return KVM_PGT_FN(kvm_pgtable_stage2_test_clear_young)(kvm->arch.mmu.pgt,
>> range->start << PAGE_SHIFT,
>> size, false);
>> }
>> +bool kvm_test_age_gfn(struct kvm *kvm, struct kvm_gfn_range *range)
>> +{
>> +
>> + if (!kvm->arch.mmu.pgt || !kvm->arch.vm_s2_ops->vm_test_age_gfn)
>> + return false;
>> +
>> + return kvm->arch.vm_s2_ops->vm_test_age_gfn(kvm, range);
>> +}
>> +
>> phys_addr_t kvm_mmu_get_httbr(void)
>> {
>> return __pa(hyp_pgtable->pgd);
>> @@ -2796,3 +2845,49 @@ void kvm_toggle_cache(struct kvm_vcpu *vcpu, bool was_enabled)
>> trace_kvm_toggle_cache(*vcpu_pc(vcpu), was_enabled, now_enabled);
>> }
>> +
>> +static const struct kvm_vm_s2_ops protected_pkvm_vm_s2_ops = {
>> + .vm_flush_remote_tlbs = pkvm_flush_remote_tlbs,
>> + .vm_flush_remote_tlbs_range = pkvm_flush_remote_tlbs_range,
>> + /*
>> + * Not supported for Protected VMs under pKVM
>> + * .vm_age_gfn
>> + * .vm_test_age_gfn
>> + * .vm_stage2_unmap_range
>> + */
>> +};
>> +
>> +static const struct kvm_vm_s2_ops pkvm_vm_s2_ops = {
>> + .vm_flush_remote_tlbs = pkvm_flush_remote_tlbs,
>> + .vm_flush_remote_tlbs_range = pkvm_flush_remote_tlbs_range,
>> + .vm_age_gfn = kvm_vm_age_gfn,
>> + .vm_test_age_gfn = kvm_vm_test_age_gfn,
>> + .vm_stage2_unmap_range = kvm_vm_stage2_unmap_range,
>> +};
>> +
>> +static const struct kvm_vm_s2_ops kvm_default_vm_s2_ops = {
>> + .vm_flush_remote_tlbs = kvm_vm_flush_remote_tlbs,
>> + .vm_flush_remote_tlbs_range = kvm_vm_flush_remote_tlbs_range,
>> + .vm_age_gfn = kvm_vm_age_gfn,
>> + .vm_test_age_gfn = kvm_vm_test_age_gfn,
>> + .vm_stage2_unmap_range = kvm_vm_stage2_unmap_range,
>> +};
>> +
>> +#define KVM_VM_S2_OPS(flavor, ops) \
>> + [flavor] = ops
>
> Parentheses are needed, to be consistent with KVM_VCPU_OPS at least.
>
> #define KVM_VM_S2_OPS(flavor, ops) \
> [(flavor)] = (ops)
>
> Actually, KVM_{VCPU, VM_S2}_OPS() can be combined to one in kvm_host.h as below.
>
> #define KVM_FLAVOR_OPS() [(flavor)] = (ops)
>
>> +static const struct kvm_vm_s2_ops *arm64_vm_s2_ops[] = {
>> + KVM_VM_S2_OPS(VM_VHE, &kvm_default_vm_s2_ops),
>> + KVM_VM_S2_OPS(VM_NVHE, &kvm_default_vm_s2_ops),
>> + KVM_VM_S2_OPS(VM_PKVM, &pkvm_vm_s2_ops),
>> + KVM_VM_S2_OPS(VM_PROTECTED_PKVM, &protected_pkvm_vm_s2_ops),
>> +};
>> +
>> +static int kvm_vm_init_vm_s2_ops(struct kvm *kvm)
>> +{
>> + BUILD_BUG_ON(ARRAY_SIZE(arm64_vm_s2_ops) != VM_FLAVOR_MAX);
>> +
>> + kvm->arch.vm_s2_ops = arm64_vm_s2_ops[kvm->arch.vm_flavor];
>> + if (WARN_ON(!kvm->arch.vm_s2_ops))
>> + return -EINVAL;
>> + return 0;
>> +}
Thanks,
Gavin
next prev parent reply other threads:[~2026-09-28 1:25 UTC|newest]
Thread overview: 77+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-20 21:28 [PATCH v19 00/20] KVM: arm64: CCA: Add basic plumbing for Realms Suzuki K Poulose
2026-09-20 21:28 ` [PATCH v19 01/20] KVM: arm64: protected VM: Handle user writes to CNTVCT_EL0/CNTPCT_EL0 Suzuki K Poulose
2026-09-22 19:25 ` Jonathan Cameron
2026-09-22 21:53 ` Suzuki K Poulose
2026-09-23 16:48 ` Jonathan Cameron
2026-09-22 22:04 ` Suzuki K Poulose
2026-09-23 16:51 ` Jonathan Cameron
2026-09-20 21:28 ` [PATCH v19 02/20] KVM: arm64: Disable Steal time accounting for protected guests Suzuki K Poulose
2026-09-20 21:44 ` sashiko-bot
2026-09-20 22:24 ` Suzuki K Poulose
2026-09-21 23:07 ` Suzuki K Poulose
2026-09-28 0:14 ` Gavin Shan
2026-09-20 21:28 ` [PATCH v19 03/20] KVM: arm64: Include kvm_emulate.h in kvm/arm_psci.h Suzuki K Poulose
2026-09-22 19:32 ` Jonathan Cameron
2026-09-20 21:28 ` [PATCH v19 04/20] KVM: arm64: Avoid including linux/kvm_host.h in kvm_pgtable.h Suzuki K Poulose
2026-09-20 21:38 ` sashiko-bot
2026-09-21 8:25 ` Suzuki K Poulose
2026-09-20 21:28 ` [PATCH v19 05/20] KVM: arm64: Track the type of VM in kvm_arch Suzuki K Poulose
2026-09-20 21:38 ` sashiko-bot
2026-09-21 8:18 ` Suzuki K Poulose
2026-09-22 19:40 ` Jonathan Cameron
2026-09-23 6:05 ` Gavin Shan
2026-09-23 6:19 ` Gavin Shan
2026-09-23 10:24 ` Suzuki K Poulose
2026-09-23 13:23 ` Gavin Shan
2026-09-23 13:29 ` Gavin Shan
2026-09-23 13:54 ` Suzuki K Poulose
2026-09-23 16:27 ` Suzuki K Poulose
2026-09-23 21:37 ` Suzuki K Poulose
2026-09-24 1:11 ` Gavin Shan
2026-09-24 8:48 ` Suzuki K Poulose
2026-09-24 10:37 ` Gavin Shan
2026-09-20 21:28 ` [PATCH v19 06/20] KVM: arm64: Refactor the vcpu_load to allow for VM specific callbacks Suzuki K Poulose
2026-09-22 19:57 ` Jonathan Cameron
2026-09-22 22:09 ` Suzuki K Poulose
2026-09-20 21:28 ` [PATCH v19 07/20] KVM: arm64: Add vcpu load/put call backs for flavors Suzuki K Poulose
2026-09-22 22:12 ` Jonathan Cameron
2026-09-20 21:28 ` [PATCH v19 08/20] KVM: arm64: Reuse kvm_stage2_unmap_range in kvm_unmap_gfn_range Suzuki K Poulose
2026-09-22 22:15 ` Jonathan Cameron
2026-09-28 0:17 ` Gavin Shan
2026-09-20 21:28 ` [PATCH v19 09/20] KVM: arm64: Add VM specific callback for S2 MMU operations Suzuki K Poulose
2026-09-22 22:29 ` Jonathan Cameron
2026-09-22 23:21 ` Suzuki K Poulose
2026-09-23 16:54 ` Jonathan Cameron
2026-09-24 15:11 ` Suzuki K Poulose
2026-09-28 1:09 ` Gavin Shan
2026-09-28 1:25 ` Gavin Shan [this message]
2026-09-28 8:13 ` Suzuki K Poulose
2026-09-28 8:10 ` Suzuki K Poulose
2026-09-20 21:28 ` [PATCH v19 10/20] KVM: arm64: Abstract out memory abort handling Suzuki K Poulose
2026-09-22 22:38 ` Jonathan Cameron
2026-09-22 23:55 ` Suzuki K Poulose
2026-09-20 21:28 ` [PATCH v19 11/20] KVM: arm64: Mandate VGIC v3 for for VMs running on hyp that don't trust the host Suzuki K Poulose
2026-09-22 22:42 ` Jonathan Cameron
2026-09-22 23:38 ` Suzuki K Poulose
2026-09-28 1:10 ` Gavin Shan
2026-09-20 21:28 ` [PATCH v19 12/20] KVM: arm64: CCA: Add a new mode for supporting Realm guests Suzuki K Poulose
2026-09-22 22:43 ` Jonathan Cameron
2026-09-20 21:28 ` [PATCH v19 13/20] KVM: arm64: CCA: Add VCPU load/put for Realms Suzuki K Poulose
2026-09-28 1:22 ` Gavin Shan
2026-09-20 21:28 ` [PATCH v19 14/20] KVM: arm64: CCA: Add bare minimal S2 operations for Realm Suzuki K Poulose
2026-09-28 1:27 ` Gavin Shan
2026-09-20 21:28 ` [PATCH v19 15/20] KVM: arm64: CCA: Introduce Realms Suzuki K Poulose
2026-09-22 22:49 ` Jonathan Cameron
2026-09-28 1:27 ` Gavin Shan
2026-09-20 21:28 ` [PATCH v19 16/20] KVM: arm64: CCA: Don't expose unsupported capabilities for realm guests Suzuki K Poulose
2026-09-22 22:53 ` Jonathan Cameron
2026-09-20 21:28 ` [PATCH v19 17/20] KVM: arm64: CCA: WARN on injected undef exceptions Suzuki K Poulose
2026-09-22 22:54 ` Jonathan Cameron
2026-09-28 1:28 ` Gavin Shan
2026-09-20 21:28 ` [PATCH v19 18/20] KVM: arm64: CCA: Support timers in realm RECs Suzuki K Poulose
2026-09-20 21:28 ` [PATCH v19 19/20] KVM: arm64: CCA: Expose SVE VL register before VCPU finalization Suzuki K Poulose
2026-09-28 1:29 ` Gavin Shan
2026-09-20 21:28 ` [PATCH v19 20/20] KVM: arm64: CCA: Control user register access for Realms Suzuki K Poulose
2026-09-28 1:30 ` Gavin Shan
2026-09-24 10:40 ` [PATCH v19 00/20] KVM: arm64: CCA: Add basic plumbing " Gavin Shan
2026-09-24 10:49 ` Suzuki K Poulose
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=df7e9870-1176-4679-bd79-807c0d87a4f2@redhat.com \
--to=gshan@redhat.com \
--cc=WeiLin.Chang@arm.com \
--cc=alpergun@google.com \
--cc=aneesh.kumar@kernel.org \
--cc=catalin.marinas@arm.com \
--cc=enju.kohei@fujitsu.com \
--cc=fj0570is@fujitsu.com \
--cc=gankulkarni@os.amperecomputing.com \
--cc=joey.gouly@arm.com \
--cc=kvm@vger.kernel.org \
--cc=kvmarm@lists.linux.dev \
--cc=linux-arm-kernel@lists.infradead.org \
--cc=linux-coco@lists.linux.dev \
--cc=linux-kernel@vger.kernel.org \
--cc=lpieralisi@kernel.org \
--cc=maz@kernel.org \
--cc=oupton@kernel.org \
--cc=sdonthineni@nvidia.com \
--cc=steven.price@arm.com \
--cc=suzuki.poulose@arm.com \
--cc=tabba@google.com \
--cc=will@kernel.org \
--cc=yuzenghui@huawei.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.