From: Pasha Tatashin <pasha.tatashin@soleen.com>
To: linux-kselftest@vger.kernel.org, legion@kernel.org,
kees@kernel.org, will@kernel.org, ruanjinjie@huawei.com,
atomlin@atomlin.com, rppt@kernel.org, jani.nikula@intel.com,
hamzamahfooz@linux.microsoft.com, joey.gouly@arm.com,
tglx@kernel.org, nsc@kernel.org, alexandre.chartre@oracle.com,
james.morse@arm.com, dianders@chromium.org, bp@alien8.de,
jpoimboe@kernel.org, shuah@kernel.org, catalin.marinas@arm.com,
linux-kbuild@vger.kernel.org, linux-arch@vger.kernel.org,
kvmarm@lists.linux.dev, jaredwhite@microsoft.com,
johan@kernel.org, pbonzini@redhat.com, mingo@redhat.com,
linux-mm@kvack.org, seanjc@google.com, mark.rutland@arm.com,
vdonnefort@google.com, tarunsahu@google.com, gshan@redhat.com,
skhan@linuxfoundation.org, linux-doc@vger.kernel.org,
xur@google.com, djbw@kernel.org, oupton@kernel.org,
nogikh@google.com, sumitg@nvidia.com,
linux-kernel@vger.kernel.org, zengheng4@huawei.com,
peterz@infradead.org, corbet@lwn.net, suzuki.poulose@arm.com,
luto@kernel.org, hpa@zytor.com, zhangpengjie2@huawei.com,
x86@kernel.org, yuzenghui@huawei.com, jic23@kernel.org,
ardb@kernel.org, pasha.tatashin@soleen.com, petr.pavlu@suse.com,
ryan.roberts@arm.com, kexec@lists.infradead.org,
pratyush@kernel.org, dave.hansen@linux.intel.com,
rdunlap@infradead.org, kvm@vger.kernel.org, fuad.tabba@linux.dev,
maz@kernel.org, mbenes@suse.cz, jgross@suse.com,
seiden@linux.ibm.com, pierre.gondois@arm.com, song@kernel.org,
nathan@kernel.org, pmladek@suse.com, graf@amazon.com,
chao.gao@intel.com, zhenglifeng1@huawei.com, arnd@arndb.de,
sidnayyar@google.com, linux-arm-kernel@lists.infradead.org,
vladimir.murzin@arm.com, kas@kernel.org
Subject: [RFC PATCH 32/46] KVM: x86: Add TDP MMU KHO preservation helpers
Date: Sun, 20 Sep 2026 15:36:36 -0400 [thread overview]
Message-ID: <20260920193650.3373435-33-pasha.tatashin@soleen.com> (raw)
In-Reply-To: <20260920193650.3373435-1-pasha.tatashin@soleen.com>
Add arch/x86/kvm/mmu/kho.c to preserve and adopt TDP MMU EPT/NPT
root page tables across Kexec Handover (KHO) live updates.
Signed-off-by: Pasha Tatashin <pasha.tatashin@soleen.com>
---
arch/x86/kvm/mmu.h | 7 ++
arch/x86/kvm/mmu/kho.c | 193 +++++++++++++++++++++++++++++++++++++++++
2 files changed, 200 insertions(+)
create mode 100644 arch/x86/kvm/mmu/kho.c
diff --git a/arch/x86/kvm/mmu.h b/arch/x86/kvm/mmu.h
index 2ae7f9ed4cf8..f35948f0906c 100644
--- a/arch/x86/kvm/mmu.h
+++ b/arch/x86/kvm/mmu.h
@@ -410,4 +410,11 @@ static inline bool kvm_is_gfn_alias(struct kvm *kvm, gfn_t gfn)
{
return gfn & kvm_gfn_direct_bits(kvm);
}
+
+/*
+ * Declared here rather than in asm/kvm_host.h: it is internal to
+ * arch/x86/kvm and has no callers outside it. Defined in mmu/kho.c.
+ */
+int kvm_mmu_preserve_kho(struct kvm *kvm);
+
#endif
diff --git a/arch/x86/kvm/mmu/kho.c b/arch/x86/kvm/mmu/kho.c
new file mode 100644
index 000000000000..a04600f0c3e0
--- /dev/null
+++ b/arch/x86/kvm/mmu/kho.c
@@ -0,0 +1,193 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Copyright (c) 2026, Google LLC.
+ * Pasha Tatashin <pasha.tatashin@soleen.com>
+ *
+ * KHO preservation of the x86 KVM MMU page tables.
+ *
+ * An orphaned vCPU keeps running its guest out of the shadow/TDP page tables
+ * while the VM is detached, so every page those tables are built from has to
+ * survive the kexec.
+ */
+
+#include <linux/kexec_handover.h>
+#include <linux/kvm_host.h>
+
+#include "mmu.h"
+#include "mmu_internal.h"
+#include "spte.h"
+#include "tdp_iter.h"
+#include "tdp_mmu.h"
+
+/*
+ * Page-pointer accumulator.
+ *
+ * kho_preserve_pages() cannot be called while holding kvm->mmu_lock: it is a
+ * rwlock_t, so the section is atomic, whereas kho_radix_add_key() below it
+ * calls might_sleep(), takes a mutex and allocates with GFP_KERNEL. So the
+ * walk runs in two phases -- collect the pages under the lock, preserve them
+ * after dropping it.
+ *
+ * A NULL @pages simply counts, which is how the caller sizes the array.
+ */
+struct kvm_mmu_kho_pages {
+ struct page **pages;
+ unsigned long nr;
+ unsigned long capacity;
+ bool overflow;
+};
+
+static void kvm_mmu_kho_add(struct kvm_mmu_kho_pages *acc, struct page *page)
+{
+ if (!acc->pages) {
+ acc->nr++;
+ return;
+ }
+
+ if (acc->nr >= acc->capacity) {
+ acc->overflow = true;
+ return;
+ }
+
+ acc->pages[acc->nr++] = page;
+}
+
+static void kvm_tdp_mmu_collect(struct kvm *kvm,
+ struct kvm_mmu_kho_pages *acc)
+{
+ gfn_t end = kvm_mmu_max_gfn() + 1;
+ struct kvm_mmu_page *root;
+ struct tdp_iter iter;
+
+ lockdep_assert_held_write(&kvm->mmu_lock);
+
+ rcu_read_lock();
+ list_for_each_entry_rcu(root, &kvm->arch.tdp_mmu_roots, link) {
+ if (root->spt)
+ kvm_mmu_kho_add(acc, virt_to_page(root->spt));
+
+ for_each_tdp_pte(iter, kvm, root, 0, end) {
+ struct page *page;
+
+ if (!is_shadow_present_pte(iter.old_spte) ||
+ is_last_spte(iter.old_spte, iter.level))
+ continue;
+
+ page = pfn_to_page(spte_to_pfn(iter.old_spte));
+ kvm_mmu_kho_add(acc, page);
+ }
+ }
+ rcu_read_unlock();
+}
+
+/*
+ * The per-vCPU root page tables are not linked into active_mmu_pages, so they
+ * have to be walked separately. pae_root, pml4_root and pml5_root are each
+ * NULL unless the corresponding paging mode is in use.
+ */
+static void kvm_mmu_collect_roots(struct kvm_mmu *mmu,
+ struct kvm_mmu_kho_pages *acc)
+{
+ void *const roots[] = { mmu->pae_root, mmu->pml4_root, mmu->pml5_root };
+ int i;
+
+ for (i = 0; i < ARRAY_SIZE(roots); i++) {
+ if (roots[i])
+ kvm_mmu_kho_add(acc, virt_to_page(roots[i]));
+ }
+}
+
+/* Collect every page backing this VM's MMU. Must be called under mmu_lock. */
+static void kvm_mmu_collect_all(struct kvm *kvm,
+ struct kvm_mmu_kho_pages *acc)
+{
+ struct kvm_mmu_page *sp;
+ struct kvm_vcpu *vcpu;
+ unsigned long i;
+
+ lockdep_assert_held_write(&kvm->mmu_lock);
+
+ acc->nr = 0;
+ acc->overflow = false;
+
+ if (tdp_mmu_enabled)
+ kvm_tdp_mmu_collect(kvm, acc);
+
+ list_for_each_entry(sp, &kvm->arch.active_mmu_pages, link) {
+ if (sp->spt)
+ kvm_mmu_kho_add(acc, virt_to_page(sp->spt));
+ }
+
+ kvm_for_each_vcpu(i, vcpu, kvm) {
+ if (vcpu->arch.mmu)
+ kvm_mmu_collect_roots(vcpu->arch.mmu, acc);
+ kvm_mmu_collect_roots(&vcpu->arch.guest_mmu, acc);
+ }
+}
+
+int kvm_mmu_preserve_kho(struct kvm *kvm)
+{
+ struct kvm_mmu_kho_pages acc = {};
+ struct kvm_kho_folios_ser *kp;
+ unsigned long i;
+ int ret = 0;
+ int attempt;
+
+ /*
+ * Size the array, then fill it. The guest can fault in new page
+ * tables between the two passes, so re-check for overflow and retry
+ * with a larger array; the slack makes repeated growth unlikely.
+ */
+ for (attempt = 0; attempt < 5; attempt++) {
+ write_lock(&kvm->mmu_lock);
+ kvm_mmu_collect_all(kvm, &acc);
+ write_unlock(&kvm->mmu_lock);
+
+ if (acc.pages && !acc.overflow)
+ break;
+
+ acc.capacity = acc.nr + (acc.nr >> 2) + 16;
+ kvfree(acc.pages);
+ acc.pages = kvmalloc_array(acc.capacity, sizeof(*acc.pages),
+ GFP_KERNEL);
+ if (!acc.pages)
+ return -ENOMEM;
+ }
+
+ if (acc.overflow) {
+ ret = -EAGAIN;
+ goto out;
+ }
+
+ if (!acc.nr)
+ goto out;
+
+ kp = kvm_kho_folios_alloc(acc.nr);
+ if (IS_ERR(kp)) {
+ ret = PTR_ERR(kp);
+ goto out;
+ }
+
+ for (i = 0; i < acc.nr; i++) {
+ ret = kho_preserve_folio(page_folio(acc.pages[i]));
+ if (ret) {
+ /*
+ * Undo the partial preservation: leaving pages marked
+ * would pin them in the incoming kernel forever with
+ * nothing owning them.
+ */
+ while (i--)
+ kho_unpreserve_folio(page_folio(acc.pages[i]));
+ kho_unpreserve_free(kp);
+ goto out;
+ }
+ kp->folios_pa[i] = page_to_phys(acc.pages[i]);
+ }
+
+ kp->nr_folios = acc.nr;
+ kvm->kho_folios = kp;
+
+out:
+ kvfree(acc.pages);
+ return ret;
+}
--
2.55.0.1082.g2b9226bbc0-goog
next prev parent reply other threads:[~2026-09-20 19:40 UTC|newest]
Thread overview: 49+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-20 19:36 [RFC PATCH 00/46] Orphaned Virtual Machines Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 01/46] KVM: luo: Delegate VM creation type to kvm_arch_vm_luo_preserve Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 02/46] KVM: arm64: Split demux_c15_{get,set}_val from userspace accessors Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 03/46] KVM: arm64: Split kvm_sys_reg_{get,set}_user from kernel accessors Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 04/46] x86/mm/ident_map: Add force_pte to support 4K PTE identity mappings Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 05/46] arm64: mm: Add trans_pgd_map_range() support Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 06/46] x86/smp: Skip offline CPUs for REBOOT_VECTOR in native_stop_other_cpus() Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 07/46] KVM: luo: Support vCPU file preservation across live updates Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 08/46] KVM: x86: Add x86 vCPU LUO preservation ABI and register helpers Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 09/46] KVM: x86: Implement architectural vCPU state preservation via LUO Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 10/46] KVM: arm64: " Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 11/46] liveupdate: Define CPU preservation linker sections Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 12/46] liveupdate: Add liveupdate_session_name() helper Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 13/46] cpu_preserve: Add physical CPU preservation ABI and core API headers Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 14/46] cpu_preserve: Add core physical CPU preservation state and park loop Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 15/46] cpu_preserve: Add physical CPU preservation lifecycle and build rules Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 16/46] liveupdate: cpu_preserve: Add sysfs interface Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 17/46] liveupdate: cpu_preserve: Add isolated address space management API Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 18/46] liveupdate: cpu_preserve: Add LUO file handler for preserved physical CPUs Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 19/46] x86: liveupdate: Add low-level physical CPU preservation assembly Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 20/46] x86: liveupdate: Add physical CPU preservation context and page table support Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 21/46] selftests: liveupdate: Add physical CPU preservation unit tests Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 22/46] selftests: liveupdate: Add physical CPU preservation live update tests Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 23/46] Documentation: liveupdate: Add physical CPU preservation documentation Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 24/46] MAINTAINERS: Add entry for KVM Caretaker Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 25/46] arm64: liveupdate: Add support for physical CPU preservation Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 26/46] oncore: Add on-core KHO ABI and public framework headers Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 27/46] oncore: Implement on-core session lifecycle and scheduling loop Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 28/46] KVM: caretaker: Add Caretaker control block and architecture ops headers Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 29/46] KVM: caretaker: Implement Caretaker session memory mapping helpers Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 30/46] KVM: caretaker: Integrate Caretaker vCPU detach, attach, and cancel with KVM Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 31/46] KVM: caretaker: Add generic KHO ABI telemetry and debugfs reporting Pasha Tatashin
2026-09-20 19:36 ` Pasha Tatashin [this message]
2026-09-20 19:36 ` [RFC PATCH 33/46] KVM: x86: Add Caretaker x86 KHO ABI and runtime context headers Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 34/46] KVM: x86: Implement Caretaker LAPIC timer and interrupt injection Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 35/46] KVM: x86: Implement Caretaker VM-exit dispatch and instruction decoders Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 36/46] KVM: x86: Implement Caretaker run loop and LUO detach/attach lifecycle Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 37/46] KVM: VMX: Add Caretaker VMX assembly guest entry/exit routine and helpers Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 38/46] KVM: VMX: Implement Caretaker VMX VMCS lifecycle and exit dispatch Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 39/46] KVM: VMX: Integrate Caretaker VMX detach serialization and KVM registration Pasha Tatashin
2026-09-21 7:42 ` [RFC PATCH 00/46] Orphaned Virtual Machines Graf (AWS), Alexander
2026-09-21 21:00 ` [RFC PATCH 39/46] KVM: VMX: Integrate Caretaker VMX detach serialization and KVM registration Pasha Tatashin
2026-09-21 21:00 ` [RFC PATCH 40/46] KVM: SVM: Add Caretaker SVM assembly guest entry/exit routine Pasha Tatashin
2026-09-21 21:00 ` [RFC PATCH 41/46] KVM: SVM: Implement Caretaker SVM VMCB lifecycle and exit dispatch Pasha Tatashin
2026-09-21 21:00 ` [RFC PATCH 42/46] KVM: arm64: Add Caretaker arm64 KHO ABI and runtime context headers Pasha Tatashin
2026-09-21 21:00 ` [RFC PATCH 43/46] KVM: arm64: Add Caretaker EL2 exception vectors and guest entry/exit assembly Pasha Tatashin
2026-09-21 21:00 ` [RFC PATCH 44/46] KVM: arm64: Implement Caretaker GICv3 CPU interface and arch timer emulation Pasha Tatashin
2026-09-21 21:00 ` [RFC PATCH 45/46] KVM: arm64: Implement Caretaker system register trap and exception handlers Pasha Tatashin
2026-09-21 21:00 ` [RFC PATCH 46/46] KVM: arm64: Implement Caretaker vCPU run loop and LUO detach/attach lifecycle Pasha Tatashin
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260920193650.3373435-33-pasha.tatashin@soleen.com \
--to=pasha.tatashin@soleen.com \
--cc=alexandre.chartre@oracle.com \
--cc=ardb@kernel.org \
--cc=arnd@arndb.de \
--cc=atomlin@atomlin.com \
--cc=bp@alien8.de \
--cc=catalin.marinas@arm.com \
--cc=chao.gao@intel.com \
--cc=corbet@lwn.net \
--cc=dave.hansen@linux.intel.com \
--cc=dianders@chromium.org \
--cc=djbw@kernel.org \
--cc=fuad.tabba@linux.dev \
--cc=graf@amazon.com \
--cc=gshan@redhat.com \
--cc=hamzamahfooz@linux.microsoft.com \
--cc=hpa@zytor.com \
--cc=james.morse@arm.com \
--cc=jani.nikula@intel.com \
--cc=jaredwhite@microsoft.com \
--cc=jgross@suse.com \
--cc=jic23@kernel.org \
--cc=joey.gouly@arm.com \
--cc=johan@kernel.org \
--cc=jpoimboe@kernel.org \
--cc=kas@kernel.org \
--cc=kees@kernel.org \
--cc=kexec@lists.infradead.org \
--cc=kvm@vger.kernel.org \
--cc=kvmarm@lists.linux.dev \
--cc=legion@kernel.org \
--cc=linux-arch@vger.kernel.org \
--cc=linux-arm-kernel@lists.infradead.org \
--cc=linux-doc@vger.kernel.org \
--cc=linux-kbuild@vger.kernel.org \
--cc=linux-kernel@vger.kernel.org \
--cc=linux-kselftest@vger.kernel.org \
--cc=linux-mm@kvack.org \
--cc=luto@kernel.org \
--cc=mark.rutland@arm.com \
--cc=maz@kernel.org \
--cc=mbenes@suse.cz \
--cc=mingo@redhat.com \
--cc=nathan@kernel.org \
--cc=nogikh@google.com \
--cc=nsc@kernel.org \
--cc=oupton@kernel.org \
--cc=pbonzini@redhat.com \
--cc=peterz@infradead.org \
--cc=petr.pavlu@suse.com \
--cc=pierre.gondois@arm.com \
--cc=pmladek@suse.com \
--cc=pratyush@kernel.org \
--cc=rdunlap@infradead.org \
--cc=rppt@kernel.org \
--cc=ruanjinjie@huawei.com \
--cc=ryan.roberts@arm.com \
--cc=seanjc@google.com \
--cc=seiden@linux.ibm.com \
--cc=shuah@kernel.org \
--cc=sidnayyar@google.com \
--cc=skhan@linuxfoundation.org \
--cc=song@kernel.org \
--cc=sumitg@nvidia.com \
--cc=suzuki.poulose@arm.com \
--cc=tarunsahu@google.com \
--cc=tglx@kernel.org \
--cc=vdonnefort@google.com \
--cc=vladimir.murzin@arm.com \
--cc=will@kernel.org \
--cc=x86@kernel.org \
--cc=xur@google.com \
--cc=yuzenghui@huawei.com \
--cc=zengheng4@huawei.com \
--cc=zhangpengjie2@huawei.com \
--cc=zhenglifeng1@huawei.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox