Linux Documentation
 help / color / mirror / Atom feed
From: Pasha Tatashin <pasha.tatashin@soleen.com>
To: linux-kselftest@vger.kernel.org, legion@kernel.org,
	kees@kernel.org, will@kernel.org, ruanjinjie@huawei.com,
	atomlin@atomlin.com, rppt@kernel.org, jani.nikula@intel.com,
	hamzamahfooz@linux.microsoft.com, joey.gouly@arm.com,
	tglx@kernel.org, nsc@kernel.org, alexandre.chartre@oracle.com,
	james.morse@arm.com, dianders@chromium.org, bp@alien8.de,
	jpoimboe@kernel.org, shuah@kernel.org, catalin.marinas@arm.com,
	linux-kbuild@vger.kernel.org, linux-arch@vger.kernel.org,
	kvmarm@lists.linux.dev, jaredwhite@microsoft.com,
	johan@kernel.org, pbonzini@redhat.com, mingo@redhat.com,
	linux-mm@kvack.org, seanjc@google.com, mark.rutland@arm.com,
	vdonnefort@google.com, tarunsahu@google.com, gshan@redhat.com,
	skhan@linuxfoundation.org, linux-doc@vger.kernel.org,
	xur@google.com, djbw@kernel.org, oupton@kernel.org,
	nogikh@google.com, sumitg@nvidia.com,
	linux-kernel@vger.kernel.org, zengheng4@huawei.com,
	peterz@infradead.org, corbet@lwn.net, suzuki.poulose@arm.com,
	luto@kernel.org, hpa@zytor.com, zhangpengjie2@huawei.com,
	x86@kernel.org, yuzenghui@huawei.com, jic23@kernel.org,
	ardb@kernel.org, pasha.tatashin@soleen.com, petr.pavlu@suse.com,
	ryan.roberts@arm.com, kexec@lists.infradead.org,
	pratyush@kernel.org, dave.hansen@linux.intel.com,
	rdunlap@infradead.org, kvm@vger.kernel.org, fuad.tabba@linux.dev,
	maz@kernel.org, mbenes@suse.cz, jgross@suse.com,
	seiden@linux.ibm.com, pierre.gondois@arm.com, song@kernel.org,
	nathan@kernel.org, pmladek@suse.com, graf@amazon.com,
	chao.gao@intel.com, zhenglifeng1@huawei.com, arnd@arndb.de,
	sidnayyar@google.com, linux-arm-kernel@lists.infradead.org,
	vladimir.murzin@arm.com, kas@kernel.org
Subject: [RFC PATCH 32/46] KVM: x86: Add TDP MMU KHO preservation helpers
Date: Sun, 20 Sep 2026 15:36:36 -0400	[thread overview]
Message-ID: <20260920193650.3373435-33-pasha.tatashin@soleen.com> (raw)
In-Reply-To: <20260920193650.3373435-1-pasha.tatashin@soleen.com>

Add arch/x86/kvm/mmu/kho.c to preserve and adopt TDP MMU EPT/NPT
root page tables across Kexec Handover (KHO) live updates.

Signed-off-by: Pasha Tatashin <pasha.tatashin@soleen.com>
---
 arch/x86/kvm/mmu.h     |   7 ++
 arch/x86/kvm/mmu/kho.c | 193 +++++++++++++++++++++++++++++++++++++++++
 2 files changed, 200 insertions(+)
 create mode 100644 arch/x86/kvm/mmu/kho.c

diff --git a/arch/x86/kvm/mmu.h b/arch/x86/kvm/mmu.h
index 2ae7f9ed4cf8..f35948f0906c 100644
--- a/arch/x86/kvm/mmu.h
+++ b/arch/x86/kvm/mmu.h
@@ -410,4 +410,11 @@ static inline bool kvm_is_gfn_alias(struct kvm *kvm, gfn_t gfn)
 {
 	return gfn & kvm_gfn_direct_bits(kvm);
 }
+
+/*
+ * Declared here rather than in asm/kvm_host.h: it is internal to
+ * arch/x86/kvm and has no callers outside it.  Defined in mmu/kho.c.
+ */
+int kvm_mmu_preserve_kho(struct kvm *kvm);
+
 #endif
diff --git a/arch/x86/kvm/mmu/kho.c b/arch/x86/kvm/mmu/kho.c
new file mode 100644
index 000000000000..a04600f0c3e0
--- /dev/null
+++ b/arch/x86/kvm/mmu/kho.c
@@ -0,0 +1,193 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Copyright (c) 2026, Google LLC.
+ * Pasha Tatashin <pasha.tatashin@soleen.com>
+ *
+ * KHO preservation of the x86 KVM MMU page tables.
+ *
+ * An orphaned vCPU keeps running its guest out of the shadow/TDP page tables
+ * while the VM is detached, so every page those tables are built from has to
+ * survive the kexec.
+ */
+
+#include <linux/kexec_handover.h>
+#include <linux/kvm_host.h>
+
+#include "mmu.h"
+#include "mmu_internal.h"
+#include "spte.h"
+#include "tdp_iter.h"
+#include "tdp_mmu.h"
+
+/*
+ * Page-pointer accumulator.
+ *
+ * kho_preserve_pages() cannot be called while holding kvm->mmu_lock: it is a
+ * rwlock_t, so the section is atomic, whereas kho_radix_add_key() below it
+ * calls might_sleep(), takes a mutex and allocates with GFP_KERNEL.  So the
+ * walk runs in two phases -- collect the pages under the lock, preserve them
+ * after dropping it.
+ *
+ * A NULL @pages simply counts, which is how the caller sizes the array.
+ */
+struct kvm_mmu_kho_pages {
+	struct page **pages;
+	unsigned long nr;
+	unsigned long capacity;
+	bool overflow;
+};
+
+static void kvm_mmu_kho_add(struct kvm_mmu_kho_pages *acc, struct page *page)
+{
+	if (!acc->pages) {
+		acc->nr++;
+		return;
+	}
+
+	if (acc->nr >= acc->capacity) {
+		acc->overflow = true;
+		return;
+	}
+
+	acc->pages[acc->nr++] = page;
+}
+
+static void kvm_tdp_mmu_collect(struct kvm *kvm,
+				struct kvm_mmu_kho_pages *acc)
+{
+	gfn_t end = kvm_mmu_max_gfn() + 1;
+	struct kvm_mmu_page *root;
+	struct tdp_iter iter;
+
+	lockdep_assert_held_write(&kvm->mmu_lock);
+
+	rcu_read_lock();
+	list_for_each_entry_rcu(root, &kvm->arch.tdp_mmu_roots, link) {
+		if (root->spt)
+			kvm_mmu_kho_add(acc, virt_to_page(root->spt));
+
+		for_each_tdp_pte(iter, kvm, root, 0, end) {
+			struct page *page;
+
+			if (!is_shadow_present_pte(iter.old_spte) ||
+			    is_last_spte(iter.old_spte, iter.level))
+				continue;
+
+			page = pfn_to_page(spte_to_pfn(iter.old_spte));
+			kvm_mmu_kho_add(acc, page);
+		}
+	}
+	rcu_read_unlock();
+}
+
+/*
+ * The per-vCPU root page tables are not linked into active_mmu_pages, so they
+ * have to be walked separately.  pae_root, pml4_root and pml5_root are each
+ * NULL unless the corresponding paging mode is in use.
+ */
+static void kvm_mmu_collect_roots(struct kvm_mmu *mmu,
+				  struct kvm_mmu_kho_pages *acc)
+{
+	void *const roots[] = { mmu->pae_root, mmu->pml4_root, mmu->pml5_root };
+	int i;
+
+	for (i = 0; i < ARRAY_SIZE(roots); i++) {
+		if (roots[i])
+			kvm_mmu_kho_add(acc, virt_to_page(roots[i]));
+	}
+}
+
+/* Collect every page backing this VM's MMU.  Must be called under mmu_lock. */
+static void kvm_mmu_collect_all(struct kvm *kvm,
+				struct kvm_mmu_kho_pages *acc)
+{
+	struct kvm_mmu_page *sp;
+	struct kvm_vcpu *vcpu;
+	unsigned long i;
+
+	lockdep_assert_held_write(&kvm->mmu_lock);
+
+	acc->nr = 0;
+	acc->overflow = false;
+
+	if (tdp_mmu_enabled)
+		kvm_tdp_mmu_collect(kvm, acc);
+
+	list_for_each_entry(sp, &kvm->arch.active_mmu_pages, link) {
+		if (sp->spt)
+			kvm_mmu_kho_add(acc, virt_to_page(sp->spt));
+	}
+
+	kvm_for_each_vcpu(i, vcpu, kvm) {
+		if (vcpu->arch.mmu)
+			kvm_mmu_collect_roots(vcpu->arch.mmu, acc);
+		kvm_mmu_collect_roots(&vcpu->arch.guest_mmu, acc);
+	}
+}
+
+int kvm_mmu_preserve_kho(struct kvm *kvm)
+{
+	struct kvm_mmu_kho_pages acc = {};
+	struct kvm_kho_folios_ser *kp;
+	unsigned long i;
+	int ret = 0;
+	int attempt;
+
+	/*
+	 * Size the array, then fill it.  The guest can fault in new page
+	 * tables between the two passes, so re-check for overflow and retry
+	 * with a larger array; the slack makes repeated growth unlikely.
+	 */
+	for (attempt = 0; attempt < 5; attempt++) {
+		write_lock(&kvm->mmu_lock);
+		kvm_mmu_collect_all(kvm, &acc);
+		write_unlock(&kvm->mmu_lock);
+
+		if (acc.pages && !acc.overflow)
+			break;
+
+		acc.capacity = acc.nr + (acc.nr >> 2) + 16;
+		kvfree(acc.pages);
+		acc.pages = kvmalloc_array(acc.capacity, sizeof(*acc.pages),
+					   GFP_KERNEL);
+		if (!acc.pages)
+			return -ENOMEM;
+	}
+
+	if (acc.overflow) {
+		ret = -EAGAIN;
+		goto out;
+	}
+
+	if (!acc.nr)
+		goto out;
+
+	kp = kvm_kho_folios_alloc(acc.nr);
+	if (IS_ERR(kp)) {
+		ret = PTR_ERR(kp);
+		goto out;
+	}
+
+	for (i = 0; i < acc.nr; i++) {
+		ret = kho_preserve_folio(page_folio(acc.pages[i]));
+		if (ret) {
+			/*
+			 * Undo the partial preservation: leaving pages marked
+			 * would pin them in the incoming kernel forever with
+			 * nothing owning them.
+			 */
+			while (i--)
+				kho_unpreserve_folio(page_folio(acc.pages[i]));
+			kho_unpreserve_free(kp);
+			goto out;
+		}
+		kp->folios_pa[i] = page_to_phys(acc.pages[i]);
+	}
+
+	kp->nr_folios = acc.nr;
+	kvm->kho_folios = kp;
+
+out:
+	kvfree(acc.pages);
+	return ret;
+}
-- 
2.55.0.1082.g2b9226bbc0-goog


  parent reply	other threads:[~2026-09-20 19:40 UTC|newest]

Thread overview: 49+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-20 19:36 [RFC PATCH 00/46] Orphaned Virtual Machines Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 01/46] KVM: luo: Delegate VM creation type to kvm_arch_vm_luo_preserve Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 02/46] KVM: arm64: Split demux_c15_{get,set}_val from userspace accessors Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 03/46] KVM: arm64: Split kvm_sys_reg_{get,set}_user from kernel accessors Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 04/46] x86/mm/ident_map: Add force_pte to support 4K PTE identity mappings Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 05/46] arm64: mm: Add trans_pgd_map_range() support Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 06/46] x86/smp: Skip offline CPUs for REBOOT_VECTOR in native_stop_other_cpus() Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 07/46] KVM: luo: Support vCPU file preservation across live updates Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 08/46] KVM: x86: Add x86 vCPU LUO preservation ABI and register helpers Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 09/46] KVM: x86: Implement architectural vCPU state preservation via LUO Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 10/46] KVM: arm64: " Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 11/46] liveupdate: Define CPU preservation linker sections Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 12/46] liveupdate: Add liveupdate_session_name() helper Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 13/46] cpu_preserve: Add physical CPU preservation ABI and core API headers Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 14/46] cpu_preserve: Add core physical CPU preservation state and park loop Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 15/46] cpu_preserve: Add physical CPU preservation lifecycle and build rules Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 16/46] liveupdate: cpu_preserve: Add sysfs interface Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 17/46] liveupdate: cpu_preserve: Add isolated address space management API Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 18/46] liveupdate: cpu_preserve: Add LUO file handler for preserved physical CPUs Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 19/46] x86: liveupdate: Add low-level physical CPU preservation assembly Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 20/46] x86: liveupdate: Add physical CPU preservation context and page table support Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 21/46] selftests: liveupdate: Add physical CPU preservation unit tests Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 22/46] selftests: liveupdate: Add physical CPU preservation live update tests Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 23/46] Documentation: liveupdate: Add physical CPU preservation documentation Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 24/46] MAINTAINERS: Add entry for KVM Caretaker Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 25/46] arm64: liveupdate: Add support for physical CPU preservation Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 26/46] oncore: Add on-core KHO ABI and public framework headers Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 27/46] oncore: Implement on-core session lifecycle and scheduling loop Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 28/46] KVM: caretaker: Add Caretaker control block and architecture ops headers Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 29/46] KVM: caretaker: Implement Caretaker session memory mapping helpers Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 30/46] KVM: caretaker: Integrate Caretaker vCPU detach, attach, and cancel with KVM Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 31/46] KVM: caretaker: Add generic KHO ABI telemetry and debugfs reporting Pasha Tatashin
2026-09-20 19:36 ` Pasha Tatashin [this message]
2026-09-20 19:36 ` [RFC PATCH 33/46] KVM: x86: Add Caretaker x86 KHO ABI and runtime context headers Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 34/46] KVM: x86: Implement Caretaker LAPIC timer and interrupt injection Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 35/46] KVM: x86: Implement Caretaker VM-exit dispatch and instruction decoders Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 36/46] KVM: x86: Implement Caretaker run loop and LUO detach/attach lifecycle Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 37/46] KVM: VMX: Add Caretaker VMX assembly guest entry/exit routine and helpers Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 38/46] KVM: VMX: Implement Caretaker VMX VMCS lifecycle and exit dispatch Pasha Tatashin
2026-09-20 19:36 ` [RFC PATCH 39/46] KVM: VMX: Integrate Caretaker VMX detach serialization and KVM registration Pasha Tatashin
2026-09-21  7:42 ` [RFC PATCH 00/46] Orphaned Virtual Machines Graf (AWS), Alexander
2026-09-21 21:00 ` [RFC PATCH 39/46] KVM: VMX: Integrate Caretaker VMX detach serialization and KVM registration Pasha Tatashin
2026-09-21 21:00   ` [RFC PATCH 40/46] KVM: SVM: Add Caretaker SVM assembly guest entry/exit routine Pasha Tatashin
2026-09-21 21:00   ` [RFC PATCH 41/46] KVM: SVM: Implement Caretaker SVM VMCB lifecycle and exit dispatch Pasha Tatashin
2026-09-21 21:00   ` [RFC PATCH 42/46] KVM: arm64: Add Caretaker arm64 KHO ABI and runtime context headers Pasha Tatashin
2026-09-21 21:00   ` [RFC PATCH 43/46] KVM: arm64: Add Caretaker EL2 exception vectors and guest entry/exit assembly Pasha Tatashin
2026-09-21 21:00   ` [RFC PATCH 44/46] KVM: arm64: Implement Caretaker GICv3 CPU interface and arch timer emulation Pasha Tatashin
2026-09-21 21:00   ` [RFC PATCH 45/46] KVM: arm64: Implement Caretaker system register trap and exception handlers Pasha Tatashin
2026-09-21 21:00   ` [RFC PATCH 46/46] KVM: arm64: Implement Caretaker vCPU run loop and LUO detach/attach lifecycle Pasha Tatashin

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260920193650.3373435-33-pasha.tatashin@soleen.com \
    --to=pasha.tatashin@soleen.com \
    --cc=alexandre.chartre@oracle.com \
    --cc=ardb@kernel.org \
    --cc=arnd@arndb.de \
    --cc=atomlin@atomlin.com \
    --cc=bp@alien8.de \
    --cc=catalin.marinas@arm.com \
    --cc=chao.gao@intel.com \
    --cc=corbet@lwn.net \
    --cc=dave.hansen@linux.intel.com \
    --cc=dianders@chromium.org \
    --cc=djbw@kernel.org \
    --cc=fuad.tabba@linux.dev \
    --cc=graf@amazon.com \
    --cc=gshan@redhat.com \
    --cc=hamzamahfooz@linux.microsoft.com \
    --cc=hpa@zytor.com \
    --cc=james.morse@arm.com \
    --cc=jani.nikula@intel.com \
    --cc=jaredwhite@microsoft.com \
    --cc=jgross@suse.com \
    --cc=jic23@kernel.org \
    --cc=joey.gouly@arm.com \
    --cc=johan@kernel.org \
    --cc=jpoimboe@kernel.org \
    --cc=kas@kernel.org \
    --cc=kees@kernel.org \
    --cc=kexec@lists.infradead.org \
    --cc=kvm@vger.kernel.org \
    --cc=kvmarm@lists.linux.dev \
    --cc=legion@kernel.org \
    --cc=linux-arch@vger.kernel.org \
    --cc=linux-arm-kernel@lists.infradead.org \
    --cc=linux-doc@vger.kernel.org \
    --cc=linux-kbuild@vger.kernel.org \
    --cc=linux-kernel@vger.kernel.org \
    --cc=linux-kselftest@vger.kernel.org \
    --cc=linux-mm@kvack.org \
    --cc=luto@kernel.org \
    --cc=mark.rutland@arm.com \
    --cc=maz@kernel.org \
    --cc=mbenes@suse.cz \
    --cc=mingo@redhat.com \
    --cc=nathan@kernel.org \
    --cc=nogikh@google.com \
    --cc=nsc@kernel.org \
    --cc=oupton@kernel.org \
    --cc=pbonzini@redhat.com \
    --cc=peterz@infradead.org \
    --cc=petr.pavlu@suse.com \
    --cc=pierre.gondois@arm.com \
    --cc=pmladek@suse.com \
    --cc=pratyush@kernel.org \
    --cc=rdunlap@infradead.org \
    --cc=rppt@kernel.org \
    --cc=ruanjinjie@huawei.com \
    --cc=ryan.roberts@arm.com \
    --cc=seanjc@google.com \
    --cc=seiden@linux.ibm.com \
    --cc=shuah@kernel.org \
    --cc=sidnayyar@google.com \
    --cc=skhan@linuxfoundation.org \
    --cc=song@kernel.org \
    --cc=sumitg@nvidia.com \
    --cc=suzuki.poulose@arm.com \
    --cc=tarunsahu@google.com \
    --cc=tglx@kernel.org \
    --cc=vdonnefort@google.com \
    --cc=vladimir.murzin@arm.com \
    --cc=will@kernel.org \
    --cc=x86@kernel.org \
    --cc=xur@google.com \
    --cc=yuzenghui@huawei.com \
    --cc=zengheng4@huawei.com \
    --cc=zhangpengjie2@huawei.com \
    --cc=zhenglifeng1@huawei.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox