[RFC PATCH 32/46] KVM: x86: Add TDP MMU KHO preservation helpers
From: Pasha Tatashin
Date: Sun Sep 20 2026 - 15:58:58 EST
Add arch/x86/kvm/mmu/kho.c to preserve and adopt TDP MMU EPT/NPT
root page tables across Kexec Handover (KHO) live updates.
Signed-off-by: Pasha Tatashin <pasha.tatashin@xxxxxxxxxx>
---
arch/x86/kvm/mmu.h | 7 ++
arch/x86/kvm/mmu/kho.c | 193 +++++++++++++++++++++++++++++++++++++++++
2 files changed, 200 insertions(+)
create mode 100644 arch/x86/kvm/mmu/kho.c
diff --git a/arch/x86/kvm/mmu.h b/arch/x86/kvm/mmu.h
index 2ae7f9ed4cf8..f35948f0906c 100644
--- a/arch/x86/kvm/mmu.h
+++ b/arch/x86/kvm/mmu.h
@@ -410,4 +410,11 @@ static inline bool kvm_is_gfn_alias(struct kvm *kvm, gfn_t gfn)
{
return gfn & kvm_gfn_direct_bits(kvm);
}
+
+/*
+ * Declared here rather than in asm/kvm_host.h: it is internal to
+ * arch/x86/kvm and has no callers outside it. Defined in mmu/kho.c.
+ */
+int kvm_mmu_preserve_kho(struct kvm *kvm);
+
#endif
diff --git a/arch/x86/kvm/mmu/kho.c b/arch/x86/kvm/mmu/kho.c
new file mode 100644
index 000000000000..a04600f0c3e0
--- /dev/null
+++ b/arch/x86/kvm/mmu/kho.c
@@ -0,0 +1,193 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Copyright (c) 2026, Google LLC.
+ * Pasha Tatashin <pasha.tatashin@xxxxxxxxxx>
+ *
+ * KHO preservation of the x86 KVM MMU page tables.
+ *
+ * An orphaned vCPU keeps running its guest out of the shadow/TDP page tables
+ * while the VM is detached, so every page those tables are built from has to
+ * survive the kexec.
+ */
+
+#include <linux/kexec_handover.h>
+#include <linux/kvm_host.h>
+
+#include "mmu.h"
+#include "mmu_internal.h"
+#include "spte.h"
+#include "tdp_iter.h"
+#include "tdp_mmu.h"
+
+/*
+ * Page-pointer accumulator.
+ *
+ * kho_preserve_pages() cannot be called while holding kvm->mmu_lock: it is a
+ * rwlock_t, so the section is atomic, whereas kho_radix_add_key() below it
+ * calls might_sleep(), takes a mutex and allocates with GFP_KERNEL. So the
+ * walk runs in two phases -- collect the pages under the lock, preserve them
+ * after dropping it.
+ *
+ * A NULL @pages simply counts, which is how the caller sizes the array.
+ */
+struct kvm_mmu_kho_pages {
+ struct page **pages;
+ unsigned long nr;
+ unsigned long capacity;
+ bool overflow;
+};
+
+static void kvm_mmu_kho_add(struct kvm_mmu_kho_pages *acc, struct page *page)
+{
+ if (!acc->pages) {
+ acc->nr++;
+ return;
+ }
+
+ if (acc->nr >= acc->capacity) {
+ acc->overflow = true;
+ return;
+ }
+
+ acc->pages[acc->nr++] = page;
+}
+
+static void kvm_tdp_mmu_collect(struct kvm *kvm,
+ struct kvm_mmu_kho_pages *acc)
+{
+ gfn_t end = kvm_mmu_max_gfn() + 1;
+ struct kvm_mmu_page *root;
+ struct tdp_iter iter;
+
+ lockdep_assert_held_write(&kvm->mmu_lock);
+
+ rcu_read_lock();
+ list_for_each_entry_rcu(root, &kvm->arch.tdp_mmu_roots, link) {
+ if (root->spt)
+ kvm_mmu_kho_add(acc, virt_to_page(root->spt));
+
+ for_each_tdp_pte(iter, kvm, root, 0, end) {
+ struct page *page;
+
+ if (!is_shadow_present_pte(iter.old_spte) ||
+ is_last_spte(iter.old_spte, iter.level))
+ continue;
+
+ page = pfn_to_page(spte_to_pfn(iter.old_spte));
+ kvm_mmu_kho_add(acc, page);
+ }
+ }
+ rcu_read_unlock();
+}
+
+/*
+ * The per-vCPU root page tables are not linked into active_mmu_pages, so they
+ * have to be walked separately. pae_root, pml4_root and pml5_root are each
+ * NULL unless the corresponding paging mode is in use.
+ */
+static void kvm_mmu_collect_roots(struct kvm_mmu *mmu,
+ struct kvm_mmu_kho_pages *acc)
+{
+ void *const roots[] = { mmu->pae_root, mmu->pml4_root, mmu->pml5_root };
+ int i;
+
+ for (i = 0; i < ARRAY_SIZE(roots); i++) {
+ if (roots[i])
+ kvm_mmu_kho_add(acc, virt_to_page(roots[i]));
+ }
+}
+
+/* Collect every page backing this VM's MMU. Must be called under mmu_lock. */
+static void kvm_mmu_collect_all(struct kvm *kvm,
+ struct kvm_mmu_kho_pages *acc)
+{
+ struct kvm_mmu_page *sp;
+ struct kvm_vcpu *vcpu;
+ unsigned long i;
+
+ lockdep_assert_held_write(&kvm->mmu_lock);
+
+ acc->nr = 0;
+ acc->overflow = false;
+
+ if (tdp_mmu_enabled)
+ kvm_tdp_mmu_collect(kvm, acc);
+
+ list_for_each_entry(sp, &kvm->arch.active_mmu_pages, link) {
+ if (sp->spt)
+ kvm_mmu_kho_add(acc, virt_to_page(sp->spt));
+ }
+
+ kvm_for_each_vcpu(i, vcpu, kvm) {
+ if (vcpu->arch.mmu)
+ kvm_mmu_collect_roots(vcpu->arch.mmu, acc);
+ kvm_mmu_collect_roots(&vcpu->arch.guest_mmu, acc);
+ }
+}
+
+int kvm_mmu_preserve_kho(struct kvm *kvm)
+{
+ struct kvm_mmu_kho_pages acc = {};
+ struct kvm_kho_folios_ser *kp;
+ unsigned long i;
+ int ret = 0;
+ int attempt;
+
+ /*
+ * Size the array, then fill it. The guest can fault in new page
+ * tables between the two passes, so re-check for overflow and retry
+ * with a larger array; the slack makes repeated growth unlikely.
+ */
+ for (attempt = 0; attempt < 5; attempt++) {
+ write_lock(&kvm->mmu_lock);
+ kvm_mmu_collect_all(kvm, &acc);
+ write_unlock(&kvm->mmu_lock);
+
+ if (acc.pages && !acc.overflow)
+ break;
+
+ acc.capacity = acc.nr + (acc.nr >> 2) + 16;
+ kvfree(acc.pages);
+ acc.pages = kvmalloc_array(acc.capacity, sizeof(*acc.pages),
+ GFP_KERNEL);
+ if (!acc.pages)
+ return -ENOMEM;
+ }
+
+ if (acc.overflow) {
+ ret = -EAGAIN;
+ goto out;
+ }
+
+ if (!acc.nr)
+ goto out;
+
+ kp = kvm_kho_folios_alloc(acc.nr);
+ if (IS_ERR(kp)) {
+ ret = PTR_ERR(kp);
+ goto out;
+ }
+
+ for (i = 0; i < acc.nr; i++) {
+ ret = kho_preserve_folio(page_folio(acc.pages[i]));
+ if (ret) {
+ /*
+ * Undo the partial preservation: leaving pages marked
+ * would pin them in the incoming kernel forever with
+ * nothing owning them.
+ */
+ while (i--)
+ kho_unpreserve_folio(page_folio(acc.pages[i]));
+ kho_unpreserve_free(kp);
+ goto out;
+ }
+ kp->folios_pa[i] = page_to_phys(acc.pages[i]);
+ }
+
+ kp->nr_folios = acc.nr;
+ kvm->kho_folios = kp;
+
+out:
+ kvfree(acc.pages);
+ return ret;
+}
--
2.55.0.1082.g2b9226bbc0-goog