[RFC PATCH 20/46] x86: liveupdate: Add physical CPU preservation context and page table support
From: Pasha Tatashin
Date: Sun Sep 20 2026 - 15:41:38 EST
Implement x86 physical CPU preservation context management, APIC wakeup,
and isolated transition page tables in arch/x86/kernel/cpu_preserve.c.
Signed-off-by: Pasha Tatashin <pasha.tatashin@xxxxxxxxxx>
---
arch/x86/Kconfig | 1 +
arch/x86/include/asm/cpu_preserve.h | 5 +-
arch/x86/kernel/Makefile | 7 +
arch/x86/kernel/cpu_preserve.c | 432 ++++++++++++++++++++++++++++
4 files changed, 444 insertions(+), 1 deletion(-)
create mode 100644 arch/x86/kernel/cpu_preserve.c
diff --git a/arch/x86/Kconfig b/arch/x86/Kconfig
index 15fd9ec5ecac..c57b54b2b2b0 100644
--- a/arch/x86/Kconfig
+++ b/arch/x86/Kconfig
@@ -71,6 +71,7 @@ config X86
select ARCH_ENABLE_MEMORY_HOTPLUG if X86_64
select ARCH_ENABLE_SPLIT_PMD_PTLOCK if (PGTABLE_LEVELS > 2) && (X86_64 || X86_PAE)
select ARCH_HAS_PMD_SOFTLEAVES if X86_64 && TRANSPARENT_HUGEPAGE
+ select ARCH_SUPPORTS_LIVEUPDATE_CPU if LIVEUPDATE
select ARCH_HAS_ACPI_TABLE_UPGRADE if ACPI
select ARCH_HAS_CPU_ATTACK_VECTORS if CPU_MITIGATIONS
select ARCH_HAS_CACHE_LINE_SIZE
diff --git a/arch/x86/include/asm/cpu_preserve.h b/arch/x86/include/asm/cpu_preserve.h
index 969903eb70cb..83f72d39a44b 100644
--- a/arch/x86/include/asm/cpu_preserve.h
+++ b/arch/x86/include/asm/cpu_preserve.h
@@ -10,9 +10,12 @@
#define ARCH_CPU_PRESERVED_STACK_ORDER THREAD_SIZE_ORDER
-#ifdef CONFIG_CC_IS_GCC
+#if __has_attribute(indirect_branch) && __has_attribute(function_return)
#define ARCH_CPU_PRESERVED_TEXT \
__attribute__((indirect_branch("keep"), function_return("keep")))
+#elif __has_attribute(indirect_branch)
+#define ARCH_CPU_PRESERVED_TEXT \
+ __attribute__((indirect_branch("keep")))
#else
#define ARCH_CPU_PRESERVED_TEXT
#endif
diff --git a/arch/x86/kernel/Makefile b/arch/x86/kernel/Makefile
index 31f46fd00527..6e238444039d 100644
--- a/arch/x86/kernel/Makefile
+++ b/arch/x86/kernel/Makefile
@@ -78,6 +78,13 @@ obj-$(CONFIG_IRQ_WORK) += irq_work.o
obj-y += probe_roms.o
obj-$(CONFIG_X86_32) += sys_ia32.o
obj-$(CONFIG_IA32_EMULATION) += sys_ia32.o signal_32.o
+KASAN_SANITIZE_cpu_preserve.o := n
+KCSAN_SANITIZE_cpu_preserve.o := n
+UBSAN_SANITIZE_cpu_preserve.o := n
+KCOV_INSTRUMENT_cpu_preserve.o := n
+CFLAGS_REMOVE_cpu_preserve.o = $(CC_FLAGS_FTRACE)
+CFLAGS_cpu_preserve.o += -fno-stack-protector $(call cc-option,-ftrivial-auto-var-init=uninitialized) $(call cc-option,-fno-jump-tables)
+obj-$(CONFIG_LIVEUPDATE_CPU) += cpu_preserve.o preserve_cpu.o
obj-$(CONFIG_X86_64) += sys_x86_64.o
obj-$(CONFIG_X86_ESPFIX64) += espfix_64.o
obj-$(CONFIG_SYSFS) += ksysfs.o
diff --git a/arch/x86/kernel/cpu_preserve.c b/arch/x86/kernel/cpu_preserve.c
new file mode 100644
index 000000000000..15327bf469db
--- /dev/null
+++ b/arch/x86/kernel/cpu_preserve.c
@@ -0,0 +1,432 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Copyright (c) 2026, Google LLC.
+ * Pasha Tatashin <pasha.tatashin@xxxxxxxxxx>
+ *
+ * Architecture specific CPU preservation support for x86.
+ */
+#include <linux/cpu_preserve.h>
+#include <linux/kexec_handover.h>
+#include <linux/mm.h>
+#include <linux/nospec.h>
+#include <linux/objtool.h>
+#include <linux/sched/mm.h>
+
+#include <asm/apic.h>
+#include <linux/cacheflush.h>
+#include <linux/cpufeature.h>
+#include <asm/desc.h>
+#include <asm/fixmap.h>
+#include <asm/msr.h>
+#include <linux/pgtable.h>
+#include <asm/set_memory.h>
+#include <linux/smp.h>
+#include <asm/tlbflush.h>
+#include <asm/trapnr.h>
+#include <asm/init.h>
+
+static u32 x86_preserved_apicid[NR_CPUS] __cpu_preserved_data = {
+ [0 ... NR_CPUS - 1] = BAD_APICID,
+};
+
+static bool x86_preserved_x2apic __cpu_preserved_data;
+
+/*
+ * Signal or wake up a preserved physical CPU via APIC ICR NMI.
+ */
+void __cpu_preserved_text arch_cpu_preserved_kick(int cpu)
+{
+ u32 apicid;
+ u64 val;
+
+ if ((unsigned int)cpu >= NR_CPUS || !cpu_is_preserved(cpu))
+ return;
+
+ cpu = array_index_nospec(cpu, NR_CPUS);
+ apicid = x86_preserved_apicid[cpu];
+ if (apicid == BAD_APICID)
+ return;
+
+ val = ((u64)apicid << 32) | APIC_DM_NMI;
+ native_wrmsrq(APIC_BASE_MSR + (APIC_ICR >> 4), val);
+}
+
+/*
+ * Low-power wait in parking loop.
+ */
+void __cpu_preserved_text arch_cpu_preserved_park_wait(void)
+{
+ cpu_relax();
+}
+
+static gate_desc x86_preserved_idt[IDT_ENTRIES] __cpu_preserved_data __aligned(PAGE_SIZE);
+static struct desc_ptr x86_preserved_idt_desc __cpu_preserved_data;
+
+static struct desc_struct x86_preserved_gdt[GDT_ENTRIES] __cpu_preserved_data __aligned(PAGE_SIZE);
+static struct desc_ptr x86_preserved_gdt_desc __cpu_preserved_data;
+static bool x86_preserved_has_svm __cpu_preserved_data;
+
+/*
+ * Initialize the preserved IDT with stub handlers.
+ *
+ * Vectors 0..31 are x86 architecture exceptions/traps:
+ * - Exceptions with error codes (#DF, #TS, #NP, #SS, #GP, #PF, #AC, #CP, #VC,
+ * and #SX / vector 30) push an 8-byte error code onto the stack before the
+ * IRET frame. They must use iret_err_handler (x86_preserved_iret_err_stub)
+ * to pop the error code before iretq.
+ * - Other exceptions push no error code and return directly via iret_handler
+ * (x86_preserved_iret_stub).
+ *
+ * Vectors >= FIRST_EXTERNAL_VECTOR (32) are device and IPI interrupts:
+ * - External interrupts push no error code but require an APIC EOI before
+ * returning, handled by eoi_handler (x86_preserved_apic_eoi_stub), so the
+ * Local APIC does not block subsequent interrupts.
+ */
+static void init_preserved_idt(void)
+{
+ unsigned long iret_err_handler = (unsigned long)&x86_preserved_iret_err_stub;
+ unsigned long eoi_handler = (unsigned long)&x86_preserved_apic_eoi_stub;
+ unsigned long iret_handler = (unsigned long)&x86_preserved_iret_stub;
+ int v;
+
+ for (v = 0; v < IDT_ENTRIES; v++) {
+ bool has_err = (v == X86_TRAP_DF ||
+ (v >= X86_TRAP_TS && v <= X86_TRAP_PF) ||
+ v == X86_TRAP_AC || v == X86_TRAP_CP ||
+ v == X86_TRAP_VC || v == 30); /* 30: #SX */
+ unsigned long handler = (v >= FIRST_EXTERNAL_VECTOR) ? eoi_handler :
+ (has_err ? iret_err_handler : iret_handler);
+
+ pack_gate(&x86_preserved_idt[v], GATE_INTERRUPT, handler, 0,
+ 0, __KERNEL_CS);
+ }
+ x86_preserved_idt_desc.size = sizeof(x86_preserved_idt) - 1;
+ x86_preserved_idt_desc.address = (unsigned long)&x86_preserved_idt[0];
+ cpu_preserved_clean(&x86_preserved_idt);
+ cpu_preserved_clean(&x86_preserved_idt_desc);
+}
+
+static void init_preserved_gdt(void)
+{
+ struct desc_struct *gdt;
+ int i;
+
+ gdt = get_current_gdt_rw();
+ for (i = 0; i < GDT_ENTRIES; i++)
+ x86_preserved_gdt[i] = gdt[i];
+ x86_preserved_gdt_desc.size = GDT_SIZE - 1;
+ x86_preserved_gdt_desc.address = (unsigned long)&x86_preserved_gdt[0];
+ cpu_preserved_clean(&x86_preserved_gdt);
+ cpu_preserved_clean(&x86_preserved_gdt_desc);
+}
+
+void __cpu_preserved_text arch_cpu_preserved_load_desc(void)
+{
+ native_load_gdt(&x86_preserved_gdt_desc);
+ native_load_idt(&x86_preserved_idt_desc);
+}
+EXPORT_SYMBOL_GPL(arch_cpu_preserved_load_desc);
+
+/*
+ * Disables local interrupts on the physical core and loads preserved IDT and GDT.
+ */
+void __cpu_preserved_text arch_cpu_preserved_park_init(int cpu)
+{
+ u32 spiv;
+
+ local_irq_disable();
+ arch_cpu_preserved_load_desc();
+
+ if (cpu >= 0 && cpu < CONFIG_NR_CPUS && x86_preserved_x2apic) {
+ x86_preserved_apicid[cpu] =
+ (u32)native_rdmsrq(APIC_BASE_MSR + (APIC_ID >> 4));
+ cpu_preserved_clean(&x86_preserved_apicid[cpu]);
+ }
+
+ spiv = (u32)native_rdmsrq(APIC_BASE_MSR + (APIC_SPIV >> 4));
+ if (!(spiv & APIC_SPIV_APIC_ENABLED)) {
+ spiv |= APIC_SPIV_APIC_ENABLED;
+ native_wrmsrq(APIC_BASE_MSR + (APIC_SPIV >> 4), spiv);
+ }
+}
+
+void arch_cpu_preserved_early_init(void)
+{
+ int i;
+
+ x86_preserved_has_svm = boot_cpu_has(X86_FEATURE_SVM);
+ x86_preserved_x2apic = x2apic_mode != 0;
+ cpu_preserved_clean(&x86_preserved_x2apic);
+
+ for (i = 0; i < nr_cpu_ids; i++) {
+ u32 apicid = cpu_physical_id(i);
+
+ if (apicid == BAD_APICID)
+ apicid = cpuid_to_apicid[i];
+ x86_preserved_apicid[i] = apicid;
+ }
+ cpu_preserved_clean(&x86_preserved_apicid);
+
+ init_preserved_idt();
+ init_preserved_gdt();
+}
+EXPORT_SYMBOL_GPL(arch_cpu_preserved_early_init);
+
+/*
+ * Disable hardware virtualization on physical core so INIT is recognized.
+ */
+static void __cpu_preserved_text arch_cpu_preserved_virt_teardown(void)
+{
+ if (__read_cr4() & X86_CR4_VMXE) {
+ asm volatile("1: vmxoff\n\t"
+ "2:\n\t"
+ _ASM_EXTABLE(1b, 2b)
+ : : : "memory", "cc");
+ asm volatile("mov %0, %%cr4" : : "r" (__read_cr4() & ~X86_CR4_VMXE) : "memory");
+ }
+
+ if (x86_preserved_has_svm) {
+ u64 efer = native_rdmsrq(MSR_EFER);
+
+ if (efer & EFER_SVME) {
+ asm volatile("stgi" : : : "memory");
+ native_wrmsrq(MSR_EFER, efer & ~EFER_SVME);
+ }
+ }
+}
+
+/*
+ * Architecture cleanup on park loop exit.
+ */
+void __cpu_preserved_text arch_cpu_preserved_park_finish(int cpu __maybe_unused)
+{
+ arch_cpu_preserved_load_desc();
+ arch_cpu_preserved_virt_teardown();
+}
+
+phys_addr_t __cpu_preserved_data x86_caretaker_pgd_pa;
+EXPORT_SYMBOL_GPL(x86_caretaker_pgd_pa);
+
+bool __cpu_preserved_text arch_cpu_preserved_is_active(void)
+{
+ struct cpu_preserved_stack_context *sctx = cpu_preserved_get_stack_context();
+ unsigned long cr3 = __read_cr3();
+
+ if (sctx && sctx->session_pgd_pa && cr3 == sctx->session_pgd_pa)
+ return true;
+
+ if (x86_caretaker_pgd_pa)
+ return cr3 == x86_caretaker_pgd_pa;
+
+ return false;
+}
+EXPORT_SYMBOL_GPL(arch_cpu_preserved_is_active);
+
+void __cpu_preserved_text arch_cpu_preserved_switch_pgd(phys_addr_t pgd_pa)
+{
+ if (pgd_pa && __read_cr3() != pgd_pa)
+ write_cr3(pgd_pa);
+}
+EXPORT_SYMBOL_GPL(arch_cpu_preserved_switch_pgd);
+
+asmlinkage void arch_cpu_preserved_call_on_stack(int cpu, unsigned long stack,
+ void (*fn)(int cpu));
+
+static void __cpu_preserved_text arch_cpu_preserved_park_worker(int cpu)
+{
+ struct cpu_preserved_stack_context *sctx = cpu_preserved_get_stack_context();
+ phys_addr_t pgd_pa = 0;
+
+ arch_cpu_preserved_park_init(cpu);
+
+ if (sctx && sctx->session_pgd_pa)
+ pgd_pa = sctx->session_pgd_pa;
+ else
+ pgd_pa = cpu_preserved_get_pgd(cpu);
+
+ if (!pgd_pa)
+ pgd_pa = x86_caretaker_pgd_pa;
+
+ if (pgd_pa)
+ write_cr3(pgd_pa);
+
+ cpu_preserved_park_loop(cpu);
+
+ arch_cpu_preserved_park_finish(cpu);
+ native_irq_disable();
+ cpu_preserved_set_dead(cpu);
+ while (1) {
+ native_irq_disable();
+ asm volatile("hlt");
+ }
+}
+STACK_FRAME_NON_STANDARD(arch_cpu_preserved_park_worker);
+
+static void arch_cpu_preserved_set_max_perf(void)
+{
+ u64 cap;
+
+ /* Intel HWP (Speed Shift): autonomously request maximum performance */
+ if (boot_cpu_has(X86_FEATURE_HWP) &&
+ !rdmsrq_safe(MSR_HWP_CAPABILITIES, &cap)) {
+ u8 highest = HWP_HIGHEST_PERF(cap);
+
+ if (highest) {
+ wrmsrq_safe(MSR_HWP_REQUEST, HWP_MIN_PERF(highest) |
+ HWP_MAX_PERF(highest) |
+ HWP_DESIRED_PERF(highest));
+ }
+ }
+
+ /* Intel Energy Performance Bias: hint for maximum performance */
+ if (boot_cpu_has(X86_FEATURE_EPB))
+ wrmsrq_safe(MSR_IA32_ENERGY_PERF_BIAS, ENERGY_PERF_BIAS_PERFORMANCE);
+
+ /* AMD CPPC: request maximum performance ratio and zero energy preference */
+ if (boot_cpu_has(X86_FEATURE_CPPC)) {
+ wrmsrq_safe(MSR_AMD_CPPC_REQ, AMD_CPPC_MAX_PERF_MASK |
+ AMD_CPPC_MIN_PERF_MASK | AMD_CPPC_DES_PERF_MASK);
+ }
+}
+
+/*
+ * Switch stack and enter park loop.
+ */
+void arch_cpu_preserved_park_on_stack(int cpu, unsigned long stack_top)
+{
+ if (cpu >= 0 && cpu < CONFIG_NR_CPUS) {
+ if (x86_preserved_x2apic)
+ x86_preserved_apicid[cpu] =
+ (u32)native_rdmsrq(APIC_BASE_MSR + (APIC_ID >> 4));
+ else
+ x86_preserved_apicid[cpu] = cpu_physical_id(cpu);
+ cpu_preserved_clean(&x86_preserved_apicid[cpu]);
+ }
+ arch_cpu_preserved_set_max_perf();
+ arch_cpu_preserved_call_on_stack(cpu, stack_top, arch_cpu_preserved_park_worker);
+}
+EXPORT_SYMBOL_GPL(arch_cpu_preserved_park_on_stack);
+
+/*
+ * Clean data cache for address range. x86 has hardware coherent caches,
+ * so a memory barrier suffices without calling unpreserved external routines.
+ */
+void __cpu_preserved_text arch_cpu_preserved_dcache_clean(unsigned long start, unsigned long end)
+{
+ /* Memory barrier to serialize cache operations on x86 */
+ mb();
+}
+
+/*
+ * Invalidate data cache for address range.
+ */
+void __cpu_preserved_text arch_cpu_preserved_dcache_inval(unsigned long start, unsigned long end)
+{
+ /* Memory barrier to serialize cache operations on x86 */
+ mb();
+}
+
+/**
+ * arch_cpu_preserved_as_map - Populate an isolated page table on x86
+ * @as: Address space to map into.
+ * @pa: Physical address of the range.
+ * @va: Virtual address the range must appear at.
+ * @size: Size of the range in bytes.
+ * @prot: Protection to apply.
+ *
+ * Return: 0 on success, or a negative errno on failure.
+ */
+int arch_cpu_preserved_as_map(struct cpu_preserved_as *as, phys_addr_t pa,
+ unsigned long va, size_t size, pgprot_t prot)
+{
+ unsigned long offset = va & ~PAGE_MASK;
+ size_t page_size = PAGE_ALIGN(offset + size);
+ unsigned long page_va = va & PAGE_MASK;
+ phys_addr_t page_pa = (pa & PAGE_MASK);
+ struct x86_mapping_info info = {
+ .alloc_pgt_page = cpu_preserved_as_alloc_page,
+ .context = as,
+ .page_flag = pgprot_val(prot),
+ .offset = page_va - page_pa,
+ .force_pte = true,
+ };
+
+ return kernel_ident_mapping_init(&info, as->pgd, page_pa,
+ page_pa + page_size);
+}
+
+/*
+ * A preserved CPU only ever enters one of these address spaces by loading
+ * CR3, which flushes everything that is not global, and nothing in them is
+ * mapped global.
+ */
+void arch_cpu_preserved_as_flush_tlb(void)
+{
+}
+
+void arch_cpu_preserved_set_transition_as(struct cpu_preserved_as *as)
+{
+ x86_caretaker_pgd_pa = as ? as->pgd_pa : 0;
+}
+
+int arch_cpu_preserved_setup_buffer(struct page *text_page,
+ unsigned int text_nr_pages,
+ struct page *data_page,
+ unsigned int data_nr_pages)
+{
+ unsigned long text_start = (unsigned long)__cpu_preserved_text_start;
+ unsigned long data_start = (unsigned long)__cpu_preserved_data_start;
+ unsigned int i;
+ int ret;
+
+ if (!x2apic_enabled()) {
+ pr_warn("cpu_preserve: x2APIC is required\n");
+ return -EOPNOTSUPP;
+ }
+
+ /* Split kernel large pages into 4K PTEs */
+ ret = set_memory_4k(text_start, text_nr_pages);
+ if (ret)
+ return ret;
+
+ ret = set_memory_4k(data_start, data_nr_pages);
+ if (ret)
+ return ret;
+
+ /* Remap init_mm kernel mappings to point to allocated buffer pages */
+ for (i = 0; i < text_nr_pages; i++) {
+ unsigned int level;
+ pte_t *pte = lookup_address(text_start + i * PAGE_SIZE, &level);
+
+ if (pte && level == PG_LEVEL_4K) {
+ phys_addr_t pa = page_to_phys(text_page) + i * PAGE_SIZE;
+
+ set_pte(pte, pfn_pte(PHYS_PFN(pa), pte_pgprot(*pte)));
+ }
+ }
+
+ for (i = 0; i < data_nr_pages; i++) {
+ unsigned int level;
+ pte_t *pte = lookup_address(data_start + i * PAGE_SIZE, &level);
+
+ if (pte && level == PG_LEVEL_4K) {
+ phys_addr_t pa = page_to_phys(data_page) + i * PAGE_SIZE;
+
+ set_pte(pte, pfn_pte(PHYS_PFN(pa), pte_pgprot(*pte)));
+ }
+ }
+
+ flush_tlb_all();
+
+ /* Ensure preserved GDT, IDT, and arch flags are initialized */
+ arch_cpu_preserved_early_init();
+ init_preserved_idt();
+ init_preserved_gdt();
+ cpu_preserved_clean(&x86_preserved_has_svm);
+
+ return 0;
+}
+
+void arch_cpu_preserved_wait_dead(int cpu)
+{
+}
--
2.55.0.1082.g2b9226bbc0-goog