[RFC PATCH 15/46] cpu_preserve: Add physical CPU preservation lifecycle and build rules

From: Pasha Tatashin

Date: Sun Sep 20 2026 - 15:39:51 EST


Implement physical CPU preservation lifecycle management, stack allocation,
and build rules with -mbranch-protection=none and -fno-stack-protector.

Signed-off-by: Pasha Tatashin <pasha.tatashin@xxxxxxxxxx>
---
kernel/cpu.c | 2 +
kernel/liveupdate/Kconfig | 35 +++
kernel/liveupdate/Makefile | 3 +
kernel/liveupdate/cpu_preserve.c | 359 +++++++++++++++++++++++++++++++
4 files changed, 399 insertions(+)

diff --git a/kernel/cpu.c b/kernel/cpu.c
index b3c8553d7bd6..8637b5e7f2de 100644
--- a/kernel/cpu.c
+++ b/kernel/cpu.c
@@ -14,6 +14,7 @@
#include <linux/sched/smt.h>
#include <linux/unistd.h>
#include <linux/cpu.h>
+#include <linux/cpu_preserve.h>
#include <linux/oom.h>
#include <linux/rcupdate.h>
#include <linux/delay.h>
@@ -349,6 +350,7 @@ static inline void cpuhp_ap_update_sync_state(enum cpuhp_sync_state state) { }
void cpuhp_ap_report_dead(void)
{
cpuhp_ap_update_sync_state(SYNC_STATE_DEAD);
+ cpu_preserved_report_dead();
}

void __weak arch_cpuhp_cleanup_dead_cpu(unsigned int cpu) { }
diff --git a/kernel/liveupdate/Kconfig b/kernel/liveupdate/Kconfig
index 2490f9a9b838..52c5fd7a620b 100644
--- a/kernel/liveupdate/Kconfig
+++ b/kernel/liveupdate/Kconfig
@@ -101,4 +101,39 @@ config LIVEUPDATE_GUEST_MEMFD

If unsure, say N.

+config ARCH_SUPPORTS_LIVEUPDATE_CPU
+ bool
+
+config LIVEUPDATE_CPU
+ bool "Live update support for preserved CPUs"
+ depends on LIVEUPDATE
+ depends on ARCH_SUPPORTS_LIVEUPDATE_CPU
+ depends on HOTPLUG_CPU
+ default LIVEUPDATE
+ help
+ Enable live update support for physical CPU preservation.
+
+ This feature allows physical CPU cores to be preserved across
+ kexec-based live updates without undergoing a hardware reset or
+ full kernel offline/online cycle.
+
+ When a core is preserved, it is isolated from host scheduling and
+ parked in a dedicated execution loop within preserved memory pages.
+ Its execution context, state, and necessary page tables are handed
+ over across the kexec transition using the KHO
+ framework. The incoming kernel detects preserved cores during early
+ boot and avoids bringing them up during smp_init(), keeping them
+ isolated and running their assigned workloads.
+
+ This capability is a core building block for live-updating virtual
+ machine hosts (e.g. Caretaker/OrphanVM) without pausing or
+ disrupting running guest workloads, and can also speed up kernel
+ live updates by skipping CPU re-initialization.
+
+ Preservation lifecycle is managed from userspace via LUO sessions
+ associated with /sys/devices/system/cpu/cpu<N>/preserve file
+ descriptors, and monitored via sysfs.
+
+ If unsure, say N.
+
endmenu
diff --git a/kernel/liveupdate/Makefile b/kernel/liveupdate/Makefile
index d9f469462556..486f7854bb75 100644
--- a/kernel/liveupdate/Makefile
+++ b/kernel/liveupdate/Makefile
@@ -1,5 +1,7 @@
# SPDX-License-Identifier: GPL-2.0

+CFLAGS_cpu_preserve.o += $(call cc-option,-mbranch-protection=none) -fno-stack-protector $(call cc-option,-ftrivial-auto-var-init=uninitialized) $(call cc-option,-fno-jump-tables)
+
luo-y := \
kho_block.o \
luo_core.o \
@@ -11,3 +13,4 @@ obj-$(CONFIG_KEXEC_HANDOVER) += kexec_handover.o
obj-$(CONFIG_KEXEC_HANDOVER_DEBUGFS) += kexec_handover_debugfs.o

obj-$(CONFIG_LIVEUPDATE) += luo.o
+obj-$(CONFIG_LIVEUPDATE_CPU) += cpu_preserve.o
diff --git a/kernel/liveupdate/cpu_preserve.c b/kernel/liveupdate/cpu_preserve.c
index e19208670c3e..c5fd3a07ccb5 100644
--- a/kernel/liveupdate/cpu_preserve.c
+++ b/kernel/liveupdate/cpu_preserve.c
@@ -247,3 +247,362 @@ static phys_addr_t cpu_preserved_get_data_pa(void)
return cpu_preserved_data_pages ? page_to_phys(cpu_preserved_data_pages) : 0;
}

+static void cpu_preserved_sync_global_ser(void)
+{
+ struct cpu_preserved_global_ser *ser = cpu_preserved_global_ser;
+
+ if (!ser)
+ return;
+
+ bitmap_to_arr64(ser->cpu_preserved_bitmap,
+ cpumask_bits(&cpu_preserved_mask), nr_cpu_ids);
+ if (cpu_preserved_text_pages) {
+ ser->text_runtime_pa = page_to_phys(cpu_preserved_text_pages);
+ ser->text_runtime_size =
+ (1UL << cpu_preserved_text_order) * PAGE_SIZE;
+ }
+ if (cpu_preserved_data_pages) {
+ ser->data_runtime_pa = page_to_phys(cpu_preserved_data_pages);
+ ser->data_runtime_size =
+ (1UL << cpu_preserved_data_order) * PAGE_SIZE;
+ }
+ KHOSER_STORE_PTR(ser->pcpus_runtime, cpu_preserved_outgoing.pcpus_ser);
+ cpu_preserved_clean_sz(ser,
+ struct_size(ser, cpu_preserved_bitmap, ser->nr_cpu_words));
+}
+
+static void cpu_preserved_free_kho(void *va, bool is_incoming)
+{
+ if (!va)
+ return;
+
+ if (is_incoming)
+ kho_restore_free(va);
+ else
+ kho_unpreserve_free(va);
+}
+
+static void cpu_preserved_free_stack(phys_addr_t stack_pa, bool is_incoming)
+{
+ if (stack_pa)
+ cpu_preserved_free_kho(phys_to_virt(stack_pa), is_incoming);
+}
+
+static void cpu_preserved_state_cleanup(struct cpu_preserved_state *st,
+ bool is_incoming)
+{
+ if (!cpumask_empty(&st->mask))
+ return;
+
+ cpu_preserved_free_kho(st->pcpus_ser, is_incoming);
+ st->pcpus_ser = NULL;
+ kfree(st->pcpus);
+ st->pcpus = NULL;
+}
+
+/*
+ * Drop @cpu out of the preserved state, free its preserved stack, and
+ * republish the globals a parked core may still be reading. The caller holds
+ * cpu_preserved_lock and has already made the core leave the park loop.
+ */
+static void __cpu_unpreserve_locked(unsigned int cpu)
+{
+ struct cpu_preserved_state *incoming = &cpu_preserved_incoming;
+ struct cpu_preserved_state *outgoing = &cpu_preserved_outgoing;
+ struct cpu_preserved_pcpu_ser *ser = cpu_preserved_get_pcpu_ser(cpu);
+ struct cpu_preserved_pcpu *pcpu = cpu_preserved_get_pcpu(cpu);
+ bool is_incoming = cpu_preserved_is_incoming(cpu);
+ phys_addr_t stack_pa = 0;
+
+ lockdep_assert_held(&cpu_preserved_lock);
+
+ cpumask_clear_cpu(cpu, &outgoing->mask);
+ cpumask_clear_cpu(cpu, &incoming->mask);
+ cpumask_clear_cpu(cpu, &cpu_preserved_mask);
+ cpu_preserved_clean(&cpu_preserved_mask);
+ set_cpu_present(cpu, true);
+
+ if (ser)
+ WRITE_ONCE(ser->workload, 0);
+
+ if (pcpu) {
+ stack_pa = pcpu->stack_pa;
+ memset(pcpu, 0, sizeof(*pcpu));
+ }
+
+ cpu_preserved_free_stack(stack_pa, is_incoming);
+
+ /* @pcpu and @ser point into these arrays: do not touch past this point. */
+ cpu_preserved_state_cleanup(outgoing, false);
+ cpu_preserved_state_cleanup(incoming, true);
+
+ if (cpumask_empty(&cpu_preserved_mask)) {
+ WRITE_ONCE(cpu_preserved_pcpus_va, NULL);
+ WRITE_ONCE(cpu_preserved_host_pcpus_va, NULL);
+ cpu_preserved_clean(&cpu_preserved_pcpus_va);
+ cpu_preserved_clean(&cpu_preserved_host_pcpus_va);
+ }
+
+ cpu_preserved_sync_global_ser();
+}
+
+/**
+ * cpu_unpreserve - Unpreserve a physical CPU and restore it to online state
+ * @cpu: Logical CPU identifier.
+ *
+ * Signals the CPU to exit the parking loop, cleans up preserved stack memory,
+ * and restores the core to host scheduling via standard add_cpu().
+ */
+static void cpu_unpreserve(unsigned int cpu)
+{
+ int ret;
+
+ scoped_guard(mutex, &cpu_preserved_lock) {
+ if (!cpu_is_preserved(cpu))
+ return;
+
+ cpu_signal_exit(cpu);
+ arch_cpu_preserved_kick(cpu);
+ }
+
+ /*
+ * cpu_wait_dead() busy-polls for up to 20 seconds. Do not hold
+ * cpu_preserved_lock across it: the poll only reads pcpu->state, which
+ * stays valid for as long as the CPU is preserved, and holding the lock
+ * here would stall every other preservation operation and every sysfs
+ * reader for the entire window.
+ */
+ if (cpu_wait_dead(cpu))
+ return;
+
+ scoped_guard(mutex, &cpu_preserved_lock) {
+ if (!cpu_is_preserved(cpu))
+ return;
+
+ __cpu_unpreserve_locked(cpu);
+ }
+
+ ret = add_cpu(cpu);
+ if (ret < 0)
+ pr_err("Failed to bring unpreserved cpu %u back online: %d\n",
+ cpu, ret);
+}
+
+/*
+ * FLB Ops for Preserved CPUs
+ */
+static int cpu_preserved_flb_preserve(struct liveupdate_flb_op_args *argp)
+{
+ unsigned int nr_words = BITS_TO_U64(nr_cpu_ids);
+ struct cpu_preserved_global_ser *ser;
+ size_t ser_sz;
+ int ret;
+
+ ret = cpu_preserved_init_runtime_buffer();
+ if (ret)
+ return ret;
+
+ ser_sz = struct_size(ser, cpu_preserved_bitmap, nr_words);
+
+ mutex_lock(&cpu_preserved_lock);
+ ser = kho_alloc_preserve(ser_sz);
+ if (IS_ERR(ser)) {
+ mutex_unlock(&cpu_preserved_lock);
+ return PTR_ERR(ser);
+ }
+
+ memset(ser, 0, ser_sz);
+ ser->nr_cpu_words = nr_words;
+ cpu_preserved_global_ser = ser;
+ cpu_preserved_sync_global_ser();
+ mutex_unlock(&cpu_preserved_lock);
+
+ argp->data = virt_to_phys(ser);
+ argp->obj = ser;
+ return 0;
+}
+
+static void cpu_preserved_flb_unpreserve(struct liveupdate_flb_op_args *argp)
+{
+ struct cpu_preserved_global_ser *ser;
+
+ if (!argp->data)
+ return;
+
+ ser = phys_to_virt(argp->data);
+ mutex_lock(&cpu_preserved_lock);
+ cpu_preserved_global_ser = NULL;
+ mutex_unlock(&cpu_preserved_lock);
+
+ cpu_preserved_unpreserve_runtime_buffer();
+ kho_unpreserve_free(ser);
+}
+
+static int cpu_preserved_flb_retrieve(struct liveupdate_flb_op_args *argp)
+{
+ struct cpu_preserved_global_ser *ser;
+ struct cpu_preserved_pcpu_ser *pcpus;
+ u64 nr_bits;
+ int cpu;
+
+ if (!argp->data)
+ return -EINVAL;
+
+ ser = phys_to_virt(argp->data);
+ arch_cpu_preserved_early_init();
+
+ /*
+ * The outgoing kernel may have been built with a larger NR_CPUS. Any
+ * preserved CPU we cannot represent would be silently forgotten and
+ * left spinning in its park loop forever, so refuse the handover
+ * instead.
+ */
+ nr_bits = (u64)ser->nr_cpu_words * BITS_PER_TYPE(u64);
+ if (nr_bits > nr_cpu_ids &&
+ find_next_bit((const unsigned long *)ser->cpu_preserved_bitmap,
+ nr_bits, nr_cpu_ids) < nr_bits) {
+ pr_err("preserved CPU above nr_cpu_ids=%u in handover data\n",
+ nr_cpu_ids);
+ return -ERANGE;
+ }
+
+ mutex_lock(&cpu_preserved_lock);
+ bitmap_from_arr64(cpumask_bits(&cpu_preserved_mask),
+ ser->cpu_preserved_bitmap, min_t(u64, nr_bits, nr_cpu_ids));
+ cpumask_copy(&cpu_preserved_incoming.mask, &cpu_preserved_mask);
+ pcpus = KHOSER_LOAD_PTR(ser->pcpus_runtime);
+
+ if (pcpus) {
+ cpu_preserved_incoming.pcpus_ser = pcpus;
+ WRITE_ONCE(cpu_preserved_pcpus_va, pcpus);
+ cpu_preserved_clean(&cpu_preserved_pcpus_va);
+ }
+ cpu_preserved_clean(&cpu_preserved_mask);
+ for_each_cpu(cpu, &cpu_preserved_mask)
+ set_cpu_present(cpu, false);
+ mutex_unlock(&cpu_preserved_lock);
+
+ argp->obj = ser;
+ return 0;
+}
+
+static void cpu_preserved_flb_finish(struct liveupdate_flb_op_args *argp)
+{
+ struct cpu_preserved_global_ser *ser;
+
+ if (!argp->obj)
+ return;
+
+ ser = argp->obj;
+
+ scoped_guard(mutex, &cpu_preserved_lock) {
+ if (cpu_preserved_incoming.pcpus_ser) {
+ kho_restore_free(cpu_preserved_incoming.pcpus_ser);
+ cpu_preserved_incoming.pcpus_ser = NULL;
+ }
+ kfree(cpu_preserved_incoming.pcpus);
+ cpu_preserved_incoming.pcpus = NULL;
+ }
+
+ if (ser->text_runtime_pa && ser->text_runtime_size) {
+ unsigned long nr_pages = ser->text_runtime_size >> PAGE_SHIFT;
+ struct page *page = kho_restore_pages(ser->text_runtime_pa, nr_pages);
+
+ if (page) {
+ for (unsigned long i = 0; i < nr_pages; i++)
+ __free_page(page + i);
+ }
+ }
+
+ if (ser->data_runtime_pa && ser->data_runtime_size) {
+ unsigned long nr_pages = ser->data_runtime_size >> PAGE_SHIFT;
+ struct page *page = kho_restore_pages(ser->data_runtime_pa, nr_pages);
+
+ if (page) {
+ for (unsigned long i = 0; i < nr_pages; i++)
+ __free_page(page + i);
+ }
+ }
+
+ kho_restore_free(ser);
+}
+
+static const struct liveupdate_flb_ops cpu_preserved_flb_ops = {
+ .preserve = cpu_preserved_flb_preserve,
+ .unpreserve = cpu_preserved_flb_unpreserve,
+ .retrieve = cpu_preserved_flb_retrieve,
+ .finish = cpu_preserved_flb_finish,
+ .owner = THIS_MODULE,
+};
+
+static struct liveupdate_flb cpu_preserved_flb = {
+ .ops = &cpu_preserved_flb_ops,
+ .compatible = CPU_PRESERVED_LUO_FLB_COMPATIBLE,
+};
+
+static int cpu_preserve_reboot_notify(struct notifier_block *nb,
+ unsigned long action, void *data)
+{
+ int cpu;
+
+ scoped_guard(mutex, &cpu_preserved_lock) {
+ for_each_cpu(cpu, &cpu_preserved_mask) {
+ /*
+ * If this CPU is not being preserved across an outgoing
+ * live update, signal it to exit the park loop and
+ * offline it.
+ */
+ if (kexec_in_progress && liveupdate_enabled() &&
+ !cpu_preserved_is_incoming(cpu))
+ continue;
+
+ cpu_signal_exit(cpu);
+ arch_cpu_preserved_kick(cpu);
+ if (cpu_wait_dead(cpu))
+ continue;
+
+ __cpu_unpreserve_locked(cpu);
+ }
+ }
+
+ return NOTIFY_OK;
+}
+
+static struct notifier_block cpu_preserve_reboot_nb = {
+ .notifier_call = cpu_preserve_reboot_notify,
+ .priority = 0,
+};
+
+/**
+ * cpu_preserve_early_init - Early boot registration & retrieval of CPUs
+ *
+ * Registers the preserved CPU file handler and FLB with LUO, retrieves incoming
+ * preserved CPU state prior to secondary SMP bringup, and registers the reboot
+ * notifier.
+ *
+ * Return: 0 on success, or negative error code on failure.
+ */
+static int __init cpu_preserve_early_init(void)
+{
+ void *obj;
+ int err;
+
+ if (!liveupdate_enabled())
+ cpumask_clear(&cpu_preserved_mask);
+ cpumask_clear(&cpu_preserved_outgoing.mask);
+ cpumask_clear(&cpu_preserved_incoming.mask);
+ cpu_preserved_outgoing.pcpus = NULL;
+ cpu_preserved_outgoing.pcpus_ser = NULL;
+ cpu_preserved_incoming.pcpus = NULL;
+ cpu_preserved_incoming.pcpus_ser = NULL;
+ cpu_preserved_global_ser = NULL;
+
+ /* Retrieve incoming preserved CPUs before secondary CPU bringup */
+ if (liveupdate_enabled())
+ liveupdate_flb_get_incoming(&cpu_preserved_flb, &obj);
+
+ register_reboot_notifier(&cpu_preserve_reboot_nb);
+
+ return 0;
+}
+early_initcall(cpu_preserve_early_init);
--
2.55.0.1082.g2b9226bbc0-goog