Enable physical CPU preservation on arm64 by selecting ARCH_SUPPORTS_LIVEUPDATE_CPU.
Integrate preserved CPU handling into the arm64 SMP and CPU hotplug paths with isolated transition page tables: - Build isolated transition page tables mapping exclusively preserved text/rodata (ROX), data (RW), stacks (RW), and preserved buffers. - Completely eliminate bulk cloning of the kernel linear direct map. - Limit KHO page table memory footprint to < 10 pages total. - Include CPU_PRESERVED_TEXT in arm64 vmlinux.lds.S. - Compile cpu_preserve.o with -mbranch-protection=none and -fno-stack-protector to prevent PAC key mismatch aborts. - In __cpu_disable(), avoid tearing down IPIs for preserved CPUs. - In op_cpu_kill(), skip waiting for the CPU to die if preserved. - In smp_send_stop(), exclude preserved CPUs from stop IPIs. - Add cache maintenance to PoC for secondary boot data and page tables. Signed-off-by: Pasha Tatashin <[email protected]> --- arch/arm64/Kconfig | 1 + arch/arm64/include/asm/cpu_preserve.h | 44 ++++ arch/arm64/kernel/Makefile | 7 + arch/arm64/kernel/cpu_preserve.c | 365 ++++++++++++++++++++++++++ arch/arm64/kernel/preserve_cpu.S | 44 ++++ arch/arm64/kernel/smp.c | 8 +- arch/arm64/kernel/vmlinux.lds.S | 1 + 7 files changed, 469 insertions(+), 1 deletion(-) create mode 100644 arch/arm64/include/asm/cpu_preserve.h create mode 100644 arch/arm64/kernel/cpu_preserve.c create mode 100644 arch/arm64/kernel/preserve_cpu.S diff --git a/arch/arm64/Kconfig b/arch/arm64/Kconfig index b5a51b0ef944..957ec38a9680 100644 --- a/arch/arm64/Kconfig +++ b/arch/arm64/Kconfig @@ -38,6 +38,7 @@ config ARM64 select ARCH_HAS_MEMBARRIER_SYNC_CORE select ARCH_HAS_MEM_ENCRYPT select ARCH_SUPPORTS_MSEAL_SYSTEM_MAPPINGS + select ARCH_SUPPORTS_LIVEUPDATE_CPU if LIVEUPDATE select ARCH_HAS_NMI_SAFE_THIS_CPU_OPS select ARCH_HAS_NON_OVERLAPPING_ADDRESS_SPACE select ARCH_HAS_NONLEAF_PMD_YOUNG if ARM64_HAFT diff --git a/arch/arm64/include/asm/cpu_preserve.h b/arch/arm64/include/asm/cpu_preserve.h new file mode 100644 index 000000000000..7c7c94f21ccc --- /dev/null +++ b/arch/arm64/include/asm/cpu_preserve.h @@ -0,0 +1,44 @@ +/* SPDX-License-Identifier: GPL-2.0 */ +/* + * Copyright (c) 2026, Google LLC. + * Pasha Tatashin <[email protected]> + */ +#ifndef __ASM_ARM64_CPU_PRESERVE_H +#define __ASM_ARM64_CPU_PRESERVE_H + +#include <asm/memory.h> +#include <asm/tlbflush.h> +#include <asm/virt.h> + +#define ARCH_CPU_PRESERVED_STACK_ORDER (THREAD_SIZE_ORDER + 1) + +bool arch_cpu_preserved_is_active(void); +asmlinkage void __arch_cpu_preserved_dcache_clean(unsigned long start, unsigned long end); +asmlinkage void arch_cpu_preserved_dcache_clean(unsigned long start, unsigned long end); +asmlinkage void arch_cpu_preserved_dcache_inval(unsigned long start, unsigned long end); + +static inline void arm64_flush_host_tlb_local(void) +{ + dsb(nshst); + if (is_kernel_in_hyp_mode()) { + asm volatile("tlbi alle2\n" + "dsb nsh\n" + "isb\n" ::: "memory"); + } else { + local_flush_tlb_all(); + } +} + +static inline void arm64_flush_host_tlb_all(void) +{ + dsb(ishst); + if (is_kernel_in_hyp_mode()) { + asm volatile("tlbi alle2is\n" + "dsb ish\n" + "isb\n" ::: "memory"); + } else { + flush_tlb_all(); + } +} + +#endif /* __ASM_ARM64_CPU_PRESERVE_H */ diff --git a/arch/arm64/kernel/Makefile b/arch/arm64/kernel/Makefile index d2690c3ec528..151513a5350b 100644 --- a/arch/arm64/kernel/Makefile +++ b/arch/arm64/kernel/Makefile @@ -70,6 +70,13 @@ obj-$(CONFIG_ARM_SDE_INTERFACE) += sdei.o obj-$(CONFIG_ARM64_PTR_AUTH) += pointer_auth.o obj-$(CONFIG_ARM64_MPAM) += mpam.o obj-$(CONFIG_ARM64_MTE) += mte.o +obj-$(CONFIG_LIVEUPDATE_CPU) += cpu_preserve.o preserve_cpu.o +KASAN_SANITIZE_cpu_preserve.o := n +KCSAN_SANITIZE_cpu_preserve.o := n +UBSAN_SANITIZE_cpu_preserve.o := n +KCOV_INSTRUMENT_cpu_preserve.o := n +CFLAGS_REMOVE_cpu_preserve.o = $(CC_FLAGS_FTRACE) +CFLAGS_cpu_preserve.o += $(call cc-option,-mbranch-protection=none) -fno-stack-protector $(call cc-option,-ftrivial-auto-var-init=uninitialized) $(call cc-option,-fno-jump-tables) obj-y += vdso-wrap.o obj-$(CONFIG_COMPAT_VDSO) += vdso32-wrap.o diff --git a/arch/arm64/kernel/cpu_preserve.c b/arch/arm64/kernel/cpu_preserve.c new file mode 100644 index 000000000000..15de1ca09062 --- /dev/null +++ b/arch/arm64/kernel/cpu_preserve.c @@ -0,0 +1,365 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Copyright (c) 2026, Google LLC. + * Pasha Tatashin <[email protected]> + * + * Architecture specific CPU preservation support for ARM64. + */ +#include <linux/arm-smccc.h> +#include <linux/cpu_preserve.h> +#include <linux/irqchip/arm-gic-v3.h> +#include <linux/kexec_handover.h> +#include <linux/kho/abi/cpu.h> +#include <linux/mm.h> +#include <linux/psci.h> +#include <linux/sched/mm.h> +#include <uapi/linux/psci.h> + +#include <asm/barrier.h> +#include <linux/cacheflush.h> +#include <asm/cpu_ops.h> +#include <asm/daifflags.h> +#include <asm/kernel-pgtable.h> +#include <asm/kvm_asm.h> +#include <linux/pgtable.h> +#include <asm/sysreg.h> +#include <asm/tlbflush.h> +#include <asm/trans_pgd.h> +#include <asm/virt.h> + +static enum arm_smccc_conduit arm64_psci_conduit __cpu_preserved_data; +phys_addr_t arm64_caretaker_pgd_pa __cpu_preserved_data; +static u64 arm64_cpu_mpidr[NR_CPUS] __cpu_preserved_data; + +/* + * Signal or wake up a preserved physical CPU via SEV. + */ +void __cpu_preserved_text arch_cpu_preserved_kick(int cpu) +{ + dsb(ishst); + sev(); + isb(); +} + +/* + * Low-power wait in parking loop. + */ +void __cpu_preserved_text arch_cpu_preserved_park_wait(void) +{ + wfe(); +} + +int __cpu_preserved_text arch_cpu_preserved_mpidr_to_cpu(u64 mpidr) +{ + int c; + + for (c = 0; c < ARRAY_SIZE(arm64_cpu_mpidr); c++) { + if ((arm64_cpu_mpidr[c] & MPIDR_HWID_BITMASK) == (mpidr & MPIDR_HWID_BITMASK)) + return c; + } + return -EINVAL; +} +EXPORT_SYMBOL_GPL(arch_cpu_preserved_mpidr_to_cpu); + +bool __cpu_preserved_text arch_cpu_preserved_is_active(void) +{ + struct cpu_preserved_stack_context *sctx = cpu_preserved_get_stack_context(); + u64 ttbr1 = read_sysreg(ttbr1_el1); + + if (sctx && sctx->session_pgd_pa && ttbr1 == sctx->session_pgd_pa) + return true; + + if (arm64_caretaker_pgd_pa) + return ttbr1 == arm64_caretaker_pgd_pa; + + return false; +} +EXPORT_SYMBOL_GPL(arch_cpu_preserved_is_active); + +void __cpu_preserved_text arch_cpu_preserved_switch_pgd(phys_addr_t pgd_pa) +{ + if (pgd_pa && read_sysreg(ttbr1_el1) != pgd_pa) { + write_sysreg(pgd_pa, ttbr1_el1); + isb(); + arm64_flush_host_tlb_local(); + } +} +EXPORT_SYMBOL_GPL(arch_cpu_preserved_switch_pgd); + +static pte_t *arm64_get_kernel_pte(unsigned long addr) +{ + pgd_t *pgdp = pgd_offset_k(addr); + p4d_t *p4dp; + pud_t *pudp; + pmd_t *pmdp; + + if (pgd_none(READ_ONCE(*pgdp))) + return NULL; + + p4dp = p4d_offset(pgdp, addr); + if (p4d_none(READ_ONCE(*p4dp))) + return NULL; + + pudp = pud_offset(p4dp, addr); + if (pud_none(READ_ONCE(*pudp)) || pud_leaf(READ_ONCE(*pudp))) + return NULL; + + pmdp = pmd_offset(pudp, addr); + if (pmd_none(READ_ONCE(*pmdp)) || pmd_leaf(READ_ONCE(*pmdp))) + return NULL; + + return pte_offset_kernel(pmdp, addr); +} + +/** + * arch_cpu_preserved_as_map - Populate an isolated page table on arm64 + * @as: Address space to map into. + * @pa: Physical address of the range. + * @va: Virtual address the range must appear at. + * @size: Size of the range in bytes. + * @prot: Protection to apply. + * + * Return: 0 on success, or a negative errno on failure. + */ +int arch_cpu_preserved_as_map(struct cpu_preserved_as *as, phys_addr_t pa, + unsigned long va, size_t size, pgprot_t prot) +{ + struct trans_pgd_info info = { + .trans_alloc_page = cpu_preserved_as_alloc_page, + .trans_alloc_arg = as, + }; + unsigned long offset = va & ~PAGE_MASK; + size_t page_size = PAGE_ALIGN(offset + size); + unsigned long page_va = va & PAGE_MASK; + phys_addr_t page_pa = (pa & PAGE_MASK); + + return trans_pgd_map_range(&info, as->pgd, page_pa, + page_va, page_size, prot); +} + +void arch_cpu_preserved_as_flush_tlb(void) +{ + arm64_flush_host_tlb_all(); +} + +void arch_cpu_preserved_set_transition_as(struct cpu_preserved_as *as) +{ + arm64_caretaker_pgd_pa = as ? as->pgd_pa : 0; + cpu_preserved_clean(&arm64_caretaker_pgd_pa); +} + +/** + * arch_cpu_preserved_setup_buffer - Set up runtime buffer and page tables + * @text_page: Runtime-allocated physical page backing preserved text + * @text_nr_pages: Number of pages in text buffer + * @data_page: Runtime-allocated physical page backing preserved data + * @data_nr_pages: Number of pages in data buffer + * + * Remap init_mm kernel mappings for __cpu_preserved_text and + * __cpu_preserved_data to point to the runtime-allocated pages outside + * Scratch. + * + * Return: 0 on success, or -ENOMEM on failure. + */ +static void arm64_split_contpte_range(unsigned long start, unsigned long end) +{ + unsigned long addr; + + if (start >= end) + return; + + for (addr = ALIGN_DOWN(start, CONT_PTE_SIZE); addr < end; addr += CONT_PTE_SIZE) { + pte_t *ptep = arm64_get_kernel_pte(addr); + int i; + + if (!ptep) + continue; + + ptep = PTR_ALIGN_DOWN(ptep, sizeof(*ptep) * CONT_PTES); + + for (i = 0; i < CONT_PTES; i++) { + pte_t pte = __ptep_get(&ptep[i]); + + if (pte_valid_cont(pte)) + __set_pte(&ptep[i], pte_mknoncont(pte)); + } + } + + flush_tlb_kernel_range(ALIGN_DOWN(start, CONT_PTE_SIZE), + ALIGN(end, CONT_PTE_SIZE)); + arm64_flush_host_tlb_all(); +} + +int arch_cpu_preserved_setup_buffer(struct page *text_page, + unsigned int text_nr_pages, + struct page *data_page, + unsigned int data_nr_pages) +{ + unsigned long text_start = (unsigned long)__cpu_preserved_text_start; + unsigned long data_start = (unsigned long)__cpu_preserved_data_start; + unsigned int i; + + /* Split any contiguous 64KB mappings before replacing individual PTEs */ + arm64_split_contpte_range(text_start, text_start + text_nr_pages * PAGE_SIZE); + arm64_split_contpte_range(data_start, data_start + data_nr_pages * PAGE_SIZE); + + /* Clean old mappings before switching PTEs */ + __arch_cpu_preserved_dcache_clean(text_start, text_start + text_nr_pages * PAGE_SIZE); + __arch_cpu_preserved_dcache_clean(data_start, data_start + data_nr_pages * PAGE_SIZE); + + /* Remap init_mm kernel mappings to point to allocated buffer pages */ + for (i = 0; i < text_nr_pages; i++) { + unsigned long va = text_start + i * PAGE_SIZE; + pte_t *ptep = arm64_get_kernel_pte(va); + + if (!ptep) + return -EINVAL; + + pgprot_t prot = __pgprot(pgprot_val(pte_pgprot(*ptep)) & ~PTE_CONT); + phys_addr_t pa = page_to_phys(text_page) + i * PAGE_SIZE; + + set_pte_at(&init_mm, va, ptep, pfn_pte(PHYS_PFN(pa), prot)); + } + + for (i = 0; i < data_nr_pages; i++) { + unsigned long va = data_start + i * PAGE_SIZE; + pte_t *ptep = arm64_get_kernel_pte(va); + + if (!ptep) + return -EINVAL; + + pgprot_t prot = __pgprot(pgprot_val(pte_pgprot(*ptep)) & ~PTE_CONT); + phys_addr_t pa = page_to_phys(data_page) + i * PAGE_SIZE; + + set_pte_at(&init_mm, va, ptep, pfn_pte(PHYS_PFN(pa), prot)); + } + + arm64_flush_host_tlb_all(); + flush_icache_range(text_start, text_start + (text_nr_pages * PAGE_SIZE)); + + for (i = 0; i < nr_cpu_ids; i++) + arm64_cpu_mpidr[i] = cpu_logical_map(i); + cpu_preserved_clean(&arm64_cpu_mpidr); + + return 0; +} + +/* + * Masks DAIF interrupts and enables GIC CPU interface for WFx wakeups. + */ +void __cpu_preserved_text arch_cpu_preserved_park_init(int cpu) +{ + struct cpu_preserved_stack_context *sctx = cpu_preserved_get_stack_context(); + phys_addr_t pgd_pa = 0; + + if (sctx && sctx->session_pgd_pa) + pgd_pa = sctx->session_pgd_pa; + else + pgd_pa = cpu_preserved_get_pgd(cpu); + + local_daif_mask(); + cpu_preserved_inval(&arm64_psci_conduit); + cpu_preserved_inval(&arm64_cpu_mpidr); + if (!pgd_pa) { + cpu_preserved_inval(&arm64_caretaker_pgd_pa); + pgd_pa = READ_ONCE(arm64_caretaker_pgd_pa); + } + + write_sysreg(0, ttbr0_el1); + if (pgd_pa) + write_sysreg(pgd_pa, ttbr1_el1); + isb(); + arm64_flush_host_tlb_local(); + + write_sysreg_s(0xff, SYS_ICC_PMR_EL1); + write_sysreg_s(1, SYS_ICC_IGRPEN1_EL1); + isb(); +} + +void arch_cpu_preserved_early_init(void) +{ + int c; + + for (c = 0; c < ARRAY_SIZE(arm64_cpu_mpidr); c++) + arm64_cpu_mpidr[c] = cpu_logical_map(c); + cpu_preserved_clean(&arm64_cpu_mpidr); + + cpu_preserved_inval(&arm64_psci_conduit); + if (arm64_psci_conduit == SMCCC_CONDUIT_NONE) { + arm64_psci_conduit = arm_smccc_1_1_get_conduit(); + cpu_preserved_clean(&arm64_psci_conduit); + } +} +EXPORT_SYMBOL_GPL(arch_cpu_preserved_early_init); + +void __cpu_preserved_text arch_cpu_preserved_park_finish(int cpu) +{ + u32 el = (read_sysreg(CurrentEL) >> 2) & 3; + enum arm_smccc_conduit conduit; + + cpu_preserved_inval(&arm64_psci_conduit); + conduit = READ_ONCE(arm64_psci_conduit); + + local_daif_mask(); + write_sysreg_s(0, SYS_ICC_PMR_EL1); + write_sysreg_s(0, SYS_ICC_IGRPEN1_EL1); + isb(); + + if (el == 2 || conduit == SMCCC_CONDUIT_NONE) + conduit = (el == 2) ? SMCCC_CONDUIT_SMC : SMCCC_CONDUIT_HVC; + + cpu_preserved_set_dead(cpu); + + /* + * Direct PSCI CPU_OFF call in preserved text without relying on + * unpreserved kernel data structures or function pointers. + * + * x0: PSCI_0_2_FN_CPU_OFF (0x84000002) + * x1: Power down state (0x00010000) + */ + if (conduit == SMCCC_CONDUIT_HVC) { + asm volatile("mov x0, #0x0002\n" + "movk x0, #0x8400, lsl #16\n" + "mov x1, #0\n" + "mov x2, #0\n" + "mov x3, #0\n" + "mov x4, #0\n" + "mov x5, #0\n" + "mov x6, #0\n" + "mov x7, #0\n" + "hvc #0\n" + : + : + : "x0", "x1", "x2", "x3", "x4", "x5", "x6", "x7", "memory" + ); + } else { + asm volatile("mov x0, #0x0002\n" + "movk x0, #0x8400, lsl #16\n" + "mov x1, #0\n" + "mov x2, #0\n" + "mov x3, #0\n" + "mov x4, #0\n" + "mov x5, #0\n" + "mov x6, #0\n" + "mov x7, #0\n" + "smc #0\n" + : + : + : "x0", "x1", "x2", "x3", "x4", "x5", "x6", "x7", "memory" + ); + } + + while (1) { + wfi(); + wfe(); + } +} + +void arch_cpu_preserved_wait_dead(int cpu) +{ + const struct cpu_operations *ops = get_cpu_ops(cpu); + + if (ops && ops->cpu_kill) + ops->cpu_kill(cpu); +} + diff --git a/arch/arm64/kernel/preserve_cpu.S b/arch/arm64/kernel/preserve_cpu.S new file mode 100644 index 000000000000..c7e3cd8133f0 --- /dev/null +++ b/arch/arm64/kernel/preserve_cpu.S @@ -0,0 +1,44 @@ +/* SPDX-License-Identifier: GPL-2.0-only */ +/* + * Low-level assembly routines for ARM64 physical CPU preservation. + * + * Copyright (c) 2026, Google LLC. + * Pasha Tatashin <[email protected]> + */ +#include <linux/linkage.h> +#include <asm/assembler.h> + + .section .text.cpu_preserved, "ax" + +/* + * arch_cpu_preserved_park_on_stack - Switch stack and enter park loop + * @cpu: Logical CPU identifier (x0) + * @stack_top: Top of runtime allocated stack (x1) + */ +SYM_FUNC_START(arch_cpu_preserved_park_on_stack) + mov sp, x1 + mov x19, x0 + bl cpu_preserved_park_loop + mov x0, x19 + bl arch_cpu_preserved_park_finish + b . +SYM_FUNC_END(arch_cpu_preserved_park_on_stack) +EXPORT_SYMBOL(arch_cpu_preserved_park_on_stack) + +/* + * __arch_cpu_preserved_dcache_clean - Clean and invalidate data cache to PoC + * @start: Virtual start address (x0) + * @end: Virtual end address (x1) + */ +SYM_FUNC_START(__arch_cpu_preserved_dcache_clean) + dsb sy + raw_dcache_line_size x2, x3 + dcache_by_myline_op_nosync civac, x0, x1, x2, x3 + dsb sy + ret +SYM_FUNC_END(__arch_cpu_preserved_dcache_clean) +SYM_FUNC_ALIAS(arch_cpu_preserved_dcache_clean, __arch_cpu_preserved_dcache_clean) +SYM_FUNC_ALIAS(arch_cpu_preserved_dcache_inval, __arch_cpu_preserved_dcache_clean) +EXPORT_SYMBOL(__arch_cpu_preserved_dcache_clean) +EXPORT_SYMBOL(arch_cpu_preserved_dcache_clean) +EXPORT_SYMBOL(arch_cpu_preserved_dcache_inval) diff --git a/arch/arm64/kernel/smp.c b/arch/arm64/kernel/smp.c index a61dc3016a11..43651eceb09c 100644 --- a/arch/arm64/kernel/smp.c +++ b/arch/arm64/kernel/smp.c @@ -21,6 +21,7 @@ #include <linux/mm.h> #include <linux/err.h> #include <linux/cpu.h> +#include <linux/cpu_preserve.h> #include <linux/smp.h> #include <linux/seq_file.h> #include <linux/irq.h> @@ -327,7 +328,12 @@ int __cpu_disable(void) static int op_cpu_kill(unsigned int cpu) { - const struct cpu_operations *ops = get_cpu_ops(cpu); + const struct cpu_operations *ops; + + if (cpu_is_preserved(cpu)) + return 0; + + ops = get_cpu_ops(cpu); /* * If we have no means of synchronising with the dying CPU, then assume diff --git a/arch/arm64/kernel/vmlinux.lds.S b/arch/arm64/kernel/vmlinux.lds.S index af1d72020976..9e809b6908a1 100644 --- a/arch/arm64/kernel/vmlinux.lds.S +++ b/arch/arm64/kernel/vmlinux.lds.S @@ -197,6 +197,7 @@ SECTIONS IRQENTRY_TEXT SOFTIRQENTRY_TEXT ENTRY_TEXT + CPU_PRESERVED_TEXT TEXT_TEXT SCHED_TEXT LOCK_TEXT -- 2.55.0.1082.g2b9226bbc0-goog

