From 0567dd7e947d10a237405e4a9d965c57dd6b473e Mon Sep 17 00:00:00 2001 From: Bradley Morgan Date: Sun, 4 Oct 2026 05:04:50 +0000 Subject: tashaboot: serror resets, secondary release scrub, gic group truth An SError while the loader runs means the machine is already broken, handing the kernel a cpu that lost is worse than stopping. The handler reports the syndrome then drives the same reset domain PSCI SYSTEM_RESET does, with a park as the fallback when the reset request is ignored. Secondaries leave the pen in the manual's boot state now, interrupts masked, and CNTVOFF_EL2 zeroed at EL2 so every PE reads the same virtual counter. A loader cannot repair a per cpu counter offset below EL2, and the kernel has no way to repair it at all, whatever ran before could have left one. The gic group registers are deliberately untouched. The writes looked like firmware duty, but the group routing is the secure world's: a non-secure loader's IGROUPR writes are dropped on hardware implementing the security extension, and on the emulator here they accept the write and the timer per cpu interrupts stop reaching the kernel, the tick dies and the boot hangs past the console handoff. Group config belongs to the EL3 monitor, this loader runs without one, the comment says so at the register level. receipt: gic 8000000 off, smp brought up 1 node 4 cpus, run /init, busybox shell, two consecutive boots, the pen scrub exercised in the qemu spin table path. --- arch/arm64/kernel/start.S | 27 +++++++++++++++++++++++++++ 1 file changed, 27 insertions(+) (limited to 'arch/arm64/kernel') diff --git a/arch/arm64/kernel/start.S b/arch/arm64/kernel/start.S index 6ee9941..3cf58d2 100644 --- a/arch/arm64/kernel/start.S +++ b/arch/arm64/kernel/start.S @@ -222,6 +222,24 @@ park: wfe b 1b 2: + /* interrupts masked at release, the manual's boot state */ + msr daifset, #0xf + /* + * every PE must read the same virtual counter. whatever + * ran before this loader could have left a per cpu offset + * in the virtual counter view, the kernel has no way to + * repair that itself. CNTVOFF_EL2 is writable at EL2 and + * the write holds for the EL1 virtual timer the kernel + * runs on. below EL2 it is out of reach, the reset value + * is the best a lower EL can do. + */ + mrs x4, CurrentEL + lsr x4, x4, #2 + cmp x4, #2 + b.lt 3f + msr cntvoff_el2, xzr + isb +3: mov x0, xzr /* secondaries enter with x0-x3 zero */ mov x1, xzr mov x2, xzr @@ -396,6 +414,15 @@ exc_serr: mov x1, #0 mov x2, lr bl exc_report + /* + * an SError while this loader runs means the machine is + * broken. handing the kernel a cpu that already lost is + * worse than stopping: report, then drive the reset domain + * the same way PSCI SYSTEM_RESET does. the reset call does + * not return, the park below is the fallback if a reset + * domain ignores the request. + */ + bl tb_system_reset ldp x29, x30, [sp], #16 b park -- cgit v1.2.3