diff options
Diffstat (limited to 'arch/arm64/kernel/start.S')
| -rw-r--r-- | arch/arm64/kernel/start.S | 456 |
1 files changed, 456 insertions, 0 deletions
diff --git a/arch/arm64/kernel/start.S b/arch/arm64/kernel/start.S new file mode 100644 index 0000000..2a5e2ae --- /dev/null +++ b/arch/arm64/kernel/start.S @@ -0,0 +1,456 @@ +/* SPDX-License-Identifier: GPL-2.0+ */ +/* + * tashaboot arm64 entry. handles whatever EL the firmware left us in, + * EL3, EL2 or EL1, with the MMU either on or off, and arrives at a + * clean EL1 with the MMU off before calling C. + * + * the secondary cores park, spin table bringup is a later problem. + * + * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org> + */ + +#include <asm/macro.h> + +.section .text.boot +.globl _start +_start: + /* code0: branch over the 64 byte Image header to reset */ + b reset + + .balign 8 +/* + * the arm64 Image header fields, per Documentation/arch/arm64/ + * booting.rst: text_offset 0x08, image_size 0x10, flags 0x18, + * magic 0x38. code0 above branches over all of it. text_offset 0 + * and image_size filled after link by tools/fillsize.py, the + * magic pins it as a proper Image so qemu -kernel enters at + * RAMBASE instead of guessing +0x80000. + */ + .quad 0x0 /* text_offset, 0x08, filled below */ + .quad 0x0 /* image_size, 0x10, filled below */ + .quad 0x0 /* flags, 0x18: LE, 4k pages, unset */ + .quad 0x0 /* reserved 0x20 */ + .quad 0x0 /* reserved 0x28 */ + .quad 0x0 /* reserved 0x30 */ + .quad 0x644d5241 /* magic, 0x38: ARM\x64 */ + +reset: + /* keep the dtb pointer before anything clobbers x0 */ + mov x19, x0 + + /* + * park secondary cores, they have nothing to do yet. at + * EL3 they still get the monitor: a firmware call on any + * PE must land in a handler, a secondary with no EL3 + * vectors traps into nothing. + */ + mrs x0, mpidr_el1 + and x0, x0, #0xff + cbnz x0, secondary_boot + + mrs x0, CurrentEL + lsr x0, x0, #2 + cmp x0, #3 + b.eq from_el3 + cmp x0, #2 + b.eq from_el2 + cmp x0, #1 + b.eq mmu_check + b park + +secondary_boot: + mrs x0, CurrentEL + lsr x0, x0, #2 + cmp x0, #3 + b.ne park + /* + * the same security state as the primary: SCR_EL3.NS + * clear leaves a PE secure, and a secondary released + * into the kernel secure is the inconsistent mode boot + * the kernel warns about, its calls trap to EL3 as if + * they were firmware's own. + */ + mrs x0, scr_el3 + orr x0, x0, #1 + msr scr_el3, x0 + isb + bl tb_monitor_init + b park + +from_el3: + /* + * EL3 holds the security state, so the monitor lives here: + * vectors, its own stack, the SMC conduit. it is resident + * after this, the kernel's firmware calls trap into it. + */ + bl tb_monitor_init + + /* the kernel runs non-secure, drop to the EL2 it prefers */ + mrs x0, scr_el3 + orr x0, x0, #1 /* SCR_EL3.NS = 1, non-secure */ + msr scr_el3, x0 + isb + + mov x0, #0x3c9 /* EL2h, DAIF masked */ + msr spsr_el3, x0 + adr x0, from_el2 + msr elr_el3, x0 + eret + +from_el2: + /* + * scrub the EL2 state and drop to EL1 for the C runtime. the + * semihosting hlt trap is an EL1 service on qemu, calling it + * from EL2 corrupts the return state. the kernel handoff goes + * back to EL2, booting.rst prefers it there, through the + * trampoline in boot.S. + */ + + /* EL1 will be aarch64 */ + mov x0, #(1 << 31) /* HCR_EL2.RW = 1 */ + msr hcr_el2, x0 + + /* let EL1 reach the counter, booting.rst demands it */ + mrs x0, cnthctl_el2 + orr x0, x0, #(3 << 0) /* EL1PCTEN | EL1PCEN */ + msr cnthctl_el2, x0 + + /* no traps to EL2 behind EL1's back */ + msr cptr_el2, xzr + msr hstr_el2, xzr + msr vpidr_el2, xzr + + /* drop to EL1, SPSR EL1h with DAIF masked */ + mov x0, #0x3c5 + msr spsr_el2, x0 + adr x0, mmu_check + msr elr_el2, x0 + eret + +mmu_check: + /* + * whether the firmware left an MMU on: M bit, bit 0, of sctlr at + * the current EL. writing the register off would not fault, but + * the page tables it built are in its own memory, better to kill + * it here than trip over a stale mapping. + */ + mrs x0, CurrentEL + lsr x0, x0, #2 + cmp x0, #2 + b.lt mmu_el1 + mrs x0, sctlr_el2 + tbz x0, #0, c_entry + + mov x0, xzr + msr sctlr_el2, x0 + isb + tlbi alle2 + dsb sy + isb + b c_entry + +mmu_el1: + mrs x0, sctlr_el1 + tbz x0, #0, c_entry + + mov x0, xzr + msr sctlr_el1, x0 + isb + ic iallu + dsb sy + tlbi vmalle1 + dsb sy + isb + +c_entry: + /* + * program the counter frequency, the kernel reads CNTFRQ right + * away (booting.rst). qemu virt runs the system counter at + * 62.5 MHz. the register is RW only at the highest implemented EL. + */ + mrs x0, CurrentEL + lsr x0, x0, #2 + cmp x0, #2 + b.lt 1f + ldr x0, =62500000 + msr cntfrq_el0, x0 + isb +1: + /* our own vectors, so aborts print instead of vanishing */ + adr x0, vectors + mrs x1, CurrentEL + lsr x1, x1, #2 + cmp x1, #2 + b.lt 2f + msr vbar_el2, x0 + b 3f +2: + msr vbar_el1, x0 +3: + isb + + /* stack for the bootloader, its own region above the bss */ + ldr x0, =__stack_top + mov sp, x0 + + /* export the spin gate array address for the dtb patcher */ + adr x0, tb_spin_gates + adrp x1, tb_spin_gates_ptr + str x0, [x1, #:lo12:tb_spin_gates_ptr] + + /* clear bss */ + ldr x0, =__bss_start + ldr x1, =__bss_end +1: cmp x0, x1 + b.hs 2f + str xzr, [x0], #8 + b 1b +2: + + /* FP/SIMD access, some kernels assume it is on */ + mov x0, #(3 << 20) + msr cpacr_el1, x0 + isb + + /* dtb pointer into C arg 0 */ + mov x0, x19 + bl tashaboot_main + + /* if main returns there is nothing sensible to do */ +/* + * the spin table pen, the Wait For Event mechanism from the manual + * (B2-144, D1-2255). each secondary watches its own gate, the + * cpu-release-addr the dtb names. WFE clears the event register and + * sleeps, the kernel writes the secondary entry to the gate, makes + * it visible, then SEV sets the event register on every PE. the load + * recheck after each wake covers a release that lands between the + * load and the WFE. entered with MMU and caches off, left the same. + */ +.globl park_ret +park_ret: +park: + adr x0, tb_spin_gates + mrs x1, mpidr_el1 + and x1, x1, #0xff /* affinity 0, the core number */ + add x0, x0, x1, lsl #3 /* gate = gates + core * 8 */ + + /* diagnostic: stamp arrival, primary prints it later */ + adr x3, tb_pen_stamps + strb w1, [x3, x1] + sevl + wfe + sevl + wfe + +1: + ldr x2, [x0] + cbnz x2, 2f + wfe + b 1b +2: + /* interrupts masked at release, the manual's boot state */ + msr daifset, #0xf + /* + * every PE must read the same virtual counter. whatever + * ran before this loader could have left a per cpu offset + * in the virtual counter view, the kernel has no way to + * repair that itself. CNTVOFF_EL2 is writable at EL2 and + * the write holds for the EL1 virtual timer the kernel + * runs on. below EL2 it is out of reach, the reset value + * is the best a lower EL can do. + */ + mrs x4, CurrentEL + lsr x4, x4, #2 + cmp x4, #2 + b.lt 3f + msr cntvoff_el2, xzr + isb +3: + mov x0, xzr /* secondaries enter with x0-x3 zero */ + mov x1, xzr + mov x2, xzr + mov x3, xzr + dsb sy + isb + br x2 + +/* + * exception vectors, the armv8 layout: 16 slots, 128 bytes each, in + * the order the manual fixes. taken from EL1h the interesting slots + * are 0x200 sync and 0x380 SError, irq and fiq just park, the + * bootloader never enables interrupts on purpose. + */ + .balign 2048 +vectors: + /* 0x000: current EL, SP_EL0 */ + .align 7 + b exc_sync + .align 7 + b exc_park_irq + .align 7 + b exc_park_irq + .align 7 + b exc_serr + + /* 0x200: current EL, SP_ELx */ + .align 7 + b exc_sync + .align 7 + b exc_park_irq + .align 7 + b exc_park_irq + .align 7 + b exc_serr + + /* 0x400: lower EL, AArch64 */ + .align 7 + b exc_sync + .align 7 + b exc_park_irq + .align 7 + b exc_park_irq + .align 7 + b exc_serr + + /* 0x600: lower EL, AArch32 */ + .align 7 + b exc_sync + .align 7 + b exc_park_irq + .align 7 + b exc_park_irq + .align 7 + b exc_serr + +.pushsection .data.tb_spin, "aw" +.align 3 +.globl tb_spin_gates +tb_spin_gates: + .quad 0, 0, 0, 0, 0, 0, 0, 0 +.globl tb_spin_gates_ptr +tb_spin_gates_ptr: + .quad 0 +.globl tb_pen_stamps +tb_pen_stamps: + .byte 0, 0, 0, 0, 0, 0, 0, 0 +.popsection + +exc_sync: + stp x29, x30, [sp, #-16]! + mov x29, sp + mrs x3, CurrentEL + lsr x3, x3, #2 + cmp x3, #2 + b.lt 1f + mrs x0, esr_el2 + mrs x2, elr_el2 + lsr x1, x0, #26 + cmp x1, #0x16 /* HVC from lower EL */ + b.eq hvc_from_el1 + mrs x1, far_el2 + b 2f +1: + mrs x0, esr_el1 + mrs x1, far_el1 +2: + /* x2 = the faulting PC when it is the sync path */ + mrs x4, CurrentEL + lsr x4, x4, #2 + cmp x4, #2 + b.lt 3f + mrs x2, elr_el2 + b 4f +3: + mrs x2, elr_el1 +4: + bl exc_report + ldp x29, x30, [sp], #16 + b park + +/* + * HVC from EL1, the PSCI conduit. x0-x3 are the PSCI args in the + * caller registers, dispatch and return in x0. ELR_EL2 is already + * the resume point, eret takes it back. + */ +hvc_from_el1: + /* + * the lower EL sync slot. three arrivals share it: PSCI hvc + * from the kernel (EC 0x16, PSCI id in x0), our own boot + * handoff (hvc with the payload entry in x8), and semihosting + * hlt #0xf000 from the EL1 C runtime (EC 0x14). qemu only + * answers the hlt when it executes at EL2, so the handler + * replays the trap at EL2 and erets home with the result. + */ + mrs x1, esr_el2 + lsr x1, x1, #26 /* EC */ + cmp x1, #0x14 /* HLT from lower EL, semihosting */ + b.eq smh_replay + + /* + * the hvc arrives with either a PSCI function id in x0 (the + * kernel calling) or the boot handoff staging the payload + * entry in x8 and the dtb in x0. PSCI ids have the 0x84/0xc4 + * prefix, a dtb pointer never does. + */ + lsr x1, x0, #24 + cmp x1, #0x84 + b.eq psci_call + cmp x1, #0xc4 + b.eq psci_call + + /* the boot handoff: ELR_EL2 = entry, eret to the payload */ + msr elr_el2, x8 + eret + +smh_replay: + /* + * x0 holds the semihosting syscall number, x1 the parameter + * block, both live in the caller's registers. replay the hlt + * here at EL2 where qemu answers it, then eret back. + */ + hlt #0xf000 + eret + +psci_call: + stp x4, x5, [sp, #-16]! + stp x6, x7, [sp, #-16]! + stp x29, x30, [sp, #-16]! + mov x29, sp + + bl tb_psci_dispatch + + ldp x29, x30, [sp], #16 + ldp x6, x7, [sp], #16 + ldp x4, x5, [sp], #16 + ldp x29, x30, [sp], #16 + eret + +exc_serr: + stp x29, x30, [sp, #-16]! + mov x29, sp + mrs x3, CurrentEL + lsr x3, x3, #2 + cmp x3, #2 + b.lt 1f + mrs x0, esr_el2 + b 2f +1: + mrs x0, esr_el1 +2: + mov x1, #0 + mov x2, lr + bl exc_report + /* + * an SError while this loader runs means the machine is + * broken. handing the kernel a cpu that already lost is + * worse than stopping: report, then drive the reset domain + * the same way PSCI SYSTEM_RESET does. the reset call does + * not return, the park below is the fallback if a reset + * domain ignores the request. + */ + bl tb_system_reset + ldp x29, x30, [sp], #16 + b park + +exc_park_irq: + b park |
