summaryrefslogtreecommitdiff
path: root/arch/arm64/kernel/start.S
diff options
context:
space:
mode:
Diffstat (limited to 'arch/arm64/kernel/start.S')
-rw-r--r--arch/arm64/kernel/start.S456
1 files changed, 456 insertions, 0 deletions
diff --git a/arch/arm64/kernel/start.S b/arch/arm64/kernel/start.S
new file mode 100644
index 0000000..2a5e2ae
--- /dev/null
+++ b/arch/arm64/kernel/start.S
@@ -0,0 +1,456 @@
+/* SPDX-License-Identifier: GPL-2.0+ */
+/*
+ * tashaboot arm64 entry. handles whatever EL the firmware left us in,
+ * EL3, EL2 or EL1, with the MMU either on or off, and arrives at a
+ * clean EL1 with the MMU off before calling C.
+ *
+ * the secondary cores park, spin table bringup is a later problem.
+ *
+ * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org>
+ */
+
+#include <asm/macro.h>
+
+.section .text.boot
+.globl _start
+_start:
+ /* code0: branch over the 64 byte Image header to reset */
+ b reset
+
+ .balign 8
+/*
+ * the arm64 Image header fields, per Documentation/arch/arm64/
+ * booting.rst: text_offset 0x08, image_size 0x10, flags 0x18,
+ * magic 0x38. code0 above branches over all of it. text_offset 0
+ * and image_size filled after link by tools/fillsize.py, the
+ * magic pins it as a proper Image so qemu -kernel enters at
+ * RAMBASE instead of guessing +0x80000.
+ */
+ .quad 0x0 /* text_offset, 0x08, filled below */
+ .quad 0x0 /* image_size, 0x10, filled below */
+ .quad 0x0 /* flags, 0x18: LE, 4k pages, unset */
+ .quad 0x0 /* reserved 0x20 */
+ .quad 0x0 /* reserved 0x28 */
+ .quad 0x0 /* reserved 0x30 */
+ .quad 0x644d5241 /* magic, 0x38: ARM\x64 */
+
+reset:
+ /* keep the dtb pointer before anything clobbers x0 */
+ mov x19, x0
+
+ /*
+ * park secondary cores, they have nothing to do yet. at
+ * EL3 they still get the monitor: a firmware call on any
+ * PE must land in a handler, a secondary with no EL3
+ * vectors traps into nothing.
+ */
+ mrs x0, mpidr_el1
+ and x0, x0, #0xff
+ cbnz x0, secondary_boot
+
+ mrs x0, CurrentEL
+ lsr x0, x0, #2
+ cmp x0, #3
+ b.eq from_el3
+ cmp x0, #2
+ b.eq from_el2
+ cmp x0, #1
+ b.eq mmu_check
+ b park
+
+secondary_boot:
+ mrs x0, CurrentEL
+ lsr x0, x0, #2
+ cmp x0, #3
+ b.ne park
+ /*
+ * the same security state as the primary: SCR_EL3.NS
+ * clear leaves a PE secure, and a secondary released
+ * into the kernel secure is the inconsistent mode boot
+ * the kernel warns about, its calls trap to EL3 as if
+ * they were firmware's own.
+ */
+ mrs x0, scr_el3
+ orr x0, x0, #1
+ msr scr_el3, x0
+ isb
+ bl tb_monitor_init
+ b park
+
+from_el3:
+ /*
+ * EL3 holds the security state, so the monitor lives here:
+ * vectors, its own stack, the SMC conduit. it is resident
+ * after this, the kernel's firmware calls trap into it.
+ */
+ bl tb_monitor_init
+
+ /* the kernel runs non-secure, drop to the EL2 it prefers */
+ mrs x0, scr_el3
+ orr x0, x0, #1 /* SCR_EL3.NS = 1, non-secure */
+ msr scr_el3, x0
+ isb
+
+ mov x0, #0x3c9 /* EL2h, DAIF masked */
+ msr spsr_el3, x0
+ adr x0, from_el2
+ msr elr_el3, x0
+ eret
+
+from_el2:
+ /*
+ * scrub the EL2 state and drop to EL1 for the C runtime. the
+ * semihosting hlt trap is an EL1 service on qemu, calling it
+ * from EL2 corrupts the return state. the kernel handoff goes
+ * back to EL2, booting.rst prefers it there, through the
+ * trampoline in boot.S.
+ */
+
+ /* EL1 will be aarch64 */
+ mov x0, #(1 << 31) /* HCR_EL2.RW = 1 */
+ msr hcr_el2, x0
+
+ /* let EL1 reach the counter, booting.rst demands it */
+ mrs x0, cnthctl_el2
+ orr x0, x0, #(3 << 0) /* EL1PCTEN | EL1PCEN */
+ msr cnthctl_el2, x0
+
+ /* no traps to EL2 behind EL1's back */
+ msr cptr_el2, xzr
+ msr hstr_el2, xzr
+ msr vpidr_el2, xzr
+
+ /* drop to EL1, SPSR EL1h with DAIF masked */
+ mov x0, #0x3c5
+ msr spsr_el2, x0
+ adr x0, mmu_check
+ msr elr_el2, x0
+ eret
+
+mmu_check:
+ /*
+ * whether the firmware left an MMU on: M bit, bit 0, of sctlr at
+ * the current EL. writing the register off would not fault, but
+ * the page tables it built are in its own memory, better to kill
+ * it here than trip over a stale mapping.
+ */
+ mrs x0, CurrentEL
+ lsr x0, x0, #2
+ cmp x0, #2
+ b.lt mmu_el1
+ mrs x0, sctlr_el2
+ tbz x0, #0, c_entry
+
+ mov x0, xzr
+ msr sctlr_el2, x0
+ isb
+ tlbi alle2
+ dsb sy
+ isb
+ b c_entry
+
+mmu_el1:
+ mrs x0, sctlr_el1
+ tbz x0, #0, c_entry
+
+ mov x0, xzr
+ msr sctlr_el1, x0
+ isb
+ ic iallu
+ dsb sy
+ tlbi vmalle1
+ dsb sy
+ isb
+
+c_entry:
+ /*
+ * program the counter frequency, the kernel reads CNTFRQ right
+ * away (booting.rst). qemu virt runs the system counter at
+ * 62.5 MHz. the register is RW only at the highest implemented EL.
+ */
+ mrs x0, CurrentEL
+ lsr x0, x0, #2
+ cmp x0, #2
+ b.lt 1f
+ ldr x0, =62500000
+ msr cntfrq_el0, x0
+ isb
+1:
+ /* our own vectors, so aborts print instead of vanishing */
+ adr x0, vectors
+ mrs x1, CurrentEL
+ lsr x1, x1, #2
+ cmp x1, #2
+ b.lt 2f
+ msr vbar_el2, x0
+ b 3f
+2:
+ msr vbar_el1, x0
+3:
+ isb
+
+ /* stack for the bootloader, its own region above the bss */
+ ldr x0, =__stack_top
+ mov sp, x0
+
+ /* export the spin gate array address for the dtb patcher */
+ adr x0, tb_spin_gates
+ adrp x1, tb_spin_gates_ptr
+ str x0, [x1, #:lo12:tb_spin_gates_ptr]
+
+ /* clear bss */
+ ldr x0, =__bss_start
+ ldr x1, =__bss_end
+1: cmp x0, x1
+ b.hs 2f
+ str xzr, [x0], #8
+ b 1b
+2:
+
+ /* FP/SIMD access, some kernels assume it is on */
+ mov x0, #(3 << 20)
+ msr cpacr_el1, x0
+ isb
+
+ /* dtb pointer into C arg 0 */
+ mov x0, x19
+ bl tashaboot_main
+
+ /* if main returns there is nothing sensible to do */
+/*
+ * the spin table pen, the Wait For Event mechanism from the manual
+ * (B2-144, D1-2255). each secondary watches its own gate, the
+ * cpu-release-addr the dtb names. WFE clears the event register and
+ * sleeps, the kernel writes the secondary entry to the gate, makes
+ * it visible, then SEV sets the event register on every PE. the load
+ * recheck after each wake covers a release that lands between the
+ * load and the WFE. entered with MMU and caches off, left the same.
+ */
+.globl park_ret
+park_ret:
+park:
+ adr x0, tb_spin_gates
+ mrs x1, mpidr_el1
+ and x1, x1, #0xff /* affinity 0, the core number */
+ add x0, x0, x1, lsl #3 /* gate = gates + core * 8 */
+
+ /* diagnostic: stamp arrival, primary prints it later */
+ adr x3, tb_pen_stamps
+ strb w1, [x3, x1]
+ sevl
+ wfe
+ sevl
+ wfe
+
+1:
+ ldr x2, [x0]
+ cbnz x2, 2f
+ wfe
+ b 1b
+2:
+ /* interrupts masked at release, the manual's boot state */
+ msr daifset, #0xf
+ /*
+ * every PE must read the same virtual counter. whatever
+ * ran before this loader could have left a per cpu offset
+ * in the virtual counter view, the kernel has no way to
+ * repair that itself. CNTVOFF_EL2 is writable at EL2 and
+ * the write holds for the EL1 virtual timer the kernel
+ * runs on. below EL2 it is out of reach, the reset value
+ * is the best a lower EL can do.
+ */
+ mrs x4, CurrentEL
+ lsr x4, x4, #2
+ cmp x4, #2
+ b.lt 3f
+ msr cntvoff_el2, xzr
+ isb
+3:
+ mov x0, xzr /* secondaries enter with x0-x3 zero */
+ mov x1, xzr
+ mov x2, xzr
+ mov x3, xzr
+ dsb sy
+ isb
+ br x2
+
+/*
+ * exception vectors, the armv8 layout: 16 slots, 128 bytes each, in
+ * the order the manual fixes. taken from EL1h the interesting slots
+ * are 0x200 sync and 0x380 SError, irq and fiq just park, the
+ * bootloader never enables interrupts on purpose.
+ */
+ .balign 2048
+vectors:
+ /* 0x000: current EL, SP_EL0 */
+ .align 7
+ b exc_sync
+ .align 7
+ b exc_park_irq
+ .align 7
+ b exc_park_irq
+ .align 7
+ b exc_serr
+
+ /* 0x200: current EL, SP_ELx */
+ .align 7
+ b exc_sync
+ .align 7
+ b exc_park_irq
+ .align 7
+ b exc_park_irq
+ .align 7
+ b exc_serr
+
+ /* 0x400: lower EL, AArch64 */
+ .align 7
+ b exc_sync
+ .align 7
+ b exc_park_irq
+ .align 7
+ b exc_park_irq
+ .align 7
+ b exc_serr
+
+ /* 0x600: lower EL, AArch32 */
+ .align 7
+ b exc_sync
+ .align 7
+ b exc_park_irq
+ .align 7
+ b exc_park_irq
+ .align 7
+ b exc_serr
+
+.pushsection .data.tb_spin, "aw"
+.align 3
+.globl tb_spin_gates
+tb_spin_gates:
+ .quad 0, 0, 0, 0, 0, 0, 0, 0
+.globl tb_spin_gates_ptr
+tb_spin_gates_ptr:
+ .quad 0
+.globl tb_pen_stamps
+tb_pen_stamps:
+ .byte 0, 0, 0, 0, 0, 0, 0, 0
+.popsection
+
+exc_sync:
+ stp x29, x30, [sp, #-16]!
+ mov x29, sp
+ mrs x3, CurrentEL
+ lsr x3, x3, #2
+ cmp x3, #2
+ b.lt 1f
+ mrs x0, esr_el2
+ mrs x2, elr_el2
+ lsr x1, x0, #26
+ cmp x1, #0x16 /* HVC from lower EL */
+ b.eq hvc_from_el1
+ mrs x1, far_el2
+ b 2f
+1:
+ mrs x0, esr_el1
+ mrs x1, far_el1
+2:
+ /* x2 = the faulting PC when it is the sync path */
+ mrs x4, CurrentEL
+ lsr x4, x4, #2
+ cmp x4, #2
+ b.lt 3f
+ mrs x2, elr_el2
+ b 4f
+3:
+ mrs x2, elr_el1
+4:
+ bl exc_report
+ ldp x29, x30, [sp], #16
+ b park
+
+/*
+ * HVC from EL1, the PSCI conduit. x0-x3 are the PSCI args in the
+ * caller registers, dispatch and return in x0. ELR_EL2 is already
+ * the resume point, eret takes it back.
+ */
+hvc_from_el1:
+ /*
+ * the lower EL sync slot. three arrivals share it: PSCI hvc
+ * from the kernel (EC 0x16, PSCI id in x0), our own boot
+ * handoff (hvc with the payload entry in x8), and semihosting
+ * hlt #0xf000 from the EL1 C runtime (EC 0x14). qemu only
+ * answers the hlt when it executes at EL2, so the handler
+ * replays the trap at EL2 and erets home with the result.
+ */
+ mrs x1, esr_el2
+ lsr x1, x1, #26 /* EC */
+ cmp x1, #0x14 /* HLT from lower EL, semihosting */
+ b.eq smh_replay
+
+ /*
+ * the hvc arrives with either a PSCI function id in x0 (the
+ * kernel calling) or the boot handoff staging the payload
+ * entry in x8 and the dtb in x0. PSCI ids have the 0x84/0xc4
+ * prefix, a dtb pointer never does.
+ */
+ lsr x1, x0, #24
+ cmp x1, #0x84
+ b.eq psci_call
+ cmp x1, #0xc4
+ b.eq psci_call
+
+ /* the boot handoff: ELR_EL2 = entry, eret to the payload */
+ msr elr_el2, x8
+ eret
+
+smh_replay:
+ /*
+ * x0 holds the semihosting syscall number, x1 the parameter
+ * block, both live in the caller's registers. replay the hlt
+ * here at EL2 where qemu answers it, then eret back.
+ */
+ hlt #0xf000
+ eret
+
+psci_call:
+ stp x4, x5, [sp, #-16]!
+ stp x6, x7, [sp, #-16]!
+ stp x29, x30, [sp, #-16]!
+ mov x29, sp
+
+ bl tb_psci_dispatch
+
+ ldp x29, x30, [sp], #16
+ ldp x6, x7, [sp], #16
+ ldp x4, x5, [sp], #16
+ ldp x29, x30, [sp], #16
+ eret
+
+exc_serr:
+ stp x29, x30, [sp, #-16]!
+ mov x29, sp
+ mrs x3, CurrentEL
+ lsr x3, x3, #2
+ cmp x3, #2
+ b.lt 1f
+ mrs x0, esr_el2
+ b 2f
+1:
+ mrs x0, esr_el1
+2:
+ mov x1, #0
+ mov x2, lr
+ bl exc_report
+ /*
+ * an SError while this loader runs means the machine is
+ * broken. handing the kernel a cpu that already lost is
+ * worse than stopping: report, then drive the reset domain
+ * the same way PSCI SYSTEM_RESET does. the reset call does
+ * not return, the park below is the fallback if a reset
+ * domain ignores the request.
+ */
+ bl tb_system_reset
+ ldp x29, x30, [sp], #16
+ b park
+
+exc_park_irq:
+ b park