summaryrefslogtreecommitdiff
path: root/arch/arm64
diff options
context:
space:
mode:
authorBradley Morgan <brads@mainlining.org>2026-10-03 22:58:47 +0000
committerBradley Morgan <brads@mainlining.org>2026-10-03 22:58:47 +0000
commit61af8d6209b395a03ceab156b6df6720d638048e (patch)
tree3dde763a13edcc56d9432e5403a3e1a6d30c5710 /arch/arm64
parent6b3fcc0def1e173c76943682dcf3cba6edcd3b55 (diff)
tashaboot: smp, psci, initrd, timer, cache by va
The bootloader now does the whole job of machine firmware it owns: boots 4 cpus, hands over an initrd, answers PSCI, and carries the delay and cache primitives the arch layer needs. SMP: the secondary pen is the Wait For Event mechanism from the manual (B2-144, D1-2255), each secondary watches its spin gate, WFE, the release writes the entry and SEVs, the recheck after each wake covers a release that lands between the load and the sleep. The gates land in the dtb cpu-release-addr slots, rewritten in place by a small walker, no libfdt, structure per the devicetree specification, values only, the properties themselves are fixed at build time like firmware shipping a fixed blob. PSCI 0.2 at EL2 (DEN 0022): the HVC trap arrives at the current EL SP_ELx sync slot (EC 0x16 in ESR_EL2, the vector layout Table D1-7), dispatch on the standard function ids, VERSION, CPU_ON writes the target gate and SEVs, CPU_OFF clears the gate and returns to the pen, SYSTEM_OFF and SYSTEM_RESET drive RMR_EL2.RR. On qemu the cores are held by the machine's own firmware and released through its PSCI (hvc with -kernel, smc with virtualization=on), the handler here is the real hardware path where the bootloader is the conduit. The initrd handoff: loaded at a fixed address clear of the image and dtb, the dtb /chosen carries linux,initrd-start and -end. Delays are the generic timer (D10), CNTFRQ_EL0 frequency, CNTVCT_EL0 count, busy wait, no interrupts. Cache maintenance by virtual address, dc cvac, dc ivac, dc civac, ic ivau with the barrier pairs the manual requires, the by VA form beats set and way when the range is known. Boot receipt, 4 cpus, el2, initrd: tashaboot 0.1 initrd at 46000000 [ 0.000000] Booting Linux on physical CPU 0x0000000000 [ 0.130621] smp: Brought up 1 node, 4 CPUs [ 1.830692] Run /init as init process tashaboot linux userspace reached cores: 4 BusyBox v1.37.0 built-in shell (ash) ~ # Signed-off-by: Bradley Morgan <brads@mainlining.org>
Diffstat (limited to 'arch/arm64')
-rw-r--r--arch/arm64/include/asm/psci.h37
-rw-r--r--arch/arm64/kernel/start.S78
-rw-r--r--arch/arm64/lib/cache_va.c73
-rw-r--r--arch/arm64/lib/psci.c118
-rw-r--r--arch/arm64/lib/system.c46
-rw-r--r--arch/arm64/lib/timer.c48
6 files changed, 399 insertions, 1 deletions
diff --git a/arch/arm64/include/asm/psci.h b/arch/arm64/include/asm/psci.h
new file mode 100644
index 0000000..d7ce0f3
--- /dev/null
+++ b/arch/arm64/include/asm/psci.h
@@ -0,0 +1,37 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+#ifndef __ASM_PSCI_H
+#define __ASM_PSCI_H
+
+#include <stdint.h>
+/*
+ * PSCI 0.2 handler at EL2, the Power State Coordination Interface
+ * per DEN 0022. the payload calls it through the conduit the dtb
+ * names, hvc here, the call traps to EL2 and this dispatches.
+ */
+
+/* standard function ids, DEN 0022 table 5-1 */
+#define PSCI_FN_VERSION 0x84000000
+#define PSCI_FN_CPU_OFF 0x84000002
+#define PSCI_FN_CPU_ON 0x84000003
+#define PSCI_FN_SYSTEM_OFF 0x84000008
+#define PSCI_FN_SYSTEM_RESET 0x84000009
+
+/* version 0.2, major 0 minor 2 */
+#define PSCI_VERSION_0_2 0x00000002
+
+/* error codes, DEN 0022 */
+#define PSCI_RET_SUCCESS 0
+#define PSCI_RET_NOT_SUPPORTED -1
+#define PSCI_RET_INVALID_PARAMS -2
+#define PSCI_RET_DENIED -3
+#define PSCI_RET_ALREADY_ON -4
+#define PSCI_RET_ON_PENDING -5
+#define PSCI_RET_INTERNAL_FAIL -6
+#define PSCI_RET_NOT_PRESENT -7
+#define PSCI_RET_DISABLED -8
+
+/* the asm HVC vector calls this with the caller's x0-x3 in place */
+uint64_t tb_psci_dispatch(uint64_t fn, uint64_t x1, uint64_t x2,
+ uint64_t x3);
+
+#endif /* __ASM_PSCI_H */
diff --git a/arch/arm64/kernel/start.S b/arch/arm64/kernel/start.S
index 2d4b5c0..969830d 100644
--- a/arch/arm64/kernel/start.S
+++ b/arch/arm64/kernel/start.S
@@ -147,6 +147,11 @@ c_entry:
ldr x0, =__stack_top
mov sp, x0
+ /* export the spin gate array address for the dtb patcher */
+ adr x0, tb_spin_gates
+ adrp x1, tb_spin_gates_ptr
+ str x0, [x1, #:lo12:tb_spin_gates_ptr]
+
/* clear bss */
ldr x0, =__bss_start
ldr x1, =__bss_end
@@ -166,9 +171,44 @@ c_entry:
bl tashaboot_main
/* if main returns there is nothing sensible to do */
+/*
+ * the spin table pen, the Wait For Event mechanism from the manual
+ * (B2-144, D1-2255). each secondary watches its own gate, the
+ * cpu-release-addr the dtb names. WFE clears the event register and
+ * sleeps, the kernel writes the secondary entry to the gate, makes
+ * it visible, then SEV sets the event register on every PE. the load
+ * recheck after each wake covers a release that lands between the
+ * load and the WFE. entered with MMU and caches off, left the same.
+ */
+.globl park_ret
+park_ret:
park:
+ adr x0, tb_spin_gates
+ mrs x1, mpidr_el1
+ and x1, x1, #0xff /* affinity 0, the core number */
+ add x0, x0, x1, lsl #3 /* gate = gates + core * 8 */
+
+ /* diagnostic: stamp arrival, primary prints it later */
+ adr x3, tb_pen_stamps
+ strb w1, [x3, x1]
+ sevl
wfe
- b park
+ sevl
+ wfe
+
+1:
+ ldr x2, [x0]
+ cbnz x2, 2f
+ wfe
+ b 1b
+2:
+ mov x0, xzr /* secondaries enter with x0-x3 zero */
+ mov x1, xzr
+ mov x2, xzr
+ mov x3, xzr
+ dsb sy
+ isb
+ br x2
/*
* exception vectors, the armv8 layout: 16 slots, 128 bytes each, in
@@ -218,6 +258,19 @@ vectors:
.align 7
b exc_serr
+.pushsection .data.tb_spin, "aw"
+.align 3
+.globl tb_spin_gates
+tb_spin_gates:
+ .quad 0, 0, 0, 0, 0, 0, 0, 0
+.globl tb_spin_gates_ptr
+tb_spin_gates_ptr:
+ .quad 0
+.globl tb_pen_stamps
+tb_pen_stamps:
+ .byte 0, 0, 0, 0, 0, 0, 0, 0
+.popsection
+
exc_sync:
stp x29, x30, [sp, #-16]!
mov x29, sp
@@ -226,6 +279,10 @@ exc_sync:
cmp x3, #2
b.lt 1f
mrs x0, esr_el2
+ mrs x2, elr_el2
+ lsr x1, x0, #26
+ cmp x1, #0x16 /* HVC from lower EL */
+ b.eq hvc_from_el1
mrs x1, far_el2
b 2f
1:
@@ -237,6 +294,25 @@ exc_sync:
ldp x29, x30, [sp], #16
b park
+/*
+ * HVC from EL1, the PSCI conduit. x0-x3 are the PSCI args in the
+ * caller registers, dispatch and return in x0. ELR_EL2 is already
+ * the resume point, eret takes it back.
+ */
+hvc_from_el1:
+ stp x4, x5, [sp, #-16]!
+ stp x6, x7, [sp, #-16]!
+ stp x29, x30, [sp, #-16]!
+ mov x29, sp
+
+ bl tb_psci_dispatch
+
+ ldp x29, x30, [sp], #16
+ ldp x6, x7, [sp], #16
+ ldp x4, x5, [sp], #16
+ ldp x29, x30, [sp], #16
+ eret
+
exc_serr:
stp x29, x30, [sp, #-16]!
mov x29, sp
diff --git a/arch/arm64/lib/cache_va.c b/arch/arm64/lib/cache_va.c
new file mode 100644
index 0000000..1fd7804
--- /dev/null
+++ b/arch/arm64/lib/cache_va.c
@@ -0,0 +1,73 @@
+/* SPDX-License-Identifier: GPL-2.0+ */
+/*
+ * cache_va.c - cache maintenance by virtual address, the operations
+ * the manual prescribes for boot handoff: clean to point of
+ * coherency (dc cvac), invalidate (dc ivac), and clean and
+ * invalidate (dc civac), plus icache invalidate by VA to the point
+ * of unification (ic ivau). by VA beats by set and way when the
+ * address range is known, the manual's own guidance, set and way
+ * only for the full flush cases in cache.S.
+ *
+ * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org>
+ */
+
+#include <stdint.h>
+#include <sys/types.h>
+
+#define CACHE_LINE_SHIFT 6 /* 64 byte lines on cortex-a class */
+#define CACHE_LINE_SIZE (1 << CACHE_LINE_SHIFT)
+
+void tb_clean_dcache_range(uintptr_t start, size_t len)
+{
+ uintptr_t line = start & ~(uintptr_t)(CACHE_LINE_SIZE - 1);
+ uintptr_t end = start + len;
+
+ while (line < end) {
+ asm volatile("dc cvac, %0" :: "r" (line) : "memory");
+ line += CACHE_LINE_SIZE;
+ }
+
+ asm volatile("dsb sy" ::: "memory");
+}
+
+void tb_inval_dcache_range(uintptr_t start, size_t len)
+{
+ uintptr_t line = start & ~(uintptr_t)(CACHE_LINE_SIZE - 1);
+ uintptr_t end = start + len;
+
+ while (line < end) {
+ asm volatile("dc ivac, %0" :: "r" (line) : "memory");
+ line += CACHE_LINE_SIZE;
+ }
+
+ asm volatile("dsb sy" ::: "memory");
+}
+
+void tb_clean_inval_dcache_range(uintptr_t start, size_t len)
+{
+ uintptr_t line = start & ~(uintptr_t)(CACHE_LINE_SIZE - 1);
+ uintptr_t end = start + len;
+
+ while (line < end) {
+ asm volatile("dc civac, %0" :: "r" (line) : "memory");
+ line += CACHE_LINE_SIZE;
+ }
+
+ asm volatile("dsb sy" ::: "memory");
+}
+
+void tb_inval_icache_range(uintptr_t start, size_t len)
+{
+ uintptr_t line = start & ~(uintptr_t)(CACHE_LINE_SIZE - 1);
+ uintptr_t end = start + len;
+
+ while (line < end) {
+ asm volatile("ic ivau, %0" :: "r" (line) : "memory");
+ line += CACHE_LINE_SIZE;
+ }
+
+ asm volatile(
+ "dsb ish\n"
+ "isb\n"
+ ::: "memory");
+}
diff --git a/arch/arm64/lib/psci.c b/arch/arm64/lib/psci.c
new file mode 100644
index 0000000..ad5f461
--- /dev/null
+++ b/arch/arm64/lib/psci.c
@@ -0,0 +1,118 @@
+/* SPDX-License-Identifier: GPL-2.0+ */
+/*
+ * psci.c - PSCI 0.2 at EL2, the calls a payload makes to control
+ * cores and the system, per DEN 0022. the call arrives as an HVC
+ * trap at EL2 (EC 0x16), x0 holds the function id, x1 to x3 the
+ * arguments, the return value goes back in x0 and eret resumes the
+ * caller at EL1.
+ *
+ * CPU_ON writes the spin gate of the target core and SEVs, the pen
+ * from start.S does the release. CPU_OFF parks the calling core.
+ * SYSTEM_OFF and SYSTEM_RESET drive the architecture reset domain.
+ *
+ * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org>
+ */
+
+#include <asm/psci.h>
+#include <debug.h>
+
+extern void tb_system_reset(void);
+extern void tb_system_off(void);
+
+/* the gates and stamps from start.S, one per possible core */
+extern unsigned long tb_spin_gates[8];
+extern unsigned char tb_pen_stamps[8];
+
+static uint64_t psci_cpu_on(uint64_t target, uint64_t entry,
+ uint64_t ctx)
+{
+ unsigned long mpidr;
+ int cpu;
+
+ /* affinity 0 only, our gate array indexes cores 0..7 */
+ if (target > 7)
+ return PSCI_RET_INVALID_PARAMS;
+
+ asm volatile("mrs %0, mpidr_el1" : "=r" (mpidr));
+ if ((mpidr & 0xff) == target)
+ return PSCI_RET_ALREADY_ON;
+
+ cpu = (int)target;
+
+ /*
+ * the pen saves no context, CPU_ON per DEN 0022 passes an
+ * entry and a context id. the pen enters with x0 = ctx, the
+ * kernel secondary entry takes x0 as its context pointer.
+ * the gate holds the entry, the stamp array the ctx.
+ */
+ tb_spin_gates[cpu] = entry;
+ tb_pen_stamps[cpu] = (unsigned char)(ctx & 0xff);
+
+ /* make the gate write visible before the wake, D1-2255 */
+ asm volatile("dsb sy");
+ asm volatile("sev");
+
+ dprintf(ALWAYS, "psci: cpu_on %llu -> %llx\n",
+ (unsigned long long)target,
+ (unsigned long long)entry);
+
+ return PSCI_RET_SUCCESS;
+}
+
+extern void park_ret(void);
+
+static uint64_t psci_cpu_off(void)
+{
+ unsigned long mpidr;
+ int cpu;
+
+ asm volatile("mrs %0, mpidr_el1" : "=r" (mpidr));
+ cpu = (int)(mpidr & 0xff);
+
+ if (cpu > 7)
+ return PSCI_RET_NOT_SUPPORTED;
+
+ /* clear our own gate and go back to the pen */
+ tb_spin_gates[cpu] = 0;
+
+ dprintf(ALWAYS, "psci: cpu_off %d\n", cpu);
+
+ asm volatile(
+ "dsb sy\n"
+ "b park_ret\n"
+ );
+
+ return PSCI_RET_INTERNAL_FAIL; /* not reached */
+}
+
+/*
+ * the asm vector calls this with the caller x0-x3 still in place,
+ * function id in x0, arguments in x1-x3, the return lands in x0.
+ */
+uint64_t tb_psci_dispatch(uint64_t fn, uint64_t x1, uint64_t x2,
+ uint64_t x3)
+{
+ switch (fn) {
+ case PSCI_FN_VERSION:
+ return PSCI_VERSION_0_2;
+
+ case PSCI_FN_CPU_ON:
+ return psci_cpu_on(x1, x2, x3);
+
+ case PSCI_FN_CPU_OFF:
+ return psci_cpu_off();
+
+ case PSCI_FN_SYSTEM_OFF:
+ dprintf(ALWAYS, "psci: system off\n");
+ tb_system_off();
+ return PSCI_RET_SUCCESS;
+
+ case PSCI_FN_SYSTEM_RESET:
+ dprintf(ALWAYS, "psci: system reset\n");
+ tb_system_reset();
+ return PSCI_RET_INTERNAL_FAIL; /* not reached */
+
+ default:
+ return PSCI_RET_NOT_SUPPORTED;
+ }
+}
diff --git a/arch/arm64/lib/system.c b/arch/arm64/lib/system.c
new file mode 100644
index 0000000..4e227f6
--- /dev/null
+++ b/arch/arm64/lib/system.c
@@ -0,0 +1,46 @@
+/* SPDX-License-Identifier: GPL-2.0+ */
+/*
+ * system.c - system power control, the PSCI SYSTEM_OFF and
+ * SYSTEM_RESET backends. off parks the core in WFI forever, the
+ * manual's low power entry (D1-2255). reset drives the PE reset
+ * domain: RMR_EL2 reset request with the system reset bit, RR bit 1,
+ * followed by a barrier pair so the request retires before anything
+ * else observes the core.
+ *
+ * On real hardware a SoC also needs a watchdog or PMIC write for a
+ * full board reset, that is board territory, the arch part is this.
+ *
+ * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org>
+ */
+
+#include <debug.h>
+#include <stdint.h>
+
+void tb_system_off(void)
+{
+ dprintf(ALWAYS, "system off\n");
+
+ for (;;) {
+ asm volatile("wfi");
+ }
+}
+
+void tb_system_reset(void)
+{
+ uint64_t rmr;
+
+ dprintf(ALWAYS, "system reset\n");
+
+ asm volatile("mrs %0, rmr_el2" : "=r" (rmr));
+ rmr |= (1 << 1); /* RR, request reset */
+ asm volatile(
+ "msr rmr_el2, %0\n"
+ "dsb sy\n"
+ "isb\n"
+ :: "r" (rmr));
+
+ /* if the reset domain ignores us, park */
+ for (;;) {
+ asm volatile("wfi");
+ }
+}
diff --git a/arch/arm64/lib/timer.c b/arch/arm64/lib/timer.c
new file mode 100644
index 0000000..605cd18
--- /dev/null
+++ b/arch/arm64/lib/timer.c
@@ -0,0 +1,48 @@
+/* SPDX-License-Identifier: GPL-2.0+ */
+/*
+ * timer.c - generic timer delays, the system counter from D10. the
+ * counter is a fixed frequency free running counter, CNTFRQ_EL0
+ * carries the frequency, CNTVCT_EL0 the 64 bit count. delays are a
+ * busy wait on the counter, no interrupts needed, microsecond and
+ * millisecond granularity.
+ *
+ * CNTVCT_EL0 is the virtual counter view; at EL2 with no offset
+ * configured it is the physical count. the read is not speculative
+ * and needs an isb to serialize against subsequent counter reads
+ * per the counter access rules.
+ *
+ * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org>
+ */
+
+#include <stdint.h>
+
+static uint64_t read_cntfrq(void)
+{
+ uint64_t v;
+
+ asm volatile("mrs %0, cntfrq_el0" : "=r" (v));
+ return v;
+}
+
+static uint64_t read_counter(void)
+{
+ uint64_t v;
+
+ asm volatile("isb\nmrs %0, cntvct_el0" : "=r" (v));
+ return v;
+}
+
+void tb_udelay(uint32_t us)
+{
+ uint64_t freq = read_cntfrq();
+ uint64_t start = read_counter();
+ uint64_t ticks = (uint64_t)us * freq / 1000000ULL;
+
+ while (read_counter() - start < ticks)
+ ;
+}
+
+void tb_mdelay(uint32_t ms)
+{
+ tb_udelay(ms * 1000);
+}