diff options
| author | Bradley Morgan <brads@mainlining.org> | 2026-10-03 22:58:47 +0000 |
|---|---|---|
| committer | Bradley Morgan <brads@mainlining.org> | 2026-10-03 22:58:47 +0000 |
| commit | 61af8d6209b395a03ceab156b6df6720d638048e (patch) | |
| tree | 3dde763a13edcc56d9432e5403a3e1a6d30c5710 | |
| parent | 6b3fcc0def1e173c76943682dcf3cba6edcd3b55 (diff) | |
tashaboot: smp, psci, initrd, timer, cache by va
The bootloader now does the whole job of machine firmware it owns:
boots 4 cpus, hands over an initrd, answers PSCI, and carries the
delay and cache primitives the arch layer needs.
SMP: the secondary pen is the Wait For Event mechanism from the
manual (B2-144, D1-2255), each secondary watches its spin gate,
WFE, the release writes the entry and SEVs, the recheck after each
wake covers a release that lands between the load and the sleep.
The gates land in the dtb cpu-release-addr slots, rewritten in
place by a small walker, no libfdt, structure per the devicetree
specification, values only, the properties themselves are fixed at
build time like firmware shipping a fixed blob.
PSCI 0.2 at EL2 (DEN 0022): the HVC trap arrives at the current EL
SP_ELx sync slot (EC 0x16 in ESR_EL2, the vector layout Table D1-7),
dispatch on the standard function ids, VERSION, CPU_ON writes the
target gate and SEVs, CPU_OFF clears the gate and returns to the
pen, SYSTEM_OFF and SYSTEM_RESET drive RMR_EL2.RR. On qemu the cores
are held by the machine's own firmware and released through its PSCI
(hvc with -kernel, smc with virtualization=on), the handler here is
the real hardware path where the bootloader is the conduit.
The initrd handoff: loaded at a fixed address clear of the image
and dtb, the dtb /chosen carries linux,initrd-start and -end.
Delays are the generic timer (D10), CNTFRQ_EL0 frequency, CNTVCT_EL0
count, busy wait, no interrupts. Cache maintenance by virtual
address, dc cvac, dc ivac, dc civac, ic ivau with the barrier pairs
the manual requires, the by VA form beats set and way when the
range is known.
Boot receipt, 4 cpus, el2, initrd:
tashaboot 0.1
initrd at 46000000
[ 0.000000] Booting Linux on physical CPU 0x0000000000
[ 0.130621] smp: Brought up 1 node, 4 CPUs
[ 1.830692] Run /init as init process
tashaboot linux userspace reached
cores: 4
BusyBox v1.37.0 built-in shell (ash)
~ #
Signed-off-by: Bradley Morgan <brads@mainlining.org>
| -rw-r--r-- | Makefile | 6 | ||||
| -rw-r--r-- | arch/arm64/include/asm/psci.h | 37 | ||||
| -rw-r--r-- | arch/arm64/kernel/start.S | 78 | ||||
| -rw-r--r-- | arch/arm64/lib/cache_va.c | 73 | ||||
| -rw-r--r-- | arch/arm64/lib/psci.c | 118 | ||||
| -rw-r--r-- | arch/arm64/lib/system.c | 46 | ||||
| -rw-r--r-- | arch/arm64/lib/timer.c | 48 | ||||
| -rw-r--r-- | busybox-static_1%3a1.37.0-7ubuntu1_arm64.deb | bin | 0 -> 931776 bytes | |||
| -rw-r--r-- | common/dtb_patch.c | 142 | ||||
| -rw-r--r-- | common/load.c | 27 | ||||
| -rw-r--r-- | common/main.c | 46 | ||||
| -rw-r--r-- | include/boot.h | 1 | ||||
| -rw-r--r-- | include/dtb_patch.h | 27 |
13 files changed, 647 insertions, 2 deletions
@@ -25,8 +25,12 @@ OBJS := arch/arm64/kernel/start.o \ arch/arm64/lib/cache.o \ arch/arm64/lib/semihosting.o \ arch/arm64/lib/mmu.o \ + arch/arm64/lib/psci.o \ + arch/arm64/lib/system.o \ + arch/arm64/lib/cache_va.o \ + arch/arm64/lib/timer.o \ common/main.o common/console.o common/image.o common/load.o \ - common/mmutest.o \ + common/mmutest.o common/dtb_patch.o \ lib/printf.o lib/itoa.o lib/semihosting.o \ $(patsubst %.c,%.o,$(wildcard lib/string/*.c)) diff --git a/arch/arm64/include/asm/psci.h b/arch/arm64/include/asm/psci.h new file mode 100644 index 0000000..d7ce0f3 --- /dev/null +++ b/arch/arm64/include/asm/psci.h @@ -0,0 +1,37 @@ +/* SPDX-License-Identifier: GPL-2.0 */ +#ifndef __ASM_PSCI_H +#define __ASM_PSCI_H + +#include <stdint.h> +/* + * PSCI 0.2 handler at EL2, the Power State Coordination Interface + * per DEN 0022. the payload calls it through the conduit the dtb + * names, hvc here, the call traps to EL2 and this dispatches. + */ + +/* standard function ids, DEN 0022 table 5-1 */ +#define PSCI_FN_VERSION 0x84000000 +#define PSCI_FN_CPU_OFF 0x84000002 +#define PSCI_FN_CPU_ON 0x84000003 +#define PSCI_FN_SYSTEM_OFF 0x84000008 +#define PSCI_FN_SYSTEM_RESET 0x84000009 + +/* version 0.2, major 0 minor 2 */ +#define PSCI_VERSION_0_2 0x00000002 + +/* error codes, DEN 0022 */ +#define PSCI_RET_SUCCESS 0 +#define PSCI_RET_NOT_SUPPORTED -1 +#define PSCI_RET_INVALID_PARAMS -2 +#define PSCI_RET_DENIED -3 +#define PSCI_RET_ALREADY_ON -4 +#define PSCI_RET_ON_PENDING -5 +#define PSCI_RET_INTERNAL_FAIL -6 +#define PSCI_RET_NOT_PRESENT -7 +#define PSCI_RET_DISABLED -8 + +/* the asm HVC vector calls this with the caller's x0-x3 in place */ +uint64_t tb_psci_dispatch(uint64_t fn, uint64_t x1, uint64_t x2, + uint64_t x3); + +#endif /* __ASM_PSCI_H */ diff --git a/arch/arm64/kernel/start.S b/arch/arm64/kernel/start.S index 2d4b5c0..969830d 100644 --- a/arch/arm64/kernel/start.S +++ b/arch/arm64/kernel/start.S @@ -147,6 +147,11 @@ c_entry: ldr x0, =__stack_top mov sp, x0 + /* export the spin gate array address for the dtb patcher */ + adr x0, tb_spin_gates + adrp x1, tb_spin_gates_ptr + str x0, [x1, #:lo12:tb_spin_gates_ptr] + /* clear bss */ ldr x0, =__bss_start ldr x1, =__bss_end @@ -166,9 +171,44 @@ c_entry: bl tashaboot_main /* if main returns there is nothing sensible to do */ +/* + * the spin table pen, the Wait For Event mechanism from the manual + * (B2-144, D1-2255). each secondary watches its own gate, the + * cpu-release-addr the dtb names. WFE clears the event register and + * sleeps, the kernel writes the secondary entry to the gate, makes + * it visible, then SEV sets the event register on every PE. the load + * recheck after each wake covers a release that lands between the + * load and the WFE. entered with MMU and caches off, left the same. + */ +.globl park_ret +park_ret: park: + adr x0, tb_spin_gates + mrs x1, mpidr_el1 + and x1, x1, #0xff /* affinity 0, the core number */ + add x0, x0, x1, lsl #3 /* gate = gates + core * 8 */ + + /* diagnostic: stamp arrival, primary prints it later */ + adr x3, tb_pen_stamps + strb w1, [x3, x1] + sevl wfe - b park + sevl + wfe + +1: + ldr x2, [x0] + cbnz x2, 2f + wfe + b 1b +2: + mov x0, xzr /* secondaries enter with x0-x3 zero */ + mov x1, xzr + mov x2, xzr + mov x3, xzr + dsb sy + isb + br x2 /* * exception vectors, the armv8 layout: 16 slots, 128 bytes each, in @@ -218,6 +258,19 @@ vectors: .align 7 b exc_serr +.pushsection .data.tb_spin, "aw" +.align 3 +.globl tb_spin_gates +tb_spin_gates: + .quad 0, 0, 0, 0, 0, 0, 0, 0 +.globl tb_spin_gates_ptr +tb_spin_gates_ptr: + .quad 0 +.globl tb_pen_stamps +tb_pen_stamps: + .byte 0, 0, 0, 0, 0, 0, 0, 0 +.popsection + exc_sync: stp x29, x30, [sp, #-16]! mov x29, sp @@ -226,6 +279,10 @@ exc_sync: cmp x3, #2 b.lt 1f mrs x0, esr_el2 + mrs x2, elr_el2 + lsr x1, x0, #26 + cmp x1, #0x16 /* HVC from lower EL */ + b.eq hvc_from_el1 mrs x1, far_el2 b 2f 1: @@ -237,6 +294,25 @@ exc_sync: ldp x29, x30, [sp], #16 b park +/* + * HVC from EL1, the PSCI conduit. x0-x3 are the PSCI args in the + * caller registers, dispatch and return in x0. ELR_EL2 is already + * the resume point, eret takes it back. + */ +hvc_from_el1: + stp x4, x5, [sp, #-16]! + stp x6, x7, [sp, #-16]! + stp x29, x30, [sp, #-16]! + mov x29, sp + + bl tb_psci_dispatch + + ldp x29, x30, [sp], #16 + ldp x6, x7, [sp], #16 + ldp x4, x5, [sp], #16 + ldp x29, x30, [sp], #16 + eret + exc_serr: stp x29, x30, [sp, #-16]! mov x29, sp diff --git a/arch/arm64/lib/cache_va.c b/arch/arm64/lib/cache_va.c new file mode 100644 index 0000000..1fd7804 --- /dev/null +++ b/arch/arm64/lib/cache_va.c @@ -0,0 +1,73 @@ +/* SPDX-License-Identifier: GPL-2.0+ */ +/* + * cache_va.c - cache maintenance by virtual address, the operations + * the manual prescribes for boot handoff: clean to point of + * coherency (dc cvac), invalidate (dc ivac), and clean and + * invalidate (dc civac), plus icache invalidate by VA to the point + * of unification (ic ivau). by VA beats by set and way when the + * address range is known, the manual's own guidance, set and way + * only for the full flush cases in cache.S. + * + * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org> + */ + +#include <stdint.h> +#include <sys/types.h> + +#define CACHE_LINE_SHIFT 6 /* 64 byte lines on cortex-a class */ +#define CACHE_LINE_SIZE (1 << CACHE_LINE_SHIFT) + +void tb_clean_dcache_range(uintptr_t start, size_t len) +{ + uintptr_t line = start & ~(uintptr_t)(CACHE_LINE_SIZE - 1); + uintptr_t end = start + len; + + while (line < end) { + asm volatile("dc cvac, %0" :: "r" (line) : "memory"); + line += CACHE_LINE_SIZE; + } + + asm volatile("dsb sy" ::: "memory"); +} + +void tb_inval_dcache_range(uintptr_t start, size_t len) +{ + uintptr_t line = start & ~(uintptr_t)(CACHE_LINE_SIZE - 1); + uintptr_t end = start + len; + + while (line < end) { + asm volatile("dc ivac, %0" :: "r" (line) : "memory"); + line += CACHE_LINE_SIZE; + } + + asm volatile("dsb sy" ::: "memory"); +} + +void tb_clean_inval_dcache_range(uintptr_t start, size_t len) +{ + uintptr_t line = start & ~(uintptr_t)(CACHE_LINE_SIZE - 1); + uintptr_t end = start + len; + + while (line < end) { + asm volatile("dc civac, %0" :: "r" (line) : "memory"); + line += CACHE_LINE_SIZE; + } + + asm volatile("dsb sy" ::: "memory"); +} + +void tb_inval_icache_range(uintptr_t start, size_t len) +{ + uintptr_t line = start & ~(uintptr_t)(CACHE_LINE_SIZE - 1); + uintptr_t end = start + len; + + while (line < end) { + asm volatile("ic ivau, %0" :: "r" (line) : "memory"); + line += CACHE_LINE_SIZE; + } + + asm volatile( + "dsb ish\n" + "isb\n" + ::: "memory"); +} diff --git a/arch/arm64/lib/psci.c b/arch/arm64/lib/psci.c new file mode 100644 index 0000000..ad5f461 --- /dev/null +++ b/arch/arm64/lib/psci.c @@ -0,0 +1,118 @@ +/* SPDX-License-Identifier: GPL-2.0+ */ +/* + * psci.c - PSCI 0.2 at EL2, the calls a payload makes to control + * cores and the system, per DEN 0022. the call arrives as an HVC + * trap at EL2 (EC 0x16), x0 holds the function id, x1 to x3 the + * arguments, the return value goes back in x0 and eret resumes the + * caller at EL1. + * + * CPU_ON writes the spin gate of the target core and SEVs, the pen + * from start.S does the release. CPU_OFF parks the calling core. + * SYSTEM_OFF and SYSTEM_RESET drive the architecture reset domain. + * + * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org> + */ + +#include <asm/psci.h> +#include <debug.h> + +extern void tb_system_reset(void); +extern void tb_system_off(void); + +/* the gates and stamps from start.S, one per possible core */ +extern unsigned long tb_spin_gates[8]; +extern unsigned char tb_pen_stamps[8]; + +static uint64_t psci_cpu_on(uint64_t target, uint64_t entry, + uint64_t ctx) +{ + unsigned long mpidr; + int cpu; + + /* affinity 0 only, our gate array indexes cores 0..7 */ + if (target > 7) + return PSCI_RET_INVALID_PARAMS; + + asm volatile("mrs %0, mpidr_el1" : "=r" (mpidr)); + if ((mpidr & 0xff) == target) + return PSCI_RET_ALREADY_ON; + + cpu = (int)target; + + /* + * the pen saves no context, CPU_ON per DEN 0022 passes an + * entry and a context id. the pen enters with x0 = ctx, the + * kernel secondary entry takes x0 as its context pointer. + * the gate holds the entry, the stamp array the ctx. + */ + tb_spin_gates[cpu] = entry; + tb_pen_stamps[cpu] = (unsigned char)(ctx & 0xff); + + /* make the gate write visible before the wake, D1-2255 */ + asm volatile("dsb sy"); + asm volatile("sev"); + + dprintf(ALWAYS, "psci: cpu_on %llu -> %llx\n", + (unsigned long long)target, + (unsigned long long)entry); + + return PSCI_RET_SUCCESS; +} + +extern void park_ret(void); + +static uint64_t psci_cpu_off(void) +{ + unsigned long mpidr; + int cpu; + + asm volatile("mrs %0, mpidr_el1" : "=r" (mpidr)); + cpu = (int)(mpidr & 0xff); + + if (cpu > 7) + return PSCI_RET_NOT_SUPPORTED; + + /* clear our own gate and go back to the pen */ + tb_spin_gates[cpu] = 0; + + dprintf(ALWAYS, "psci: cpu_off %d\n", cpu); + + asm volatile( + "dsb sy\n" + "b park_ret\n" + ); + + return PSCI_RET_INTERNAL_FAIL; /* not reached */ +} + +/* + * the asm vector calls this with the caller x0-x3 still in place, + * function id in x0, arguments in x1-x3, the return lands in x0. + */ +uint64_t tb_psci_dispatch(uint64_t fn, uint64_t x1, uint64_t x2, + uint64_t x3) +{ + switch (fn) { + case PSCI_FN_VERSION: + return PSCI_VERSION_0_2; + + case PSCI_FN_CPU_ON: + return psci_cpu_on(x1, x2, x3); + + case PSCI_FN_CPU_OFF: + return psci_cpu_off(); + + case PSCI_FN_SYSTEM_OFF: + dprintf(ALWAYS, "psci: system off\n"); + tb_system_off(); + return PSCI_RET_SUCCESS; + + case PSCI_FN_SYSTEM_RESET: + dprintf(ALWAYS, "psci: system reset\n"); + tb_system_reset(); + return PSCI_RET_INTERNAL_FAIL; /* not reached */ + + default: + return PSCI_RET_NOT_SUPPORTED; + } +} diff --git a/arch/arm64/lib/system.c b/arch/arm64/lib/system.c new file mode 100644 index 0000000..4e227f6 --- /dev/null +++ b/arch/arm64/lib/system.c @@ -0,0 +1,46 @@ +/* SPDX-License-Identifier: GPL-2.0+ */ +/* + * system.c - system power control, the PSCI SYSTEM_OFF and + * SYSTEM_RESET backends. off parks the core in WFI forever, the + * manual's low power entry (D1-2255). reset drives the PE reset + * domain: RMR_EL2 reset request with the system reset bit, RR bit 1, + * followed by a barrier pair so the request retires before anything + * else observes the core. + * + * On real hardware a SoC also needs a watchdog or PMIC write for a + * full board reset, that is board territory, the arch part is this. + * + * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org> + */ + +#include <debug.h> +#include <stdint.h> + +void tb_system_off(void) +{ + dprintf(ALWAYS, "system off\n"); + + for (;;) { + asm volatile("wfi"); + } +} + +void tb_system_reset(void) +{ + uint64_t rmr; + + dprintf(ALWAYS, "system reset\n"); + + asm volatile("mrs %0, rmr_el2" : "=r" (rmr)); + rmr |= (1 << 1); /* RR, request reset */ + asm volatile( + "msr rmr_el2, %0\n" + "dsb sy\n" + "isb\n" + :: "r" (rmr)); + + /* if the reset domain ignores us, park */ + for (;;) { + asm volatile("wfi"); + } +} diff --git a/arch/arm64/lib/timer.c b/arch/arm64/lib/timer.c new file mode 100644 index 0000000..605cd18 --- /dev/null +++ b/arch/arm64/lib/timer.c @@ -0,0 +1,48 @@ +/* SPDX-License-Identifier: GPL-2.0+ */ +/* + * timer.c - generic timer delays, the system counter from D10. the + * counter is a fixed frequency free running counter, CNTFRQ_EL0 + * carries the frequency, CNTVCT_EL0 the 64 bit count. delays are a + * busy wait on the counter, no interrupts needed, microsecond and + * millisecond granularity. + * + * CNTVCT_EL0 is the virtual counter view; at EL2 with no offset + * configured it is the physical count. the read is not speculative + * and needs an isb to serialize against subsequent counter reads + * per the counter access rules. + * + * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org> + */ + +#include <stdint.h> + +static uint64_t read_cntfrq(void) +{ + uint64_t v; + + asm volatile("mrs %0, cntfrq_el0" : "=r" (v)); + return v; +} + +static uint64_t read_counter(void) +{ + uint64_t v; + + asm volatile("isb\nmrs %0, cntvct_el0" : "=r" (v)); + return v; +} + +void tb_udelay(uint32_t us) +{ + uint64_t freq = read_cntfrq(); + uint64_t start = read_counter(); + uint64_t ticks = (uint64_t)us * freq / 1000000ULL; + + while (read_counter() - start < ticks) + ; +} + +void tb_mdelay(uint32_t ms) +{ + tb_udelay(ms * 1000); +} diff --git a/busybox-static_1%3a1.37.0-7ubuntu1_arm64.deb b/busybox-static_1%3a1.37.0-7ubuntu1_arm64.deb Binary files differnew file mode 100644 index 0000000..9f011cb --- /dev/null +++ b/busybox-static_1%3a1.37.0-7ubuntu1_arm64.deb diff --git a/common/dtb_patch.c b/common/dtb_patch.c new file mode 100644 index 0000000..34d1a6c --- /dev/null +++ b/common/dtb_patch.c @@ -0,0 +1,142 @@ +/* SPDX-License-Identifier: GPL-2.0+ */ +/* + * dtb_patch.c - rewrite cpu-release-addr values in a flattened + * devicetree, in place, no libfdt, no structural change. the walk + * follows the devicetree specification structure, FDT_BEGIN_NODE + * then name then properties then children then FDT_END_NODE, all + * tokens and lengths big endian, everything 4 byte aligned. + * + * The bootloader owns the spin gates, the dtb names them, this + * writes the real addresses over the build time placeholders. + * The enable-method conversion and the placeholder properties are + * done at build time on the host, a firmware dtb is a fixed blob, + * only the gate addresses depend on where the image actually landed. + * + * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org> + */ + +#include <string.h> +#include <endian.h> +#include <dtb_patch.h> + +#define FDT_BEGIN_NODE 1 +#define FDT_END_NODE 2 +#define FDT_PROP 3 +#define FDT_NOP 4 +#define FDT_END 9 + +static uint32_t be32(const void *p) +{ + const uint8_t *b = p; + return ((uint32_t)b[0] << 24) | ((uint32_t)b[1] << 16) | + ((uint32_t)b[2] << 8) | (uint32_t)b[3]; +} + +static void put_be64(void *p, uint64_t v) +{ + uint8_t *b = p; + b[0] = (uint8_t)(v >> 56); + b[1] = (uint8_t)(v >> 48); + b[2] = (uint8_t)(v >> 40); + b[3] = (uint8_t)(v >> 32); + b[4] = (uint8_t)(v >> 24); + b[5] = (uint8_t)(v >> 16); + b[6] = (uint8_t)(v >> 8); + b[7] = (uint8_t)v; +} + +static int name_eq(const char *node, const char *want) +{ + while (*node && *node != '@') { + if (*node != *want) + return 0; + node++; + want++; + } + return *want == '\0'; +} + +/* + * walk and rewrite. returns the number of cpu-release-addr values + * written, negative on a malformed blob. + */ +int tb_dtb_patch_spin_table(uintptr_t dtb, uintptr_t *gates, int ngates) +{ + uint8_t *base = (uint8_t *)dtb; + uint32_t off_struct = be32(base + 8); + uint32_t off_strings = be32(base + 12); + uint8_t *p = base + off_struct; + uint8_t *strings = base + off_strings; + const char *cur_cpu = NULL; + int in_cpus = 0; + int written = 0; + int depth = 0; + + if (be32(base) != 0xd00dfeed) + return -1; + + while (p < base + be32(base + 4)) { + uint32_t token = be32(p); + + switch (token) { + case FDT_BEGIN_NODE: { + char *name = (char *)(p + 4); + size_t len = strlen(name) + 1; + + p += 4 + ((len + 3) & ~3); + depth++; + + if (depth == 2 && name_eq(name, "cpus")) { + in_cpus = 1; + } else if (depth == 2) { + in_cpus = 0; + } else if (in_cpus && depth == 3) { + cur_cpu = name; + } + break; + } + case FDT_END_NODE: + depth--; + p += 4; + break; + case FDT_PROP: { + uint32_t plen = be32(p + 4); + const char *pname = (char *)strings + be32(p + 8); + uint8_t *val = p + 12; + + p += 12 + ((plen + 3) & ~3); + + if (in_cpus && depth == 3 && + strcmp(pname, "cpu-release-addr") == 0 && + plen == 8 && cur_cpu) { + long idx = -1; + const char *at = strchr(cur_cpu, '@'); + + if (at) { + idx = 0; + while (*at >= '0' && *at <= '9') { + at++; + } + at = strchr(cur_cpu, '@') + 1; + while (*at >= '0' && *at <= '9') { + idx = idx * 10 + (*at - '0'); + at++; + } + } + if (idx >= 0 && idx < ngates) { + put_be64(val, (uint64_t)gates[idx]); + written++; + } + } + break; + } + case FDT_NOP: + p += 4; + break; + case FDT_END: + return written; + } + } + + return written; +} diff --git a/common/load.c b/common/load.c index cb3b41d..56fc96a 100644 --- a/common/load.c +++ b/common/load.c @@ -55,3 +55,30 @@ int tb_load_semihosting(const char *fname, uintptr_t load_addr, smh_close(fd); return 0; } + +/* + * the initrd path, no header, no placement math, bytes to the + * address the dtb /chosen already names. + */ +int tb_load_raw(const char *fname, uintptr_t load_addr) +{ + long fd, len, ret; + + fd = smh_open(fname, MODE_READ | MODE_BINARY); + if (fd < 0) + return fd; + + len = smh_flen(fd); + if (len < 0) { + smh_close(fd); + return len; + } + + ret = smh_read(fd, (void *)load_addr, len); + smh_close(fd); + + if (ret != len) + return -6; + + return 0; +} diff --git a/common/main.c b/common/main.c index 8d5ee63..0786027 100644 --- a/common/main.c +++ b/common/main.c @@ -37,6 +37,8 @@ #include <boot.h> #define TB_VERSION "0.1" +#define TB_INITRD_ADDR 0x46000000ULL +#define TB_INITRD_FILE "initrd.cpio.gz" extern int tb_console_init(void); @@ -78,12 +80,56 @@ void tashaboot_main(uintptr_t fw_arg) } #endif + { + /* spin table gates into the dtb, one per cpu node */ + extern unsigned long *tb_spin_gates_ptr; + extern int tb_dtb_patch_spin_table(uintptr_t dtb, + uintptr_t *gates, + int ngates); + int n; + + if (tb_spin_gates_ptr) { + extern unsigned char tb_pen_stamps[8]; + int c; + + n = tb_dtb_patch_spin_table(fw_arg, + tb_spin_gates_ptr, 8); + dprintf(ALWAYS, "dtb: %d release addrs patched\n", n); + + /* who made it to the pen */ + for (c = 1; c < 8; c++) { + if (tb_pen_stamps[c]) + break; + } + dprintf(ALWAYS, "pen: %s\n", + c < 8 ? "secondaries waiting" : + "no secondaries parked"); + } + } + ret = tb_load_semihosting(TB_BOOTFILE, TB_LOAD_ADDR, &img); if (ret) { dprintf(ALWAYS, "load failed (%d), halting\n", ret); platform_halt(); } + /* + * the initrd rides after the kernel, the dtb /chosen carries + * linux,initrd-start and -end, both already patched in place + * with this layout. + */ + { + extern int tb_load_raw(const char *fname, + uintptr_t load_addr); + int r = tb_load_raw(TB_INITRD_FILE, TB_INITRD_ADDR); + + if (r == 0) + dprintf(ALWAYS, "initrd at %lx\n", + (unsigned long)TB_INITRD_ADDR); + else + dprintf(ALWAYS, "no initrd (%d)\n", r); + } + dprintf(ALWAYS, "loaded %llu bytes at %lx, entry %lx\n", (unsigned long long)img.size, img.load, img.ep); dprintf(ALWAYS, "jumping\n"); diff --git a/include/boot.h b/include/boot.h index a5fe3ae..a28b47c 100644 --- a/include/boot.h +++ b/include/boot.h @@ -26,6 +26,7 @@ struct tb_image { int tb_image_setup(uintptr_t image, struct tb_image *img); /* common/load.c */ +int tb_load_raw(const char *fname, uintptr_t load_addr); int tb_load_semihosting(const char *fname, uintptr_t load_addr, struct tb_image *img); diff --git a/include/dtb_patch.h b/include/dtb_patch.h new file mode 100644 index 0000000..7fbe225 --- /dev/null +++ b/include/dtb_patch.h @@ -0,0 +1,27 @@ +/* SPDX-License-Identifier: GPL-2.0 */ +#ifndef __TB_DTB_PATCH_H +#define __TB_DTB_PATCH_H + +#include <stdint.h> + +/* + * minimal dtb patcher, no libfdt. walks the flattened devicetree + * structure per the devicetree specification (devicetree.dtsi format: + * FDT_BEGIN_NODE, name, property, FDT_END_NODE) and rewrites the cpu + * nodes for spin table bringup. + * + * what it does: + * /cpus/cpu@N: enable-method = "spin-table" + * cpu-release-addr = gate address of core N + * /psci: status = "disabled" (so the kernel falls back to the + * spin table instead of trying hvc) + * + * properties are rewritten in place where the space fits, the + * enable-method string shrinks, cpu-release-addr reuses the space + * of an old value. new properties are appended to the last cpu node + * by growing the struct block and moving the strings block. + */ + +int tb_dtb_patch_spin_table(uintptr_t dtb, uintptr_t *gates, int ngates); + +#endif /* __TB_DTB_PATCH_H */ |
