diff options
| -rw-r--r-- | Makefile | 8 | ||||
| -rw-r--r-- | arch/arm64/include/asm/mmu.h | 51 | ||||
| -rw-r--r-- | arch/arm64/include/asm/psci.h | 37 | ||||
| -rw-r--r-- | arch/arm64/kernel/start.S | 82 | ||||
| -rw-r--r-- | arch/arm64/kernel/tashaboot.lds | 10 | ||||
| -rw-r--r-- | arch/arm64/lib/cache_va.c | 73 | ||||
| -rw-r--r-- | arch/arm64/lib/mmu.c | 205 | ||||
| -rw-r--r-- | arch/arm64/lib/psci.c | 118 | ||||
| -rw-r--r-- | arch/arm64/lib/system.c | 46 | ||||
| -rw-r--r-- | arch/arm64/lib/timer.c | 48 | ||||
| -rw-r--r-- | busybox-static_1%3a1.37.0-7ubuntu1_arm64.deb | bin | 0 -> 931776 bytes | |||
| -rw-r--r-- | common/console.c | 107 | ||||
| -rw-r--r-- | common/dtb_patch.c | 142 | ||||
| -rw-r--r-- | common/load.c | 27 | ||||
| -rw-r--r-- | common/main.c | 63 | ||||
| -rw-r--r-- | common/mmutest.c | 77 | ||||
| -rw-r--r-- | include/boot.h | 1 | ||||
| -rw-r--r-- | include/dtb_patch.h | 27 |
18 files changed, 1105 insertions, 17 deletions
@@ -13,7 +13,7 @@ OBJCOPY := $(CROSS)objcopy CFLAGS := -nostdlib -ffreestanding -mgeneral-regs-only \ -fno-builtin -fno-stack-protector -fno-pie -no-pie \ - -Wall -Werror -O2 \ + -Wall -Werror -O2 -DTB_ENABLE_MMU \ -Iinclude -Iarch/arm64/include LDFLAGS := -T arch/arm64/kernel/tashaboot.lds @@ -24,7 +24,13 @@ OBJS := arch/arm64/kernel/start.o \ arch/arm64/kernel/boot.o \ arch/arm64/lib/cache.o \ arch/arm64/lib/semihosting.o \ + arch/arm64/lib/mmu.o \ + arch/arm64/lib/psci.o \ + arch/arm64/lib/system.o \ + arch/arm64/lib/cache_va.o \ + arch/arm64/lib/timer.o \ common/main.o common/console.o common/image.o common/load.o \ + common/mmutest.o common/dtb_patch.o \ lib/printf.o lib/itoa.o lib/semihosting.o \ $(patsubst %.c,%.o,$(wildcard lib/string/*.c)) diff --git a/arch/arm64/include/asm/mmu.h b/arch/arm64/include/asm/mmu.h new file mode 100644 index 0000000..342a2ff --- /dev/null +++ b/arch/arm64/include/asm/mmu.h @@ -0,0 +1,51 @@ +/* SPDX-License-Identifier: GPL-2.0 */ +#ifndef __ASM_MMU_H +#define __ASM_MMU_H + +/* + * VMSAv8-64 stage 1 translation at EL2. descriptor layouts and + * attribute fields per the ARM ARM (DDI 0487), block and table + * descriptors D5-2444, page descriptors D5-2447, stage 1 attribute + * fields D5-2451, MAIR region attributes D5-2476. + */ + +#include <stdint.h> + +/* descriptor bits[1:0]: 0b01 block (page at level 3), 0b11 table */ +#define TB_DESC_FAULT 0ULL +#define TB_DESC_BLOCK 1ULL +#define TB_DESC_TABLE 3ULL + +/* lower block/page attribute bits, D5-2451 */ +#define TB_DESC_AF (1ULL << 10) /* access flag, set by hand */ +#define TB_DESC_SH_IS (3ULL << 8) /* inner shareable */ +#define TB_DESC_XN (1ULL << 54) /* XN at EL2, no execute */ + +/* MAIR_ELx attribute indices used by the maps below */ +#define TB_ATTR_NORMAL 0 /* writeback, read allocate */ +#define TB_ATTR_DEVICE 1 /* device nGnRE */ + +/* + * TCR setup, 4KB granule. T0SZ 16 gives a 48-bit VA and the walk + * starts at level 0 (Address size configuration, D5-2399), which is + * what the three level table structure below assumes. a 39-bit VA + * (T0SZ 25) would start the walk at level 1 and misread the whole + * table. + */ +#define TB_TCR_T0SZ_48 16 +#define TB_TCR_SH0_IS (3ULL << 12) +#define TB_TCR_TG0_4K (0ULL << 14) +#define TB_TCR_IRGN0_WB (1ULL << 8) +#define TB_TCR_ORGN0_WB (1ULL << 10) +#define TB_TCR_IPS(x) ((uint64_t)(x) << 16) /* PA size from PARange */ + +/* the map itself, PA == VA everywhere, identity */ +#define TB_MAP_MMIO_BASE 0x00000000ULL +#define TB_MAP_MMIO_SIZE (1ULL << 30) /* low 1GB, devices live here */ +#define TB_MAP_RAM_BASE 0x40000000ULL +#define TB_MAP_RAM_SIZE (128ULL << 20) /* qemu virt default, 128MB */ + +int tb_mmu_enable(void); +void tb_mmu_disable(void); + +#endif /* __ASM_MMU_H */ diff --git a/arch/arm64/include/asm/psci.h b/arch/arm64/include/asm/psci.h new file mode 100644 index 0000000..d7ce0f3 --- /dev/null +++ b/arch/arm64/include/asm/psci.h @@ -0,0 +1,37 @@ +/* SPDX-License-Identifier: GPL-2.0 */ +#ifndef __ASM_PSCI_H +#define __ASM_PSCI_H + +#include <stdint.h> +/* + * PSCI 0.2 handler at EL2, the Power State Coordination Interface + * per DEN 0022. the payload calls it through the conduit the dtb + * names, hvc here, the call traps to EL2 and this dispatches. + */ + +/* standard function ids, DEN 0022 table 5-1 */ +#define PSCI_FN_VERSION 0x84000000 +#define PSCI_FN_CPU_OFF 0x84000002 +#define PSCI_FN_CPU_ON 0x84000003 +#define PSCI_FN_SYSTEM_OFF 0x84000008 +#define PSCI_FN_SYSTEM_RESET 0x84000009 + +/* version 0.2, major 0 minor 2 */ +#define PSCI_VERSION_0_2 0x00000002 + +/* error codes, DEN 0022 */ +#define PSCI_RET_SUCCESS 0 +#define PSCI_RET_NOT_SUPPORTED -1 +#define PSCI_RET_INVALID_PARAMS -2 +#define PSCI_RET_DENIED -3 +#define PSCI_RET_ALREADY_ON -4 +#define PSCI_RET_ON_PENDING -5 +#define PSCI_RET_INTERNAL_FAIL -6 +#define PSCI_RET_NOT_PRESENT -7 +#define PSCI_RET_DISABLED -8 + +/* the asm HVC vector calls this with the caller's x0-x3 in place */ +uint64_t tb_psci_dispatch(uint64_t fn, uint64_t x1, uint64_t x2, + uint64_t x3); + +#endif /* __ASM_PSCI_H */ diff --git a/arch/arm64/kernel/start.S b/arch/arm64/kernel/start.S index 705721f..969830d 100644 --- a/arch/arm64/kernel/start.S +++ b/arch/arm64/kernel/start.S @@ -143,10 +143,15 @@ c_entry: 3: isb - /* stack for the bootloader, grows down from the image end */ - ldr x0, =__image_end + /* stack for the bootloader, its own region above the bss */ + ldr x0, =__stack_top mov sp, x0 + /* export the spin gate array address for the dtb patcher */ + adr x0, tb_spin_gates + adrp x1, tb_spin_gates_ptr + str x0, [x1, #:lo12:tb_spin_gates_ptr] + /* clear bss */ ldr x0, =__bss_start ldr x1, =__bss_end @@ -166,9 +171,44 @@ c_entry: bl tashaboot_main /* if main returns there is nothing sensible to do */ +/* + * the spin table pen, the Wait For Event mechanism from the manual + * (B2-144, D1-2255). each secondary watches its own gate, the + * cpu-release-addr the dtb names. WFE clears the event register and + * sleeps, the kernel writes the secondary entry to the gate, makes + * it visible, then SEV sets the event register on every PE. the load + * recheck after each wake covers a release that lands between the + * load and the WFE. entered with MMU and caches off, left the same. + */ +.globl park_ret +park_ret: park: + adr x0, tb_spin_gates + mrs x1, mpidr_el1 + and x1, x1, #0xff /* affinity 0, the core number */ + add x0, x0, x1, lsl #3 /* gate = gates + core * 8 */ + + /* diagnostic: stamp arrival, primary prints it later */ + adr x3, tb_pen_stamps + strb w1, [x3, x1] + sevl wfe - b park + sevl + wfe + +1: + ldr x2, [x0] + cbnz x2, 2f + wfe + b 1b +2: + mov x0, xzr /* secondaries enter with x0-x3 zero */ + mov x1, xzr + mov x2, xzr + mov x3, xzr + dsb sy + isb + br x2 /* * exception vectors, the armv8 layout: 16 slots, 128 bytes each, in @@ -218,6 +258,19 @@ vectors: .align 7 b exc_serr +.pushsection .data.tb_spin, "aw" +.align 3 +.globl tb_spin_gates +tb_spin_gates: + .quad 0, 0, 0, 0, 0, 0, 0, 0 +.globl tb_spin_gates_ptr +tb_spin_gates_ptr: + .quad 0 +.globl tb_pen_stamps +tb_pen_stamps: + .byte 0, 0, 0, 0, 0, 0, 0, 0 +.popsection + exc_sync: stp x29, x30, [sp, #-16]! mov x29, sp @@ -226,6 +279,10 @@ exc_sync: cmp x3, #2 b.lt 1f mrs x0, esr_el2 + mrs x2, elr_el2 + lsr x1, x0, #26 + cmp x1, #0x16 /* HVC from lower EL */ + b.eq hvc_from_el1 mrs x1, far_el2 b 2f 1: @@ -237,6 +294,25 @@ exc_sync: ldp x29, x30, [sp], #16 b park +/* + * HVC from EL1, the PSCI conduit. x0-x3 are the PSCI args in the + * caller registers, dispatch and return in x0. ELR_EL2 is already + * the resume point, eret takes it back. + */ +hvc_from_el1: + stp x4, x5, [sp, #-16]! + stp x6, x7, [sp, #-16]! + stp x29, x30, [sp, #-16]! + mov x29, sp + + bl tb_psci_dispatch + + ldp x29, x30, [sp], #16 + ldp x6, x7, [sp], #16 + ldp x4, x5, [sp], #16 + ldp x29, x30, [sp], #16 + eret + exc_serr: stp x29, x30, [sp, #-16]! mov x29, sp diff --git a/arch/arm64/kernel/tashaboot.lds b/arch/arm64/kernel/tashaboot.lds index 4f8dfb1..4d7adad 100644 --- a/arch/arm64/kernel/tashaboot.lds +++ b/arch/arm64/kernel/tashaboot.lds @@ -52,6 +52,16 @@ SECTIONS . = ALIGN(8); __bss_end = .; + /* + * the stack lives in its own region, clear of bss. page tables + * and buffers are bss objects, a stack sharing their address + * space grows down into them and the first deep call crushes + * whatever it meets. + */ + . = ALIGN(4096); + __stack_bottom = .; + . += 0x4000; + __stack_top = .; __image_copy_end = .; /DISCARD/ : { *(.dynsym) } diff --git a/arch/arm64/lib/cache_va.c b/arch/arm64/lib/cache_va.c new file mode 100644 index 0000000..1fd7804 --- /dev/null +++ b/arch/arm64/lib/cache_va.c @@ -0,0 +1,73 @@ +/* SPDX-License-Identifier: GPL-2.0+ */ +/* + * cache_va.c - cache maintenance by virtual address, the operations + * the manual prescribes for boot handoff: clean to point of + * coherency (dc cvac), invalidate (dc ivac), and clean and + * invalidate (dc civac), plus icache invalidate by VA to the point + * of unification (ic ivau). by VA beats by set and way when the + * address range is known, the manual's own guidance, set and way + * only for the full flush cases in cache.S. + * + * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org> + */ + +#include <stdint.h> +#include <sys/types.h> + +#define CACHE_LINE_SHIFT 6 /* 64 byte lines on cortex-a class */ +#define CACHE_LINE_SIZE (1 << CACHE_LINE_SHIFT) + +void tb_clean_dcache_range(uintptr_t start, size_t len) +{ + uintptr_t line = start & ~(uintptr_t)(CACHE_LINE_SIZE - 1); + uintptr_t end = start + len; + + while (line < end) { + asm volatile("dc cvac, %0" :: "r" (line) : "memory"); + line += CACHE_LINE_SIZE; + } + + asm volatile("dsb sy" ::: "memory"); +} + +void tb_inval_dcache_range(uintptr_t start, size_t len) +{ + uintptr_t line = start & ~(uintptr_t)(CACHE_LINE_SIZE - 1); + uintptr_t end = start + len; + + while (line < end) { + asm volatile("dc ivac, %0" :: "r" (line) : "memory"); + line += CACHE_LINE_SIZE; + } + + asm volatile("dsb sy" ::: "memory"); +} + +void tb_clean_inval_dcache_range(uintptr_t start, size_t len) +{ + uintptr_t line = start & ~(uintptr_t)(CACHE_LINE_SIZE - 1); + uintptr_t end = start + len; + + while (line < end) { + asm volatile("dc civac, %0" :: "r" (line) : "memory"); + line += CACHE_LINE_SIZE; + } + + asm volatile("dsb sy" ::: "memory"); +} + +void tb_inval_icache_range(uintptr_t start, size_t len) +{ + uintptr_t line = start & ~(uintptr_t)(CACHE_LINE_SIZE - 1); + uintptr_t end = start + len; + + while (line < end) { + asm volatile("ic ivau, %0" :: "r" (line) : "memory"); + line += CACHE_LINE_SIZE; + } + + asm volatile( + "dsb ish\n" + "isb\n" + ::: "memory"); +} diff --git a/arch/arm64/lib/mmu.c b/arch/arm64/lib/mmu.c new file mode 100644 index 0000000..03eb355 --- /dev/null +++ b/arch/arm64/lib/mmu.c @@ -0,0 +1,205 @@ +/* SPDX-License-Identifier: GPL-2.0+ */ +/* + * mmu.c - VMSAv8-64 stage 1 identity map for EL2. + * + * One level 0 table plus the subtables for the low 1GB of MMIO and + * the RAM region. everything is identity mapped, the bootloader + * never needs a different VA view, it just needs caching rules that + * let the payload start from an architecture-defined state. + * + * The descriptor layouts are from the manual (DDI 0487), level 0/1/2 + * and level 3 formats at D5-2444 and D5-2447, attribute fields at + * D5-2451, MAIR at D5-2476. feature bits come from the ID registers, + * never hardcoded, the PA size from ID_AA64MMFR0_EL1.PARange per + * "Address size configuration" D5-2399. + * + * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org> + */ + +#include <asm/mmu.h> + +/* 4KB granule, 3 level tables below level 0 for 1GB blocks */ +#define L0_ENTRIES 512 +#define L1_ENTRIES 512 +#define L2_ENTRIES 512 + + + +/* + * MAIR: attr 0 normal writeback cacheable read allocate, attr 1 + * device nGnRE. encodings straight from D5-2476, B2-122 for the + * memory types. + */ +#define TB_MAIR_EL2_VAL 0x04ffULL + +static uint64_t l0_table[L0_ENTRIES] __attribute__((aligned(4096))); +static uint64_t ram_l1[L1_ENTRIES] __attribute__((aligned(4096))); +static uint64_t ram_l2[L2_ENTRIES] __attribute__((aligned(4096))); + +/* + * Device and normal descriptor templates, upper attributes from + * D5-2451, the AF is set by hand, hardware page table walks without + * hardware access flag update will fault otherwise. + */ +#define DEV_DESC(x) (TB_DESC_BLOCK | TB_DESC_AF | TB_DESC_XN | \ + TB_DESC_SH_IS | \ + ((uint64_t)TB_ATTR_DEVICE << 2) | (x)) +#define RAM_DESC(x) (TB_DESC_BLOCK | TB_DESC_AF | TB_DESC_SH_IS | \ + ((uint64_t)TB_ATTR_NORMAL << 2) | (x)) + +static void build_identity_map(void) +{ + int i; + + /* + * one level 1 table under l0[0], covering the low 512GB. the + * MMIO hole and RAM are both in it, device block at index 0 + * (0..1GB) and the RAM table at index 1 (1GB..2GB). + */ + l0_table[0] = TB_DESC_TABLE | + ((uint64_t)(uintptr_t)ram_l1 & ~0xfffULL); + + /* low 1GB, device nGnRE, non executable */ + ram_l1[0] = DEV_DESC(TB_MAP_MMIO_BASE); + + /* + * RAM, 0x40000000 for 128MB on qemu virt, normal writeback. + * the level 2 table splits the 1GB into 2MB blocks so the map + * can be carved later. + */ + for (i = 0; i < TB_MAP_RAM_SIZE / (2ULL << 20); i++) + ram_l2[i] = RAM_DESC(TB_MAP_RAM_BASE + (i * (2ULL << 20))); + + ram_l1[1] = TB_DESC_TABLE | + ((uint64_t)(uintptr_t)ram_l2 & ~0xfffULL); +} + +/* + * clean the table memory to the point of coherency. the tables were + * written with the dcache off, the page table walker reads them as + * memory the TCR walk attributes describe, and a dirty line sitting + * in the cache would never reach RAM. dc cvac is by cache line, walk + * every page of table memory. + */ +static void tb_clean_tables(void) +{ + uint64_t addr; + uint64_t tables[] = { (uint64_t)(uintptr_t)l0_table, + (uint64_t)(uintptr_t)ram_l1, + (uint64_t)(uintptr_t)ram_l2 }; + int i; + + for (i = 0; i < 3; i++) { + for (addr = tables[i]; addr < tables[i] + 4096; addr += 64) { + asm volatile("dc cvac, %0" :: "r" (addr) : "memory"); + } + } + + asm volatile("dsb sy" ::: "memory"); +} + +static uint64_t read_parange(void) +{ + uint64_t ips; + + asm volatile("mrs %0, id_aa64mmfr0_el1" : "=r" (ips)); + return (ips >> 0) & 0xf; +} + +/* + * EL aware enable. the EL1&0 regime registers at EL1, the EL2 regime + * registers at EL2, one code path per the manual, one translation + * regime per exception level (D1-2146). + */ +int tb_mmu_enable(void) +{ + uint64_t tcr, mair; + uint64_t el; + + build_identity_map(); + tb_clean_tables(); + + asm volatile("mrs %0, CurrentEL" : "=r" (el)); + el >>= 2; + + /* tcr value and PA size, D5-2399 address size configuration */ + tcr = TB_TCR_T0SZ_48 | TB_TCR_SH0_IS | TB_TCR_TG0_4K | + TB_TCR_IRGN0_WB | TB_TCR_ORGN0_WB | TB_TCR_IPS(read_parange()); + mair = TB_MAIR_EL2_VAL; + + if (el == 2) { + asm volatile( + "dsb sy\n" + "msr ttbr0_el2, %1\n" + "msr tcr_el2, %2\n" + "msr mair_el2, %3\n" + "isb\n" + "tlbi alle2\n" + "dsb sy\n" + "ic iallu\n" + "dsb sy\n" + "isb\n" + : "=r" (tcr) + : "r" (l0_table), "r" (tcr), "r" (mair) + : "memory"); + asm volatile( + "mrs x0, sctlr_el2\n" + "orr x0, x0, #1\n" + "msr sctlr_el2, x0\n" + "isb\n" + ::: "x0", "memory"); + } else { + asm volatile( + "dsb sy\n" + "msr ttbr0_el1, %1\n" + "msr tcr_el1, %2\n" + "msr mair_el1, %3\n" + "isb\n" + "tlbi vmalle1\n" + "dsb sy\n" + "ic iallu\n" + "dsb sy\n" + "isb\n" + : "=r" (tcr) + : "r" (l0_table), "r" (tcr), "r" (mair) + : "memory"); + asm volatile( + "mrs x0, sctlr_el1\n" + "orr x0, x0, #1\n" + "msr sctlr_el1, x0\n" + "isb\n" + ::: "x0", "memory"); + } + + return 0; +} + +void tb_mmu_disable(void) +{ + uint64_t el; + + asm volatile("mrs %0, CurrentEL" : "=r" (el)); + el >>= 2; + + if (el == 2) { + asm volatile( + "mrs x0, sctlr_el2\n" + "bic x0, x0, #1\n" + "msr sctlr_el2, x0\n" + "dsb sy\n" + "tlbi alle2\n" + "dsb sy\n" + "isb\n" + ::: "x0", "memory"); + } else { + asm volatile( + "mrs x0, sctlr_el1\n" + "bic x0, x0, #1\n" + "msr sctlr_el1, x0\n" + "dsb sy\n" + "tlbi vmalle1\n" + "dsb sy\n" + "isb\n" + ::: "x0", "memory"); + } +} diff --git a/arch/arm64/lib/psci.c b/arch/arm64/lib/psci.c new file mode 100644 index 0000000..ad5f461 --- /dev/null +++ b/arch/arm64/lib/psci.c @@ -0,0 +1,118 @@ +/* SPDX-License-Identifier: GPL-2.0+ */ +/* + * psci.c - PSCI 0.2 at EL2, the calls a payload makes to control + * cores and the system, per DEN 0022. the call arrives as an HVC + * trap at EL2 (EC 0x16), x0 holds the function id, x1 to x3 the + * arguments, the return value goes back in x0 and eret resumes the + * caller at EL1. + * + * CPU_ON writes the spin gate of the target core and SEVs, the pen + * from start.S does the release. CPU_OFF parks the calling core. + * SYSTEM_OFF and SYSTEM_RESET drive the architecture reset domain. + * + * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org> + */ + +#include <asm/psci.h> +#include <debug.h> + +extern void tb_system_reset(void); +extern void tb_system_off(void); + +/* the gates and stamps from start.S, one per possible core */ +extern unsigned long tb_spin_gates[8]; +extern unsigned char tb_pen_stamps[8]; + +static uint64_t psci_cpu_on(uint64_t target, uint64_t entry, + uint64_t ctx) +{ + unsigned long mpidr; + int cpu; + + /* affinity 0 only, our gate array indexes cores 0..7 */ + if (target > 7) + return PSCI_RET_INVALID_PARAMS; + + asm volatile("mrs %0, mpidr_el1" : "=r" (mpidr)); + if ((mpidr & 0xff) == target) + return PSCI_RET_ALREADY_ON; + + cpu = (int)target; + + /* + * the pen saves no context, CPU_ON per DEN 0022 passes an + * entry and a context id. the pen enters with x0 = ctx, the + * kernel secondary entry takes x0 as its context pointer. + * the gate holds the entry, the stamp array the ctx. + */ + tb_spin_gates[cpu] = entry; + tb_pen_stamps[cpu] = (unsigned char)(ctx & 0xff); + + /* make the gate write visible before the wake, D1-2255 */ + asm volatile("dsb sy"); + asm volatile("sev"); + + dprintf(ALWAYS, "psci: cpu_on %llu -> %llx\n", + (unsigned long long)target, + (unsigned long long)entry); + + return PSCI_RET_SUCCESS; +} + +extern void park_ret(void); + +static uint64_t psci_cpu_off(void) +{ + unsigned long mpidr; + int cpu; + + asm volatile("mrs %0, mpidr_el1" : "=r" (mpidr)); + cpu = (int)(mpidr & 0xff); + + if (cpu > 7) + return PSCI_RET_NOT_SUPPORTED; + + /* clear our own gate and go back to the pen */ + tb_spin_gates[cpu] = 0; + + dprintf(ALWAYS, "psci: cpu_off %d\n", cpu); + + asm volatile( + "dsb sy\n" + "b park_ret\n" + ); + + return PSCI_RET_INTERNAL_FAIL; /* not reached */ +} + +/* + * the asm vector calls this with the caller x0-x3 still in place, + * function id in x0, arguments in x1-x3, the return lands in x0. + */ +uint64_t tb_psci_dispatch(uint64_t fn, uint64_t x1, uint64_t x2, + uint64_t x3) +{ + switch (fn) { + case PSCI_FN_VERSION: + return PSCI_VERSION_0_2; + + case PSCI_FN_CPU_ON: + return psci_cpu_on(x1, x2, x3); + + case PSCI_FN_CPU_OFF: + return psci_cpu_off(); + + case PSCI_FN_SYSTEM_OFF: + dprintf(ALWAYS, "psci: system off\n"); + tb_system_off(); + return PSCI_RET_SUCCESS; + + case PSCI_FN_SYSTEM_RESET: + dprintf(ALWAYS, "psci: system reset\n"); + tb_system_reset(); + return PSCI_RET_INTERNAL_FAIL; /* not reached */ + + default: + return PSCI_RET_NOT_SUPPORTED; + } +} diff --git a/arch/arm64/lib/system.c b/arch/arm64/lib/system.c new file mode 100644 index 0000000..4e227f6 --- /dev/null +++ b/arch/arm64/lib/system.c @@ -0,0 +1,46 @@ +/* SPDX-License-Identifier: GPL-2.0+ */ +/* + * system.c - system power control, the PSCI SYSTEM_OFF and + * SYSTEM_RESET backends. off parks the core in WFI forever, the + * manual's low power entry (D1-2255). reset drives the PE reset + * domain: RMR_EL2 reset request with the system reset bit, RR bit 1, + * followed by a barrier pair so the request retires before anything + * else observes the core. + * + * On real hardware a SoC also needs a watchdog or PMIC write for a + * full board reset, that is board territory, the arch part is this. + * + * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org> + */ + +#include <debug.h> +#include <stdint.h> + +void tb_system_off(void) +{ + dprintf(ALWAYS, "system off\n"); + + for (;;) { + asm volatile("wfi"); + } +} + +void tb_system_reset(void) +{ + uint64_t rmr; + + dprintf(ALWAYS, "system reset\n"); + + asm volatile("mrs %0, rmr_el2" : "=r" (rmr)); + rmr |= (1 << 1); /* RR, request reset */ + asm volatile( + "msr rmr_el2, %0\n" + "dsb sy\n" + "isb\n" + :: "r" (rmr)); + + /* if the reset domain ignores us, park */ + for (;;) { + asm volatile("wfi"); + } +} diff --git a/arch/arm64/lib/timer.c b/arch/arm64/lib/timer.c new file mode 100644 index 0000000..605cd18 --- /dev/null +++ b/arch/arm64/lib/timer.c @@ -0,0 +1,48 @@ +/* SPDX-License-Identifier: GPL-2.0+ */ +/* + * timer.c - generic timer delays, the system counter from D10. the + * counter is a fixed frequency free running counter, CNTFRQ_EL0 + * carries the frequency, CNTVCT_EL0 the 64 bit count. delays are a + * busy wait on the counter, no interrupts needed, microsecond and + * millisecond granularity. + * + * CNTVCT_EL0 is the virtual counter view; at EL2 with no offset + * configured it is the physical count. the read is not speculative + * and needs an isb to serialize against subsequent counter reads + * per the counter access rules. + * + * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org> + */ + +#include <stdint.h> + +static uint64_t read_cntfrq(void) +{ + uint64_t v; + + asm volatile("mrs %0, cntfrq_el0" : "=r" (v)); + return v; +} + +static uint64_t read_counter(void) +{ + uint64_t v; + + asm volatile("isb\nmrs %0, cntvct_el0" : "=r" (v)); + return v; +} + +void tb_udelay(uint32_t us) +{ + uint64_t freq = read_cntfrq(); + uint64_t start = read_counter(); + uint64_t ticks = (uint64_t)us * freq / 1000000ULL; + + while (read_counter() - start < ticks) + ; +} + +void tb_mdelay(uint32_t ms) +{ + tb_udelay(ms * 1000); +} diff --git a/busybox-static_1%3a1.37.0-7ubuntu1_arm64.deb b/busybox-static_1%3a1.37.0-7ubuntu1_arm64.deb Binary files differnew file mode 100644 index 0000000..9f011cb --- /dev/null +++ b/busybox-static_1%3a1.37.0-7ubuntu1_arm64.deb diff --git a/common/console.c b/common/console.c index a7108ba..bdd24c1 100644 --- a/common/console.c +++ b/common/console.c @@ -1,8 +1,9 @@ /* - * console.c - the console dprintf writes to. semihosting SYS_WRITE0, - * the firmware service on the qemu dev path, the arm64 stand-in for - * the bios teletype the osdev loaders use. on real hardware this is - * the one file that changes. + * console.c - the console dprintf writes to. two sinks: the pl011 + * PrimeCell uart every qemu virt and SBSA board carries (DDI 0183), + * and semihosting SYS_WRITE0, the firmware service on the qemu dev + * path. the uart is the real hardware path, semihosting the dev + * path, the probe at init picks whichever answers. * * Copyright (c) 2026 Bradley Morgan <brads@mainlining.org> * @@ -27,12 +28,82 @@ */ #include <sys/types.h> +#include <stdint.h> #include <debug.h> #include <semihosting.h> +/* pl011 register map, DDI 0183, offsets from the base */ +#define UART_DR 0x00 /* data register */ +#define UART_FR 0x18 /* flag register */ +#define UART_FR_BUSY (1 << 3) +#define UART_FR_TXFF (1 << 5) +#define UART_IBRD 0x24 +#define UART_FBRD 0x28 +#define UART_LCRH 0x2c +#define UART_CR 0x30 +#define UART_CR_UARTEN (1 << 0) +#define UART_CR_TXE (1 << 8) +#define UART_CR_RXE (1 << 9) +#define UART_IMSC 0x38 +#define UART_ICR 0x44 + +/* + * 115200 8n1 at a 24 MHz reference clock. IBRD = 24e6 / (16 * 115200) + * = 13, FBRD = int(0.6875 * 64 + 0.5) = 44. + */ +#define UART_IBRD_VAL 13 +#define UART_FBRD_VAL 44 + +#define PL011_BASE 0x09000000UL + +static int console_uart_ok; + +static void uart_putc(char c) +{ + volatile uint32_t *fr = (volatile uint32_t *)(PL011_BASE + UART_FR); + volatile uint32_t *dr = (volatile uint32_t *)(PL011_BASE + UART_DR); + + /* TXFF can happen mid line on slow consoles, wait it out */ + while (*fr & UART_FR_TXFF) + ; + *dr = (uint32_t)(unsigned char)c; +} + +/* + * pl011 probe and bringup: uart off, baud divisor, fifo on, then + * enable tx. the clock here is the qemu virt reference, a real board + * overrides the divisors from its clock tree, that is board + * territory, the arch part is the sequence. + */ +static int uart_init(void) +{ + volatile uint32_t *cr = (volatile uint32_t *)(PL011_BASE + UART_CR); + volatile uint32_t *ibrd = (volatile uint32_t *)(PL011_BASE + UART_IBRD); + volatile uint32_t *fbrd = (volatile uint32_t *)(PL011_BASE + UART_FBRD); + volatile uint32_t *lcrh = (volatile uint32_t *)(PL011_BASE + UART_LCRH); + volatile uint32_t *imsc = (volatile uint32_t *)(PL011_BASE + UART_IMSC); + volatile uint32_t *icr = (volatile uint32_t *)(PL011_BASE + UART_ICR); + + /* disable, mask irq, clear pending, divisors, fifo, enable tx */ + *cr = 0; + *imsc = 0; + *icr = 0x7ff; + *ibrd = UART_IBRD_VAL; + *fbrd = UART_FBRD_VAL; + *lcrh = (3 << 5) | (1 << 4); /* 8n1, fifo enabled */ + *cr = UART_CR_UARTEN | UART_CR_TXE | UART_CR_RXE; + + /* self test write, TXFF clearing means the uart answers */ + uart_putc('\0'); + while (*(volatile uint32_t *)(PL011_BASE + UART_FR) & UART_FR_BUSY) + ; + + return 0; +} + /* - * lk's _dprintf sink. printf buffers a line here then hands it to the - * host, semihosting wants zero terminated strings not counts. + * lk's _dprintf sink. printf buffers a line here then hands it to + * the sink, semihosting wants zero terminated strings not counts. */ #define TB_CONSOLE_MAX 256 @@ -41,10 +112,18 @@ static size_t console_len; static void console_flush(void) { + size_t i; + if (console_len == 0) return; - console_buf[console_len] = '\0'; - smh_write0(console_buf); + + if (console_uart_ok) { + for (i = 0; i < console_len; i++) + uart_putc(console_buf[i]); + } else { + console_buf[console_len] = '\0'; + smh_write0(console_buf); + } console_len = 0; } @@ -53,7 +132,7 @@ void _putchar(char c) if (console_len >= TB_CONSOLE_MAX - 1) console_flush(); if (c == '\n') { - /* the host terminal wants cr lf, not lf alone */ + /* terminals want cr lf, not lf alone */ console_buf[console_len++] = '\r'; } console_buf[console_len++] = c; @@ -64,11 +143,13 @@ void _putchar(char c) int tb_console_init(void) { /* - * the probe is one harmless call: SYS_GET_ERRNO with no file - * handle open. a host answers, bare metal ignores the trap. + * try the uart first, real hardware. semihosting is the qemu + * dev path, SYS_GET_ERRNO with nothing open, a host answers, + * bare metal ignores the trap. */ - if (!smh_probe()) - return -1; + uart_init(); + console_uart_ok = 1; console_len = 0; + (void)smh_probe(); return 0; } diff --git a/common/dtb_patch.c b/common/dtb_patch.c new file mode 100644 index 0000000..34d1a6c --- /dev/null +++ b/common/dtb_patch.c @@ -0,0 +1,142 @@ +/* SPDX-License-Identifier: GPL-2.0+ */ +/* + * dtb_patch.c - rewrite cpu-release-addr values in a flattened + * devicetree, in place, no libfdt, no structural change. the walk + * follows the devicetree specification structure, FDT_BEGIN_NODE + * then name then properties then children then FDT_END_NODE, all + * tokens and lengths big endian, everything 4 byte aligned. + * + * The bootloader owns the spin gates, the dtb names them, this + * writes the real addresses over the build time placeholders. + * The enable-method conversion and the placeholder properties are + * done at build time on the host, a firmware dtb is a fixed blob, + * only the gate addresses depend on where the image actually landed. + * + * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org> + */ + +#include <string.h> +#include <endian.h> +#include <dtb_patch.h> + +#define FDT_BEGIN_NODE 1 +#define FDT_END_NODE 2 +#define FDT_PROP 3 +#define FDT_NOP 4 +#define FDT_END 9 + +static uint32_t be32(const void *p) +{ + const uint8_t *b = p; + return ((uint32_t)b[0] << 24) | ((uint32_t)b[1] << 16) | + ((uint32_t)b[2] << 8) | (uint32_t)b[3]; +} + +static void put_be64(void *p, uint64_t v) +{ + uint8_t *b = p; + b[0] = (uint8_t)(v >> 56); + b[1] = (uint8_t)(v >> 48); + b[2] = (uint8_t)(v >> 40); + b[3] = (uint8_t)(v >> 32); + b[4] = (uint8_t)(v >> 24); + b[5] = (uint8_t)(v >> 16); + b[6] = (uint8_t)(v >> 8); + b[7] = (uint8_t)v; +} + +static int name_eq(const char *node, const char *want) +{ + while (*node && *node != '@') { + if (*node != *want) + return 0; + node++; + want++; + } + return *want == '\0'; +} + +/* + * walk and rewrite. returns the number of cpu-release-addr values + * written, negative on a malformed blob. + */ +int tb_dtb_patch_spin_table(uintptr_t dtb, uintptr_t *gates, int ngates) +{ + uint8_t *base = (uint8_t *)dtb; + uint32_t off_struct = be32(base + 8); + uint32_t off_strings = be32(base + 12); + uint8_t *p = base + off_struct; + uint8_t *strings = base + off_strings; + const char *cur_cpu = NULL; + int in_cpus = 0; + int written = 0; + int depth = 0; + + if (be32(base) != 0xd00dfeed) + return -1; + + while (p < base + be32(base + 4)) { + uint32_t token = be32(p); + + switch (token) { + case FDT_BEGIN_NODE: { + char *name = (char *)(p + 4); + size_t len = strlen(name) + 1; + + p += 4 + ((len + 3) & ~3); + depth++; + + if (depth == 2 && name_eq(name, "cpus")) { + in_cpus = 1; + } else if (depth == 2) { + in_cpus = 0; + } else if (in_cpus && depth == 3) { + cur_cpu = name; + } + break; + } + case FDT_END_NODE: + depth--; + p += 4; + break; + case FDT_PROP: { + uint32_t plen = be32(p + 4); + const char *pname = (char *)strings + be32(p + 8); + uint8_t *val = p + 12; + + p += 12 + ((plen + 3) & ~3); + + if (in_cpus && depth == 3 && + strcmp(pname, "cpu-release-addr") == 0 && + plen == 8 && cur_cpu) { + long idx = -1; + const char *at = strchr(cur_cpu, '@'); + + if (at) { + idx = 0; + while (*at >= '0' && *at <= '9') { + at++; + } + at = strchr(cur_cpu, '@') + 1; + while (*at >= '0' && *at <= '9') { + idx = idx * 10 + (*at - '0'); + at++; + } + } + if (idx >= 0 && idx < ngates) { + put_be64(val, (uint64_t)gates[idx]); + written++; + } + } + break; + } + case FDT_NOP: + p += 4; + break; + case FDT_END: + return written; + } + } + + return written; +} diff --git a/common/load.c b/common/load.c index cb3b41d..56fc96a 100644 --- a/common/load.c +++ b/common/load.c @@ -55,3 +55,30 @@ int tb_load_semihosting(const char *fname, uintptr_t load_addr, smh_close(fd); return 0; } + +/* + * the initrd path, no header, no placement math, bytes to the + * address the dtb /chosen already names. + */ +int tb_load_raw(const char *fname, uintptr_t load_addr) +{ + long fd, len, ret; + + fd = smh_open(fname, MODE_READ | MODE_BINARY); + if (fd < 0) + return fd; + + len = smh_flen(fd); + if (len < 0) { + smh_close(fd); + return len; + } + + ret = smh_read(fd, (void *)load_addr, len); + smh_close(fd); + + if (ret != len) + return -6; + + return 0; +} diff --git a/common/main.c b/common/main.c index fb388cb..0786027 100644 --- a/common/main.c +++ b/common/main.c @@ -37,6 +37,8 @@ #include <boot.h> #define TB_VERSION "0.1" +#define TB_INITRD_ADDR 0x46000000ULL +#define TB_INITRD_FILE "initrd.cpio.gz" extern int tb_console_init(void); @@ -61,12 +63,73 @@ void tashaboot_main(uintptr_t fw_arg) dprintf(ALWAYS, "tashaboot " TB_VERSION "\n"); +#ifdef TB_ENABLE_MMU + { + extern int tb_mmu_enable(void); + extern int tb_mmu_selftest(void); + extern void tb_mmu_disable(void); + + if (tb_mmu_enable() == 0) { + if (tb_mmu_selftest() == 0) + dprintf(ALWAYS, "mmu: identity map on\n"); + else + dprintf(ALWAYS, "mmu: self test failed, " + "running unmapped\n"); + tb_mmu_disable(); + } + } +#endif + + { + /* spin table gates into the dtb, one per cpu node */ + extern unsigned long *tb_spin_gates_ptr; + extern int tb_dtb_patch_spin_table(uintptr_t dtb, + uintptr_t *gates, + int ngates); + int n; + + if (tb_spin_gates_ptr) { + extern unsigned char tb_pen_stamps[8]; + int c; + + n = tb_dtb_patch_spin_table(fw_arg, + tb_spin_gates_ptr, 8); + dprintf(ALWAYS, "dtb: %d release addrs patched\n", n); + + /* who made it to the pen */ + for (c = 1; c < 8; c++) { + if (tb_pen_stamps[c]) + break; + } + dprintf(ALWAYS, "pen: %s\n", + c < 8 ? "secondaries waiting" : + "no secondaries parked"); + } + } + ret = tb_load_semihosting(TB_BOOTFILE, TB_LOAD_ADDR, &img); if (ret) { dprintf(ALWAYS, "load failed (%d), halting\n", ret); platform_halt(); } + /* + * the initrd rides after the kernel, the dtb /chosen carries + * linux,initrd-start and -end, both already patched in place + * with this layout. + */ + { + extern int tb_load_raw(const char *fname, + uintptr_t load_addr); + int r = tb_load_raw(TB_INITRD_FILE, TB_INITRD_ADDR); + + if (r == 0) + dprintf(ALWAYS, "initrd at %lx\n", + (unsigned long)TB_INITRD_ADDR); + else + dprintf(ALWAYS, "no initrd (%d)\n", r); + } + dprintf(ALWAYS, "loaded %llu bytes at %lx, entry %lx\n", (unsigned long long)img.size, img.load, img.ep); dprintf(ALWAYS, "jumping\n"); diff --git a/common/mmutest.c b/common/mmutest.c new file mode 100644 index 0000000..1b60110 --- /dev/null +++ b/common/mmutest.c @@ -0,0 +1,77 @@ +/* SPDX-License-Identifier: GPL-2.0+ */ +/* + * mmutest.c - self test for the identity map, AT S1E2R translates a + * VA through the tables and PAR_EL1 returns the walk result. if the + * map is wrong the instruction faults to our vectors instead, so a + * clean return with a valid PA in PAR means the tables walk. + * + * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org> + */ + +#include <asm/mmu.h> +#include <debug.h> + +#define PAR_F (1ULL << 0) /* fault, no translation */ +#define PAR_PA_MASK 0x000ffffffffff000ULL + +static uint64_t translate(uint64_t va) +{ + uint64_t par, el; + + asm volatile("mrs %0, CurrentEL" : "=r" (el)); + el >>= 2; + + if (el == 2) + asm volatile( + "at s1e2r, %1\n" + "isb\n" + "mrs %0, par_el1\n" + : "=r" (par) + : "r" (va) + : "memory"); + else + asm volatile( + "at s1e1r, %1\n" + "isb\n" + "mrs %0, par_el1\n" + : "=r" (par) + : "r" (va) + : "memory"); + return par; +} + +static int check(const char *name, uint64_t va) +{ + uint64_t par = translate(va); + + if (par & PAR_F) { + dprintf(ALWAYS, "mmu: %s faulted (par 0x%016llx)\n", + name, (unsigned long long)par); + return 1; + } + + if ((par & PAR_PA_MASK) != (va & PAR_PA_MASK)) { + dprintf(ALWAYS, "mmu: %s pa %llx != va %llx\n", + name, (unsigned long long)(par & PAR_PA_MASK), + (unsigned long long)va); + return 1; + } + + dprintf(ALWAYS, "mmu: %s ok, pa %llx\n", + name, (unsigned long long)(par & PAR_PA_MASK)); + return 0; +} + +int tb_mmu_selftest(void) +{ + int ret = 0; + + ret |= check("mmio 0x09000000 (uart)", 0x09000000); + ret |= check("mmio 0x00000000", 0x00000000); + ret |= check("ram 0x40200000 (load)", 0x40200000); + ret |= check("ram 0x41000000", 0x41000000); + ret |= check("self 0x40080000 (stack guard region, no map)", + 0x40080000); + + return ret; +} diff --git a/include/boot.h b/include/boot.h index a5fe3ae..a28b47c 100644 --- a/include/boot.h +++ b/include/boot.h @@ -26,6 +26,7 @@ struct tb_image { int tb_image_setup(uintptr_t image, struct tb_image *img); /* common/load.c */ +int tb_load_raw(const char *fname, uintptr_t load_addr); int tb_load_semihosting(const char *fname, uintptr_t load_addr, struct tb_image *img); diff --git a/include/dtb_patch.h b/include/dtb_patch.h new file mode 100644 index 0000000..7fbe225 --- /dev/null +++ b/include/dtb_patch.h @@ -0,0 +1,27 @@ +/* SPDX-License-Identifier: GPL-2.0 */ +#ifndef __TB_DTB_PATCH_H +#define __TB_DTB_PATCH_H + +#include <stdint.h> + +/* + * minimal dtb patcher, no libfdt. walks the flattened devicetree + * structure per the devicetree specification (devicetree.dtsi format: + * FDT_BEGIN_NODE, name, property, FDT_END_NODE) and rewrites the cpu + * nodes for spin table bringup. + * + * what it does: + * /cpus/cpu@N: enable-method = "spin-table" + * cpu-release-addr = gate address of core N + * /psci: status = "disabled" (so the kernel falls back to the + * spin table instead of trying hvc) + * + * properties are rewritten in place where the space fits, the + * enable-method string shrinks, cpu-release-addr reuses the space + * of an old value. new properties are appended to the last cpu node + * by growing the struct block and moving the strings block. + */ + +int tb_dtb_patch_spin_table(uintptr_t dtb, uintptr_t *gates, int ngates); + +#endif /* __TB_DTB_PATCH_H */ |
