summaryrefslogtreecommitdiff
diff options
context:
space:
mode:
-rw-r--r--Makefile8
-rw-r--r--arch/arm64/include/asm/mmu.h51
-rw-r--r--arch/arm64/include/asm/psci.h37
-rw-r--r--arch/arm64/kernel/start.S82
-rw-r--r--arch/arm64/kernel/tashaboot.lds10
-rw-r--r--arch/arm64/lib/cache_va.c73
-rw-r--r--arch/arm64/lib/mmu.c205
-rw-r--r--arch/arm64/lib/psci.c118
-rw-r--r--arch/arm64/lib/system.c46
-rw-r--r--arch/arm64/lib/timer.c48
-rw-r--r--busybox-static_1%3a1.37.0-7ubuntu1_arm64.debbin0 -> 931776 bytes
-rw-r--r--common/console.c107
-rw-r--r--common/dtb_patch.c142
-rw-r--r--common/load.c27
-rw-r--r--common/main.c63
-rw-r--r--common/mmutest.c77
-rw-r--r--include/boot.h1
-rw-r--r--include/dtb_patch.h27
18 files changed, 1105 insertions, 17 deletions
diff --git a/Makefile b/Makefile
index be2cb63..ae9877b 100644
--- a/Makefile
+++ b/Makefile
@@ -13,7 +13,7 @@ OBJCOPY := $(CROSS)objcopy
CFLAGS := -nostdlib -ffreestanding -mgeneral-regs-only \
-fno-builtin -fno-stack-protector -fno-pie -no-pie \
- -Wall -Werror -O2 \
+ -Wall -Werror -O2 -DTB_ENABLE_MMU \
-Iinclude -Iarch/arm64/include
LDFLAGS := -T arch/arm64/kernel/tashaboot.lds
@@ -24,7 +24,13 @@ OBJS := arch/arm64/kernel/start.o \
arch/arm64/kernel/boot.o \
arch/arm64/lib/cache.o \
arch/arm64/lib/semihosting.o \
+ arch/arm64/lib/mmu.o \
+ arch/arm64/lib/psci.o \
+ arch/arm64/lib/system.o \
+ arch/arm64/lib/cache_va.o \
+ arch/arm64/lib/timer.o \
common/main.o common/console.o common/image.o common/load.o \
+ common/mmutest.o common/dtb_patch.o \
lib/printf.o lib/itoa.o lib/semihosting.o \
$(patsubst %.c,%.o,$(wildcard lib/string/*.c))
diff --git a/arch/arm64/include/asm/mmu.h b/arch/arm64/include/asm/mmu.h
new file mode 100644
index 0000000..342a2ff
--- /dev/null
+++ b/arch/arm64/include/asm/mmu.h
@@ -0,0 +1,51 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+#ifndef __ASM_MMU_H
+#define __ASM_MMU_H
+
+/*
+ * VMSAv8-64 stage 1 translation at EL2. descriptor layouts and
+ * attribute fields per the ARM ARM (DDI 0487), block and table
+ * descriptors D5-2444, page descriptors D5-2447, stage 1 attribute
+ * fields D5-2451, MAIR region attributes D5-2476.
+ */
+
+#include <stdint.h>
+
+/* descriptor bits[1:0]: 0b01 block (page at level 3), 0b11 table */
+#define TB_DESC_FAULT 0ULL
+#define TB_DESC_BLOCK 1ULL
+#define TB_DESC_TABLE 3ULL
+
+/* lower block/page attribute bits, D5-2451 */
+#define TB_DESC_AF (1ULL << 10) /* access flag, set by hand */
+#define TB_DESC_SH_IS (3ULL << 8) /* inner shareable */
+#define TB_DESC_XN (1ULL << 54) /* XN at EL2, no execute */
+
+/* MAIR_ELx attribute indices used by the maps below */
+#define TB_ATTR_NORMAL 0 /* writeback, read allocate */
+#define TB_ATTR_DEVICE 1 /* device nGnRE */
+
+/*
+ * TCR setup, 4KB granule. T0SZ 16 gives a 48-bit VA and the walk
+ * starts at level 0 (Address size configuration, D5-2399), which is
+ * what the three level table structure below assumes. a 39-bit VA
+ * (T0SZ 25) would start the walk at level 1 and misread the whole
+ * table.
+ */
+#define TB_TCR_T0SZ_48 16
+#define TB_TCR_SH0_IS (3ULL << 12)
+#define TB_TCR_TG0_4K (0ULL << 14)
+#define TB_TCR_IRGN0_WB (1ULL << 8)
+#define TB_TCR_ORGN0_WB (1ULL << 10)
+#define TB_TCR_IPS(x) ((uint64_t)(x) << 16) /* PA size from PARange */
+
+/* the map itself, PA == VA everywhere, identity */
+#define TB_MAP_MMIO_BASE 0x00000000ULL
+#define TB_MAP_MMIO_SIZE (1ULL << 30) /* low 1GB, devices live here */
+#define TB_MAP_RAM_BASE 0x40000000ULL
+#define TB_MAP_RAM_SIZE (128ULL << 20) /* qemu virt default, 128MB */
+
+int tb_mmu_enable(void);
+void tb_mmu_disable(void);
+
+#endif /* __ASM_MMU_H */
diff --git a/arch/arm64/include/asm/psci.h b/arch/arm64/include/asm/psci.h
new file mode 100644
index 0000000..d7ce0f3
--- /dev/null
+++ b/arch/arm64/include/asm/psci.h
@@ -0,0 +1,37 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+#ifndef __ASM_PSCI_H
+#define __ASM_PSCI_H
+
+#include <stdint.h>
+/*
+ * PSCI 0.2 handler at EL2, the Power State Coordination Interface
+ * per DEN 0022. the payload calls it through the conduit the dtb
+ * names, hvc here, the call traps to EL2 and this dispatches.
+ */
+
+/* standard function ids, DEN 0022 table 5-1 */
+#define PSCI_FN_VERSION 0x84000000
+#define PSCI_FN_CPU_OFF 0x84000002
+#define PSCI_FN_CPU_ON 0x84000003
+#define PSCI_FN_SYSTEM_OFF 0x84000008
+#define PSCI_FN_SYSTEM_RESET 0x84000009
+
+/* version 0.2, major 0 minor 2 */
+#define PSCI_VERSION_0_2 0x00000002
+
+/* error codes, DEN 0022 */
+#define PSCI_RET_SUCCESS 0
+#define PSCI_RET_NOT_SUPPORTED -1
+#define PSCI_RET_INVALID_PARAMS -2
+#define PSCI_RET_DENIED -3
+#define PSCI_RET_ALREADY_ON -4
+#define PSCI_RET_ON_PENDING -5
+#define PSCI_RET_INTERNAL_FAIL -6
+#define PSCI_RET_NOT_PRESENT -7
+#define PSCI_RET_DISABLED -8
+
+/* the asm HVC vector calls this with the caller's x0-x3 in place */
+uint64_t tb_psci_dispatch(uint64_t fn, uint64_t x1, uint64_t x2,
+ uint64_t x3);
+
+#endif /* __ASM_PSCI_H */
diff --git a/arch/arm64/kernel/start.S b/arch/arm64/kernel/start.S
index 705721f..969830d 100644
--- a/arch/arm64/kernel/start.S
+++ b/arch/arm64/kernel/start.S
@@ -143,10 +143,15 @@ c_entry:
3:
isb
- /* stack for the bootloader, grows down from the image end */
- ldr x0, =__image_end
+ /* stack for the bootloader, its own region above the bss */
+ ldr x0, =__stack_top
mov sp, x0
+ /* export the spin gate array address for the dtb patcher */
+ adr x0, tb_spin_gates
+ adrp x1, tb_spin_gates_ptr
+ str x0, [x1, #:lo12:tb_spin_gates_ptr]
+
/* clear bss */
ldr x0, =__bss_start
ldr x1, =__bss_end
@@ -166,9 +171,44 @@ c_entry:
bl tashaboot_main
/* if main returns there is nothing sensible to do */
+/*
+ * the spin table pen, the Wait For Event mechanism from the manual
+ * (B2-144, D1-2255). each secondary watches its own gate, the
+ * cpu-release-addr the dtb names. WFE clears the event register and
+ * sleeps, the kernel writes the secondary entry to the gate, makes
+ * it visible, then SEV sets the event register on every PE. the load
+ * recheck after each wake covers a release that lands between the
+ * load and the WFE. entered with MMU and caches off, left the same.
+ */
+.globl park_ret
+park_ret:
park:
+ adr x0, tb_spin_gates
+ mrs x1, mpidr_el1
+ and x1, x1, #0xff /* affinity 0, the core number */
+ add x0, x0, x1, lsl #3 /* gate = gates + core * 8 */
+
+ /* diagnostic: stamp arrival, primary prints it later */
+ adr x3, tb_pen_stamps
+ strb w1, [x3, x1]
+ sevl
wfe
- b park
+ sevl
+ wfe
+
+1:
+ ldr x2, [x0]
+ cbnz x2, 2f
+ wfe
+ b 1b
+2:
+ mov x0, xzr /* secondaries enter with x0-x3 zero */
+ mov x1, xzr
+ mov x2, xzr
+ mov x3, xzr
+ dsb sy
+ isb
+ br x2
/*
* exception vectors, the armv8 layout: 16 slots, 128 bytes each, in
@@ -218,6 +258,19 @@ vectors:
.align 7
b exc_serr
+.pushsection .data.tb_spin, "aw"
+.align 3
+.globl tb_spin_gates
+tb_spin_gates:
+ .quad 0, 0, 0, 0, 0, 0, 0, 0
+.globl tb_spin_gates_ptr
+tb_spin_gates_ptr:
+ .quad 0
+.globl tb_pen_stamps
+tb_pen_stamps:
+ .byte 0, 0, 0, 0, 0, 0, 0, 0
+.popsection
+
exc_sync:
stp x29, x30, [sp, #-16]!
mov x29, sp
@@ -226,6 +279,10 @@ exc_sync:
cmp x3, #2
b.lt 1f
mrs x0, esr_el2
+ mrs x2, elr_el2
+ lsr x1, x0, #26
+ cmp x1, #0x16 /* HVC from lower EL */
+ b.eq hvc_from_el1
mrs x1, far_el2
b 2f
1:
@@ -237,6 +294,25 @@ exc_sync:
ldp x29, x30, [sp], #16
b park
+/*
+ * HVC from EL1, the PSCI conduit. x0-x3 are the PSCI args in the
+ * caller registers, dispatch and return in x0. ELR_EL2 is already
+ * the resume point, eret takes it back.
+ */
+hvc_from_el1:
+ stp x4, x5, [sp, #-16]!
+ stp x6, x7, [sp, #-16]!
+ stp x29, x30, [sp, #-16]!
+ mov x29, sp
+
+ bl tb_psci_dispatch
+
+ ldp x29, x30, [sp], #16
+ ldp x6, x7, [sp], #16
+ ldp x4, x5, [sp], #16
+ ldp x29, x30, [sp], #16
+ eret
+
exc_serr:
stp x29, x30, [sp, #-16]!
mov x29, sp
diff --git a/arch/arm64/kernel/tashaboot.lds b/arch/arm64/kernel/tashaboot.lds
index 4f8dfb1..4d7adad 100644
--- a/arch/arm64/kernel/tashaboot.lds
+++ b/arch/arm64/kernel/tashaboot.lds
@@ -52,6 +52,16 @@ SECTIONS
. = ALIGN(8);
__bss_end = .;
+ /*
+ * the stack lives in its own region, clear of bss. page tables
+ * and buffers are bss objects, a stack sharing their address
+ * space grows down into them and the first deep call crushes
+ * whatever it meets.
+ */
+ . = ALIGN(4096);
+ __stack_bottom = .;
+ . += 0x4000;
+ __stack_top = .;
__image_copy_end = .;
/DISCARD/ : { *(.dynsym) }
diff --git a/arch/arm64/lib/cache_va.c b/arch/arm64/lib/cache_va.c
new file mode 100644
index 0000000..1fd7804
--- /dev/null
+++ b/arch/arm64/lib/cache_va.c
@@ -0,0 +1,73 @@
+/* SPDX-License-Identifier: GPL-2.0+ */
+/*
+ * cache_va.c - cache maintenance by virtual address, the operations
+ * the manual prescribes for boot handoff: clean to point of
+ * coherency (dc cvac), invalidate (dc ivac), and clean and
+ * invalidate (dc civac), plus icache invalidate by VA to the point
+ * of unification (ic ivau). by VA beats by set and way when the
+ * address range is known, the manual's own guidance, set and way
+ * only for the full flush cases in cache.S.
+ *
+ * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org>
+ */
+
+#include <stdint.h>
+#include <sys/types.h>
+
+#define CACHE_LINE_SHIFT 6 /* 64 byte lines on cortex-a class */
+#define CACHE_LINE_SIZE (1 << CACHE_LINE_SHIFT)
+
+void tb_clean_dcache_range(uintptr_t start, size_t len)
+{
+ uintptr_t line = start & ~(uintptr_t)(CACHE_LINE_SIZE - 1);
+ uintptr_t end = start + len;
+
+ while (line < end) {
+ asm volatile("dc cvac, %0" :: "r" (line) : "memory");
+ line += CACHE_LINE_SIZE;
+ }
+
+ asm volatile("dsb sy" ::: "memory");
+}
+
+void tb_inval_dcache_range(uintptr_t start, size_t len)
+{
+ uintptr_t line = start & ~(uintptr_t)(CACHE_LINE_SIZE - 1);
+ uintptr_t end = start + len;
+
+ while (line < end) {
+ asm volatile("dc ivac, %0" :: "r" (line) : "memory");
+ line += CACHE_LINE_SIZE;
+ }
+
+ asm volatile("dsb sy" ::: "memory");
+}
+
+void tb_clean_inval_dcache_range(uintptr_t start, size_t len)
+{
+ uintptr_t line = start & ~(uintptr_t)(CACHE_LINE_SIZE - 1);
+ uintptr_t end = start + len;
+
+ while (line < end) {
+ asm volatile("dc civac, %0" :: "r" (line) : "memory");
+ line += CACHE_LINE_SIZE;
+ }
+
+ asm volatile("dsb sy" ::: "memory");
+}
+
+void tb_inval_icache_range(uintptr_t start, size_t len)
+{
+ uintptr_t line = start & ~(uintptr_t)(CACHE_LINE_SIZE - 1);
+ uintptr_t end = start + len;
+
+ while (line < end) {
+ asm volatile("ic ivau, %0" :: "r" (line) : "memory");
+ line += CACHE_LINE_SIZE;
+ }
+
+ asm volatile(
+ "dsb ish\n"
+ "isb\n"
+ ::: "memory");
+}
diff --git a/arch/arm64/lib/mmu.c b/arch/arm64/lib/mmu.c
new file mode 100644
index 0000000..03eb355
--- /dev/null
+++ b/arch/arm64/lib/mmu.c
@@ -0,0 +1,205 @@
+/* SPDX-License-Identifier: GPL-2.0+ */
+/*
+ * mmu.c - VMSAv8-64 stage 1 identity map for EL2.
+ *
+ * One level 0 table plus the subtables for the low 1GB of MMIO and
+ * the RAM region. everything is identity mapped, the bootloader
+ * never needs a different VA view, it just needs caching rules that
+ * let the payload start from an architecture-defined state.
+ *
+ * The descriptor layouts are from the manual (DDI 0487), level 0/1/2
+ * and level 3 formats at D5-2444 and D5-2447, attribute fields at
+ * D5-2451, MAIR at D5-2476. feature bits come from the ID registers,
+ * never hardcoded, the PA size from ID_AA64MMFR0_EL1.PARange per
+ * "Address size configuration" D5-2399.
+ *
+ * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org>
+ */
+
+#include <asm/mmu.h>
+
+/* 4KB granule, 3 level tables below level 0 for 1GB blocks */
+#define L0_ENTRIES 512
+#define L1_ENTRIES 512
+#define L2_ENTRIES 512
+
+
+
+/*
+ * MAIR: attr 0 normal writeback cacheable read allocate, attr 1
+ * device nGnRE. encodings straight from D5-2476, B2-122 for the
+ * memory types.
+ */
+#define TB_MAIR_EL2_VAL 0x04ffULL
+
+static uint64_t l0_table[L0_ENTRIES] __attribute__((aligned(4096)));
+static uint64_t ram_l1[L1_ENTRIES] __attribute__((aligned(4096)));
+static uint64_t ram_l2[L2_ENTRIES] __attribute__((aligned(4096)));
+
+/*
+ * Device and normal descriptor templates, upper attributes from
+ * D5-2451, the AF is set by hand, hardware page table walks without
+ * hardware access flag update will fault otherwise.
+ */
+#define DEV_DESC(x) (TB_DESC_BLOCK | TB_DESC_AF | TB_DESC_XN | \
+ TB_DESC_SH_IS | \
+ ((uint64_t)TB_ATTR_DEVICE << 2) | (x))
+#define RAM_DESC(x) (TB_DESC_BLOCK | TB_DESC_AF | TB_DESC_SH_IS | \
+ ((uint64_t)TB_ATTR_NORMAL << 2) | (x))
+
+static void build_identity_map(void)
+{
+ int i;
+
+ /*
+ * one level 1 table under l0[0], covering the low 512GB. the
+ * MMIO hole and RAM are both in it, device block at index 0
+ * (0..1GB) and the RAM table at index 1 (1GB..2GB).
+ */
+ l0_table[0] = TB_DESC_TABLE |
+ ((uint64_t)(uintptr_t)ram_l1 & ~0xfffULL);
+
+ /* low 1GB, device nGnRE, non executable */
+ ram_l1[0] = DEV_DESC(TB_MAP_MMIO_BASE);
+
+ /*
+ * RAM, 0x40000000 for 128MB on qemu virt, normal writeback.
+ * the level 2 table splits the 1GB into 2MB blocks so the map
+ * can be carved later.
+ */
+ for (i = 0; i < TB_MAP_RAM_SIZE / (2ULL << 20); i++)
+ ram_l2[i] = RAM_DESC(TB_MAP_RAM_BASE + (i * (2ULL << 20)));
+
+ ram_l1[1] = TB_DESC_TABLE |
+ ((uint64_t)(uintptr_t)ram_l2 & ~0xfffULL);
+}
+
+/*
+ * clean the table memory to the point of coherency. the tables were
+ * written with the dcache off, the page table walker reads them as
+ * memory the TCR walk attributes describe, and a dirty line sitting
+ * in the cache would never reach RAM. dc cvac is by cache line, walk
+ * every page of table memory.
+ */
+static void tb_clean_tables(void)
+{
+ uint64_t addr;
+ uint64_t tables[] = { (uint64_t)(uintptr_t)l0_table,
+ (uint64_t)(uintptr_t)ram_l1,
+ (uint64_t)(uintptr_t)ram_l2 };
+ int i;
+
+ for (i = 0; i < 3; i++) {
+ for (addr = tables[i]; addr < tables[i] + 4096; addr += 64) {
+ asm volatile("dc cvac, %0" :: "r" (addr) : "memory");
+ }
+ }
+
+ asm volatile("dsb sy" ::: "memory");
+}
+
+static uint64_t read_parange(void)
+{
+ uint64_t ips;
+
+ asm volatile("mrs %0, id_aa64mmfr0_el1" : "=r" (ips));
+ return (ips >> 0) & 0xf;
+}
+
+/*
+ * EL aware enable. the EL1&0 regime registers at EL1, the EL2 regime
+ * registers at EL2, one code path per the manual, one translation
+ * regime per exception level (D1-2146).
+ */
+int tb_mmu_enable(void)
+{
+ uint64_t tcr, mair;
+ uint64_t el;
+
+ build_identity_map();
+ tb_clean_tables();
+
+ asm volatile("mrs %0, CurrentEL" : "=r" (el));
+ el >>= 2;
+
+ /* tcr value and PA size, D5-2399 address size configuration */
+ tcr = TB_TCR_T0SZ_48 | TB_TCR_SH0_IS | TB_TCR_TG0_4K |
+ TB_TCR_IRGN0_WB | TB_TCR_ORGN0_WB | TB_TCR_IPS(read_parange());
+ mair = TB_MAIR_EL2_VAL;
+
+ if (el == 2) {
+ asm volatile(
+ "dsb sy\n"
+ "msr ttbr0_el2, %1\n"
+ "msr tcr_el2, %2\n"
+ "msr mair_el2, %3\n"
+ "isb\n"
+ "tlbi alle2\n"
+ "dsb sy\n"
+ "ic iallu\n"
+ "dsb sy\n"
+ "isb\n"
+ : "=r" (tcr)
+ : "r" (l0_table), "r" (tcr), "r" (mair)
+ : "memory");
+ asm volatile(
+ "mrs x0, sctlr_el2\n"
+ "orr x0, x0, #1\n"
+ "msr sctlr_el2, x0\n"
+ "isb\n"
+ ::: "x0", "memory");
+ } else {
+ asm volatile(
+ "dsb sy\n"
+ "msr ttbr0_el1, %1\n"
+ "msr tcr_el1, %2\n"
+ "msr mair_el1, %3\n"
+ "isb\n"
+ "tlbi vmalle1\n"
+ "dsb sy\n"
+ "ic iallu\n"
+ "dsb sy\n"
+ "isb\n"
+ : "=r" (tcr)
+ : "r" (l0_table), "r" (tcr), "r" (mair)
+ : "memory");
+ asm volatile(
+ "mrs x0, sctlr_el1\n"
+ "orr x0, x0, #1\n"
+ "msr sctlr_el1, x0\n"
+ "isb\n"
+ ::: "x0", "memory");
+ }
+
+ return 0;
+}
+
+void tb_mmu_disable(void)
+{
+ uint64_t el;
+
+ asm volatile("mrs %0, CurrentEL" : "=r" (el));
+ el >>= 2;
+
+ if (el == 2) {
+ asm volatile(
+ "mrs x0, sctlr_el2\n"
+ "bic x0, x0, #1\n"
+ "msr sctlr_el2, x0\n"
+ "dsb sy\n"
+ "tlbi alle2\n"
+ "dsb sy\n"
+ "isb\n"
+ ::: "x0", "memory");
+ } else {
+ asm volatile(
+ "mrs x0, sctlr_el1\n"
+ "bic x0, x0, #1\n"
+ "msr sctlr_el1, x0\n"
+ "dsb sy\n"
+ "tlbi vmalle1\n"
+ "dsb sy\n"
+ "isb\n"
+ ::: "x0", "memory");
+ }
+}
diff --git a/arch/arm64/lib/psci.c b/arch/arm64/lib/psci.c
new file mode 100644
index 0000000..ad5f461
--- /dev/null
+++ b/arch/arm64/lib/psci.c
@@ -0,0 +1,118 @@
+/* SPDX-License-Identifier: GPL-2.0+ */
+/*
+ * psci.c - PSCI 0.2 at EL2, the calls a payload makes to control
+ * cores and the system, per DEN 0022. the call arrives as an HVC
+ * trap at EL2 (EC 0x16), x0 holds the function id, x1 to x3 the
+ * arguments, the return value goes back in x0 and eret resumes the
+ * caller at EL1.
+ *
+ * CPU_ON writes the spin gate of the target core and SEVs, the pen
+ * from start.S does the release. CPU_OFF parks the calling core.
+ * SYSTEM_OFF and SYSTEM_RESET drive the architecture reset domain.
+ *
+ * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org>
+ */
+
+#include <asm/psci.h>
+#include <debug.h>
+
+extern void tb_system_reset(void);
+extern void tb_system_off(void);
+
+/* the gates and stamps from start.S, one per possible core */
+extern unsigned long tb_spin_gates[8];
+extern unsigned char tb_pen_stamps[8];
+
+static uint64_t psci_cpu_on(uint64_t target, uint64_t entry,
+ uint64_t ctx)
+{
+ unsigned long mpidr;
+ int cpu;
+
+ /* affinity 0 only, our gate array indexes cores 0..7 */
+ if (target > 7)
+ return PSCI_RET_INVALID_PARAMS;
+
+ asm volatile("mrs %0, mpidr_el1" : "=r" (mpidr));
+ if ((mpidr & 0xff) == target)
+ return PSCI_RET_ALREADY_ON;
+
+ cpu = (int)target;
+
+ /*
+ * the pen saves no context, CPU_ON per DEN 0022 passes an
+ * entry and a context id. the pen enters with x0 = ctx, the
+ * kernel secondary entry takes x0 as its context pointer.
+ * the gate holds the entry, the stamp array the ctx.
+ */
+ tb_spin_gates[cpu] = entry;
+ tb_pen_stamps[cpu] = (unsigned char)(ctx & 0xff);
+
+ /* make the gate write visible before the wake, D1-2255 */
+ asm volatile("dsb sy");
+ asm volatile("sev");
+
+ dprintf(ALWAYS, "psci: cpu_on %llu -> %llx\n",
+ (unsigned long long)target,
+ (unsigned long long)entry);
+
+ return PSCI_RET_SUCCESS;
+}
+
+extern void park_ret(void);
+
+static uint64_t psci_cpu_off(void)
+{
+ unsigned long mpidr;
+ int cpu;
+
+ asm volatile("mrs %0, mpidr_el1" : "=r" (mpidr));
+ cpu = (int)(mpidr & 0xff);
+
+ if (cpu > 7)
+ return PSCI_RET_NOT_SUPPORTED;
+
+ /* clear our own gate and go back to the pen */
+ tb_spin_gates[cpu] = 0;
+
+ dprintf(ALWAYS, "psci: cpu_off %d\n", cpu);
+
+ asm volatile(
+ "dsb sy\n"
+ "b park_ret\n"
+ );
+
+ return PSCI_RET_INTERNAL_FAIL; /* not reached */
+}
+
+/*
+ * the asm vector calls this with the caller x0-x3 still in place,
+ * function id in x0, arguments in x1-x3, the return lands in x0.
+ */
+uint64_t tb_psci_dispatch(uint64_t fn, uint64_t x1, uint64_t x2,
+ uint64_t x3)
+{
+ switch (fn) {
+ case PSCI_FN_VERSION:
+ return PSCI_VERSION_0_2;
+
+ case PSCI_FN_CPU_ON:
+ return psci_cpu_on(x1, x2, x3);
+
+ case PSCI_FN_CPU_OFF:
+ return psci_cpu_off();
+
+ case PSCI_FN_SYSTEM_OFF:
+ dprintf(ALWAYS, "psci: system off\n");
+ tb_system_off();
+ return PSCI_RET_SUCCESS;
+
+ case PSCI_FN_SYSTEM_RESET:
+ dprintf(ALWAYS, "psci: system reset\n");
+ tb_system_reset();
+ return PSCI_RET_INTERNAL_FAIL; /* not reached */
+
+ default:
+ return PSCI_RET_NOT_SUPPORTED;
+ }
+}
diff --git a/arch/arm64/lib/system.c b/arch/arm64/lib/system.c
new file mode 100644
index 0000000..4e227f6
--- /dev/null
+++ b/arch/arm64/lib/system.c
@@ -0,0 +1,46 @@
+/* SPDX-License-Identifier: GPL-2.0+ */
+/*
+ * system.c - system power control, the PSCI SYSTEM_OFF and
+ * SYSTEM_RESET backends. off parks the core in WFI forever, the
+ * manual's low power entry (D1-2255). reset drives the PE reset
+ * domain: RMR_EL2 reset request with the system reset bit, RR bit 1,
+ * followed by a barrier pair so the request retires before anything
+ * else observes the core.
+ *
+ * On real hardware a SoC also needs a watchdog or PMIC write for a
+ * full board reset, that is board territory, the arch part is this.
+ *
+ * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org>
+ */
+
+#include <debug.h>
+#include <stdint.h>
+
+void tb_system_off(void)
+{
+ dprintf(ALWAYS, "system off\n");
+
+ for (;;) {
+ asm volatile("wfi");
+ }
+}
+
+void tb_system_reset(void)
+{
+ uint64_t rmr;
+
+ dprintf(ALWAYS, "system reset\n");
+
+ asm volatile("mrs %0, rmr_el2" : "=r" (rmr));
+ rmr |= (1 << 1); /* RR, request reset */
+ asm volatile(
+ "msr rmr_el2, %0\n"
+ "dsb sy\n"
+ "isb\n"
+ :: "r" (rmr));
+
+ /* if the reset domain ignores us, park */
+ for (;;) {
+ asm volatile("wfi");
+ }
+}
diff --git a/arch/arm64/lib/timer.c b/arch/arm64/lib/timer.c
new file mode 100644
index 0000000..605cd18
--- /dev/null
+++ b/arch/arm64/lib/timer.c
@@ -0,0 +1,48 @@
+/* SPDX-License-Identifier: GPL-2.0+ */
+/*
+ * timer.c - generic timer delays, the system counter from D10. the
+ * counter is a fixed frequency free running counter, CNTFRQ_EL0
+ * carries the frequency, CNTVCT_EL0 the 64 bit count. delays are a
+ * busy wait on the counter, no interrupts needed, microsecond and
+ * millisecond granularity.
+ *
+ * CNTVCT_EL0 is the virtual counter view; at EL2 with no offset
+ * configured it is the physical count. the read is not speculative
+ * and needs an isb to serialize against subsequent counter reads
+ * per the counter access rules.
+ *
+ * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org>
+ */
+
+#include <stdint.h>
+
+static uint64_t read_cntfrq(void)
+{
+ uint64_t v;
+
+ asm volatile("mrs %0, cntfrq_el0" : "=r" (v));
+ return v;
+}
+
+static uint64_t read_counter(void)
+{
+ uint64_t v;
+
+ asm volatile("isb\nmrs %0, cntvct_el0" : "=r" (v));
+ return v;
+}
+
+void tb_udelay(uint32_t us)
+{
+ uint64_t freq = read_cntfrq();
+ uint64_t start = read_counter();
+ uint64_t ticks = (uint64_t)us * freq / 1000000ULL;
+
+ while (read_counter() - start < ticks)
+ ;
+}
+
+void tb_mdelay(uint32_t ms)
+{
+ tb_udelay(ms * 1000);
+}
diff --git a/busybox-static_1%3a1.37.0-7ubuntu1_arm64.deb b/busybox-static_1%3a1.37.0-7ubuntu1_arm64.deb
new file mode 100644
index 0000000..9f011cb
--- /dev/null
+++ b/busybox-static_1%3a1.37.0-7ubuntu1_arm64.deb
Binary files differ
diff --git a/common/console.c b/common/console.c
index a7108ba..bdd24c1 100644
--- a/common/console.c
+++ b/common/console.c
@@ -1,8 +1,9 @@
/*
- * console.c - the console dprintf writes to. semihosting SYS_WRITE0,
- * the firmware service on the qemu dev path, the arm64 stand-in for
- * the bios teletype the osdev loaders use. on real hardware this is
- * the one file that changes.
+ * console.c - the console dprintf writes to. two sinks: the pl011
+ * PrimeCell uart every qemu virt and SBSA board carries (DDI 0183),
+ * and semihosting SYS_WRITE0, the firmware service on the qemu dev
+ * path. the uart is the real hardware path, semihosting the dev
+ * path, the probe at init picks whichever answers.
*
* Copyright (c) 2026 Bradley Morgan <brads@mainlining.org>
*
@@ -27,12 +28,82 @@
*/
#include <sys/types.h>
+#include <stdint.h>
#include <debug.h>
#include <semihosting.h>
+/* pl011 register map, DDI 0183, offsets from the base */
+#define UART_DR 0x00 /* data register */
+#define UART_FR 0x18 /* flag register */
+#define UART_FR_BUSY (1 << 3)
+#define UART_FR_TXFF (1 << 5)
+#define UART_IBRD 0x24
+#define UART_FBRD 0x28
+#define UART_LCRH 0x2c
+#define UART_CR 0x30
+#define UART_CR_UARTEN (1 << 0)
+#define UART_CR_TXE (1 << 8)
+#define UART_CR_RXE (1 << 9)
+#define UART_IMSC 0x38
+#define UART_ICR 0x44
+
+/*
+ * 115200 8n1 at a 24 MHz reference clock. IBRD = 24e6 / (16 * 115200)
+ * = 13, FBRD = int(0.6875 * 64 + 0.5) = 44.
+ */
+#define UART_IBRD_VAL 13
+#define UART_FBRD_VAL 44
+
+#define PL011_BASE 0x09000000UL
+
+static int console_uart_ok;
+
+static void uart_putc(char c)
+{
+ volatile uint32_t *fr = (volatile uint32_t *)(PL011_BASE + UART_FR);
+ volatile uint32_t *dr = (volatile uint32_t *)(PL011_BASE + UART_DR);
+
+ /* TXFF can happen mid line on slow consoles, wait it out */
+ while (*fr & UART_FR_TXFF)
+ ;
+ *dr = (uint32_t)(unsigned char)c;
+}
+
+/*
+ * pl011 probe and bringup: uart off, baud divisor, fifo on, then
+ * enable tx. the clock here is the qemu virt reference, a real board
+ * overrides the divisors from its clock tree, that is board
+ * territory, the arch part is the sequence.
+ */
+static int uart_init(void)
+{
+ volatile uint32_t *cr = (volatile uint32_t *)(PL011_BASE + UART_CR);
+ volatile uint32_t *ibrd = (volatile uint32_t *)(PL011_BASE + UART_IBRD);
+ volatile uint32_t *fbrd = (volatile uint32_t *)(PL011_BASE + UART_FBRD);
+ volatile uint32_t *lcrh = (volatile uint32_t *)(PL011_BASE + UART_LCRH);
+ volatile uint32_t *imsc = (volatile uint32_t *)(PL011_BASE + UART_IMSC);
+ volatile uint32_t *icr = (volatile uint32_t *)(PL011_BASE + UART_ICR);
+
+ /* disable, mask irq, clear pending, divisors, fifo, enable tx */
+ *cr = 0;
+ *imsc = 0;
+ *icr = 0x7ff;
+ *ibrd = UART_IBRD_VAL;
+ *fbrd = UART_FBRD_VAL;
+ *lcrh = (3 << 5) | (1 << 4); /* 8n1, fifo enabled */
+ *cr = UART_CR_UARTEN | UART_CR_TXE | UART_CR_RXE;
+
+ /* self test write, TXFF clearing means the uart answers */
+ uart_putc('\0');
+ while (*(volatile uint32_t *)(PL011_BASE + UART_FR) & UART_FR_BUSY)
+ ;
+
+ return 0;
+}
+
/*
- * lk's _dprintf sink. printf buffers a line here then hands it to the
- * host, semihosting wants zero terminated strings not counts.
+ * lk's _dprintf sink. printf buffers a line here then hands it to
+ * the sink, semihosting wants zero terminated strings not counts.
*/
#define TB_CONSOLE_MAX 256
@@ -41,10 +112,18 @@ static size_t console_len;
static void console_flush(void)
{
+ size_t i;
+
if (console_len == 0)
return;
- console_buf[console_len] = '\0';
- smh_write0(console_buf);
+
+ if (console_uart_ok) {
+ for (i = 0; i < console_len; i++)
+ uart_putc(console_buf[i]);
+ } else {
+ console_buf[console_len] = '\0';
+ smh_write0(console_buf);
+ }
console_len = 0;
}
@@ -53,7 +132,7 @@ void _putchar(char c)
if (console_len >= TB_CONSOLE_MAX - 1)
console_flush();
if (c == '\n') {
- /* the host terminal wants cr lf, not lf alone */
+ /* terminals want cr lf, not lf alone */
console_buf[console_len++] = '\r';
}
console_buf[console_len++] = c;
@@ -64,11 +143,13 @@ void _putchar(char c)
int tb_console_init(void)
{
/*
- * the probe is one harmless call: SYS_GET_ERRNO with no file
- * handle open. a host answers, bare metal ignores the trap.
+ * try the uart first, real hardware. semihosting is the qemu
+ * dev path, SYS_GET_ERRNO with nothing open, a host answers,
+ * bare metal ignores the trap.
*/
- if (!smh_probe())
- return -1;
+ uart_init();
+ console_uart_ok = 1;
console_len = 0;
+ (void)smh_probe();
return 0;
}
diff --git a/common/dtb_patch.c b/common/dtb_patch.c
new file mode 100644
index 0000000..34d1a6c
--- /dev/null
+++ b/common/dtb_patch.c
@@ -0,0 +1,142 @@
+/* SPDX-License-Identifier: GPL-2.0+ */
+/*
+ * dtb_patch.c - rewrite cpu-release-addr values in a flattened
+ * devicetree, in place, no libfdt, no structural change. the walk
+ * follows the devicetree specification structure, FDT_BEGIN_NODE
+ * then name then properties then children then FDT_END_NODE, all
+ * tokens and lengths big endian, everything 4 byte aligned.
+ *
+ * The bootloader owns the spin gates, the dtb names them, this
+ * writes the real addresses over the build time placeholders.
+ * The enable-method conversion and the placeholder properties are
+ * done at build time on the host, a firmware dtb is a fixed blob,
+ * only the gate addresses depend on where the image actually landed.
+ *
+ * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org>
+ */
+
+#include <string.h>
+#include <endian.h>
+#include <dtb_patch.h>
+
+#define FDT_BEGIN_NODE 1
+#define FDT_END_NODE 2
+#define FDT_PROP 3
+#define FDT_NOP 4
+#define FDT_END 9
+
+static uint32_t be32(const void *p)
+{
+ const uint8_t *b = p;
+ return ((uint32_t)b[0] << 24) | ((uint32_t)b[1] << 16) |
+ ((uint32_t)b[2] << 8) | (uint32_t)b[3];
+}
+
+static void put_be64(void *p, uint64_t v)
+{
+ uint8_t *b = p;
+ b[0] = (uint8_t)(v >> 56);
+ b[1] = (uint8_t)(v >> 48);
+ b[2] = (uint8_t)(v >> 40);
+ b[3] = (uint8_t)(v >> 32);
+ b[4] = (uint8_t)(v >> 24);
+ b[5] = (uint8_t)(v >> 16);
+ b[6] = (uint8_t)(v >> 8);
+ b[7] = (uint8_t)v;
+}
+
+static int name_eq(const char *node, const char *want)
+{
+ while (*node && *node != '@') {
+ if (*node != *want)
+ return 0;
+ node++;
+ want++;
+ }
+ return *want == '\0';
+}
+
+/*
+ * walk and rewrite. returns the number of cpu-release-addr values
+ * written, negative on a malformed blob.
+ */
+int tb_dtb_patch_spin_table(uintptr_t dtb, uintptr_t *gates, int ngates)
+{
+ uint8_t *base = (uint8_t *)dtb;
+ uint32_t off_struct = be32(base + 8);
+ uint32_t off_strings = be32(base + 12);
+ uint8_t *p = base + off_struct;
+ uint8_t *strings = base + off_strings;
+ const char *cur_cpu = NULL;
+ int in_cpus = 0;
+ int written = 0;
+ int depth = 0;
+
+ if (be32(base) != 0xd00dfeed)
+ return -1;
+
+ while (p < base + be32(base + 4)) {
+ uint32_t token = be32(p);
+
+ switch (token) {
+ case FDT_BEGIN_NODE: {
+ char *name = (char *)(p + 4);
+ size_t len = strlen(name) + 1;
+
+ p += 4 + ((len + 3) & ~3);
+ depth++;
+
+ if (depth == 2 && name_eq(name, "cpus")) {
+ in_cpus = 1;
+ } else if (depth == 2) {
+ in_cpus = 0;
+ } else if (in_cpus && depth == 3) {
+ cur_cpu = name;
+ }
+ break;
+ }
+ case FDT_END_NODE:
+ depth--;
+ p += 4;
+ break;
+ case FDT_PROP: {
+ uint32_t plen = be32(p + 4);
+ const char *pname = (char *)strings + be32(p + 8);
+ uint8_t *val = p + 12;
+
+ p += 12 + ((plen + 3) & ~3);
+
+ if (in_cpus && depth == 3 &&
+ strcmp(pname, "cpu-release-addr") == 0 &&
+ plen == 8 && cur_cpu) {
+ long idx = -1;
+ const char *at = strchr(cur_cpu, '@');
+
+ if (at) {
+ idx = 0;
+ while (*at >= '0' && *at <= '9') {
+ at++;
+ }
+ at = strchr(cur_cpu, '@') + 1;
+ while (*at >= '0' && *at <= '9') {
+ idx = idx * 10 + (*at - '0');
+ at++;
+ }
+ }
+ if (idx >= 0 && idx < ngates) {
+ put_be64(val, (uint64_t)gates[idx]);
+ written++;
+ }
+ }
+ break;
+ }
+ case FDT_NOP:
+ p += 4;
+ break;
+ case FDT_END:
+ return written;
+ }
+ }
+
+ return written;
+}
diff --git a/common/load.c b/common/load.c
index cb3b41d..56fc96a 100644
--- a/common/load.c
+++ b/common/load.c
@@ -55,3 +55,30 @@ int tb_load_semihosting(const char *fname, uintptr_t load_addr,
smh_close(fd);
return 0;
}
+
+/*
+ * the initrd path, no header, no placement math, bytes to the
+ * address the dtb /chosen already names.
+ */
+int tb_load_raw(const char *fname, uintptr_t load_addr)
+{
+ long fd, len, ret;
+
+ fd = smh_open(fname, MODE_READ | MODE_BINARY);
+ if (fd < 0)
+ return fd;
+
+ len = smh_flen(fd);
+ if (len < 0) {
+ smh_close(fd);
+ return len;
+ }
+
+ ret = smh_read(fd, (void *)load_addr, len);
+ smh_close(fd);
+
+ if (ret != len)
+ return -6;
+
+ return 0;
+}
diff --git a/common/main.c b/common/main.c
index fb388cb..0786027 100644
--- a/common/main.c
+++ b/common/main.c
@@ -37,6 +37,8 @@
#include <boot.h>
#define TB_VERSION "0.1"
+#define TB_INITRD_ADDR 0x46000000ULL
+#define TB_INITRD_FILE "initrd.cpio.gz"
extern int tb_console_init(void);
@@ -61,12 +63,73 @@ void tashaboot_main(uintptr_t fw_arg)
dprintf(ALWAYS, "tashaboot " TB_VERSION "\n");
+#ifdef TB_ENABLE_MMU
+ {
+ extern int tb_mmu_enable(void);
+ extern int tb_mmu_selftest(void);
+ extern void tb_mmu_disable(void);
+
+ if (tb_mmu_enable() == 0) {
+ if (tb_mmu_selftest() == 0)
+ dprintf(ALWAYS, "mmu: identity map on\n");
+ else
+ dprintf(ALWAYS, "mmu: self test failed, "
+ "running unmapped\n");
+ tb_mmu_disable();
+ }
+ }
+#endif
+
+ {
+ /* spin table gates into the dtb, one per cpu node */
+ extern unsigned long *tb_spin_gates_ptr;
+ extern int tb_dtb_patch_spin_table(uintptr_t dtb,
+ uintptr_t *gates,
+ int ngates);
+ int n;
+
+ if (tb_spin_gates_ptr) {
+ extern unsigned char tb_pen_stamps[8];
+ int c;
+
+ n = tb_dtb_patch_spin_table(fw_arg,
+ tb_spin_gates_ptr, 8);
+ dprintf(ALWAYS, "dtb: %d release addrs patched\n", n);
+
+ /* who made it to the pen */
+ for (c = 1; c < 8; c++) {
+ if (tb_pen_stamps[c])
+ break;
+ }
+ dprintf(ALWAYS, "pen: %s\n",
+ c < 8 ? "secondaries waiting" :
+ "no secondaries parked");
+ }
+ }
+
ret = tb_load_semihosting(TB_BOOTFILE, TB_LOAD_ADDR, &img);
if (ret) {
dprintf(ALWAYS, "load failed (%d), halting\n", ret);
platform_halt();
}
+ /*
+ * the initrd rides after the kernel, the dtb /chosen carries
+ * linux,initrd-start and -end, both already patched in place
+ * with this layout.
+ */
+ {
+ extern int tb_load_raw(const char *fname,
+ uintptr_t load_addr);
+ int r = tb_load_raw(TB_INITRD_FILE, TB_INITRD_ADDR);
+
+ if (r == 0)
+ dprintf(ALWAYS, "initrd at %lx\n",
+ (unsigned long)TB_INITRD_ADDR);
+ else
+ dprintf(ALWAYS, "no initrd (%d)\n", r);
+ }
+
dprintf(ALWAYS, "loaded %llu bytes at %lx, entry %lx\n",
(unsigned long long)img.size, img.load, img.ep);
dprintf(ALWAYS, "jumping\n");
diff --git a/common/mmutest.c b/common/mmutest.c
new file mode 100644
index 0000000..1b60110
--- /dev/null
+++ b/common/mmutest.c
@@ -0,0 +1,77 @@
+/* SPDX-License-Identifier: GPL-2.0+ */
+/*
+ * mmutest.c - self test for the identity map, AT S1E2R translates a
+ * VA through the tables and PAR_EL1 returns the walk result. if the
+ * map is wrong the instruction faults to our vectors instead, so a
+ * clean return with a valid PA in PAR means the tables walk.
+ *
+ * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org>
+ */
+
+#include <asm/mmu.h>
+#include <debug.h>
+
+#define PAR_F (1ULL << 0) /* fault, no translation */
+#define PAR_PA_MASK 0x000ffffffffff000ULL
+
+static uint64_t translate(uint64_t va)
+{
+ uint64_t par, el;
+
+ asm volatile("mrs %0, CurrentEL" : "=r" (el));
+ el >>= 2;
+
+ if (el == 2)
+ asm volatile(
+ "at s1e2r, %1\n"
+ "isb\n"
+ "mrs %0, par_el1\n"
+ : "=r" (par)
+ : "r" (va)
+ : "memory");
+ else
+ asm volatile(
+ "at s1e1r, %1\n"
+ "isb\n"
+ "mrs %0, par_el1\n"
+ : "=r" (par)
+ : "r" (va)
+ : "memory");
+ return par;
+}
+
+static int check(const char *name, uint64_t va)
+{
+ uint64_t par = translate(va);
+
+ if (par & PAR_F) {
+ dprintf(ALWAYS, "mmu: %s faulted (par 0x%016llx)\n",
+ name, (unsigned long long)par);
+ return 1;
+ }
+
+ if ((par & PAR_PA_MASK) != (va & PAR_PA_MASK)) {
+ dprintf(ALWAYS, "mmu: %s pa %llx != va %llx\n",
+ name, (unsigned long long)(par & PAR_PA_MASK),
+ (unsigned long long)va);
+ return 1;
+ }
+
+ dprintf(ALWAYS, "mmu: %s ok, pa %llx\n",
+ name, (unsigned long long)(par & PAR_PA_MASK));
+ return 0;
+}
+
+int tb_mmu_selftest(void)
+{
+ int ret = 0;
+
+ ret |= check("mmio 0x09000000 (uart)", 0x09000000);
+ ret |= check("mmio 0x00000000", 0x00000000);
+ ret |= check("ram 0x40200000 (load)", 0x40200000);
+ ret |= check("ram 0x41000000", 0x41000000);
+ ret |= check("self 0x40080000 (stack guard region, no map)",
+ 0x40080000);
+
+ return ret;
+}
diff --git a/include/boot.h b/include/boot.h
index a5fe3ae..a28b47c 100644
--- a/include/boot.h
+++ b/include/boot.h
@@ -26,6 +26,7 @@ struct tb_image {
int tb_image_setup(uintptr_t image, struct tb_image *img);
/* common/load.c */
+int tb_load_raw(const char *fname, uintptr_t load_addr);
int tb_load_semihosting(const char *fname, uintptr_t load_addr,
struct tb_image *img);
diff --git a/include/dtb_patch.h b/include/dtb_patch.h
new file mode 100644
index 0000000..7fbe225
--- /dev/null
+++ b/include/dtb_patch.h
@@ -0,0 +1,27 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+#ifndef __TB_DTB_PATCH_H
+#define __TB_DTB_PATCH_H
+
+#include <stdint.h>
+
+/*
+ * minimal dtb patcher, no libfdt. walks the flattened devicetree
+ * structure per the devicetree specification (devicetree.dtsi format:
+ * FDT_BEGIN_NODE, name, property, FDT_END_NODE) and rewrites the cpu
+ * nodes for spin table bringup.
+ *
+ * what it does:
+ * /cpus/cpu@N: enable-method = "spin-table"
+ * cpu-release-addr = gate address of core N
+ * /psci: status = "disabled" (so the kernel falls back to the
+ * spin table instead of trying hvc)
+ *
+ * properties are rewritten in place where the space fits, the
+ * enable-method string shrinks, cpu-release-addr reuses the space
+ * of an old value. new properties are appended to the last cpu node
+ * by growing the struct block and moving the strings block.
+ */
+
+int tb_dtb_patch_spin_table(uintptr_t dtb, uintptr_t *gates, int ngates);
+
+#endif /* __TB_DTB_PATCH_H */