From 5ff54a875642962fc857cee00bde17f9a465f1fa Mon Sep 17 00:00:00 2001 From: Bradley Morgan Date: Sun, 4 Oct 2026 10:42:10 +0000 Subject: tashaboot: arm64 bootloader holy shit it's here, Tashaboot, based from arm arm, enjoy reading this masterpiece Signed-off-by: Bradley Morgan --- common/console.c | 164 +++++++++++++++++++++++++++++ common/dtb_find.c | 178 +++++++++++++++++++++++++++++++ common/dtb_grow.c | 177 +++++++++++++++++++++++++++++++ common/dtb_patch.c | 288 ++++++++++++++++++++++++++++++++++++++++++++++++++ common/dtb_reloc.c | 69 ++++++++++++ common/image.c | 77 ++++++++++++++ common/load.c | 86 +++++++++++++++ common/main.c | 300 +++++++++++++++++++++++++++++++++++++++++++++++++++++ common/mmutest.c | 77 ++++++++++++++ 9 files changed, 1416 insertions(+) create mode 100644 common/console.c create mode 100644 common/dtb_find.c create mode 100644 common/dtb_grow.c create mode 100644 common/dtb_patch.c create mode 100644 common/dtb_reloc.c create mode 100644 common/image.c create mode 100644 common/load.c create mode 100644 common/main.c create mode 100644 common/mmutest.c (limited to 'common') diff --git a/common/console.c b/common/console.c new file mode 100644 index 0000000..64e5128 --- /dev/null +++ b/common/console.c @@ -0,0 +1,164 @@ +/* + * console.c - the console dprintf writes to. two sinks: the pl011 + * PrimeCell uart every qemu virt and SBSA board carries (DDI 0183), + * and semihosting SYS_WRITE0, the firmware service on the qemu dev + * path. the uart is the real hardware path, semihosting the dev + * path, the probe at init picks whichever answers. + * + * Copyright (c) 2026 Bradley Morgan + * + * Permission is hereby granted, free of charge, to any person obtaining + * a copy of this software and associated documentation files + * (the "Software"), to deal in the Software without restriction, + * including without limitation the rights to use, copy, modify, merge, + * publish, distribute, sublicense, and/or sell copies of the Software, + * and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be + * included in all copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, + * EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF + * MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. + * IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY + * CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, + * TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE + * SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ + +#include +#include +#include +#include + +/* pl011 register map, DDI 0183, offsets from the base */ +#define UART_DR 0x00 /* data register */ +#define UART_FR 0x18 /* flag register */ +#define UART_FR_BUSY (1 << 3) +#define UART_FR_TXFF (1 << 5) +#define UART_IBRD 0x24 +#define UART_FBRD 0x28 +#define UART_LCRH 0x2c +#define UART_CR 0x30 +#define UART_CR_UARTEN (1 << 0) +#define UART_CR_TXE (1 << 8) +#define UART_CR_RXE (1 << 9) +#define UART_IMSC 0x38 +#define UART_ICR 0x44 + +/* + * 115200 8n1 at a 24 MHz reference clock. IBRD = 24e6 / (16 * 115200) + * = 13, FBRD = int(0.6875 * 64 + 0.5) = 44. + */ +#define UART_IBRD_VAL 13 +#define UART_FBRD_VAL 44 + +/* the base comes from the devicetree walk, the qemu default + * only covers the dev path before the walk runs + */ +static uintptr_t pl011_base = 0x09000000UL; + +void tb_console_set_pl011(uintptr_t base) +{ + if (base) + pl011_base = base; +} + +static int console_uart_ok; + +static void uart_putc(char c) +{ + volatile uint32_t *fr = (volatile uint32_t *)(pl011_base + UART_FR); + volatile uint32_t *dr = (volatile uint32_t *)(pl011_base + UART_DR); + + /* TXFF can happen mid line on slow consoles, wait it out */ + while (*fr & UART_FR_TXFF) + ; + *dr = (uint32_t)(unsigned char)c; +} + +/* + * pl011 probe and bringup: uart off, baud divisor, fifo on, then + * enable tx. the clock here is the qemu virt reference, a real board + * overrides the divisors from its clock tree, that is board + * territory, the arch part is the sequence. + */ +static int uart_init(void) +{ + volatile uint32_t *cr = (volatile uint32_t *)(pl011_base + UART_CR); + volatile uint32_t *ibrd = (volatile uint32_t *)(pl011_base + UART_IBRD); + volatile uint32_t *fbrd = (volatile uint32_t *)(pl011_base + UART_FBRD); + volatile uint32_t *lcrh = (volatile uint32_t *)(pl011_base + UART_LCRH); + volatile uint32_t *imsc = (volatile uint32_t *)(pl011_base + UART_IMSC); + volatile uint32_t *icr = (volatile uint32_t *)(pl011_base + UART_ICR); + + /* disable, mask irq, clear pending, divisors, fifo, enable tx */ + *cr = 0; + *imsc = 0; + *icr = 0x7ff; + *ibrd = UART_IBRD_VAL; + *fbrd = UART_FBRD_VAL; + *lcrh = (3 << 5) | (1 << 4); /* 8n1, fifo enabled */ + *cr = UART_CR_UARTEN | UART_CR_TXE | UART_CR_RXE; + + /* self test write, TXFF clearing means the uart answers */ + uart_putc('\0'); + while (*(volatile uint32_t *)(pl011_base + UART_FR) & UART_FR_BUSY) + ; + + return 0; +} + +/* + * lk's _dprintf sink. printf buffers a line here then hands it to + * the sink, semihosting wants zero terminated strings not counts. + */ +#define TB_CONSOLE_MAX 256 + +static char console_buf[TB_CONSOLE_MAX]; +static size_t console_len; + +static void console_flush(void) +{ + size_t i; + + if (console_len == 0) + return; + + if (console_uart_ok) { + for (i = 0; i < console_len; i++) + uart_putc(console_buf[i]); + } else { + console_buf[console_len] = '\0'; + smh_write0(console_buf); + } + console_len = 0; +} + +void _putchar(char c) +{ + if (console_len >= TB_CONSOLE_MAX - 1) + console_flush(); + if (c == '\n') { + /* terminals want cr lf, not lf alone */ + console_buf[console_len++] = '\r'; + } + console_buf[console_len++] = c; + if (c == '\n') + console_flush(); +} + +int tb_console_init(void) +{ + /* + * try the uart first, real hardware. semihosting is the qemu + * dev path, SYS_GET_ERRNO with nothing open, a host answers, + * bare metal ignores the trap. + */ + uart_init(); + console_uart_ok = 1; + console_len = 0; + (void)smh_probe(); + return 0; +} diff --git a/common/dtb_find.c b/common/dtb_find.c new file mode 100644 index 0000000..29a67f3 --- /dev/null +++ b/common/dtb_find.c @@ -0,0 +1,178 @@ +/* + * dtb_find.c - locate nodes and read reg by walking the flat + * devicetree. the machine tells the firmware where its devices + * live, a bootloader that hardcodes the gic address breaks on + * the first board with a different map. + * + * the walk is the standard token scan, FDT_BEGIN_NODE with a + * matching name at any depth, then the reg property inside, + * the first address/size pair decoded per the parent's cell + * counts, which the root carries in #address-cells and + * #size-cells. + * + * Copyright (C) 2026 Bradley Morgan + */ + +#include +#include +#include + +#define FDT_BEGIN_NODE 1 +#define FDT_END_NODE 2 +#define FDT_PROP 3 +#define FDT_NOP 4 +#define FDT_END 9 + +static uint32_t be32(const void *p) +{ + const uint8_t *b = p; + + return ((uint32_t)b[0] << 24) | ((uint32_t)b[1] << 16) | + ((uint32_t)b[2] << 8) | (uint32_t)b[3]; +} + +static int name_eq(const char *a, const char *b) +{ + while (*a && *a != '@') { + if (*a != *b) + return 0; + a++; + b++; + } + return *b == '\0' || *b == '@'; +} + +/* + * find the first node whose name matches, at any depth. returns + * the offset of its FDT_BEGIN_NODE token or 0 when absent. + */ +static uint32_t fdt_find_node(uintptr_t dtb, const char *name) +{ + uint8_t *basep = (uint8_t *)dtb; + uint32_t off_struct = be32(basep + 8); + uint32_t totalsize = be32(basep + 4); + uint8_t *p = basep + off_struct; + + if (be32(basep) != 0xd00dfeed) + return 0; + + while (p < basep + totalsize) { + uint32_t token = be32(p); + + if (token == FDT_BEGIN_NODE) { + char *n = (char *)(p + 4); + size_t nlen = strlen(n) + 1; + + if (name_eq(n, name)) + return (uint32_t)(p - basep); + p += 4 + ((nlen + 3) & ~3); + } else if (token == FDT_PROP) { + uint32_t plen = be32(p + 4); + + p += 12 + ((plen + 3) & ~3); + } else if (token == FDT_END_NODE || + token == FDT_NOP) { + p += 4; + } else if (token == FDT_END) { + break; + } else { + return 0; + } + } + + return 0; +} + +/* + * read the first reg pair of a node at the given token offset, + * honoring the root cell counts. pairs of 2 or 4 cells are the + * ones machines carry, anything else fails. the caller reads + * more pairs off the returned cursor if it needs them. + */ +int tb_dtb_reg0(uintptr_t dtb, uint32_t node_off, uintptr_t *addr, + size_t *size) +{ + uint8_t *basep = (uint8_t *)dtb; + uint32_t off_strings = be32(basep + 12); + uint8_t *p = basep + node_off; + uint32_t totalsize = be32(basep + 4); + uint32_t ac = 2; + uint32_t sc = 2; + /* + * zero, the node's own FDT_BEGIN_NODE below brings it to + * one and the props inside sit at depth one. starting at + * one instead skips every prop in the node. + */ + int depth_open = 0; + + while (p < basep + totalsize) { + uint32_t token = be32(p); + + if (token == FDT_BEGIN_NODE) { + char *n = (char *)(p + 4); + size_t nlen = strlen(n) + 1; + + depth_open++; + p += 4 + ((nlen + 3) & ~3); + } else if (token == FDT_END_NODE) { + depth_open--; + if (!depth_open) + return -1; + p += 4; + } else if (token == FDT_PROP) { + uint32_t plen = be32(p + 4); + const char *pname = + (char *)basep + off_strings + be32(p + 8); + uint8_t *val = p + 12; + + if (depth_open == 1 && + strcmp(pname, "#address-cells") == 0) + ac = be32(val); + if (depth_open == 1 && + strcmp(pname, "#size-cells") == 0) + sc = be32(val); + if (depth_open == 1 && strcmp(pname, "reg") == 0) { + if (plen >= (ac + sc) * 4) { + uint64_t a = 0; + uint64_t s = 0; + + for (uint32_t i = 0; i < ac; i++) + a = (a << 32) | + be32(val + i * 4); + for (uint32_t i = 0; i < sc; i++) + s = (s << 32) | + be32(val + (ac + i) * 4); + *addr = (uintptr_t)a; + if (size) + *size = (size_t)s; + return 0; + } + return -1; + } + p += 12 + ((plen + 3) & ~3); + } else if (token == FDT_NOP) { + p += 4; + } else if (token == FDT_END) { + return -1; + } else { + return -1; + } + } + + return -1; +} + +/* + * the whole lookup in one call: find the node, read its first + * reg pair. + */ +int tb_dtb_find_reg0(uintptr_t dtb, const char *name, uintptr_t *addr, + size_t *size) +{ + uint32_t off = fdt_find_node(dtb, name); + + if (!off) + return -1; + + return tb_dtb_reg0(dtb, off, addr, size); +} diff --git a/common/dtb_grow.c b/common/dtb_grow.c new file mode 100644 index 0000000..309a43c --- /dev/null +++ b/common/dtb_grow.c @@ -0,0 +1,177 @@ +/* + * dtb_grow.c - add properties to a node in a devicetree that has + * room, the relocated copy from dtb_reloc.c. the insert point is + * the node's FDT_END_NODE token, everything after it moves up by + * the inserted size, the header totalsize tracks it. + * + * the insert is safe when the node sits at the end of the struct + * block, which is the common shape, /chosen is created last by + * firmware and the tail behind it is two end tokens and the + * block end. the strings block sits after the grow room and + * never moves. + * + * Copyright (C) 2026 Bradley Morgan + */ + +#include +#include +#include +#include + +#define FDT_BEGIN_NODE 1 +#define FDT_END_NODE 2 +#define FDT_PROP 3 +#define FDT_NOP 4 +#define FDT_END 9 + +static uint32_t be32(const void *p) +{ + const uint8_t *b = p; + + return ((uint32_t)b[0] << 24) | ((uint32_t)b[1] << 16) | + ((uint32_t)b[2] << 8) | (uint32_t)b[3]; +} + +static void put_be32(void *p, uint32_t v) +{ + uint8_t *b = p; + + b[0] = (uint8_t)(v >> 24); + b[1] = (uint8_t)(v >> 16); + b[2] = (uint8_t)(v >> 8); + b[3] = (uint8_t)v; +} + +static int name_eq(const char *a, const char *b) +{ + while (*a && *a != '@') { + if (*a != *b) + return 0; + a++; + b++; + } + return *b == '\0' || *b == '@'; +} + +/* + * insert one property into /chosen before its end token. value is + * copied as raw cells, len the byte count. name lands in the free + * space after the strings block. returns 0 or -1. + */ +int tb_dtb_add_chosen_prop(uintptr_t dtb, const char *name, + const void *val, size_t len) +{ + uint8_t *basep = (uint8_t *)dtb; + uint32_t off_struct = be32(basep + 8); + uint32_t off_strings = be32(basep + 12); + uint32_t totalsize = be32(basep + 4); + uint8_t *p = basep + off_struct; + uint8_t *ins; + size_t name_len = strlen(name) + 1; + size_t prop_size; + int depth = 0; + int in_chosen = 0; + + if (be32(basep) != 0xd00dfeed) + return -1; + + /* find the chosen node's end token, one level under the root */ + while (p < basep + totalsize) { + uint32_t token = be32(p); + + if (token == FDT_BEGIN_NODE) { + char *n = (char *)(p + 4); + size_t nlen = strlen(n) + 1; + + depth++; + if (depth == 2 && name_eq(n, "chosen")) + in_chosen = 1; + p += 4 + ((nlen + 3) & ~3); + } else if (token == FDT_END_NODE) { + if (in_chosen && depth == 2) { + ins = p; + break; + } + depth--; + p += 4; + } else if (token == FDT_PROP) { + uint32_t plen = be32(p + 4); + + p += 12 + ((plen + 3) & ~3); + } else if (token == FDT_NOP) { + p += 4; + } else if (token == FDT_END) { + break; + } else { + return -1; + } + } + + if (!ins) + return -2; + + ins = p; + + /* + * the insert: the strings block moves up by prop_size so the + * struct block can grow into its old place, the struct tail + * after chosen moves up by prop_size, the new name lands at + * the end of the moved strings block, and totalsize covers + * both. prop name offsets are strings relative so they keep + * resolving after the move. + */ + { + prop_size = 12 + ((len + 3) & ~3); + size_t strings_len = (size_t)be32(basep + 32); + + /* strings block up by prop_size */ + for (size_t i = strings_len; i > 0; i--) + basep[off_strings + prop_size + i - 1] = + basep[off_strings + i - 1]; + + /* struct tail after the insert point up by prop_size */ + { + size_t tail = (size_t)(basep + off_strings - ins); + + for (size_t i = tail; i > 0; i--) + ins[i + prop_size - 1] = ins[i - 1]; + } + + /* the prop token, name offset = old strings length */ + put_be32(ins, FDT_PROP); + put_be32(ins + 4, (uint32_t)len); + put_be32(ins + 8, (uint32_t)strings_len); + for (size_t i = 0; i < len; i++) + ins[12 + i] = ((const uint8_t *)val)[i]; + for (size_t i = len; i < ((len + 3) & ~3); i++) + ins[12 + i] = 0; + + /* the name at the end of the moved strings block */ + for (size_t i = 0; i < name_len; i++) + basep[off_strings + prop_size + strings_len + i] = + name[i]; + + /* + * size_dt_struct bounds the token walk, libfdt + * rejects anything past it as BADSTRUCTURE. it grows + * by the prop size here, the strings size by the name + * length, totalsize by both. + */ + put_be32(basep + 4, totalsize + (uint32_t)prop_size + + (uint32_t)name_len); + put_be32(basep + 12, off_strings + (uint32_t)prop_size); + put_be32(basep + 36, be32(basep + 36) + (uint32_t)prop_size); + /* + * size_dt_strings must grow too, libfdt validates + * name offsets against it and rejects the whole tree + * when the new names sit past the declared end. the + * kernel's early parser is the same libfdt, a stale + * field there means no memory node and a page table + * panic before the first print. + */ + put_be32(basep + 32, (uint32_t)strings_len + + (uint32_t)name_len); + } + + return 0; +} diff --git a/common/dtb_patch.c b/common/dtb_patch.c new file mode 100644 index 0000000..cd9a1f3 --- /dev/null +++ b/common/dtb_patch.c @@ -0,0 +1,288 @@ +/* SPDX-License-Identifier: GPL-2.0+ */ +/* + * dtb_patch.c - rewrite cpu-release-addr values in a flattened + * devicetree, in place, no libfdt, no structural change. the walk + * follows the devicetree specification structure, FDT_BEGIN_NODE + * then name then properties then children then FDT_END_NODE, all + * tokens and lengths big endian, everything 4 byte aligned. + * + * The bootloader owns the spin gates, the dtb names them, this + * writes the real addresses over the build time placeholders. + * The enable-method conversion and the placeholder properties are + * done at build time on the host, a firmware dtb is a fixed blob, + * only the gate addresses depend on where the image actually landed. + * + * Copyright (C) 2026 Bradley Morgan + */ + +#include +#include +#include + +#define FDT_BEGIN_NODE 1 +#define FDT_END_NODE 2 +#define FDT_PROP 3 +#define FDT_NOP 4 +#define FDT_END 9 + +static uint32_t be32(const void *p) +{ + const uint8_t *b = p; + return ((uint32_t)b[0] << 24) | ((uint32_t)b[1] << 16) | + ((uint32_t)b[2] << 8) | (uint32_t)b[3]; +} + +static void put_be32(void *p, uint32_t v) +{ + uint8_t *b = p; + b[0] = (uint8_t)(v >> 24); + b[1] = (uint8_t)(v >> 16); + b[2] = (uint8_t)(v >> 8); + b[3] = (uint8_t)v; +} + +static void put_be64(void *p, uint64_t v) +{ + uint8_t *b = p; + b[0] = (uint8_t)(v >> 56); + b[1] = (uint8_t)(v >> 48); + b[2] = (uint8_t)(v >> 40); + b[3] = (uint8_t)(v >> 32); + b[4] = (uint8_t)(v >> 24); + b[5] = (uint8_t)(v >> 16); + b[6] = (uint8_t)(v >> 8); + b[7] = (uint8_t)v; +} + +static int name_eq(const char *node, const char *want) +{ + while (*node && *node != '@') { + if (*node != *want) + return 0; + node++; + want++; + } + return *want == '\0'; +} + +/* + * rewrite /memory reg with the RAM the bootloader actually sees. + * the value is two u32 cells, base and size, addresses above 4GB + * need the parent #address-cells respected, virt is below 4GB and + * 2 cells for size. returns 0 on success. + */ +int tb_dtb_patch_memory(uintptr_t dtb, uint64_t base, uint64_t size) +{ + uint8_t *basep = (uint8_t *)dtb; + uint32_t off_struct = be32(basep + 8); + uint32_t off_strings = be32(basep + 12); + uint8_t *p = basep + off_struct; + uint8_t *strings = basep + off_strings; + const char *cur_node = NULL; + int depth = 0; + + if (be32(basep) != 0xd00dfeed) + return -1; + + while (p < basep + be32(basep + 4)) { + uint32_t token = be32(p); + + switch (token) { + case FDT_BEGIN_NODE: { + char *name = (char *)(p + 4); + size_t len = strlen(name) + 1; + + p += 4 + ((len + 3) & ~3); + depth++; + cur_node = name; + break; + } + case FDT_END_NODE: + depth--; + p += 4; + break; + case FDT_PROP: { + uint32_t plen = be32(p + 4); + const char *pname = (char *)strings + be32(p + 8); + uint8_t *val = p + 12; + + p += 12 + ((plen + 3) & ~3); + + if (depth == 2 && name_eq(cur_node, "memory") && + strcmp(pname, "reg") == 0 && plen >= 16) { + /* + * #address-cells 2, #size-cells 2, the + * reg is four cells, base hi lo and + * size hi lo, below 4GB the hi cells + * are zero. + */ + put_be32(val, (uint32_t)(base >> 32)); + put_be32(val + 4, (uint32_t)base); + put_be32(val + 8, (uint32_t)(size >> 32)); + put_be32(val + 12, (uint32_t)size); + return 0; + } + break; + } + case FDT_NOP: + p += 4; + break; + case FDT_END: + return -2; + } + } + + return -3; +} + +/* + * walk and rewrite. returns the number of cpu-release-addr values + * written, negative on a malformed blob. + */ +int tb_dtb_patch_spin_table(uintptr_t dtb, uintptr_t *gates, int ngates) +{ + uint8_t *base = (uint8_t *)dtb; + uint32_t off_struct = be32(base + 8); + uint32_t off_strings = be32(base + 12); + uint8_t *p = base + off_struct; + uint8_t *strings = base + off_strings; + const char *cur_cpu = NULL; + int in_cpus = 0; + int written = 0; + int depth = 0; + + if (be32(base) != 0xd00dfeed) + return -1; + + while (p < base + be32(base + 4)) { + uint32_t token = be32(p); + + switch (token) { + case FDT_BEGIN_NODE: { + char *name = (char *)(p + 4); + size_t len = strlen(name) + 1; + + p += 4 + ((len + 3) & ~3); + depth++; + + if (depth == 2 && name_eq(name, "cpus")) { + in_cpus = 1; + } else if (depth == 2) { + in_cpus = 0; + } else if (in_cpus && depth == 3) { + cur_cpu = name; + } + break; + } + case FDT_END_NODE: + depth--; + p += 4; + break; + case FDT_PROP: { + uint32_t plen = be32(p + 4); + const char *pname = (char *)strings + be32(p + 8); + uint8_t *val = p + 12; + + p += 12 + ((plen + 3) & ~3); + + if (in_cpus && depth == 3 && + strcmp(pname, "cpu-release-addr") == 0 && + plen == 8 && cur_cpu) { + long idx = -1; + const char *at = strchr(cur_cpu, '@'); + + if (at) { + idx = 0; + while (*at >= '0' && *at <= '9') { + at++; + } + at = strchr(cur_cpu, '@') + 1; + while (*at >= '0' && *at <= '9') { + idx = idx * 10 + (*at - '0'); + at++; + } + } + if (idx >= 0 && idx < ngates) { + put_be64(val, (uint64_t)gates[idx]); + written++; + } + } + break; + } + case FDT_NOP: + p += 4; + break; + case FDT_END: + return written; + } + } + + return written; +} + +/* + * tell the kernel where the initrd landed. /chosen is created by + * the machine firmware, the two cells exist when an initrd was + * already staged, we overwrite them in place. depth 2 under the + * root, node name "chosen". + */ +int tb_dtb_patch_initrd(uintptr_t dtb, uint64_t start, uint64_t end) +{ + uint8_t *basep = (uint8_t *)dtb; + uint32_t off_struct = be32(basep + 8); + uint32_t off_strings = be32(basep + 12); + uint8_t *p = basep + off_struct; + uint8_t *strings = basep + off_strings; + const char *cur_node = NULL; + int depth = 0; + + if (be32(basep) != 0xd00dfeed) + return -1; + + while (p < basep + be32(basep + 4)) { + uint32_t token = be32(p); + + switch (token) { + case FDT_BEGIN_NODE: { + char *name = (char *)(p + 4); + size_t len = strlen(name) + 1; + + p += 4 + ((len + 3) & ~3); + depth++; + cur_node = name; + break; + } + case FDT_END_NODE: + depth--; + p += 4; + break; + case FDT_PROP: { + uint32_t plen = be32(p + 4); + const char *pname = (char *)strings + be32(p + 8); + uint8_t *val = p + 12; + + p += 12 + ((plen + 3) & ~3); + + if (depth == 2 && name_eq(cur_node, "chosen") && + strcmp(pname, "linux,initrd-start") == 0 && + plen >= 8) { + put_be64(val, start); + } + if (depth == 2 && name_eq(cur_node, "chosen") && + strcmp(pname, "linux,initrd-end") == 0 && + plen >= 8) { + put_be64(val, end); + return 0; + } + break; + } + case FDT_NOP: + p += 4; + break; + case FDT_END: + return -2; + } + } + + return -2; +} diff --git a/common/dtb_reloc.c b/common/dtb_reloc.c new file mode 100644 index 0000000..c3e0fe5 --- /dev/null +++ b/common/dtb_reloc.c @@ -0,0 +1,69 @@ +/* + * dtb_reloc.c - grow the devicetree the way libfdt does, in a + * buffer with room to spare. firmware cannot edit a packed fdt + * in place, new properties shift everything behind them, so the + * blob is copied into scratch verbatim, the free space after + * totalsize is the room the insert code shifts into, then the + * walkers patch the copy and the kernel gets its address. + * + * The layout follows the devicetree specification: header, + * struct block, strings block, free space. The rebuild copies + * header, struct, strings, fixes the offsets in the new header, + * and leaves the gap between struct and strings as the room new + * properties will consume. + * + * Copyright (C) 2026 Bradley Morgan + */ + +#include +#include +#include +#include + +#define FDT_BEGIN_NODE 1 +#define FDT_END_NODE 2 +#define FDT_PROP 3 +#define FDT_NOP 4 +#define FDT_END 9 + +static uint32_t be32(const void *p) +{ + const uint8_t *b = p; + + return ((uint32_t)b[0] << 24) | ((uint32_t)b[1] << 16) | + ((uint32_t)b[2] << 8) | (uint32_t)b[3]; +} + +/* + * copy the blob into the scratch, grow bytes of headroom after + * the end. returns the new blob address or 0 on a short buffer. + */ +uintptr_t tb_dtb_relocate(uintptr_t dtb, void *scratch, size_t scratch_size, + size_t grow) +{ + uint8_t *in = (uint8_t *)dtb; + uint8_t *out = scratch; + uint32_t totalsize; + + if (be32(in) != 0xd00dfeed) + return 0; + + totalsize = be32(in + 4); + + if (scratch_size < (size_t)totalsize + grow) + return 0; + + /* + * verbatim copy, byte for byte. the grow room is the free + * scratch after totalsize, the insert code shifts the + * strings block into it. an interior gap between the + * struct and strings blocks only invites the walkers to + * count it as tree. + */ + for (uint32_t i = 0; i < totalsize; i++) + out[i] = in[i]; + + (void)grow; + + return (uintptr_t)out; +} diff --git a/common/image.c b/common/image.c new file mode 100644 index 0000000..640eab4 --- /dev/null +++ b/common/image.c @@ -0,0 +1,77 @@ +/* SPDX-License-Identifier: GPL-2.0+ */ +/* + * image.c - arm64 kernel Image validation and placement. + * + * The relocation rules are the ones from the kernel boot protocol, + * including the pre a2c1d73b94ed quirks, same math u-boot's + * booti_setup() runs. + * + * Copyright (C) 2026 Bradley Morgan + */ + +#include +#include +#include +#include + +#define LINUX_ARM64_IMAGE_MAGIC 0x644d5241 /* "ARM\x64" */ + +#define SZ_16M 0x01000000 +#define SZ_2M 0x00200000 + +/* the 64 byte header from Documentation/arch/arm64/booting.rst */ +struct Image_header { + uint32 code0; /* executable */ + uint32 code1; /* unused */ + uint64 text_offset; /* load offset, LE */ + uint64 image_size; /* size, LE */ + uint64 flags; /* bit 3: relocatable */ + uint64 res1; + uint64 res2; + uint64 res3; + uint32 magic; /* "ARM\x64" */ + uint32 res4; +}; + +int tb_image_setup(uintptr_t image, struct tb_image *img) +{ + const struct Image_header *ih = (const struct Image_header *)image; + uint64_t image_size, text_offset; + + if (le32_to_cpu(ih->magic) != LINUX_ARM64_IMAGE_MAGIC) + return -1; + + if (le64_to_cpu(ih->image_size) == 0) { + /* ancient image, no size field, assume the old defaults */ + image_size = SZ_16M; + text_offset = 0x80000; + } else { + image_size = le64_to_cpu(ih->image_size); + text_offset = le64_to_cpu(ih->text_offset); + } + + /* + * flag bit 3 says the image can live anywhere, honour where it + * already is. otherwise the base must be 2MB aligned, the + * physical offset from there is text_offset. + */ + if (le64_to_cpu(ih->flags) & (1ULL << 3)) { + uintptr_t base = image - text_offset; + + img->load = ((base + SZ_2M - 1) & ~(uintptr_t)(SZ_2M - 1)) + + text_offset; + } else { + /* + * no relocate flag: the image must sit text_offset from a + * 2MB aligned base. it is already staged at the fixed + * address, treat its own position as the answer. + */ + img->load = image; + } + + /* the whole image, header included, lives at load, entry is code0 */ + img->ep = img->load; + img->size = image_size; + + return 0; +} diff --git a/common/load.c b/common/load.c new file mode 100644 index 0000000..3a0f8b6 --- /dev/null +++ b/common/load.c @@ -0,0 +1,86 @@ +/* SPDX-License-Identifier: GPL-2.0+ */ +/* + * load.c - pull the payload into RAM. semihosting is the whole story + * for now, the bios INT 13h of this loader: the host serves the file, + * we read it at the fixed address and let the image header place it. + * + * Copyright (C) 2026 Bradley Morgan + */ + +#include +#include +#include +#include +#include +#include + +int tb_load_semihosting(const char *fname, uintptr_t load_addr, + struct tb_image *img) +{ + long fd, len, ret; + + fd = smh_open(fname, MODE_READ | MODE_BINARY); + if (fd < 0) + return fd; + + len = smh_flen(fd); + if (len < 0) { + smh_close(fd); + return len; + } + + /* header first so placement is known before the big copy */ + ret = smh_read(fd, (void *)load_addr, 64); + if (ret != 64) { + smh_close(fd); + return -3; + } + + if (tb_image_setup(load_addr, img)) { + smh_close(fd); + return -4; + } + + if (img->load != load_addr) { + /* the image wants to sit elsewhere, copy the header there */ + memmove((void *)img->load, (void *)load_addr, 64); + } + + ret = smh_read(fd, (void *)(img->load + 64), len - 64); + if (ret != len - 64) { + smh_close(fd); + return -5; + } + + smh_close(fd); + return 0; +} + +/* + * the initrd path, no header, no placement math, bytes to the + * address the dtb /chosen already names. + */ +int tb_load_raw(const char *fname, uintptr_t load_addr, size_t *sizep) +{ + long fd, len, ret; + + fd = smh_open(fname, MODE_READ | MODE_BINARY); + if (fd < 0) + return fd; + + len = smh_flen(fd); + if (len < 0) { + smh_close(fd); + return len; + } + + ret = smh_read(fd, (void *)load_addr, len); + smh_close(fd); + + if (ret != len) + return -6; + + if (sizep) + *sizep = (size_t)len; + return 0; +} diff --git a/common/main.c b/common/main.c new file mode 100644 index 0000000..ea85be0 --- /dev/null +++ b/common/main.c @@ -0,0 +1,300 @@ +/* + * main.c - the C entry. console up first, then load the payload and + * jump. called from start.S with x0 = whatever the firmware passed. + * + * the osdev model, arm64: the firmware services do the work, the + * kernel goes at a fixed known address, whatever x0 we were handed + * goes straight through to the payload. + * + * Copyright (c) 2026 Bradley Morgan + * + * Permission is hereby granted, free of charge, to any person obtaining + * a copy of this software and associated documentation files + * (the "Software"), to deal in the Software without restriction, + * including without limitation the rights to use, copy, modify, merge, + * publish, distribute, sublicense, and/or sell copies of the Software, + * and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be + * included in all copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, + * EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF + * MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. + * IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY + * CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, + * TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE + * SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ + +#include +#include +#include +#include +#include +#include +#include + +#define TB_VERSION "0.1" +#define TB_INITRD_ADDR 0x46000000ULL +#define TB_INITRD_FILE "initrd.cpio.gz" + +extern int tb_console_init(void); +extern void tb_system_reset(void); + +/* + * the payload goes 16MB clear of wherever this bootloader is + * actually running, ADR knows the runtime base and qemu is free + * to place us anywhere. hardcoding 0x40200000 smashed our own + * image when qemu loaded us there. + */ +extern char __image_copy_end[]; +#define TB_LOAD_ADDR ((uintptr_t)__image_copy_end + (16ULL << 20)) + +/* the file semihosting serves as the payload */ +#define TB_BOOTFILE "Image" + +extern void __NO_RETURN tb_boot_linux(uintptr_t ep, uintptr_t fw_arg); +extern uintptr_t tb_dtb_relocate(uintptr_t dtb, void *scratch, + size_t scratch_size, size_t grow); +extern int tb_gic_init(uintptr_t gicd, uintptr_t gicc); +extern int tb_dtb_add_chosen_prop(uintptr_t dtb, const char *name, + const void *val, size_t len); +static size_t initrd_size; + +void tashaboot_main(uintptr_t fw_arg) +{ + struct tb_image img; + int ret; + + /* + * the console uart the machine named, before the first + * print. the base is the first reg pair of the pl011 + * node, the same walk the gic used, no board hardcodes. + */ + { + extern int tb_dtb_find_reg0(uintptr_t dtb, + const char *name, + uintptr_t *addr, size_t *size); + extern void tb_console_set_pl011(uintptr_t base); + uintptr_t uart = 0; + size_t usz = 0; + + if (tb_dtb_find_reg0(fw_arg, "pl011", &uart, &usz) == 0) + tb_console_set_pl011(uart); + } + +#ifdef TB_HW_RECEIPT + { + uint64_t midr, el, cntfrq, mpidr; + + asm volatile("mrs %0, midr_el1" : "=r"(midr)); + asm volatile("mrs %0, CurrentEL" : "=r"(el)); + asm volatile("mrs %0, cntfrq_el0" : "=r"(cntfrq)); + asm volatile("mrs %0, mpidr_el1" : "=r"(mpidr)); + + tb_console_init(); + dprintf(ALWAYS, "tashaboot on real hardware\n"); + dprintf(ALWAYS, "midr %llx el %llx cntfrq %llx mpidr %llx\n", + (unsigned long long)midr, + (unsigned long long)el >> 2, + (unsigned long long)cntfrq, + (unsigned long long)mpidr); + + /* let the console drain before the reset domain hits */ + for (volatile int i = 0; i < 100000000; i++) + ; + + tb_system_reset(); + } +#endif + + if (tb_console_init()) + return; + + dprintf(ALWAYS, "tashaboot " TB_VERSION "\n"); + +#ifdef TB_ENABLE_MMU + { + extern int tb_mmu_enable(void); + extern int tb_mmu_selftest(void); + extern void tb_mmu_disable(void); + + if (tb_mmu_enable() == 0) { + if (tb_mmu_selftest() == 0) + dprintf(ALWAYS, "mmu: identity map on\n"); + else + dprintf(ALWAYS, "mmu: self test failed, " + "running unmapped\n"); + tb_mmu_disable(); + } + } +#endif + + { + /* report the RAM we actually live in, before any mmu */ + extern int tb_dtb_patch_memory(uintptr_t dtb, + uint64_t base, + uint64_t size); + int r; + + r = 0; (void)r; + + { + uint32_t *cells = (uint32_t *)(fw_arg + 0x16c); + int i; + + dprintf(ALWAYS, "cells after:"); + for (i = 0; i < 4; i++) + dprintf(ALWAYS, " %08x", cells[i]); + dprintf(ALWAYS, "\n"); + } + } + + { + /* spin table gates into the dtb, one per cpu node */ + extern unsigned long *tb_spin_gates_ptr; + extern int tb_dtb_patch_spin_table(uintptr_t dtb, + uintptr_t *gates, + int ngates); + int n; + + if (tb_spin_gates_ptr) { + extern unsigned char tb_pen_stamps[8]; + int c; + + n = tb_dtb_patch_spin_table(fw_arg, + tb_spin_gates_ptr, 8); + dprintf(ALWAYS, "dtb: %d release addrs patched\n", n); + + /* who made it to the pen */ + for (c = 1; c < 8; c++) { + if (tb_pen_stamps[c]) + break; + } + dprintf(ALWAYS, "pen: %s\n", + c < 8 ? "secondaries waiting" : + "no secondaries parked"); + } + } + + ret = tb_load_semihosting(TB_BOOTFILE, TB_LOAD_ADDR, &img); + if (ret) { + dprintf(ALWAYS, "load failed (%d), halting\n", ret); + platform_halt(); + } + + /* + * firmware owns the devicetree it hands the kernel. ours + * relocates into scratch with grow room, then the chosen + * properties are added there and the kernel gets the new + * address, the same flow libfdt firmware uses. + */ + { + /* + * the scratch lives at a fixed free address, clear of + * our image, the payload, and the kernel relocation + * zone. a bss array would sit inside 0x40080000+ and + * the kernel overwrites it while copying itself. + */ + uint8_t *dtb_scratch = (uint8_t *)0x45000000ULL; + uintptr_t newdtb; + + newdtb = tb_dtb_relocate(fw_arg, dtb_scratch, + 0x10000, 0x200); + if (!newdtb) { + dprintf(ALWAYS, "dtb: relocate failed\n"); + platform_halt(); + } + + fw_arg = newdtb; + + /* + * the interrupt controller the machine told us + * about, found by name, the reg pair read with the + * root cell counts. the gic goes into the defined + * off state before the kernel brings its own irq + * handling up. + */ + { + extern int tb_dtb_find_reg0(uintptr_t dtb, + const char *name, + uintptr_t *addr, + size_t *size); + uintptr_t gicd = 0; + uintptr_t gicc = 0; + size_t sz = 0; + + if (tb_dtb_find_reg0(fw_arg, "intc", &gicd, &sz) == 0 && + sz >= 0x10000) { + gicc = gicd + 0x10000; + tb_gic_init(gicd, gicc); + dprintf(ALWAYS, "gic: %lx off\n", + (unsigned long)gicd); + } + } + } + + /* + * the initrd rides after the kernel, the dtb /chosen carries + * linux,initrd-start and -end, both already patched in place + * with this layout. + */ + { + int r = tb_load_raw(TB_INITRD_FILE, TB_INITRD_ADDR, + &initrd_size); + + if (r == 0) + dprintf(ALWAYS, "initrd at %lx, %lx bytes\n", + (unsigned long)TB_INITRD_ADDR, + (unsigned long)initrd_size); + else + dprintf(ALWAYS, "no initrd (%d)\n", r); + } + + /* + * the chosen properties, written now that the initrd size + * is known. the cells are big endian, the fdt is a big + * endian format end to end. + */ + { + uint8_t start_cells[8], end_cells[8]; + uint64_t start = TB_INITRD_ADDR; + uint64_t end = TB_INITRD_ADDR + initrd_size; + int a, b; + + for (int i = 0; i < 8; i++) { + start_cells[i] = (uint8_t)(start >> (56 - 8 * i)); + end_cells[i] = (uint8_t)(end >> (56 - 8 * i)); + } + a = tb_dtb_add_chosen_prop(fw_arg, "linux,initrd-start", + start_cells, 8); + b = tb_dtb_add_chosen_prop(fw_arg, "linux,initrd-end", + end_cells, 8); + dprintf(ALWAYS, "dtb: initrd props %d %d\n", a, b); + } + + + + dprintf(ALWAYS, "loaded %llu bytes at %lx, entry %lx\n", + (unsigned long long)img.size, img.load, img.ep); + +#ifdef TB_TEST_SMC + { + register uint64_t r0 asm("x0") = 0x84000000; + register uint64_t r1 asm("x1") = 0; + register uint64_t r2 asm("x2") = 0; + register uint64_t r3 asm("x3") = 0; + + asm volatile("smc #0" + : "+r"(r0), "+r"(r1), "+r"(r2), "+r"(r3)); + dprintf(ALWAYS, "smc conduit: psci version %lx\n", + (unsigned long)r0); + } +#endif + + dprintf(ALWAYS, "jumping\n"); + + tb_boot_linux(img.ep, fw_arg); +} diff --git a/common/mmutest.c b/common/mmutest.c new file mode 100644 index 0000000..1b60110 --- /dev/null +++ b/common/mmutest.c @@ -0,0 +1,77 @@ +/* SPDX-License-Identifier: GPL-2.0+ */ +/* + * mmutest.c - self test for the identity map, AT S1E2R translates a + * VA through the tables and PAR_EL1 returns the walk result. if the + * map is wrong the instruction faults to our vectors instead, so a + * clean return with a valid PA in PAR means the tables walk. + * + * Copyright (C) 2026 Bradley Morgan + */ + +#include +#include + +#define PAR_F (1ULL << 0) /* fault, no translation */ +#define PAR_PA_MASK 0x000ffffffffff000ULL + +static uint64_t translate(uint64_t va) +{ + uint64_t par, el; + + asm volatile("mrs %0, CurrentEL" : "=r" (el)); + el >>= 2; + + if (el == 2) + asm volatile( + "at s1e2r, %1\n" + "isb\n" + "mrs %0, par_el1\n" + : "=r" (par) + : "r" (va) + : "memory"); + else + asm volatile( + "at s1e1r, %1\n" + "isb\n" + "mrs %0, par_el1\n" + : "=r" (par) + : "r" (va) + : "memory"); + return par; +} + +static int check(const char *name, uint64_t va) +{ + uint64_t par = translate(va); + + if (par & PAR_F) { + dprintf(ALWAYS, "mmu: %s faulted (par 0x%016llx)\n", + name, (unsigned long long)par); + return 1; + } + + if ((par & PAR_PA_MASK) != (va & PAR_PA_MASK)) { + dprintf(ALWAYS, "mmu: %s pa %llx != va %llx\n", + name, (unsigned long long)(par & PAR_PA_MASK), + (unsigned long long)va); + return 1; + } + + dprintf(ALWAYS, "mmu: %s ok, pa %llx\n", + name, (unsigned long long)(par & PAR_PA_MASK)); + return 0; +} + +int tb_mmu_selftest(void) +{ + int ret = 0; + + ret |= check("mmio 0x09000000 (uart)", 0x09000000); + ret |= check("mmio 0x00000000", 0x00000000); + ret |= check("ram 0x40200000 (load)", 0x40200000); + ret |= check("ram 0x41000000", 0x41000000); + ret |= check("self 0x40080000 (stack guard region, no map)", + 0x40080000); + + return ret; +} -- cgit v1.2.3