diff options
Diffstat (limited to 'common')
| -rw-r--r-- | common/console.c | 107 | ||||
| -rw-r--r-- | common/dtb_grow.c | 177 | ||||
| -rw-r--r-- | common/dtb_patch.c | 288 | ||||
| -rw-r--r-- | common/dtb_reloc.c | 69 | ||||
| -rw-r--r-- | common/load.c | 29 | ||||
| -rw-r--r-- | common/main.c | 147 | ||||
| -rw-r--r-- | common/mmutest.c | 77 |
7 files changed, 878 insertions, 16 deletions
diff --git a/common/console.c b/common/console.c index a7108ba..bdd24c1 100644 --- a/common/console.c +++ b/common/console.c @@ -1,8 +1,9 @@ /* - * console.c - the console dprintf writes to. semihosting SYS_WRITE0, - * the firmware service on the qemu dev path, the arm64 stand-in for - * the bios teletype the osdev loaders use. on real hardware this is - * the one file that changes. + * console.c - the console dprintf writes to. two sinks: the pl011 + * PrimeCell uart every qemu virt and SBSA board carries (DDI 0183), + * and semihosting SYS_WRITE0, the firmware service on the qemu dev + * path. the uart is the real hardware path, semihosting the dev + * path, the probe at init picks whichever answers. * * Copyright (c) 2026 Bradley Morgan <brads@mainlining.org> * @@ -27,12 +28,82 @@ */ #include <sys/types.h> +#include <stdint.h> #include <debug.h> #include <semihosting.h> +/* pl011 register map, DDI 0183, offsets from the base */ +#define UART_DR 0x00 /* data register */ +#define UART_FR 0x18 /* flag register */ +#define UART_FR_BUSY (1 << 3) +#define UART_FR_TXFF (1 << 5) +#define UART_IBRD 0x24 +#define UART_FBRD 0x28 +#define UART_LCRH 0x2c +#define UART_CR 0x30 +#define UART_CR_UARTEN (1 << 0) +#define UART_CR_TXE (1 << 8) +#define UART_CR_RXE (1 << 9) +#define UART_IMSC 0x38 +#define UART_ICR 0x44 + +/* + * 115200 8n1 at a 24 MHz reference clock. IBRD = 24e6 / (16 * 115200) + * = 13, FBRD = int(0.6875 * 64 + 0.5) = 44. + */ +#define UART_IBRD_VAL 13 +#define UART_FBRD_VAL 44 + +#define PL011_BASE 0x09000000UL + +static int console_uart_ok; + +static void uart_putc(char c) +{ + volatile uint32_t *fr = (volatile uint32_t *)(PL011_BASE + UART_FR); + volatile uint32_t *dr = (volatile uint32_t *)(PL011_BASE + UART_DR); + + /* TXFF can happen mid line on slow consoles, wait it out */ + while (*fr & UART_FR_TXFF) + ; + *dr = (uint32_t)(unsigned char)c; +} + +/* + * pl011 probe and bringup: uart off, baud divisor, fifo on, then + * enable tx. the clock here is the qemu virt reference, a real board + * overrides the divisors from its clock tree, that is board + * territory, the arch part is the sequence. + */ +static int uart_init(void) +{ + volatile uint32_t *cr = (volatile uint32_t *)(PL011_BASE + UART_CR); + volatile uint32_t *ibrd = (volatile uint32_t *)(PL011_BASE + UART_IBRD); + volatile uint32_t *fbrd = (volatile uint32_t *)(PL011_BASE + UART_FBRD); + volatile uint32_t *lcrh = (volatile uint32_t *)(PL011_BASE + UART_LCRH); + volatile uint32_t *imsc = (volatile uint32_t *)(PL011_BASE + UART_IMSC); + volatile uint32_t *icr = (volatile uint32_t *)(PL011_BASE + UART_ICR); + + /* disable, mask irq, clear pending, divisors, fifo, enable tx */ + *cr = 0; + *imsc = 0; + *icr = 0x7ff; + *ibrd = UART_IBRD_VAL; + *fbrd = UART_FBRD_VAL; + *lcrh = (3 << 5) | (1 << 4); /* 8n1, fifo enabled */ + *cr = UART_CR_UARTEN | UART_CR_TXE | UART_CR_RXE; + + /* self test write, TXFF clearing means the uart answers */ + uart_putc('\0'); + while (*(volatile uint32_t *)(PL011_BASE + UART_FR) & UART_FR_BUSY) + ; + + return 0; +} + /* - * lk's _dprintf sink. printf buffers a line here then hands it to the - * host, semihosting wants zero terminated strings not counts. + * lk's _dprintf sink. printf buffers a line here then hands it to + * the sink, semihosting wants zero terminated strings not counts. */ #define TB_CONSOLE_MAX 256 @@ -41,10 +112,18 @@ static size_t console_len; static void console_flush(void) { + size_t i; + if (console_len == 0) return; - console_buf[console_len] = '\0'; - smh_write0(console_buf); + + if (console_uart_ok) { + for (i = 0; i < console_len; i++) + uart_putc(console_buf[i]); + } else { + console_buf[console_len] = '\0'; + smh_write0(console_buf); + } console_len = 0; } @@ -53,7 +132,7 @@ void _putchar(char c) if (console_len >= TB_CONSOLE_MAX - 1) console_flush(); if (c == '\n') { - /* the host terminal wants cr lf, not lf alone */ + /* terminals want cr lf, not lf alone */ console_buf[console_len++] = '\r'; } console_buf[console_len++] = c; @@ -64,11 +143,13 @@ void _putchar(char c) int tb_console_init(void) { /* - * the probe is one harmless call: SYS_GET_ERRNO with no file - * handle open. a host answers, bare metal ignores the trap. + * try the uart first, real hardware. semihosting is the qemu + * dev path, SYS_GET_ERRNO with nothing open, a host answers, + * bare metal ignores the trap. */ - if (!smh_probe()) - return -1; + uart_init(); + console_uart_ok = 1; console_len = 0; + (void)smh_probe(); return 0; } diff --git a/common/dtb_grow.c b/common/dtb_grow.c new file mode 100644 index 0000000..309a43c --- /dev/null +++ b/common/dtb_grow.c @@ -0,0 +1,177 @@ +/* + * dtb_grow.c - add properties to a node in a devicetree that has + * room, the relocated copy from dtb_reloc.c. the insert point is + * the node's FDT_END_NODE token, everything after it moves up by + * the inserted size, the header totalsize tracks it. + * + * the insert is safe when the node sits at the end of the struct + * block, which is the common shape, /chosen is created last by + * firmware and the tail behind it is two end tokens and the + * block end. the strings block sits after the grow room and + * never moves. + * + * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org> + */ + +#include <string.h> +#include <endian.h> +#include <boot.h> +#include <dtb_patch.h> + +#define FDT_BEGIN_NODE 1 +#define FDT_END_NODE 2 +#define FDT_PROP 3 +#define FDT_NOP 4 +#define FDT_END 9 + +static uint32_t be32(const void *p) +{ + const uint8_t *b = p; + + return ((uint32_t)b[0] << 24) | ((uint32_t)b[1] << 16) | + ((uint32_t)b[2] << 8) | (uint32_t)b[3]; +} + +static void put_be32(void *p, uint32_t v) +{ + uint8_t *b = p; + + b[0] = (uint8_t)(v >> 24); + b[1] = (uint8_t)(v >> 16); + b[2] = (uint8_t)(v >> 8); + b[3] = (uint8_t)v; +} + +static int name_eq(const char *a, const char *b) +{ + while (*a && *a != '@') { + if (*a != *b) + return 0; + a++; + b++; + } + return *b == '\0' || *b == '@'; +} + +/* + * insert one property into /chosen before its end token. value is + * copied as raw cells, len the byte count. name lands in the free + * space after the strings block. returns 0 or -1. + */ +int tb_dtb_add_chosen_prop(uintptr_t dtb, const char *name, + const void *val, size_t len) +{ + uint8_t *basep = (uint8_t *)dtb; + uint32_t off_struct = be32(basep + 8); + uint32_t off_strings = be32(basep + 12); + uint32_t totalsize = be32(basep + 4); + uint8_t *p = basep + off_struct; + uint8_t *ins; + size_t name_len = strlen(name) + 1; + size_t prop_size; + int depth = 0; + int in_chosen = 0; + + if (be32(basep) != 0xd00dfeed) + return -1; + + /* find the chosen node's end token, one level under the root */ + while (p < basep + totalsize) { + uint32_t token = be32(p); + + if (token == FDT_BEGIN_NODE) { + char *n = (char *)(p + 4); + size_t nlen = strlen(n) + 1; + + depth++; + if (depth == 2 && name_eq(n, "chosen")) + in_chosen = 1; + p += 4 + ((nlen + 3) & ~3); + } else if (token == FDT_END_NODE) { + if (in_chosen && depth == 2) { + ins = p; + break; + } + depth--; + p += 4; + } else if (token == FDT_PROP) { + uint32_t plen = be32(p + 4); + + p += 12 + ((plen + 3) & ~3); + } else if (token == FDT_NOP) { + p += 4; + } else if (token == FDT_END) { + break; + } else { + return -1; + } + } + + if (!ins) + return -2; + + ins = p; + + /* + * the insert: the strings block moves up by prop_size so the + * struct block can grow into its old place, the struct tail + * after chosen moves up by prop_size, the new name lands at + * the end of the moved strings block, and totalsize covers + * both. prop name offsets are strings relative so they keep + * resolving after the move. + */ + { + prop_size = 12 + ((len + 3) & ~3); + size_t strings_len = (size_t)be32(basep + 32); + + /* strings block up by prop_size */ + for (size_t i = strings_len; i > 0; i--) + basep[off_strings + prop_size + i - 1] = + basep[off_strings + i - 1]; + + /* struct tail after the insert point up by prop_size */ + { + size_t tail = (size_t)(basep + off_strings - ins); + + for (size_t i = tail; i > 0; i--) + ins[i + prop_size - 1] = ins[i - 1]; + } + + /* the prop token, name offset = old strings length */ + put_be32(ins, FDT_PROP); + put_be32(ins + 4, (uint32_t)len); + put_be32(ins + 8, (uint32_t)strings_len); + for (size_t i = 0; i < len; i++) + ins[12 + i] = ((const uint8_t *)val)[i]; + for (size_t i = len; i < ((len + 3) & ~3); i++) + ins[12 + i] = 0; + + /* the name at the end of the moved strings block */ + for (size_t i = 0; i < name_len; i++) + basep[off_strings + prop_size + strings_len + i] = + name[i]; + + /* + * size_dt_struct bounds the token walk, libfdt + * rejects anything past it as BADSTRUCTURE. it grows + * by the prop size here, the strings size by the name + * length, totalsize by both. + */ + put_be32(basep + 4, totalsize + (uint32_t)prop_size + + (uint32_t)name_len); + put_be32(basep + 12, off_strings + (uint32_t)prop_size); + put_be32(basep + 36, be32(basep + 36) + (uint32_t)prop_size); + /* + * size_dt_strings must grow too, libfdt validates + * name offsets against it and rejects the whole tree + * when the new names sit past the declared end. the + * kernel's early parser is the same libfdt, a stale + * field there means no memory node and a page table + * panic before the first print. + */ + put_be32(basep + 32, (uint32_t)strings_len + + (uint32_t)name_len); + } + + return 0; +} diff --git a/common/dtb_patch.c b/common/dtb_patch.c new file mode 100644 index 0000000..cd9a1f3 --- /dev/null +++ b/common/dtb_patch.c @@ -0,0 +1,288 @@ +/* SPDX-License-Identifier: GPL-2.0+ */ +/* + * dtb_patch.c - rewrite cpu-release-addr values in a flattened + * devicetree, in place, no libfdt, no structural change. the walk + * follows the devicetree specification structure, FDT_BEGIN_NODE + * then name then properties then children then FDT_END_NODE, all + * tokens and lengths big endian, everything 4 byte aligned. + * + * The bootloader owns the spin gates, the dtb names them, this + * writes the real addresses over the build time placeholders. + * The enable-method conversion and the placeholder properties are + * done at build time on the host, a firmware dtb is a fixed blob, + * only the gate addresses depend on where the image actually landed. + * + * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org> + */ + +#include <string.h> +#include <endian.h> +#include <dtb_patch.h> + +#define FDT_BEGIN_NODE 1 +#define FDT_END_NODE 2 +#define FDT_PROP 3 +#define FDT_NOP 4 +#define FDT_END 9 + +static uint32_t be32(const void *p) +{ + const uint8_t *b = p; + return ((uint32_t)b[0] << 24) | ((uint32_t)b[1] << 16) | + ((uint32_t)b[2] << 8) | (uint32_t)b[3]; +} + +static void put_be32(void *p, uint32_t v) +{ + uint8_t *b = p; + b[0] = (uint8_t)(v >> 24); + b[1] = (uint8_t)(v >> 16); + b[2] = (uint8_t)(v >> 8); + b[3] = (uint8_t)v; +} + +static void put_be64(void *p, uint64_t v) +{ + uint8_t *b = p; + b[0] = (uint8_t)(v >> 56); + b[1] = (uint8_t)(v >> 48); + b[2] = (uint8_t)(v >> 40); + b[3] = (uint8_t)(v >> 32); + b[4] = (uint8_t)(v >> 24); + b[5] = (uint8_t)(v >> 16); + b[6] = (uint8_t)(v >> 8); + b[7] = (uint8_t)v; +} + +static int name_eq(const char *node, const char *want) +{ + while (*node && *node != '@') { + if (*node != *want) + return 0; + node++; + want++; + } + return *want == '\0'; +} + +/* + * rewrite /memory reg with the RAM the bootloader actually sees. + * the value is two u32 cells, base and size, addresses above 4GB + * need the parent #address-cells respected, virt is below 4GB and + * 2 cells for size. returns 0 on success. + */ +int tb_dtb_patch_memory(uintptr_t dtb, uint64_t base, uint64_t size) +{ + uint8_t *basep = (uint8_t *)dtb; + uint32_t off_struct = be32(basep + 8); + uint32_t off_strings = be32(basep + 12); + uint8_t *p = basep + off_struct; + uint8_t *strings = basep + off_strings; + const char *cur_node = NULL; + int depth = 0; + + if (be32(basep) != 0xd00dfeed) + return -1; + + while (p < basep + be32(basep + 4)) { + uint32_t token = be32(p); + + switch (token) { + case FDT_BEGIN_NODE: { + char *name = (char *)(p + 4); + size_t len = strlen(name) + 1; + + p += 4 + ((len + 3) & ~3); + depth++; + cur_node = name; + break; + } + case FDT_END_NODE: + depth--; + p += 4; + break; + case FDT_PROP: { + uint32_t plen = be32(p + 4); + const char *pname = (char *)strings + be32(p + 8); + uint8_t *val = p + 12; + + p += 12 + ((plen + 3) & ~3); + + if (depth == 2 && name_eq(cur_node, "memory") && + strcmp(pname, "reg") == 0 && plen >= 16) { + /* + * #address-cells 2, #size-cells 2, the + * reg is four cells, base hi lo and + * size hi lo, below 4GB the hi cells + * are zero. + */ + put_be32(val, (uint32_t)(base >> 32)); + put_be32(val + 4, (uint32_t)base); + put_be32(val + 8, (uint32_t)(size >> 32)); + put_be32(val + 12, (uint32_t)size); + return 0; + } + break; + } + case FDT_NOP: + p += 4; + break; + case FDT_END: + return -2; + } + } + + return -3; +} + +/* + * walk and rewrite. returns the number of cpu-release-addr values + * written, negative on a malformed blob. + */ +int tb_dtb_patch_spin_table(uintptr_t dtb, uintptr_t *gates, int ngates) +{ + uint8_t *base = (uint8_t *)dtb; + uint32_t off_struct = be32(base + 8); + uint32_t off_strings = be32(base + 12); + uint8_t *p = base + off_struct; + uint8_t *strings = base + off_strings; + const char *cur_cpu = NULL; + int in_cpus = 0; + int written = 0; + int depth = 0; + + if (be32(base) != 0xd00dfeed) + return -1; + + while (p < base + be32(base + 4)) { + uint32_t token = be32(p); + + switch (token) { + case FDT_BEGIN_NODE: { + char *name = (char *)(p + 4); + size_t len = strlen(name) + 1; + + p += 4 + ((len + 3) & ~3); + depth++; + + if (depth == 2 && name_eq(name, "cpus")) { + in_cpus = 1; + } else if (depth == 2) { + in_cpus = 0; + } else if (in_cpus && depth == 3) { + cur_cpu = name; + } + break; + } + case FDT_END_NODE: + depth--; + p += 4; + break; + case FDT_PROP: { + uint32_t plen = be32(p + 4); + const char *pname = (char *)strings + be32(p + 8); + uint8_t *val = p + 12; + + p += 12 + ((plen + 3) & ~3); + + if (in_cpus && depth == 3 && + strcmp(pname, "cpu-release-addr") == 0 && + plen == 8 && cur_cpu) { + long idx = -1; + const char *at = strchr(cur_cpu, '@'); + + if (at) { + idx = 0; + while (*at >= '0' && *at <= '9') { + at++; + } + at = strchr(cur_cpu, '@') + 1; + while (*at >= '0' && *at <= '9') { + idx = idx * 10 + (*at - '0'); + at++; + } + } + if (idx >= 0 && idx < ngates) { + put_be64(val, (uint64_t)gates[idx]); + written++; + } + } + break; + } + case FDT_NOP: + p += 4; + break; + case FDT_END: + return written; + } + } + + return written; +} + +/* + * tell the kernel where the initrd landed. /chosen is created by + * the machine firmware, the two cells exist when an initrd was + * already staged, we overwrite them in place. depth 2 under the + * root, node name "chosen". + */ +int tb_dtb_patch_initrd(uintptr_t dtb, uint64_t start, uint64_t end) +{ + uint8_t *basep = (uint8_t *)dtb; + uint32_t off_struct = be32(basep + 8); + uint32_t off_strings = be32(basep + 12); + uint8_t *p = basep + off_struct; + uint8_t *strings = basep + off_strings; + const char *cur_node = NULL; + int depth = 0; + + if (be32(basep) != 0xd00dfeed) + return -1; + + while (p < basep + be32(basep + 4)) { + uint32_t token = be32(p); + + switch (token) { + case FDT_BEGIN_NODE: { + char *name = (char *)(p + 4); + size_t len = strlen(name) + 1; + + p += 4 + ((len + 3) & ~3); + depth++; + cur_node = name; + break; + } + case FDT_END_NODE: + depth--; + p += 4; + break; + case FDT_PROP: { + uint32_t plen = be32(p + 4); + const char *pname = (char *)strings + be32(p + 8); + uint8_t *val = p + 12; + + p += 12 + ((plen + 3) & ~3); + + if (depth == 2 && name_eq(cur_node, "chosen") && + strcmp(pname, "linux,initrd-start") == 0 && + plen >= 8) { + put_be64(val, start); + } + if (depth == 2 && name_eq(cur_node, "chosen") && + strcmp(pname, "linux,initrd-end") == 0 && + plen >= 8) { + put_be64(val, end); + return 0; + } + break; + } + case FDT_NOP: + p += 4; + break; + case FDT_END: + return -2; + } + } + + return -2; +} diff --git a/common/dtb_reloc.c b/common/dtb_reloc.c new file mode 100644 index 0000000..c3e0fe5 --- /dev/null +++ b/common/dtb_reloc.c @@ -0,0 +1,69 @@ +/* + * dtb_reloc.c - grow the devicetree the way libfdt does, in a + * buffer with room to spare. firmware cannot edit a packed fdt + * in place, new properties shift everything behind them, so the + * blob is copied into scratch verbatim, the free space after + * totalsize is the room the insert code shifts into, then the + * walkers patch the copy and the kernel gets its address. + * + * The layout follows the devicetree specification: header, + * struct block, strings block, free space. The rebuild copies + * header, struct, strings, fixes the offsets in the new header, + * and leaves the gap between struct and strings as the room new + * properties will consume. + * + * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org> + */ + +#include <string.h> +#include <endian.h> +#include <boot.h> +#include <dtb_patch.h> + +#define FDT_BEGIN_NODE 1 +#define FDT_END_NODE 2 +#define FDT_PROP 3 +#define FDT_NOP 4 +#define FDT_END 9 + +static uint32_t be32(const void *p) +{ + const uint8_t *b = p; + + return ((uint32_t)b[0] << 24) | ((uint32_t)b[1] << 16) | + ((uint32_t)b[2] << 8) | (uint32_t)b[3]; +} + +/* + * copy the blob into the scratch, grow bytes of headroom after + * the end. returns the new blob address or 0 on a short buffer. + */ +uintptr_t tb_dtb_relocate(uintptr_t dtb, void *scratch, size_t scratch_size, + size_t grow) +{ + uint8_t *in = (uint8_t *)dtb; + uint8_t *out = scratch; + uint32_t totalsize; + + if (be32(in) != 0xd00dfeed) + return 0; + + totalsize = be32(in + 4); + + if (scratch_size < (size_t)totalsize + grow) + return 0; + + /* + * verbatim copy, byte for byte. the grow room is the free + * scratch after totalsize, the insert code shifts the + * strings block into it. an interior gap between the + * struct and strings blocks only invites the walkers to + * count it as tree. + */ + for (uint32_t i = 0; i < totalsize; i++) + out[i] = in[i]; + + (void)grow; + + return (uintptr_t)out; +} diff --git a/common/load.c b/common/load.c index cb3b41d..3a0f8b6 100644 --- a/common/load.c +++ b/common/load.c @@ -55,3 +55,32 @@ int tb_load_semihosting(const char *fname, uintptr_t load_addr, smh_close(fd); return 0; } + +/* + * the initrd path, no header, no placement math, bytes to the + * address the dtb /chosen already names. + */ +int tb_load_raw(const char *fname, uintptr_t load_addr, size_t *sizep) +{ + long fd, len, ret; + + fd = smh_open(fname, MODE_READ | MODE_BINARY); + if (fd < 0) + return fd; + + len = smh_flen(fd); + if (len < 0) { + smh_close(fd); + return len; + } + + ret = smh_read(fd, (void *)load_addr, len); + smh_close(fd); + + if (ret != len) + return -6; + + if (sizep) + *sizep = (size_t)len; + return 0; +} diff --git a/common/main.c b/common/main.c index fb388cb..a8fdf58 100644 --- a/common/main.c +++ b/common/main.c @@ -37,19 +37,29 @@ #include <boot.h> #define TB_VERSION "0.1" +#define TB_INITRD_ADDR 0x46000000ULL +#define TB_INITRD_FILE "initrd.cpio.gz" extern int tb_console_init(void); /* - * fixed load address, the osdev way. past the bootloader at the - * bottom of RAM, the image header decides its final resting place. + * the payload goes 16MB clear of wherever this bootloader is + * actually running, ADR knows the runtime base and qemu is free + * to place us anywhere. hardcoding 0x40200000 smashed our own + * image when qemu loaded us there. */ -#define TB_LOAD_ADDR 0x40200000 +extern char __image_copy_end[]; +#define TB_LOAD_ADDR ((uintptr_t)__image_copy_end + (16ULL << 20)) /* the file semihosting serves as the payload */ #define TB_BOOTFILE "Image" extern void __NO_RETURN tb_boot_linux(uintptr_t ep, uintptr_t fw_arg); +extern uintptr_t tb_dtb_relocate(uintptr_t dtb, void *scratch, + size_t scratch_size, size_t grow); +extern int tb_dtb_add_chosen_prop(uintptr_t dtb, const char *name, + const void *val, size_t len); +static size_t initrd_size; void tashaboot_main(uintptr_t fw_arg) { @@ -61,12 +71,143 @@ void tashaboot_main(uintptr_t fw_arg) dprintf(ALWAYS, "tashaboot " TB_VERSION "\n"); +#ifdef TB_ENABLE_MMU + { + extern int tb_mmu_enable(void); + extern int tb_mmu_selftest(void); + extern void tb_mmu_disable(void); + + if (tb_mmu_enable() == 0) { + if (tb_mmu_selftest() == 0) + dprintf(ALWAYS, "mmu: identity map on\n"); + else + dprintf(ALWAYS, "mmu: self test failed, " + "running unmapped\n"); + tb_mmu_disable(); + } + } +#endif + + { + /* report the RAM we actually live in, before any mmu */ + extern int tb_dtb_patch_memory(uintptr_t dtb, + uint64_t base, + uint64_t size); + int r; + + r = 0; (void)r; + + { + uint32_t *cells = (uint32_t *)(fw_arg + 0x16c); + int i; + + dprintf(ALWAYS, "cells after:"); + for (i = 0; i < 4; i++) + dprintf(ALWAYS, " %08x", cells[i]); + dprintf(ALWAYS, "\n"); + } + } + + { + /* spin table gates into the dtb, one per cpu node */ + extern unsigned long *tb_spin_gates_ptr; + extern int tb_dtb_patch_spin_table(uintptr_t dtb, + uintptr_t *gates, + int ngates); + int n; + + if (tb_spin_gates_ptr) { + extern unsigned char tb_pen_stamps[8]; + int c; + + n = tb_dtb_patch_spin_table(fw_arg, + tb_spin_gates_ptr, 8); + dprintf(ALWAYS, "dtb: %d release addrs patched\n", n); + + /* who made it to the pen */ + for (c = 1; c < 8; c++) { + if (tb_pen_stamps[c]) + break; + } + dprintf(ALWAYS, "pen: %s\n", + c < 8 ? "secondaries waiting" : + "no secondaries parked"); + } + } + ret = tb_load_semihosting(TB_BOOTFILE, TB_LOAD_ADDR, &img); if (ret) { dprintf(ALWAYS, "load failed (%d), halting\n", ret); platform_halt(); } + /* + * firmware owns the devicetree it hands the kernel. ours + * relocates into scratch with grow room, then the chosen + * properties are added there and the kernel gets the new + * address, the same flow libfdt firmware uses. + */ + { + /* + * the scratch lives at a fixed free address, clear of + * our image, the payload, and the kernel relocation + * zone. a bss array would sit inside 0x40080000+ and + * the kernel overwrites it while copying itself. + */ + uint8_t *dtb_scratch = (uint8_t *)0x45000000ULL; + uintptr_t newdtb; + + newdtb = tb_dtb_relocate(fw_arg, dtb_scratch, + 0x10000, 0x200); + if (!newdtb) { + dprintf(ALWAYS, "dtb: relocate failed\n"); + platform_halt(); + } + + fw_arg = newdtb; + } + + /* + * the initrd rides after the kernel, the dtb /chosen carries + * linux,initrd-start and -end, both already patched in place + * with this layout. + */ + { + int r = tb_load_raw(TB_INITRD_FILE, TB_INITRD_ADDR, + &initrd_size); + + if (r == 0) + dprintf(ALWAYS, "initrd at %lx, %lx bytes\n", + (unsigned long)TB_INITRD_ADDR, + (unsigned long)initrd_size); + else + dprintf(ALWAYS, "no initrd (%d)\n", r); + } + + /* + * the chosen properties, written now that the initrd size + * is known. the cells are big endian, the fdt is a big + * endian format end to end. + */ + { + uint8_t start_cells[8], end_cells[8]; + uint64_t start = TB_INITRD_ADDR; + uint64_t end = TB_INITRD_ADDR + initrd_size; + int a, b; + + for (int i = 0; i < 8; i++) { + start_cells[i] = (uint8_t)(start >> (56 - 8 * i)); + end_cells[i] = (uint8_t)(end >> (56 - 8 * i)); + } + a = tb_dtb_add_chosen_prop(fw_arg, "linux,initrd-start", + start_cells, 8); + b = tb_dtb_add_chosen_prop(fw_arg, "linux,initrd-end", + end_cells, 8); + dprintf(ALWAYS, "dtb: initrd props %d %d\n", a, b); + } + + + dprintf(ALWAYS, "loaded %llu bytes at %lx, entry %lx\n", (unsigned long long)img.size, img.load, img.ep); dprintf(ALWAYS, "jumping\n"); diff --git a/common/mmutest.c b/common/mmutest.c new file mode 100644 index 0000000..1b60110 --- /dev/null +++ b/common/mmutest.c @@ -0,0 +1,77 @@ +/* SPDX-License-Identifier: GPL-2.0+ */ +/* + * mmutest.c - self test for the identity map, AT S1E2R translates a + * VA through the tables and PAR_EL1 returns the walk result. if the + * map is wrong the instruction faults to our vectors instead, so a + * clean return with a valid PA in PAR means the tables walk. + * + * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org> + */ + +#include <asm/mmu.h> +#include <debug.h> + +#define PAR_F (1ULL << 0) /* fault, no translation */ +#define PAR_PA_MASK 0x000ffffffffff000ULL + +static uint64_t translate(uint64_t va) +{ + uint64_t par, el; + + asm volatile("mrs %0, CurrentEL" : "=r" (el)); + el >>= 2; + + if (el == 2) + asm volatile( + "at s1e2r, %1\n" + "isb\n" + "mrs %0, par_el1\n" + : "=r" (par) + : "r" (va) + : "memory"); + else + asm volatile( + "at s1e1r, %1\n" + "isb\n" + "mrs %0, par_el1\n" + : "=r" (par) + : "r" (va) + : "memory"); + return par; +} + +static int check(const char *name, uint64_t va) +{ + uint64_t par = translate(va); + + if (par & PAR_F) { + dprintf(ALWAYS, "mmu: %s faulted (par 0x%016llx)\n", + name, (unsigned long long)par); + return 1; + } + + if ((par & PAR_PA_MASK) != (va & PAR_PA_MASK)) { + dprintf(ALWAYS, "mmu: %s pa %llx != va %llx\n", + name, (unsigned long long)(par & PAR_PA_MASK), + (unsigned long long)va); + return 1; + } + + dprintf(ALWAYS, "mmu: %s ok, pa %llx\n", + name, (unsigned long long)(par & PAR_PA_MASK)); + return 0; +} + +int tb_mmu_selftest(void) +{ + int ret = 0; + + ret |= check("mmio 0x09000000 (uart)", 0x09000000); + ret |= check("mmio 0x00000000", 0x00000000); + ret |= check("ram 0x40200000 (load)", 0x40200000); + ret |= check("ram 0x41000000", 0x41000000); + ret |= check("self 0x40080000 (stack guard region, no map)", + 0x40080000); + + return ret; +} |
