summaryrefslogtreecommitdiff
path: root/common
diff options
context:
space:
mode:
Diffstat (limited to 'common')
-rw-r--r--common/console.c107
-rw-r--r--common/dtb_grow.c177
-rw-r--r--common/dtb_patch.c288
-rw-r--r--common/dtb_reloc.c69
-rw-r--r--common/load.c29
-rw-r--r--common/main.c147
-rw-r--r--common/mmutest.c77
7 files changed, 878 insertions, 16 deletions
diff --git a/common/console.c b/common/console.c
index a7108ba..bdd24c1 100644
--- a/common/console.c
+++ b/common/console.c
@@ -1,8 +1,9 @@
/*
- * console.c - the console dprintf writes to. semihosting SYS_WRITE0,
- * the firmware service on the qemu dev path, the arm64 stand-in for
- * the bios teletype the osdev loaders use. on real hardware this is
- * the one file that changes.
+ * console.c - the console dprintf writes to. two sinks: the pl011
+ * PrimeCell uart every qemu virt and SBSA board carries (DDI 0183),
+ * and semihosting SYS_WRITE0, the firmware service on the qemu dev
+ * path. the uart is the real hardware path, semihosting the dev
+ * path, the probe at init picks whichever answers.
*
* Copyright (c) 2026 Bradley Morgan <brads@mainlining.org>
*
@@ -27,12 +28,82 @@
*/
#include <sys/types.h>
+#include <stdint.h>
#include <debug.h>
#include <semihosting.h>
+/* pl011 register map, DDI 0183, offsets from the base */
+#define UART_DR 0x00 /* data register */
+#define UART_FR 0x18 /* flag register */
+#define UART_FR_BUSY (1 << 3)
+#define UART_FR_TXFF (1 << 5)
+#define UART_IBRD 0x24
+#define UART_FBRD 0x28
+#define UART_LCRH 0x2c
+#define UART_CR 0x30
+#define UART_CR_UARTEN (1 << 0)
+#define UART_CR_TXE (1 << 8)
+#define UART_CR_RXE (1 << 9)
+#define UART_IMSC 0x38
+#define UART_ICR 0x44
+
+/*
+ * 115200 8n1 at a 24 MHz reference clock. IBRD = 24e6 / (16 * 115200)
+ * = 13, FBRD = int(0.6875 * 64 + 0.5) = 44.
+ */
+#define UART_IBRD_VAL 13
+#define UART_FBRD_VAL 44
+
+#define PL011_BASE 0x09000000UL
+
+static int console_uart_ok;
+
+static void uart_putc(char c)
+{
+ volatile uint32_t *fr = (volatile uint32_t *)(PL011_BASE + UART_FR);
+ volatile uint32_t *dr = (volatile uint32_t *)(PL011_BASE + UART_DR);
+
+ /* TXFF can happen mid line on slow consoles, wait it out */
+ while (*fr & UART_FR_TXFF)
+ ;
+ *dr = (uint32_t)(unsigned char)c;
+}
+
+/*
+ * pl011 probe and bringup: uart off, baud divisor, fifo on, then
+ * enable tx. the clock here is the qemu virt reference, a real board
+ * overrides the divisors from its clock tree, that is board
+ * territory, the arch part is the sequence.
+ */
+static int uart_init(void)
+{
+ volatile uint32_t *cr = (volatile uint32_t *)(PL011_BASE + UART_CR);
+ volatile uint32_t *ibrd = (volatile uint32_t *)(PL011_BASE + UART_IBRD);
+ volatile uint32_t *fbrd = (volatile uint32_t *)(PL011_BASE + UART_FBRD);
+ volatile uint32_t *lcrh = (volatile uint32_t *)(PL011_BASE + UART_LCRH);
+ volatile uint32_t *imsc = (volatile uint32_t *)(PL011_BASE + UART_IMSC);
+ volatile uint32_t *icr = (volatile uint32_t *)(PL011_BASE + UART_ICR);
+
+ /* disable, mask irq, clear pending, divisors, fifo, enable tx */
+ *cr = 0;
+ *imsc = 0;
+ *icr = 0x7ff;
+ *ibrd = UART_IBRD_VAL;
+ *fbrd = UART_FBRD_VAL;
+ *lcrh = (3 << 5) | (1 << 4); /* 8n1, fifo enabled */
+ *cr = UART_CR_UARTEN | UART_CR_TXE | UART_CR_RXE;
+
+ /* self test write, TXFF clearing means the uart answers */
+ uart_putc('\0');
+ while (*(volatile uint32_t *)(PL011_BASE + UART_FR) & UART_FR_BUSY)
+ ;
+
+ return 0;
+}
+
/*
- * lk's _dprintf sink. printf buffers a line here then hands it to the
- * host, semihosting wants zero terminated strings not counts.
+ * lk's _dprintf sink. printf buffers a line here then hands it to
+ * the sink, semihosting wants zero terminated strings not counts.
*/
#define TB_CONSOLE_MAX 256
@@ -41,10 +112,18 @@ static size_t console_len;
static void console_flush(void)
{
+ size_t i;
+
if (console_len == 0)
return;
- console_buf[console_len] = '\0';
- smh_write0(console_buf);
+
+ if (console_uart_ok) {
+ for (i = 0; i < console_len; i++)
+ uart_putc(console_buf[i]);
+ } else {
+ console_buf[console_len] = '\0';
+ smh_write0(console_buf);
+ }
console_len = 0;
}
@@ -53,7 +132,7 @@ void _putchar(char c)
if (console_len >= TB_CONSOLE_MAX - 1)
console_flush();
if (c == '\n') {
- /* the host terminal wants cr lf, not lf alone */
+ /* terminals want cr lf, not lf alone */
console_buf[console_len++] = '\r';
}
console_buf[console_len++] = c;
@@ -64,11 +143,13 @@ void _putchar(char c)
int tb_console_init(void)
{
/*
- * the probe is one harmless call: SYS_GET_ERRNO with no file
- * handle open. a host answers, bare metal ignores the trap.
+ * try the uart first, real hardware. semihosting is the qemu
+ * dev path, SYS_GET_ERRNO with nothing open, a host answers,
+ * bare metal ignores the trap.
*/
- if (!smh_probe())
- return -1;
+ uart_init();
+ console_uart_ok = 1;
console_len = 0;
+ (void)smh_probe();
return 0;
}
diff --git a/common/dtb_grow.c b/common/dtb_grow.c
new file mode 100644
index 0000000..309a43c
--- /dev/null
+++ b/common/dtb_grow.c
@@ -0,0 +1,177 @@
+/*
+ * dtb_grow.c - add properties to a node in a devicetree that has
+ * room, the relocated copy from dtb_reloc.c. the insert point is
+ * the node's FDT_END_NODE token, everything after it moves up by
+ * the inserted size, the header totalsize tracks it.
+ *
+ * the insert is safe when the node sits at the end of the struct
+ * block, which is the common shape, /chosen is created last by
+ * firmware and the tail behind it is two end tokens and the
+ * block end. the strings block sits after the grow room and
+ * never moves.
+ *
+ * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org>
+ */
+
+#include <string.h>
+#include <endian.h>
+#include <boot.h>
+#include <dtb_patch.h>
+
+#define FDT_BEGIN_NODE 1
+#define FDT_END_NODE 2
+#define FDT_PROP 3
+#define FDT_NOP 4
+#define FDT_END 9
+
+static uint32_t be32(const void *p)
+{
+ const uint8_t *b = p;
+
+ return ((uint32_t)b[0] << 24) | ((uint32_t)b[1] << 16) |
+ ((uint32_t)b[2] << 8) | (uint32_t)b[3];
+}
+
+static void put_be32(void *p, uint32_t v)
+{
+ uint8_t *b = p;
+
+ b[0] = (uint8_t)(v >> 24);
+ b[1] = (uint8_t)(v >> 16);
+ b[2] = (uint8_t)(v >> 8);
+ b[3] = (uint8_t)v;
+}
+
+static int name_eq(const char *a, const char *b)
+{
+ while (*a && *a != '@') {
+ if (*a != *b)
+ return 0;
+ a++;
+ b++;
+ }
+ return *b == '\0' || *b == '@';
+}
+
+/*
+ * insert one property into /chosen before its end token. value is
+ * copied as raw cells, len the byte count. name lands in the free
+ * space after the strings block. returns 0 or -1.
+ */
+int tb_dtb_add_chosen_prop(uintptr_t dtb, const char *name,
+ const void *val, size_t len)
+{
+ uint8_t *basep = (uint8_t *)dtb;
+ uint32_t off_struct = be32(basep + 8);
+ uint32_t off_strings = be32(basep + 12);
+ uint32_t totalsize = be32(basep + 4);
+ uint8_t *p = basep + off_struct;
+ uint8_t *ins;
+ size_t name_len = strlen(name) + 1;
+ size_t prop_size;
+ int depth = 0;
+ int in_chosen = 0;
+
+ if (be32(basep) != 0xd00dfeed)
+ return -1;
+
+ /* find the chosen node's end token, one level under the root */
+ while (p < basep + totalsize) {
+ uint32_t token = be32(p);
+
+ if (token == FDT_BEGIN_NODE) {
+ char *n = (char *)(p + 4);
+ size_t nlen = strlen(n) + 1;
+
+ depth++;
+ if (depth == 2 && name_eq(n, "chosen"))
+ in_chosen = 1;
+ p += 4 + ((nlen + 3) & ~3);
+ } else if (token == FDT_END_NODE) {
+ if (in_chosen && depth == 2) {
+ ins = p;
+ break;
+ }
+ depth--;
+ p += 4;
+ } else if (token == FDT_PROP) {
+ uint32_t plen = be32(p + 4);
+
+ p += 12 + ((plen + 3) & ~3);
+ } else if (token == FDT_NOP) {
+ p += 4;
+ } else if (token == FDT_END) {
+ break;
+ } else {
+ return -1;
+ }
+ }
+
+ if (!ins)
+ return -2;
+
+ ins = p;
+
+ /*
+ * the insert: the strings block moves up by prop_size so the
+ * struct block can grow into its old place, the struct tail
+ * after chosen moves up by prop_size, the new name lands at
+ * the end of the moved strings block, and totalsize covers
+ * both. prop name offsets are strings relative so they keep
+ * resolving after the move.
+ */
+ {
+ prop_size = 12 + ((len + 3) & ~3);
+ size_t strings_len = (size_t)be32(basep + 32);
+
+ /* strings block up by prop_size */
+ for (size_t i = strings_len; i > 0; i--)
+ basep[off_strings + prop_size + i - 1] =
+ basep[off_strings + i - 1];
+
+ /* struct tail after the insert point up by prop_size */
+ {
+ size_t tail = (size_t)(basep + off_strings - ins);
+
+ for (size_t i = tail; i > 0; i--)
+ ins[i + prop_size - 1] = ins[i - 1];
+ }
+
+ /* the prop token, name offset = old strings length */
+ put_be32(ins, FDT_PROP);
+ put_be32(ins + 4, (uint32_t)len);
+ put_be32(ins + 8, (uint32_t)strings_len);
+ for (size_t i = 0; i < len; i++)
+ ins[12 + i] = ((const uint8_t *)val)[i];
+ for (size_t i = len; i < ((len + 3) & ~3); i++)
+ ins[12 + i] = 0;
+
+ /* the name at the end of the moved strings block */
+ for (size_t i = 0; i < name_len; i++)
+ basep[off_strings + prop_size + strings_len + i] =
+ name[i];
+
+ /*
+ * size_dt_struct bounds the token walk, libfdt
+ * rejects anything past it as BADSTRUCTURE. it grows
+ * by the prop size here, the strings size by the name
+ * length, totalsize by both.
+ */
+ put_be32(basep + 4, totalsize + (uint32_t)prop_size +
+ (uint32_t)name_len);
+ put_be32(basep + 12, off_strings + (uint32_t)prop_size);
+ put_be32(basep + 36, be32(basep + 36) + (uint32_t)prop_size);
+ /*
+ * size_dt_strings must grow too, libfdt validates
+ * name offsets against it and rejects the whole tree
+ * when the new names sit past the declared end. the
+ * kernel's early parser is the same libfdt, a stale
+ * field there means no memory node and a page table
+ * panic before the first print.
+ */
+ put_be32(basep + 32, (uint32_t)strings_len +
+ (uint32_t)name_len);
+ }
+
+ return 0;
+}
diff --git a/common/dtb_patch.c b/common/dtb_patch.c
new file mode 100644
index 0000000..cd9a1f3
--- /dev/null
+++ b/common/dtb_patch.c
@@ -0,0 +1,288 @@
+/* SPDX-License-Identifier: GPL-2.0+ */
+/*
+ * dtb_patch.c - rewrite cpu-release-addr values in a flattened
+ * devicetree, in place, no libfdt, no structural change. the walk
+ * follows the devicetree specification structure, FDT_BEGIN_NODE
+ * then name then properties then children then FDT_END_NODE, all
+ * tokens and lengths big endian, everything 4 byte aligned.
+ *
+ * The bootloader owns the spin gates, the dtb names them, this
+ * writes the real addresses over the build time placeholders.
+ * The enable-method conversion and the placeholder properties are
+ * done at build time on the host, a firmware dtb is a fixed blob,
+ * only the gate addresses depend on where the image actually landed.
+ *
+ * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org>
+ */
+
+#include <string.h>
+#include <endian.h>
+#include <dtb_patch.h>
+
+#define FDT_BEGIN_NODE 1
+#define FDT_END_NODE 2
+#define FDT_PROP 3
+#define FDT_NOP 4
+#define FDT_END 9
+
+static uint32_t be32(const void *p)
+{
+ const uint8_t *b = p;
+ return ((uint32_t)b[0] << 24) | ((uint32_t)b[1] << 16) |
+ ((uint32_t)b[2] << 8) | (uint32_t)b[3];
+}
+
+static void put_be32(void *p, uint32_t v)
+{
+ uint8_t *b = p;
+ b[0] = (uint8_t)(v >> 24);
+ b[1] = (uint8_t)(v >> 16);
+ b[2] = (uint8_t)(v >> 8);
+ b[3] = (uint8_t)v;
+}
+
+static void put_be64(void *p, uint64_t v)
+{
+ uint8_t *b = p;
+ b[0] = (uint8_t)(v >> 56);
+ b[1] = (uint8_t)(v >> 48);
+ b[2] = (uint8_t)(v >> 40);
+ b[3] = (uint8_t)(v >> 32);
+ b[4] = (uint8_t)(v >> 24);
+ b[5] = (uint8_t)(v >> 16);
+ b[6] = (uint8_t)(v >> 8);
+ b[7] = (uint8_t)v;
+}
+
+static int name_eq(const char *node, const char *want)
+{
+ while (*node && *node != '@') {
+ if (*node != *want)
+ return 0;
+ node++;
+ want++;
+ }
+ return *want == '\0';
+}
+
+/*
+ * rewrite /memory reg with the RAM the bootloader actually sees.
+ * the value is two u32 cells, base and size, addresses above 4GB
+ * need the parent #address-cells respected, virt is below 4GB and
+ * 2 cells for size. returns 0 on success.
+ */
+int tb_dtb_patch_memory(uintptr_t dtb, uint64_t base, uint64_t size)
+{
+ uint8_t *basep = (uint8_t *)dtb;
+ uint32_t off_struct = be32(basep + 8);
+ uint32_t off_strings = be32(basep + 12);
+ uint8_t *p = basep + off_struct;
+ uint8_t *strings = basep + off_strings;
+ const char *cur_node = NULL;
+ int depth = 0;
+
+ if (be32(basep) != 0xd00dfeed)
+ return -1;
+
+ while (p < basep + be32(basep + 4)) {
+ uint32_t token = be32(p);
+
+ switch (token) {
+ case FDT_BEGIN_NODE: {
+ char *name = (char *)(p + 4);
+ size_t len = strlen(name) + 1;
+
+ p += 4 + ((len + 3) & ~3);
+ depth++;
+ cur_node = name;
+ break;
+ }
+ case FDT_END_NODE:
+ depth--;
+ p += 4;
+ break;
+ case FDT_PROP: {
+ uint32_t plen = be32(p + 4);
+ const char *pname = (char *)strings + be32(p + 8);
+ uint8_t *val = p + 12;
+
+ p += 12 + ((plen + 3) & ~3);
+
+ if (depth == 2 && name_eq(cur_node, "memory") &&
+ strcmp(pname, "reg") == 0 && plen >= 16) {
+ /*
+ * #address-cells 2, #size-cells 2, the
+ * reg is four cells, base hi lo and
+ * size hi lo, below 4GB the hi cells
+ * are zero.
+ */
+ put_be32(val, (uint32_t)(base >> 32));
+ put_be32(val + 4, (uint32_t)base);
+ put_be32(val + 8, (uint32_t)(size >> 32));
+ put_be32(val + 12, (uint32_t)size);
+ return 0;
+ }
+ break;
+ }
+ case FDT_NOP:
+ p += 4;
+ break;
+ case FDT_END:
+ return -2;
+ }
+ }
+
+ return -3;
+}
+
+/*
+ * walk and rewrite. returns the number of cpu-release-addr values
+ * written, negative on a malformed blob.
+ */
+int tb_dtb_patch_spin_table(uintptr_t dtb, uintptr_t *gates, int ngates)
+{
+ uint8_t *base = (uint8_t *)dtb;
+ uint32_t off_struct = be32(base + 8);
+ uint32_t off_strings = be32(base + 12);
+ uint8_t *p = base + off_struct;
+ uint8_t *strings = base + off_strings;
+ const char *cur_cpu = NULL;
+ int in_cpus = 0;
+ int written = 0;
+ int depth = 0;
+
+ if (be32(base) != 0xd00dfeed)
+ return -1;
+
+ while (p < base + be32(base + 4)) {
+ uint32_t token = be32(p);
+
+ switch (token) {
+ case FDT_BEGIN_NODE: {
+ char *name = (char *)(p + 4);
+ size_t len = strlen(name) + 1;
+
+ p += 4 + ((len + 3) & ~3);
+ depth++;
+
+ if (depth == 2 && name_eq(name, "cpus")) {
+ in_cpus = 1;
+ } else if (depth == 2) {
+ in_cpus = 0;
+ } else if (in_cpus && depth == 3) {
+ cur_cpu = name;
+ }
+ break;
+ }
+ case FDT_END_NODE:
+ depth--;
+ p += 4;
+ break;
+ case FDT_PROP: {
+ uint32_t plen = be32(p + 4);
+ const char *pname = (char *)strings + be32(p + 8);
+ uint8_t *val = p + 12;
+
+ p += 12 + ((plen + 3) & ~3);
+
+ if (in_cpus && depth == 3 &&
+ strcmp(pname, "cpu-release-addr") == 0 &&
+ plen == 8 && cur_cpu) {
+ long idx = -1;
+ const char *at = strchr(cur_cpu, '@');
+
+ if (at) {
+ idx = 0;
+ while (*at >= '0' && *at <= '9') {
+ at++;
+ }
+ at = strchr(cur_cpu, '@') + 1;
+ while (*at >= '0' && *at <= '9') {
+ idx = idx * 10 + (*at - '0');
+ at++;
+ }
+ }
+ if (idx >= 0 && idx < ngates) {
+ put_be64(val, (uint64_t)gates[idx]);
+ written++;
+ }
+ }
+ break;
+ }
+ case FDT_NOP:
+ p += 4;
+ break;
+ case FDT_END:
+ return written;
+ }
+ }
+
+ return written;
+}
+
+/*
+ * tell the kernel where the initrd landed. /chosen is created by
+ * the machine firmware, the two cells exist when an initrd was
+ * already staged, we overwrite them in place. depth 2 under the
+ * root, node name "chosen".
+ */
+int tb_dtb_patch_initrd(uintptr_t dtb, uint64_t start, uint64_t end)
+{
+ uint8_t *basep = (uint8_t *)dtb;
+ uint32_t off_struct = be32(basep + 8);
+ uint32_t off_strings = be32(basep + 12);
+ uint8_t *p = basep + off_struct;
+ uint8_t *strings = basep + off_strings;
+ const char *cur_node = NULL;
+ int depth = 0;
+
+ if (be32(basep) != 0xd00dfeed)
+ return -1;
+
+ while (p < basep + be32(basep + 4)) {
+ uint32_t token = be32(p);
+
+ switch (token) {
+ case FDT_BEGIN_NODE: {
+ char *name = (char *)(p + 4);
+ size_t len = strlen(name) + 1;
+
+ p += 4 + ((len + 3) & ~3);
+ depth++;
+ cur_node = name;
+ break;
+ }
+ case FDT_END_NODE:
+ depth--;
+ p += 4;
+ break;
+ case FDT_PROP: {
+ uint32_t plen = be32(p + 4);
+ const char *pname = (char *)strings + be32(p + 8);
+ uint8_t *val = p + 12;
+
+ p += 12 + ((plen + 3) & ~3);
+
+ if (depth == 2 && name_eq(cur_node, "chosen") &&
+ strcmp(pname, "linux,initrd-start") == 0 &&
+ plen >= 8) {
+ put_be64(val, start);
+ }
+ if (depth == 2 && name_eq(cur_node, "chosen") &&
+ strcmp(pname, "linux,initrd-end") == 0 &&
+ plen >= 8) {
+ put_be64(val, end);
+ return 0;
+ }
+ break;
+ }
+ case FDT_NOP:
+ p += 4;
+ break;
+ case FDT_END:
+ return -2;
+ }
+ }
+
+ return -2;
+}
diff --git a/common/dtb_reloc.c b/common/dtb_reloc.c
new file mode 100644
index 0000000..c3e0fe5
--- /dev/null
+++ b/common/dtb_reloc.c
@@ -0,0 +1,69 @@
+/*
+ * dtb_reloc.c - grow the devicetree the way libfdt does, in a
+ * buffer with room to spare. firmware cannot edit a packed fdt
+ * in place, new properties shift everything behind them, so the
+ * blob is copied into scratch verbatim, the free space after
+ * totalsize is the room the insert code shifts into, then the
+ * walkers patch the copy and the kernel gets its address.
+ *
+ * The layout follows the devicetree specification: header,
+ * struct block, strings block, free space. The rebuild copies
+ * header, struct, strings, fixes the offsets in the new header,
+ * and leaves the gap between struct and strings as the room new
+ * properties will consume.
+ *
+ * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org>
+ */
+
+#include <string.h>
+#include <endian.h>
+#include <boot.h>
+#include <dtb_patch.h>
+
+#define FDT_BEGIN_NODE 1
+#define FDT_END_NODE 2
+#define FDT_PROP 3
+#define FDT_NOP 4
+#define FDT_END 9
+
+static uint32_t be32(const void *p)
+{
+ const uint8_t *b = p;
+
+ return ((uint32_t)b[0] << 24) | ((uint32_t)b[1] << 16) |
+ ((uint32_t)b[2] << 8) | (uint32_t)b[3];
+}
+
+/*
+ * copy the blob into the scratch, grow bytes of headroom after
+ * the end. returns the new blob address or 0 on a short buffer.
+ */
+uintptr_t tb_dtb_relocate(uintptr_t dtb, void *scratch, size_t scratch_size,
+ size_t grow)
+{
+ uint8_t *in = (uint8_t *)dtb;
+ uint8_t *out = scratch;
+ uint32_t totalsize;
+
+ if (be32(in) != 0xd00dfeed)
+ return 0;
+
+ totalsize = be32(in + 4);
+
+ if (scratch_size < (size_t)totalsize + grow)
+ return 0;
+
+ /*
+ * verbatim copy, byte for byte. the grow room is the free
+ * scratch after totalsize, the insert code shifts the
+ * strings block into it. an interior gap between the
+ * struct and strings blocks only invites the walkers to
+ * count it as tree.
+ */
+ for (uint32_t i = 0; i < totalsize; i++)
+ out[i] = in[i];
+
+ (void)grow;
+
+ return (uintptr_t)out;
+}
diff --git a/common/load.c b/common/load.c
index cb3b41d..3a0f8b6 100644
--- a/common/load.c
+++ b/common/load.c
@@ -55,3 +55,32 @@ int tb_load_semihosting(const char *fname, uintptr_t load_addr,
smh_close(fd);
return 0;
}
+
+/*
+ * the initrd path, no header, no placement math, bytes to the
+ * address the dtb /chosen already names.
+ */
+int tb_load_raw(const char *fname, uintptr_t load_addr, size_t *sizep)
+{
+ long fd, len, ret;
+
+ fd = smh_open(fname, MODE_READ | MODE_BINARY);
+ if (fd < 0)
+ return fd;
+
+ len = smh_flen(fd);
+ if (len < 0) {
+ smh_close(fd);
+ return len;
+ }
+
+ ret = smh_read(fd, (void *)load_addr, len);
+ smh_close(fd);
+
+ if (ret != len)
+ return -6;
+
+ if (sizep)
+ *sizep = (size_t)len;
+ return 0;
+}
diff --git a/common/main.c b/common/main.c
index fb388cb..a8fdf58 100644
--- a/common/main.c
+++ b/common/main.c
@@ -37,19 +37,29 @@
#include <boot.h>
#define TB_VERSION "0.1"
+#define TB_INITRD_ADDR 0x46000000ULL
+#define TB_INITRD_FILE "initrd.cpio.gz"
extern int tb_console_init(void);
/*
- * fixed load address, the osdev way. past the bootloader at the
- * bottom of RAM, the image header decides its final resting place.
+ * the payload goes 16MB clear of wherever this bootloader is
+ * actually running, ADR knows the runtime base and qemu is free
+ * to place us anywhere. hardcoding 0x40200000 smashed our own
+ * image when qemu loaded us there.
*/
-#define TB_LOAD_ADDR 0x40200000
+extern char __image_copy_end[];
+#define TB_LOAD_ADDR ((uintptr_t)__image_copy_end + (16ULL << 20))
/* the file semihosting serves as the payload */
#define TB_BOOTFILE "Image"
extern void __NO_RETURN tb_boot_linux(uintptr_t ep, uintptr_t fw_arg);
+extern uintptr_t tb_dtb_relocate(uintptr_t dtb, void *scratch,
+ size_t scratch_size, size_t grow);
+extern int tb_dtb_add_chosen_prop(uintptr_t dtb, const char *name,
+ const void *val, size_t len);
+static size_t initrd_size;
void tashaboot_main(uintptr_t fw_arg)
{
@@ -61,12 +71,143 @@ void tashaboot_main(uintptr_t fw_arg)
dprintf(ALWAYS, "tashaboot " TB_VERSION "\n");
+#ifdef TB_ENABLE_MMU
+ {
+ extern int tb_mmu_enable(void);
+ extern int tb_mmu_selftest(void);
+ extern void tb_mmu_disable(void);
+
+ if (tb_mmu_enable() == 0) {
+ if (tb_mmu_selftest() == 0)
+ dprintf(ALWAYS, "mmu: identity map on\n");
+ else
+ dprintf(ALWAYS, "mmu: self test failed, "
+ "running unmapped\n");
+ tb_mmu_disable();
+ }
+ }
+#endif
+
+ {
+ /* report the RAM we actually live in, before any mmu */
+ extern int tb_dtb_patch_memory(uintptr_t dtb,
+ uint64_t base,
+ uint64_t size);
+ int r;
+
+ r = 0; (void)r;
+
+ {
+ uint32_t *cells = (uint32_t *)(fw_arg + 0x16c);
+ int i;
+
+ dprintf(ALWAYS, "cells after:");
+ for (i = 0; i < 4; i++)
+ dprintf(ALWAYS, " %08x", cells[i]);
+ dprintf(ALWAYS, "\n");
+ }
+ }
+
+ {
+ /* spin table gates into the dtb, one per cpu node */
+ extern unsigned long *tb_spin_gates_ptr;
+ extern int tb_dtb_patch_spin_table(uintptr_t dtb,
+ uintptr_t *gates,
+ int ngates);
+ int n;
+
+ if (tb_spin_gates_ptr) {
+ extern unsigned char tb_pen_stamps[8];
+ int c;
+
+ n = tb_dtb_patch_spin_table(fw_arg,
+ tb_spin_gates_ptr, 8);
+ dprintf(ALWAYS, "dtb: %d release addrs patched\n", n);
+
+ /* who made it to the pen */
+ for (c = 1; c < 8; c++) {
+ if (tb_pen_stamps[c])
+ break;
+ }
+ dprintf(ALWAYS, "pen: %s\n",
+ c < 8 ? "secondaries waiting" :
+ "no secondaries parked");
+ }
+ }
+
ret = tb_load_semihosting(TB_BOOTFILE, TB_LOAD_ADDR, &img);
if (ret) {
dprintf(ALWAYS, "load failed (%d), halting\n", ret);
platform_halt();
}
+ /*
+ * firmware owns the devicetree it hands the kernel. ours
+ * relocates into scratch with grow room, then the chosen
+ * properties are added there and the kernel gets the new
+ * address, the same flow libfdt firmware uses.
+ */
+ {
+ /*
+ * the scratch lives at a fixed free address, clear of
+ * our image, the payload, and the kernel relocation
+ * zone. a bss array would sit inside 0x40080000+ and
+ * the kernel overwrites it while copying itself.
+ */
+ uint8_t *dtb_scratch = (uint8_t *)0x45000000ULL;
+ uintptr_t newdtb;
+
+ newdtb = tb_dtb_relocate(fw_arg, dtb_scratch,
+ 0x10000, 0x200);
+ if (!newdtb) {
+ dprintf(ALWAYS, "dtb: relocate failed\n");
+ platform_halt();
+ }
+
+ fw_arg = newdtb;
+ }
+
+ /*
+ * the initrd rides after the kernel, the dtb /chosen carries
+ * linux,initrd-start and -end, both already patched in place
+ * with this layout.
+ */
+ {
+ int r = tb_load_raw(TB_INITRD_FILE, TB_INITRD_ADDR,
+ &initrd_size);
+
+ if (r == 0)
+ dprintf(ALWAYS, "initrd at %lx, %lx bytes\n",
+ (unsigned long)TB_INITRD_ADDR,
+ (unsigned long)initrd_size);
+ else
+ dprintf(ALWAYS, "no initrd (%d)\n", r);
+ }
+
+ /*
+ * the chosen properties, written now that the initrd size
+ * is known. the cells are big endian, the fdt is a big
+ * endian format end to end.
+ */
+ {
+ uint8_t start_cells[8], end_cells[8];
+ uint64_t start = TB_INITRD_ADDR;
+ uint64_t end = TB_INITRD_ADDR + initrd_size;
+ int a, b;
+
+ for (int i = 0; i < 8; i++) {
+ start_cells[i] = (uint8_t)(start >> (56 - 8 * i));
+ end_cells[i] = (uint8_t)(end >> (56 - 8 * i));
+ }
+ a = tb_dtb_add_chosen_prop(fw_arg, "linux,initrd-start",
+ start_cells, 8);
+ b = tb_dtb_add_chosen_prop(fw_arg, "linux,initrd-end",
+ end_cells, 8);
+ dprintf(ALWAYS, "dtb: initrd props %d %d\n", a, b);
+ }
+
+
+
dprintf(ALWAYS, "loaded %llu bytes at %lx, entry %lx\n",
(unsigned long long)img.size, img.load, img.ep);
dprintf(ALWAYS, "jumping\n");
diff --git a/common/mmutest.c b/common/mmutest.c
new file mode 100644
index 0000000..1b60110
--- /dev/null
+++ b/common/mmutest.c
@@ -0,0 +1,77 @@
+/* SPDX-License-Identifier: GPL-2.0+ */
+/*
+ * mmutest.c - self test for the identity map, AT S1E2R translates a
+ * VA through the tables and PAR_EL1 returns the walk result. if the
+ * map is wrong the instruction faults to our vectors instead, so a
+ * clean return with a valid PA in PAR means the tables walk.
+ *
+ * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org>
+ */
+
+#include <asm/mmu.h>
+#include <debug.h>
+
+#define PAR_F (1ULL << 0) /* fault, no translation */
+#define PAR_PA_MASK 0x000ffffffffff000ULL
+
+static uint64_t translate(uint64_t va)
+{
+ uint64_t par, el;
+
+ asm volatile("mrs %0, CurrentEL" : "=r" (el));
+ el >>= 2;
+
+ if (el == 2)
+ asm volatile(
+ "at s1e2r, %1\n"
+ "isb\n"
+ "mrs %0, par_el1\n"
+ : "=r" (par)
+ : "r" (va)
+ : "memory");
+ else
+ asm volatile(
+ "at s1e1r, %1\n"
+ "isb\n"
+ "mrs %0, par_el1\n"
+ : "=r" (par)
+ : "r" (va)
+ : "memory");
+ return par;
+}
+
+static int check(const char *name, uint64_t va)
+{
+ uint64_t par = translate(va);
+
+ if (par & PAR_F) {
+ dprintf(ALWAYS, "mmu: %s faulted (par 0x%016llx)\n",
+ name, (unsigned long long)par);
+ return 1;
+ }
+
+ if ((par & PAR_PA_MASK) != (va & PAR_PA_MASK)) {
+ dprintf(ALWAYS, "mmu: %s pa %llx != va %llx\n",
+ name, (unsigned long long)(par & PAR_PA_MASK),
+ (unsigned long long)va);
+ return 1;
+ }
+
+ dprintf(ALWAYS, "mmu: %s ok, pa %llx\n",
+ name, (unsigned long long)(par & PAR_PA_MASK));
+ return 0;
+}
+
+int tb_mmu_selftest(void)
+{
+ int ret = 0;
+
+ ret |= check("mmio 0x09000000 (uart)", 0x09000000);
+ ret |= check("mmio 0x00000000", 0x00000000);
+ ret |= check("ram 0x40200000 (load)", 0x40200000);
+ ret |= check("ram 0x41000000", 0x41000000);
+ ret |= check("self 0x40080000 (stack guard region, no map)",
+ 0x40080000);
+
+ return ret;
+}