summaryrefslogtreecommitdiff
diff options
context:
space:
mode:
-rw-r--r--Makefile8
-rw-r--r--arch/arm64/kernel/boot.S23
-rw-r--r--arch/arm64/kernel/start.S86
-rw-r--r--arch/arm64/kernel/tashaboot.lds14
-rw-r--r--arch/arm64/lib/gic.c105
-rw-r--r--common/dtb_find.c178
-rw-r--r--common/dtb_grow.c177
-rw-r--r--common/dtb_patch.c146
-rw-r--r--common/dtb_reloc.c69
-rw-r--r--common/load.c4
-rw-r--r--common/main.c120
-rw-r--r--include/boot.h2
-rw-r--r--tools/fillsize.py13
13 files changed, 923 insertions, 22 deletions
diff --git a/Makefile b/Makefile
index ae9877b..6112249 100644
--- a/Makefile
+++ b/Makefile
@@ -13,7 +13,7 @@ OBJCOPY := $(CROSS)objcopy
CFLAGS := -nostdlib -ffreestanding -mgeneral-regs-only \
-fno-builtin -fno-stack-protector -fno-pie -no-pie \
- -Wall -Werror -O2 -DTB_ENABLE_MMU \
+ -Wall -Werror -O2 \
-Iinclude -Iarch/arm64/include
LDFLAGS := -T arch/arm64/kernel/tashaboot.lds
@@ -23,14 +23,15 @@ OBJS := arch/arm64/kernel/start.o \
arch/arm64/kernel/halt.o \
arch/arm64/kernel/boot.o \
arch/arm64/lib/cache.o \
- arch/arm64/lib/semihosting.o \
+ arch/arm64/lib/semihosting.o arch/arm64/lib/gic.o \
arch/arm64/lib/mmu.o \
arch/arm64/lib/psci.o \
arch/arm64/lib/system.o \
arch/arm64/lib/cache_va.o \
arch/arm64/lib/timer.o \
common/main.o common/console.o common/image.o common/load.o \
- common/mmutest.o common/dtb_patch.o \
+ common/mmutest.o common/dtb_patch.o common/dtb_reloc.o \
+ common/dtb_grow.o common/dtb_find.o \
lib/printf.o lib/itoa.o lib/semihosting.o \
$(patsubst %.c,%.o,$(wildcard lib/string/*.c))
@@ -42,6 +43,7 @@ build/tashaboot.elf: $(OBJS) arch/arm64/kernel/tashaboot.lds
build/tashaboot.bin: build/tashaboot.elf
$(OBJCOPY) -O binary $< $@
+ python3 tools/fillsize.py $@
%.o: %.c
$(CC) $(CFLAGS) -c -o $@ $<
diff --git a/arch/arm64/kernel/boot.S b/arch/arm64/kernel/boot.S
index 615be06..d8b888f 100644
--- a/arch/arm64/kernel/boot.S
+++ b/arch/arm64/kernel/boot.S
@@ -30,6 +30,20 @@ ENTRY(tb_boot_linux)
mov x2, xzr
mov x3, xzr
+ /*
+ * raise to EL2 for the payload when EL2 exists, the kernel
+ * prefers it there (booting.rst). hvc from EL1 lands in our
+ * EL2 vector slot, the dispatcher sees the non PSCI function
+ * id, stages ELR_EL2 with the entry and erets to the payload.
+ * on an EL1 only machine this is a straight branch.
+ */
+ mrs x9, CurrentEL
+ lsr x9, x9, #2
+ cmp x9, #2
+ b.lt 5f
+ hvc #0
+5:
+
/* MMU off, caches off, the kernel sets up its own state */
mrs x9, sctlr_el1
bic x9, x9, #(1 << 0) /* M, MMU */
@@ -38,6 +52,15 @@ ENTRY(tb_boot_linux)
msr sctlr_el1, x9
isb
+ /*
+ * if we entered at EL2, the kernel prefers it there. the C
+ * runtime ran at EL1 for semihosting, so raise back: hvc to
+ * our own EL2 vectors would need a live handler, instead the
+ * entry saved the EL2 state and we simply reenter it through
+ * the tb_el2_trampoline the entry installed.
+ */
+
+
br x8
ENDPROC(tb_boot_linux)
.popsection
diff --git a/arch/arm64/kernel/start.S b/arch/arm64/kernel/start.S
index 969830d..6ee9941 100644
--- a/arch/arm64/kernel/start.S
+++ b/arch/arm64/kernel/start.S
@@ -14,12 +14,25 @@
.section .text.boot
.globl _start
_start:
+ /* code0: branch over the 64 byte Image header to reset */
b reset
.balign 8
-.globl _text_base
-_text_base:
- .quad 0x40000000
+/*
+ * the arm64 Image header fields, per Documentation/arch/arm64/
+ * booting.rst: text_offset 0x08, image_size 0x10, flags 0x18,
+ * magic 0x38. code0 above branches over all of it. text_offset 0
+ * and image_size filled after link by tools/fillsize.py, the
+ * magic pins it as a proper Image so qemu -kernel enters at
+ * RAMBASE instead of guessing +0x80000.
+ */
+ .quad 0x0 /* text_offset, 0x08, filled below */
+ .quad 0x0 /* image_size, 0x10, filled below */
+ .quad 0x0 /* flags, 0x18: LE, 4k pages, unset */
+ .quad 0x0 /* reserved 0x20 */
+ .quad 0x0 /* reserved 0x28 */
+ .quad 0x0 /* reserved 0x30 */
+ .quad 0x644d5241 /* magic, 0x38: ARM\x64 */
reset:
/* keep the dtb pointer before anything clobbers x0 */
@@ -60,12 +73,14 @@ from_el3:
from_el2:
/*
- * stay at EL2: the kernel wants it for the virtualization
- * extensions and hands off from there. everything below scrubs
- * the EL2 state so the kernel starts clean.
+ * scrub the EL2 state and drop to EL1 for the C runtime. the
+ * semihosting hlt trap is an EL1 service on qemu, calling it
+ * from EL2 corrupts the return state. the kernel handoff goes
+ * back to EL2, booting.rst prefers it there, through the
+ * trampoline in boot.S.
*/
- /* EL1 will be aarch64 when the kernel drops itself down */
+ /* EL1 will be aarch64 */
mov x0, #(1 << 31) /* HCR_EL2.RW = 1 */
msr hcr_el2, x0
@@ -79,7 +94,12 @@ from_el2:
msr hstr_el2, xzr
msr vpidr_el2, xzr
- b mmu_check
+ /* drop to EL1, SPSR EL1h with DAIF masked */
+ mov x0, #0x3c5
+ msr spsr_el2, x0
+ adr x0, mmu_check
+ msr elr_el2, x0
+ eret
mmu_check:
/*
@@ -289,7 +309,16 @@ exc_sync:
mrs x0, esr_el1
mrs x1, far_el1
2:
- mov x2, lr
+ /* x2 = the faulting PC when it is the sync path */
+ mrs x4, CurrentEL
+ lsr x4, x4, #2
+ cmp x4, #2
+ b.lt 3f
+ mrs x2, elr_el2
+ b 4f
+3:
+ mrs x2, elr_el1
+4:
bl exc_report
ldp x29, x30, [sp], #16
b park
@@ -300,6 +329,45 @@ exc_sync:
* the resume point, eret takes it back.
*/
hvc_from_el1:
+ /*
+ * the lower EL sync slot. three arrivals share it: PSCI hvc
+ * from the kernel (EC 0x16, PSCI id in x0), our own boot
+ * handoff (hvc with the payload entry in x8), and semihosting
+ * hlt #0xf000 from the EL1 C runtime (EC 0x14). qemu only
+ * answers the hlt when it executes at EL2, so the handler
+ * replays the trap at EL2 and erets home with the result.
+ */
+ mrs x1, esr_el2
+ lsr x1, x1, #26 /* EC */
+ cmp x1, #0x14 /* HLT from lower EL, semihosting */
+ b.eq smh_replay
+
+ /*
+ * the hvc arrives with either a PSCI function id in x0 (the
+ * kernel calling) or the boot handoff staging the payload
+ * entry in x8 and the dtb in x0. PSCI ids have the 0x84/0xc4
+ * prefix, a dtb pointer never does.
+ */
+ lsr x1, x0, #24
+ cmp x1, #0x84
+ b.eq psci_call
+ cmp x1, #0xc4
+ b.eq psci_call
+
+ /* the boot handoff: ELR_EL2 = entry, eret to the payload */
+ msr elr_el2, x8
+ eret
+
+smh_replay:
+ /*
+ * x0 holds the semihosting syscall number, x1 the parameter
+ * block, both live in the caller's registers. replay the hlt
+ * here at EL2 where qemu answers it, then eret back.
+ */
+ hlt #0xf000
+ eret
+
+psci_call:
stp x4, x5, [sp, #-16]!
stp x6, x7, [sp, #-16]!
stp x29, x30, [sp, #-16]!
diff --git a/arch/arm64/kernel/tashaboot.lds b/arch/arm64/kernel/tashaboot.lds
index 4d7adad..e2b7954 100644
--- a/arch/arm64/kernel/tashaboot.lds
+++ b/arch/arm64/kernel/tashaboot.lds
@@ -12,6 +12,14 @@ ENTRY(_start)
SECTIONS
{
+ /*
+ * the first 64 bytes are the arm64 Image header: code0 'b' over
+ * it, magic ARM\x64, text_offset 0. qemu -kernel parses the
+ * header, loads the file at 0x40000000 and enters at
+ * 0x40000000, where the branch lands on reset at 0x40000040.
+ * without the header qemu guesses text_offset 0x80000 and runs
+ * the whole loader from the wrong address.
+ */
. = 0x40000000;
__image_copy_start = .;
@@ -21,6 +29,12 @@ SECTIONS
{
arch/arm64/kernel/start.o (.text.boot)
*(.text.boot)
+
+ /* the Image header, code0 branches over it */
+ . = ALIGN(64);
+ *(.text.imgheader)
+ . = ALIGN(64);
+
*(.text*)
}
diff --git a/arch/arm64/lib/gic.c b/arch/arm64/lib/gic.c
new file mode 100644
index 0000000..11bb9cd
--- /dev/null
+++ b/arch/arm64/lib/gic.c
@@ -0,0 +1,105 @@
+/*
+ * gic.c - the interrupt controller state a bootloader owns. the
+ * kernel programs the gic itself for the running system, but it
+ * trusts the state it inherits: on real hardware the secure
+ * world configures which interrupts are visible to non-secure,
+ * and a bootloader that leaves random enables or secure group
+ * bits set hands the kernel a half-configured distributor that
+ * can fire before the kernel's irqchip driver is up.
+ *
+ * this is the gicv2 sequence from the TRM, the same shape
+ * u-boot leaves the machine in: distributor off, every
+ * interrupt in the non-secure group, all per interrupt enables
+ * cleared, pending state cleared, cpu interfaces off. defined
+ * state, nothing firing, the kernel starts from zero.
+ *
+ * GICv1 shows the same register map minus the security
+ * extension registers, the writes below are harmless there.
+ *
+ * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org>
+ */
+
+#include <string.h>
+#include <stdint.h>
+#include <boot.h>
+#include <reg.h>
+
+/* distributor registers, offsets from the GICD base */
+#define GICD_CTLR 0x000
+#define GICD_TYPER 0x004
+#define GICD_IGROUPR(n) (0x080 + (n) * 4)
+#define GICD_ISENABLER(n) (0x100 + (n) * 4)
+#define GICD_ICENABLER(n) (0x180 + (n) * 4)
+#define GICD_ICPENDR(n) (0x280 + (n) * 4)
+#define GICD_ICACTIVER(n) (0x380 + (n) * 4)
+
+/* cpu interface registers, offsets from the GICC base */
+#define GICC_CTLR 0x000
+#define GICC_PMR 0x004
+
+/* GICD_CTLR bits */
+#define GICD_CTLR_ENABLE_GRP1 (1 << 0)
+#define GICD_CTLR_ENABLE_GRP0 (1 << 1)
+
+/* GICC_CTLR bits */
+#define GICC_CTLR_ENABLE (1 << 0)
+
+#define GICD_TYPER_ITLINES_MASK 0x1f
+
+/*
+ * how many 32-irq lines the distributor carries, TYPER.ITLines
+ * holds count of (irqs / 32) - 1, clamped per the spec because
+ * the field is 5 bits and caps at 1020 irqs.
+ */
+static int gicd_irq_lines(uintptr_t gicd)
+{
+ uint32_t typer = readl(REG32(gicd + GICD_TYPER));
+
+ return ((typer & GICD_TYPER_ITLINES_MASK) + 1);
+}
+
+/*
+ * leave the gic in the defined state the kernel expects. the
+ * addresses come from the devicetree the caller walked, qemu
+ * virt carries a gicv2 at 0x08000000 with the cpu interface at
+ * +0x10000.
+ */
+int tb_gic_init(uintptr_t gicd, uintptr_t gicc)
+{
+ int lines;
+ int n;
+
+ if (!gicd || !gicc)
+ return -1;
+
+ /* the distributor is off while it is reconfigured */
+ writel(0, REG32(gicd + GICD_CTLR));
+ writel(0, REG32(gicc + GICC_CTLR));
+
+ lines = gicd_irq_lines(gicd);
+
+ /*
+ * every interrupt in group 1, the non-secure group. the
+ * kernel does not see group 0 interrupts on non-secure
+ * hardware, and a bootloader that leaves any line in the
+ * secure group strands it.
+ */
+ for (n = 0; n < lines; n++)
+ writel(0xffffffff, REG32(gicd + GICD_IGROUPR(n)));
+
+ /* no per interrupt enables, nothing pending */
+ for (n = 0; n < lines; n++) {
+ writel(0xffffffff, REG32(gicd + GICD_ICENABLER(n)));
+ writel(0xffffffff, REG32(gicd + GICD_ICPENDR(n)));
+ }
+
+ /*
+ * the cpu interface stays off with the priority mask at
+ * the lowest priority, the kernel raises it when it
+ * brings its own irq handling up. off is the defined
+ * state, the enable is the kernel's decision to make.
+ */
+ writel(0, REG32(gicc + GICC_PMR));
+
+ return 0;
+}
diff --git a/common/dtb_find.c b/common/dtb_find.c
new file mode 100644
index 0000000..29a67f3
--- /dev/null
+++ b/common/dtb_find.c
@@ -0,0 +1,178 @@
+/*
+ * dtb_find.c - locate nodes and read reg by walking the flat
+ * devicetree. the machine tells the firmware where its devices
+ * live, a bootloader that hardcodes the gic address breaks on
+ * the first board with a different map.
+ *
+ * the walk is the standard token scan, FDT_BEGIN_NODE with a
+ * matching name at any depth, then the reg property inside,
+ * the first address/size pair decoded per the parent's cell
+ * counts, which the root carries in #address-cells and
+ * #size-cells.
+ *
+ * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org>
+ */
+
+#include <string.h>
+#include <stdint.h>
+#include <boot.h>
+
+#define FDT_BEGIN_NODE 1
+#define FDT_END_NODE 2
+#define FDT_PROP 3
+#define FDT_NOP 4
+#define FDT_END 9
+
+static uint32_t be32(const void *p)
+{
+ const uint8_t *b = p;
+
+ return ((uint32_t)b[0] << 24) | ((uint32_t)b[1] << 16) |
+ ((uint32_t)b[2] << 8) | (uint32_t)b[3];
+}
+
+static int name_eq(const char *a, const char *b)
+{
+ while (*a && *a != '@') {
+ if (*a != *b)
+ return 0;
+ a++;
+ b++;
+ }
+ return *b == '\0' || *b == '@';
+}
+
+/*
+ * find the first node whose name matches, at any depth. returns
+ * the offset of its FDT_BEGIN_NODE token or 0 when absent.
+ */
+static uint32_t fdt_find_node(uintptr_t dtb, const char *name)
+{
+ uint8_t *basep = (uint8_t *)dtb;
+ uint32_t off_struct = be32(basep + 8);
+ uint32_t totalsize = be32(basep + 4);
+ uint8_t *p = basep + off_struct;
+
+ if (be32(basep) != 0xd00dfeed)
+ return 0;
+
+ while (p < basep + totalsize) {
+ uint32_t token = be32(p);
+
+ if (token == FDT_BEGIN_NODE) {
+ char *n = (char *)(p + 4);
+ size_t nlen = strlen(n) + 1;
+
+ if (name_eq(n, name))
+ return (uint32_t)(p - basep);
+ p += 4 + ((nlen + 3) & ~3);
+ } else if (token == FDT_PROP) {
+ uint32_t plen = be32(p + 4);
+
+ p += 12 + ((plen + 3) & ~3);
+ } else if (token == FDT_END_NODE ||
+ token == FDT_NOP) {
+ p += 4;
+ } else if (token == FDT_END) {
+ break;
+ } else {
+ return 0;
+ }
+ }
+
+ return 0;
+}
+
+/*
+ * read the first reg pair of a node at the given token offset,
+ * honoring the root cell counts. pairs of 2 or 4 cells are the
+ * ones machines carry, anything else fails. the caller reads
+ * more pairs off the returned cursor if it needs them.
+ */
+int tb_dtb_reg0(uintptr_t dtb, uint32_t node_off, uintptr_t *addr,
+ size_t *size)
+{
+ uint8_t *basep = (uint8_t *)dtb;
+ uint32_t off_strings = be32(basep + 12);
+ uint8_t *p = basep + node_off;
+ uint32_t totalsize = be32(basep + 4);
+ uint32_t ac = 2;
+ uint32_t sc = 2;
+ /*
+ * zero, the node's own FDT_BEGIN_NODE below brings it to
+ * one and the props inside sit at depth one. starting at
+ * one instead skips every prop in the node.
+ */
+ int depth_open = 0;
+
+ while (p < basep + totalsize) {
+ uint32_t token = be32(p);
+
+ if (token == FDT_BEGIN_NODE) {
+ char *n = (char *)(p + 4);
+ size_t nlen = strlen(n) + 1;
+
+ depth_open++;
+ p += 4 + ((nlen + 3) & ~3);
+ } else if (token == FDT_END_NODE) {
+ depth_open--;
+ if (!depth_open)
+ return -1;
+ p += 4;
+ } else if (token == FDT_PROP) {
+ uint32_t plen = be32(p + 4);
+ const char *pname =
+ (char *)basep + off_strings + be32(p + 8);
+ uint8_t *val = p + 12;
+
+ if (depth_open == 1 &&
+ strcmp(pname, "#address-cells") == 0)
+ ac = be32(val);
+ if (depth_open == 1 &&
+ strcmp(pname, "#size-cells") == 0)
+ sc = be32(val);
+ if (depth_open == 1 && strcmp(pname, "reg") == 0) {
+ if (plen >= (ac + sc) * 4) {
+ uint64_t a = 0;
+ uint64_t s = 0;
+
+ for (uint32_t i = 0; i < ac; i++)
+ a = (a << 32) |
+ be32(val + i * 4);
+ for (uint32_t i = 0; i < sc; i++)
+ s = (s << 32) |
+ be32(val + (ac + i) * 4);
+ *addr = (uintptr_t)a;
+ if (size)
+ *size = (size_t)s;
+ return 0;
+ }
+ return -1;
+ }
+ p += 12 + ((plen + 3) & ~3);
+ } else if (token == FDT_NOP) {
+ p += 4;
+ } else if (token == FDT_END) {
+ return -1;
+ } else {
+ return -1;
+ }
+ }
+
+ return -1;
+}
+
+/*
+ * the whole lookup in one call: find the node, read its first
+ * reg pair.
+ */
+int tb_dtb_find_reg0(uintptr_t dtb, const char *name, uintptr_t *addr,
+ size_t *size)
+{
+ uint32_t off = fdt_find_node(dtb, name);
+
+ if (!off)
+ return -1;
+
+ return tb_dtb_reg0(dtb, off, addr, size);
+}
diff --git a/common/dtb_grow.c b/common/dtb_grow.c
new file mode 100644
index 0000000..309a43c
--- /dev/null
+++ b/common/dtb_grow.c
@@ -0,0 +1,177 @@
+/*
+ * dtb_grow.c - add properties to a node in a devicetree that has
+ * room, the relocated copy from dtb_reloc.c. the insert point is
+ * the node's FDT_END_NODE token, everything after it moves up by
+ * the inserted size, the header totalsize tracks it.
+ *
+ * the insert is safe when the node sits at the end of the struct
+ * block, which is the common shape, /chosen is created last by
+ * firmware and the tail behind it is two end tokens and the
+ * block end. the strings block sits after the grow room and
+ * never moves.
+ *
+ * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org>
+ */
+
+#include <string.h>
+#include <endian.h>
+#include <boot.h>
+#include <dtb_patch.h>
+
+#define FDT_BEGIN_NODE 1
+#define FDT_END_NODE 2
+#define FDT_PROP 3
+#define FDT_NOP 4
+#define FDT_END 9
+
+static uint32_t be32(const void *p)
+{
+ const uint8_t *b = p;
+
+ return ((uint32_t)b[0] << 24) | ((uint32_t)b[1] << 16) |
+ ((uint32_t)b[2] << 8) | (uint32_t)b[3];
+}
+
+static void put_be32(void *p, uint32_t v)
+{
+ uint8_t *b = p;
+
+ b[0] = (uint8_t)(v >> 24);
+ b[1] = (uint8_t)(v >> 16);
+ b[2] = (uint8_t)(v >> 8);
+ b[3] = (uint8_t)v;
+}
+
+static int name_eq(const char *a, const char *b)
+{
+ while (*a && *a != '@') {
+ if (*a != *b)
+ return 0;
+ a++;
+ b++;
+ }
+ return *b == '\0' || *b == '@';
+}
+
+/*
+ * insert one property into /chosen before its end token. value is
+ * copied as raw cells, len the byte count. name lands in the free
+ * space after the strings block. returns 0 or -1.
+ */
+int tb_dtb_add_chosen_prop(uintptr_t dtb, const char *name,
+ const void *val, size_t len)
+{
+ uint8_t *basep = (uint8_t *)dtb;
+ uint32_t off_struct = be32(basep + 8);
+ uint32_t off_strings = be32(basep + 12);
+ uint32_t totalsize = be32(basep + 4);
+ uint8_t *p = basep + off_struct;
+ uint8_t *ins;
+ size_t name_len = strlen(name) + 1;
+ size_t prop_size;
+ int depth = 0;
+ int in_chosen = 0;
+
+ if (be32(basep) != 0xd00dfeed)
+ return -1;
+
+ /* find the chosen node's end token, one level under the root */
+ while (p < basep + totalsize) {
+ uint32_t token = be32(p);
+
+ if (token == FDT_BEGIN_NODE) {
+ char *n = (char *)(p + 4);
+ size_t nlen = strlen(n) + 1;
+
+ depth++;
+ if (depth == 2 && name_eq(n, "chosen"))
+ in_chosen = 1;
+ p += 4 + ((nlen + 3) & ~3);
+ } else if (token == FDT_END_NODE) {
+ if (in_chosen && depth == 2) {
+ ins = p;
+ break;
+ }
+ depth--;
+ p += 4;
+ } else if (token == FDT_PROP) {
+ uint32_t plen = be32(p + 4);
+
+ p += 12 + ((plen + 3) & ~3);
+ } else if (token == FDT_NOP) {
+ p += 4;
+ } else if (token == FDT_END) {
+ break;
+ } else {
+ return -1;
+ }
+ }
+
+ if (!ins)
+ return -2;
+
+ ins = p;
+
+ /*
+ * the insert: the strings block moves up by prop_size so the
+ * struct block can grow into its old place, the struct tail
+ * after chosen moves up by prop_size, the new name lands at
+ * the end of the moved strings block, and totalsize covers
+ * both. prop name offsets are strings relative so they keep
+ * resolving after the move.
+ */
+ {
+ prop_size = 12 + ((len + 3) & ~3);
+ size_t strings_len = (size_t)be32(basep + 32);
+
+ /* strings block up by prop_size */
+ for (size_t i = strings_len; i > 0; i--)
+ basep[off_strings + prop_size + i - 1] =
+ basep[off_strings + i - 1];
+
+ /* struct tail after the insert point up by prop_size */
+ {
+ size_t tail = (size_t)(basep + off_strings - ins);
+
+ for (size_t i = tail; i > 0; i--)
+ ins[i + prop_size - 1] = ins[i - 1];
+ }
+
+ /* the prop token, name offset = old strings length */
+ put_be32(ins, FDT_PROP);
+ put_be32(ins + 4, (uint32_t)len);
+ put_be32(ins + 8, (uint32_t)strings_len);
+ for (size_t i = 0; i < len; i++)
+ ins[12 + i] = ((const uint8_t *)val)[i];
+ for (size_t i = len; i < ((len + 3) & ~3); i++)
+ ins[12 + i] = 0;
+
+ /* the name at the end of the moved strings block */
+ for (size_t i = 0; i < name_len; i++)
+ basep[off_strings + prop_size + strings_len + i] =
+ name[i];
+
+ /*
+ * size_dt_struct bounds the token walk, libfdt
+ * rejects anything past it as BADSTRUCTURE. it grows
+ * by the prop size here, the strings size by the name
+ * length, totalsize by both.
+ */
+ put_be32(basep + 4, totalsize + (uint32_t)prop_size +
+ (uint32_t)name_len);
+ put_be32(basep + 12, off_strings + (uint32_t)prop_size);
+ put_be32(basep + 36, be32(basep + 36) + (uint32_t)prop_size);
+ /*
+ * size_dt_strings must grow too, libfdt validates
+ * name offsets against it and rejects the whole tree
+ * when the new names sit past the declared end. the
+ * kernel's early parser is the same libfdt, a stale
+ * field there means no memory node and a page table
+ * panic before the first print.
+ */
+ put_be32(basep + 32, (uint32_t)strings_len +
+ (uint32_t)name_len);
+ }
+
+ return 0;
+}
diff --git a/common/dtb_patch.c b/common/dtb_patch.c
index 34d1a6c..cd9a1f3 100644
--- a/common/dtb_patch.c
+++ b/common/dtb_patch.c
@@ -32,6 +32,15 @@ static uint32_t be32(const void *p)
((uint32_t)b[2] << 8) | (uint32_t)b[3];
}
+static void put_be32(void *p, uint32_t v)
+{
+ uint8_t *b = p;
+ b[0] = (uint8_t)(v >> 24);
+ b[1] = (uint8_t)(v >> 16);
+ b[2] = (uint8_t)(v >> 8);
+ b[3] = (uint8_t)v;
+}
+
static void put_be64(void *p, uint64_t v)
{
uint8_t *b = p;
@@ -57,6 +66,76 @@ static int name_eq(const char *node, const char *want)
}
/*
+ * rewrite /memory reg with the RAM the bootloader actually sees.
+ * the value is two u32 cells, base and size, addresses above 4GB
+ * need the parent #address-cells respected, virt is below 4GB and
+ * 2 cells for size. returns 0 on success.
+ */
+int tb_dtb_patch_memory(uintptr_t dtb, uint64_t base, uint64_t size)
+{
+ uint8_t *basep = (uint8_t *)dtb;
+ uint32_t off_struct = be32(basep + 8);
+ uint32_t off_strings = be32(basep + 12);
+ uint8_t *p = basep + off_struct;
+ uint8_t *strings = basep + off_strings;
+ const char *cur_node = NULL;
+ int depth = 0;
+
+ if (be32(basep) != 0xd00dfeed)
+ return -1;
+
+ while (p < basep + be32(basep + 4)) {
+ uint32_t token = be32(p);
+
+ switch (token) {
+ case FDT_BEGIN_NODE: {
+ char *name = (char *)(p + 4);
+ size_t len = strlen(name) + 1;
+
+ p += 4 + ((len + 3) & ~3);
+ depth++;
+ cur_node = name;
+ break;
+ }
+ case FDT_END_NODE:
+ depth--;
+ p += 4;
+ break;
+ case FDT_PROP: {
+ uint32_t plen = be32(p + 4);
+ const char *pname = (char *)strings + be32(p + 8);
+ uint8_t *val = p + 12;
+
+ p += 12 + ((plen + 3) & ~3);
+
+ if (depth == 2 && name_eq(cur_node, "memory") &&
+ strcmp(pname, "reg") == 0 && plen >= 16) {
+ /*
+ * #address-cells 2, #size-cells 2, the
+ * reg is four cells, base hi lo and
+ * size hi lo, below 4GB the hi cells
+ * are zero.
+ */
+ put_be32(val, (uint32_t)(base >> 32));
+ put_be32(val + 4, (uint32_t)base);
+ put_be32(val + 8, (uint32_t)(size >> 32));
+ put_be32(val + 12, (uint32_t)size);
+ return 0;
+ }
+ break;
+ }
+ case FDT_NOP:
+ p += 4;
+ break;
+ case FDT_END:
+ return -2;
+ }
+ }
+
+ return -3;
+}
+
+/*
* walk and rewrite. returns the number of cpu-release-addr values
* written, negative on a malformed blob.
*/
@@ -140,3 +219,70 @@ int tb_dtb_patch_spin_table(uintptr_t dtb, uintptr_t *gates, int ngates)
return written;
}
+
+/*
+ * tell the kernel where the initrd landed. /chosen is created by
+ * the machine firmware, the two cells exist when an initrd was
+ * already staged, we overwrite them in place. depth 2 under the
+ * root, node name "chosen".
+ */
+int tb_dtb_patch_initrd(uintptr_t dtb, uint64_t start, uint64_t end)
+{
+ uint8_t *basep = (uint8_t *)dtb;
+ uint32_t off_struct = be32(basep + 8);
+ uint32_t off_strings = be32(basep + 12);
+ uint8_t *p = basep + off_struct;
+ uint8_t *strings = basep + off_strings;
+ const char *cur_node = NULL;
+ int depth = 0;
+
+ if (be32(basep) != 0xd00dfeed)
+ return -1;
+
+ while (p < basep + be32(basep + 4)) {
+ uint32_t token = be32(p);
+
+ switch (token) {
+ case FDT_BEGIN_NODE: {
+ char *name = (char *)(p + 4);
+ size_t len = strlen(name) + 1;
+
+ p += 4 + ((len + 3) & ~3);
+ depth++;
+ cur_node = name;
+ break;
+ }
+ case FDT_END_NODE:
+ depth--;
+ p += 4;
+ break;
+ case FDT_PROP: {
+ uint32_t plen = be32(p + 4);
+ const char *pname = (char *)strings + be32(p + 8);
+ uint8_t *val = p + 12;
+
+ p += 12 + ((plen + 3) & ~3);
+
+ if (depth == 2 && name_eq(cur_node, "chosen") &&
+ strcmp(pname, "linux,initrd-start") == 0 &&
+ plen >= 8) {
+ put_be64(val, start);
+ }
+ if (depth == 2 && name_eq(cur_node, "chosen") &&
+ strcmp(pname, "linux,initrd-end") == 0 &&
+ plen >= 8) {
+ put_be64(val, end);
+ return 0;
+ }
+ break;
+ }
+ case FDT_NOP:
+ p += 4;
+ break;
+ case FDT_END:
+ return -2;
+ }
+ }
+
+ return -2;
+}
diff --git a/common/dtb_reloc.c b/common/dtb_reloc.c
new file mode 100644
index 0000000..c3e0fe5
--- /dev/null
+++ b/common/dtb_reloc.c
@@ -0,0 +1,69 @@
+/*
+ * dtb_reloc.c - grow the devicetree the way libfdt does, in a
+ * buffer with room to spare. firmware cannot edit a packed fdt
+ * in place, new properties shift everything behind them, so the
+ * blob is copied into scratch verbatim, the free space after
+ * totalsize is the room the insert code shifts into, then the
+ * walkers patch the copy and the kernel gets its address.
+ *
+ * The layout follows the devicetree specification: header,
+ * struct block, strings block, free space. The rebuild copies
+ * header, struct, strings, fixes the offsets in the new header,
+ * and leaves the gap between struct and strings as the room new
+ * properties will consume.
+ *
+ * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org>
+ */
+
+#include <string.h>
+#include <endian.h>
+#include <boot.h>
+#include <dtb_patch.h>
+
+#define FDT_BEGIN_NODE 1
+#define FDT_END_NODE 2
+#define FDT_PROP 3
+#define FDT_NOP 4
+#define FDT_END 9
+
+static uint32_t be32(const void *p)
+{
+ const uint8_t *b = p;
+
+ return ((uint32_t)b[0] << 24) | ((uint32_t)b[1] << 16) |
+ ((uint32_t)b[2] << 8) | (uint32_t)b[3];
+}
+
+/*
+ * copy the blob into the scratch, grow bytes of headroom after
+ * the end. returns the new blob address or 0 on a short buffer.
+ */
+uintptr_t tb_dtb_relocate(uintptr_t dtb, void *scratch, size_t scratch_size,
+ size_t grow)
+{
+ uint8_t *in = (uint8_t *)dtb;
+ uint8_t *out = scratch;
+ uint32_t totalsize;
+
+ if (be32(in) != 0xd00dfeed)
+ return 0;
+
+ totalsize = be32(in + 4);
+
+ if (scratch_size < (size_t)totalsize + grow)
+ return 0;
+
+ /*
+ * verbatim copy, byte for byte. the grow room is the free
+ * scratch after totalsize, the insert code shifts the
+ * strings block into it. an interior gap between the
+ * struct and strings blocks only invites the walkers to
+ * count it as tree.
+ */
+ for (uint32_t i = 0; i < totalsize; i++)
+ out[i] = in[i];
+
+ (void)grow;
+
+ return (uintptr_t)out;
+}
diff --git a/common/load.c b/common/load.c
index 56fc96a..3a0f8b6 100644
--- a/common/load.c
+++ b/common/load.c
@@ -60,7 +60,7 @@ int tb_load_semihosting(const char *fname, uintptr_t load_addr,
* the initrd path, no header, no placement math, bytes to the
* address the dtb /chosen already names.
*/
-int tb_load_raw(const char *fname, uintptr_t load_addr)
+int tb_load_raw(const char *fname, uintptr_t load_addr, size_t *sizep)
{
long fd, len, ret;
@@ -80,5 +80,7 @@ int tb_load_raw(const char *fname, uintptr_t load_addr)
if (ret != len)
return -6;
+ if (sizep)
+ *sizep = (size_t)len;
return 0;
}
diff --git a/common/main.c b/common/main.c
index 0786027..a0237af 100644
--- a/common/main.c
+++ b/common/main.c
@@ -43,15 +43,24 @@
extern int tb_console_init(void);
/*
- * fixed load address, the osdev way. past the bootloader at the
- * bottom of RAM, the image header decides its final resting place.
+ * the payload goes 16MB clear of wherever this bootloader is
+ * actually running, ADR knows the runtime base and qemu is free
+ * to place us anywhere. hardcoding 0x40200000 smashed our own
+ * image when qemu loaded us there.
*/
-#define TB_LOAD_ADDR 0x40200000
+extern char __image_copy_end[];
+#define TB_LOAD_ADDR ((uintptr_t)__image_copy_end + (16ULL << 20))
/* the file semihosting serves as the payload */
#define TB_BOOTFILE "Image"
extern void __NO_RETURN tb_boot_linux(uintptr_t ep, uintptr_t fw_arg);
+extern uintptr_t tb_dtb_relocate(uintptr_t dtb, void *scratch,
+ size_t scratch_size, size_t grow);
+extern int tb_gic_init(uintptr_t gicd, uintptr_t gicc);
+extern int tb_dtb_add_chosen_prop(uintptr_t dtb, const char *name,
+ const void *val, size_t len);
+static size_t initrd_size;
void tashaboot_main(uintptr_t fw_arg)
{
@@ -81,6 +90,26 @@ void tashaboot_main(uintptr_t fw_arg)
#endif
{
+ /* report the RAM we actually live in, before any mmu */
+ extern int tb_dtb_patch_memory(uintptr_t dtb,
+ uint64_t base,
+ uint64_t size);
+ int r;
+
+ r = 0; (void)r;
+
+ {
+ uint32_t *cells = (uint32_t *)(fw_arg + 0x16c);
+ int i;
+
+ dprintf(ALWAYS, "cells after:");
+ for (i = 0; i < 4; i++)
+ dprintf(ALWAYS, " %08x", cells[i]);
+ dprintf(ALWAYS, "\n");
+ }
+ }
+
+ {
/* spin table gates into the dtb, one per cpu node */
extern unsigned long *tb_spin_gates_ptr;
extern int tb_dtb_patch_spin_table(uintptr_t dtb,
@@ -114,22 +143,97 @@ void tashaboot_main(uintptr_t fw_arg)
}
/*
+ * firmware owns the devicetree it hands the kernel. ours
+ * relocates into scratch with grow room, then the chosen
+ * properties are added there and the kernel gets the new
+ * address, the same flow libfdt firmware uses.
+ */
+ {
+ /*
+ * the scratch lives at a fixed free address, clear of
+ * our image, the payload, and the kernel relocation
+ * zone. a bss array would sit inside 0x40080000+ and
+ * the kernel overwrites it while copying itself.
+ */
+ uint8_t *dtb_scratch = (uint8_t *)0x45000000ULL;
+ uintptr_t newdtb;
+
+ newdtb = tb_dtb_relocate(fw_arg, dtb_scratch,
+ 0x10000, 0x200);
+ if (!newdtb) {
+ dprintf(ALWAYS, "dtb: relocate failed\n");
+ platform_halt();
+ }
+
+ fw_arg = newdtb;
+
+ /*
+ * the interrupt controller the machine told us
+ * about, found by name, the reg pair read with the
+ * root cell counts. the gic goes into the defined
+ * off state before the kernel brings its own irq
+ * handling up.
+ */
+ {
+ extern int tb_dtb_find_reg0(uintptr_t dtb,
+ const char *name,
+ uintptr_t *addr,
+ size_t *size);
+ uintptr_t gicd = 0;
+ uintptr_t gicc = 0;
+ size_t sz = 0;
+
+ if (tb_dtb_find_reg0(fw_arg, "intc", &gicd, &sz) == 0 &&
+ sz >= 0x10000) {
+ gicc = gicd + 0x10000;
+ tb_gic_init(gicd, gicc);
+ dprintf(ALWAYS, "gic: %lx off\n",
+ (unsigned long)gicd);
+ }
+ }
+ }
+
+ /*
* the initrd rides after the kernel, the dtb /chosen carries
* linux,initrd-start and -end, both already patched in place
* with this layout.
*/
{
- extern int tb_load_raw(const char *fname,
- uintptr_t load_addr);
- int r = tb_load_raw(TB_INITRD_FILE, TB_INITRD_ADDR);
+ int r = tb_load_raw(TB_INITRD_FILE, TB_INITRD_ADDR,
+ &initrd_size);
if (r == 0)
- dprintf(ALWAYS, "initrd at %lx\n",
- (unsigned long)TB_INITRD_ADDR);
+ dprintf(ALWAYS, "initrd at %lx, %lx bytes\n",
+ (unsigned long)TB_INITRD_ADDR,
+ (unsigned long)initrd_size);
else
dprintf(ALWAYS, "no initrd (%d)\n", r);
}
+ /*
+ * the chosen properties, written now that the initrd size
+ * is known. the cells are big endian, the fdt is a big
+ * endian format end to end.
+ */
+ {
+ uint8_t start_cells[8], end_cells[8];
+ uint64_t start = TB_INITRD_ADDR;
+ uint64_t end = TB_INITRD_ADDR + initrd_size;
+ int a, b;
+
+ for (int i = 0; i < 8; i++) {
+ start_cells[i] = (uint8_t)(start >> (56 - 8 * i));
+ end_cells[i] = (uint8_t)(end >> (56 - 8 * i));
+ }
+ a = tb_dtb_add_chosen_prop(fw_arg, "linux,initrd-start",
+ start_cells, 8);
+ b = tb_dtb_add_chosen_prop(fw_arg, "linux,initrd-end",
+ end_cells, 8);
+ dprintf(ALWAYS, "dtb: initrd props %d %d\n", a, b);
+ }
+
+
+
dprintf(ALWAYS, "loaded %llu bytes at %lx, entry %lx\n",
(unsigned long long)img.size, img.load, img.ep);
dprintf(ALWAYS, "jumping\n");
diff --git a/include/boot.h b/include/boot.h
index a28b47c..06b05d7 100644
--- a/include/boot.h
+++ b/include/boot.h
@@ -26,7 +26,7 @@ struct tb_image {
int tb_image_setup(uintptr_t image, struct tb_image *img);
/* common/load.c */
-int tb_load_raw(const char *fname, uintptr_t load_addr);
+int tb_load_raw(const char *fname, uintptr_t load_addr, size_t *sizep);
int tb_load_semihosting(const char *fname, uintptr_t load_addr,
struct tb_image *img);
diff --git a/tools/fillsize.py b/tools/fillsize.py
new file mode 100644
index 0000000..fa472d1
--- /dev/null
+++ b/tools/fillsize.py
@@ -0,0 +1,13 @@
+#!/usr/bin/env python3
+# fillsize.py - stamp image_size into the arm64 Image header of a
+# built binary. the linker cannot know the final file size, the
+# header field stays 0 through the link, this runs after objcopy.
+import struct
+import sys
+
+path = sys.argv[1]
+d = bytearray(open(path, 'rb').read())
+assert d[0x38:0x3c] == b'ARM\x64', 'no Image magic, refusing to stamp'
+struct.pack_into('<Q', d, 0x10, len(d))
+open(path, 'wb').write(d)
+print('image_size %d stamped' % len(d))