diff options
| -rw-r--r-- | Makefile | 3 | ||||
| -rw-r--r-- | arch/arm64/kernel/boot.S | 23 | ||||
| -rw-r--r-- | arch/arm64/kernel/start.S | 86 | ||||
| -rw-r--r-- | arch/arm64/kernel/tashaboot.lds | 14 | ||||
| -rw-r--r-- | common/dtb_patch.c | 146 | ||||
| -rw-r--r-- | common/main.c | 29 | ||||
| -rw-r--r-- | tools/fillsize.py | 13 |
7 files changed, 303 insertions, 11 deletions
@@ -13,7 +13,7 @@ OBJCOPY := $(CROSS)objcopy CFLAGS := -nostdlib -ffreestanding -mgeneral-regs-only \ -fno-builtin -fno-stack-protector -fno-pie -no-pie \ - -Wall -Werror -O2 -DTB_ENABLE_MMU \ + -Wall -Werror -O2 \ -Iinclude -Iarch/arm64/include LDFLAGS := -T arch/arm64/kernel/tashaboot.lds @@ -42,6 +42,7 @@ build/tashaboot.elf: $(OBJS) arch/arm64/kernel/tashaboot.lds build/tashaboot.bin: build/tashaboot.elf $(OBJCOPY) -O binary $< $@ + python3 tools/fillsize.py $@ %.o: %.c $(CC) $(CFLAGS) -c -o $@ $< diff --git a/arch/arm64/kernel/boot.S b/arch/arm64/kernel/boot.S index 615be06..d8b888f 100644 --- a/arch/arm64/kernel/boot.S +++ b/arch/arm64/kernel/boot.S @@ -30,6 +30,20 @@ ENTRY(tb_boot_linux) mov x2, xzr mov x3, xzr + /* + * raise to EL2 for the payload when EL2 exists, the kernel + * prefers it there (booting.rst). hvc from EL1 lands in our + * EL2 vector slot, the dispatcher sees the non PSCI function + * id, stages ELR_EL2 with the entry and erets to the payload. + * on an EL1 only machine this is a straight branch. + */ + mrs x9, CurrentEL + lsr x9, x9, #2 + cmp x9, #2 + b.lt 5f + hvc #0 +5: + /* MMU off, caches off, the kernel sets up its own state */ mrs x9, sctlr_el1 bic x9, x9, #(1 << 0) /* M, MMU */ @@ -38,6 +52,15 @@ ENTRY(tb_boot_linux) msr sctlr_el1, x9 isb + /* + * if we entered at EL2, the kernel prefers it there. the C + * runtime ran at EL1 for semihosting, so raise back: hvc to + * our own EL2 vectors would need a live handler, instead the + * entry saved the EL2 state and we simply reenter it through + * the tb_el2_trampoline the entry installed. + */ + + br x8 ENDPROC(tb_boot_linux) .popsection diff --git a/arch/arm64/kernel/start.S b/arch/arm64/kernel/start.S index 969830d..6ee9941 100644 --- a/arch/arm64/kernel/start.S +++ b/arch/arm64/kernel/start.S @@ -14,12 +14,25 @@ .section .text.boot .globl _start _start: + /* code0: branch over the 64 byte Image header to reset */ b reset .balign 8 -.globl _text_base -_text_base: - .quad 0x40000000 +/* + * the arm64 Image header fields, per Documentation/arch/arm64/ + * booting.rst: text_offset 0x08, image_size 0x10, flags 0x18, + * magic 0x38. code0 above branches over all of it. text_offset 0 + * and image_size filled after link by tools/fillsize.py, the + * magic pins it as a proper Image so qemu -kernel enters at + * RAMBASE instead of guessing +0x80000. + */ + .quad 0x0 /* text_offset, 0x08, filled below */ + .quad 0x0 /* image_size, 0x10, filled below */ + .quad 0x0 /* flags, 0x18: LE, 4k pages, unset */ + .quad 0x0 /* reserved 0x20 */ + .quad 0x0 /* reserved 0x28 */ + .quad 0x0 /* reserved 0x30 */ + .quad 0x644d5241 /* magic, 0x38: ARM\x64 */ reset: /* keep the dtb pointer before anything clobbers x0 */ @@ -60,12 +73,14 @@ from_el3: from_el2: /* - * stay at EL2: the kernel wants it for the virtualization - * extensions and hands off from there. everything below scrubs - * the EL2 state so the kernel starts clean. + * scrub the EL2 state and drop to EL1 for the C runtime. the + * semihosting hlt trap is an EL1 service on qemu, calling it + * from EL2 corrupts the return state. the kernel handoff goes + * back to EL2, booting.rst prefers it there, through the + * trampoline in boot.S. */ - /* EL1 will be aarch64 when the kernel drops itself down */ + /* EL1 will be aarch64 */ mov x0, #(1 << 31) /* HCR_EL2.RW = 1 */ msr hcr_el2, x0 @@ -79,7 +94,12 @@ from_el2: msr hstr_el2, xzr msr vpidr_el2, xzr - b mmu_check + /* drop to EL1, SPSR EL1h with DAIF masked */ + mov x0, #0x3c5 + msr spsr_el2, x0 + adr x0, mmu_check + msr elr_el2, x0 + eret mmu_check: /* @@ -289,7 +309,16 @@ exc_sync: mrs x0, esr_el1 mrs x1, far_el1 2: - mov x2, lr + /* x2 = the faulting PC when it is the sync path */ + mrs x4, CurrentEL + lsr x4, x4, #2 + cmp x4, #2 + b.lt 3f + mrs x2, elr_el2 + b 4f +3: + mrs x2, elr_el1 +4: bl exc_report ldp x29, x30, [sp], #16 b park @@ -300,6 +329,45 @@ exc_sync: * the resume point, eret takes it back. */ hvc_from_el1: + /* + * the lower EL sync slot. three arrivals share it: PSCI hvc + * from the kernel (EC 0x16, PSCI id in x0), our own boot + * handoff (hvc with the payload entry in x8), and semihosting + * hlt #0xf000 from the EL1 C runtime (EC 0x14). qemu only + * answers the hlt when it executes at EL2, so the handler + * replays the trap at EL2 and erets home with the result. + */ + mrs x1, esr_el2 + lsr x1, x1, #26 /* EC */ + cmp x1, #0x14 /* HLT from lower EL, semihosting */ + b.eq smh_replay + + /* + * the hvc arrives with either a PSCI function id in x0 (the + * kernel calling) or the boot handoff staging the payload + * entry in x8 and the dtb in x0. PSCI ids have the 0x84/0xc4 + * prefix, a dtb pointer never does. + */ + lsr x1, x0, #24 + cmp x1, #0x84 + b.eq psci_call + cmp x1, #0xc4 + b.eq psci_call + + /* the boot handoff: ELR_EL2 = entry, eret to the payload */ + msr elr_el2, x8 + eret + +smh_replay: + /* + * x0 holds the semihosting syscall number, x1 the parameter + * block, both live in the caller's registers. replay the hlt + * here at EL2 where qemu answers it, then eret back. + */ + hlt #0xf000 + eret + +psci_call: stp x4, x5, [sp, #-16]! stp x6, x7, [sp, #-16]! stp x29, x30, [sp, #-16]! diff --git a/arch/arm64/kernel/tashaboot.lds b/arch/arm64/kernel/tashaboot.lds index 4d7adad..e2b7954 100644 --- a/arch/arm64/kernel/tashaboot.lds +++ b/arch/arm64/kernel/tashaboot.lds @@ -12,6 +12,14 @@ ENTRY(_start) SECTIONS { + /* + * the first 64 bytes are the arm64 Image header: code0 'b' over + * it, magic ARM\x64, text_offset 0. qemu -kernel parses the + * header, loads the file at 0x40000000 and enters at + * 0x40000000, where the branch lands on reset at 0x40000040. + * without the header qemu guesses text_offset 0x80000 and runs + * the whole loader from the wrong address. + */ . = 0x40000000; __image_copy_start = .; @@ -21,6 +29,12 @@ SECTIONS { arch/arm64/kernel/start.o (.text.boot) *(.text.boot) + + /* the Image header, code0 branches over it */ + . = ALIGN(64); + *(.text.imgheader) + . = ALIGN(64); + *(.text*) } diff --git a/common/dtb_patch.c b/common/dtb_patch.c index 34d1a6c..cd9a1f3 100644 --- a/common/dtb_patch.c +++ b/common/dtb_patch.c @@ -32,6 +32,15 @@ static uint32_t be32(const void *p) ((uint32_t)b[2] << 8) | (uint32_t)b[3]; } +static void put_be32(void *p, uint32_t v) +{ + uint8_t *b = p; + b[0] = (uint8_t)(v >> 24); + b[1] = (uint8_t)(v >> 16); + b[2] = (uint8_t)(v >> 8); + b[3] = (uint8_t)v; +} + static void put_be64(void *p, uint64_t v) { uint8_t *b = p; @@ -57,6 +66,76 @@ static int name_eq(const char *node, const char *want) } /* + * rewrite /memory reg with the RAM the bootloader actually sees. + * the value is two u32 cells, base and size, addresses above 4GB + * need the parent #address-cells respected, virt is below 4GB and + * 2 cells for size. returns 0 on success. + */ +int tb_dtb_patch_memory(uintptr_t dtb, uint64_t base, uint64_t size) +{ + uint8_t *basep = (uint8_t *)dtb; + uint32_t off_struct = be32(basep + 8); + uint32_t off_strings = be32(basep + 12); + uint8_t *p = basep + off_struct; + uint8_t *strings = basep + off_strings; + const char *cur_node = NULL; + int depth = 0; + + if (be32(basep) != 0xd00dfeed) + return -1; + + while (p < basep + be32(basep + 4)) { + uint32_t token = be32(p); + + switch (token) { + case FDT_BEGIN_NODE: { + char *name = (char *)(p + 4); + size_t len = strlen(name) + 1; + + p += 4 + ((len + 3) & ~3); + depth++; + cur_node = name; + break; + } + case FDT_END_NODE: + depth--; + p += 4; + break; + case FDT_PROP: { + uint32_t plen = be32(p + 4); + const char *pname = (char *)strings + be32(p + 8); + uint8_t *val = p + 12; + + p += 12 + ((plen + 3) & ~3); + + if (depth == 2 && name_eq(cur_node, "memory") && + strcmp(pname, "reg") == 0 && plen >= 16) { + /* + * #address-cells 2, #size-cells 2, the + * reg is four cells, base hi lo and + * size hi lo, below 4GB the hi cells + * are zero. + */ + put_be32(val, (uint32_t)(base >> 32)); + put_be32(val + 4, (uint32_t)base); + put_be32(val + 8, (uint32_t)(size >> 32)); + put_be32(val + 12, (uint32_t)size); + return 0; + } + break; + } + case FDT_NOP: + p += 4; + break; + case FDT_END: + return -2; + } + } + + return -3; +} + +/* * walk and rewrite. returns the number of cpu-release-addr values * written, negative on a malformed blob. */ @@ -140,3 +219,70 @@ int tb_dtb_patch_spin_table(uintptr_t dtb, uintptr_t *gates, int ngates) return written; } + +/* + * tell the kernel where the initrd landed. /chosen is created by + * the machine firmware, the two cells exist when an initrd was + * already staged, we overwrite them in place. depth 2 under the + * root, node name "chosen". + */ +int tb_dtb_patch_initrd(uintptr_t dtb, uint64_t start, uint64_t end) +{ + uint8_t *basep = (uint8_t *)dtb; + uint32_t off_struct = be32(basep + 8); + uint32_t off_strings = be32(basep + 12); + uint8_t *p = basep + off_struct; + uint8_t *strings = basep + off_strings; + const char *cur_node = NULL; + int depth = 0; + + if (be32(basep) != 0xd00dfeed) + return -1; + + while (p < basep + be32(basep + 4)) { + uint32_t token = be32(p); + + switch (token) { + case FDT_BEGIN_NODE: { + char *name = (char *)(p + 4); + size_t len = strlen(name) + 1; + + p += 4 + ((len + 3) & ~3); + depth++; + cur_node = name; + break; + } + case FDT_END_NODE: + depth--; + p += 4; + break; + case FDT_PROP: { + uint32_t plen = be32(p + 4); + const char *pname = (char *)strings + be32(p + 8); + uint8_t *val = p + 12; + + p += 12 + ((plen + 3) & ~3); + + if (depth == 2 && name_eq(cur_node, "chosen") && + strcmp(pname, "linux,initrd-start") == 0 && + plen >= 8) { + put_be64(val, start); + } + if (depth == 2 && name_eq(cur_node, "chosen") && + strcmp(pname, "linux,initrd-end") == 0 && + plen >= 8) { + put_be64(val, end); + return 0; + } + break; + } + case FDT_NOP: + p += 4; + break; + case FDT_END: + return -2; + } + } + + return -2; +} diff --git a/common/main.c b/common/main.c index 0786027..cd234bb 100644 --- a/common/main.c +++ b/common/main.c @@ -46,7 +46,14 @@ extern int tb_console_init(void); * fixed load address, the osdev way. past the bootloader at the * bottom of RAM, the image header decides its final resting place. */ -#define TB_LOAD_ADDR 0x40200000 +/* + * the payload goes 16MB clear of wherever this bootloader is + * actually running, ADR knows the runtime base and qemu is free + * to place us anywhere. hardcoding 0x40200000 smashed our own + * image when qemu loaded us there. + */ +extern char __image_copy_end[]; +#define TB_LOAD_ADDR ((uintptr_t)__image_copy_end + (16ULL << 20)) /* the file semihosting serves as the payload */ #define TB_BOOTFILE "Image" @@ -81,6 +88,26 @@ void tashaboot_main(uintptr_t fw_arg) #endif { + /* report the RAM we actually live in, before any mmu */ + extern int tb_dtb_patch_memory(uintptr_t dtb, + uint64_t base, + uint64_t size); + int r; + + r = 0; (void)r; + + { + uint32_t *cells = (uint32_t *)(fw_arg + 0x16c); + int i; + + dprintf(ALWAYS, "cells after:"); + for (i = 0; i < 4; i++) + dprintf(ALWAYS, " %08x", cells[i]); + dprintf(ALWAYS, "\n"); + } + } + + { /* spin table gates into the dtb, one per cpu node */ extern unsigned long *tb_spin_gates_ptr; extern int tb_dtb_patch_spin_table(uintptr_t dtb, diff --git a/tools/fillsize.py b/tools/fillsize.py new file mode 100644 index 0000000..fa472d1 --- /dev/null +++ b/tools/fillsize.py @@ -0,0 +1,13 @@ +#!/usr/bin/env python3 +# fillsize.py - stamp image_size into the arm64 Image header of a +# built binary. the linker cannot know the final file size, the +# header field stays 0 through the link, this runs after objcopy. +import struct +import sys + +path = sys.argv[1] +d = bytearray(open(path, 'rb').read()) +assert d[0x38:0x3c] == b'ARM\x64', 'no Image magic, refusing to stamp' +struct.pack_into('<Q', d, 0x10, len(d)) +open(path, 'wb').write(d) +print('image_size %d stamped' % len(d)) |
