diff options
| author | Bradley Morgan <brads@mainlining.org> | 2026-10-04 00:32:49 +0000 |
|---|---|---|
| committer | Bradley Morgan <brads@mainlining.org> | 2026-10-04 00:32:49 +0000 |
| commit | d72c2f898ed4c17aba0080c8bf6a0173cca940dc (patch) | |
| tree | 7ecfd6d2b7095e61e3ce7a5ee35a8772822ca6a8 /arch/arm64/kernel/start.S | |
| parent | 4f700c280b55c551047c00c71ef14aacf5830872 (diff) | |
qemu -kernel parses a raw arm64 blob as a linux Image and enters
at RAMBASE plus whatever text_offset it guesses out of the
garbage, 0x80000 in our case. every wild PC at image+0x80000 in
the debug logs was our own code running from the wrong address.
the binary now carries a real Image header: code0 branches over
it, magic ARM\x64 at 0x38, text_offset 0, image_size stamped
after objcopy by tools/fillsize.py.
the runtime also split by exception level. the C body runs at
EL1, the semihosting hlt is answered by qemu only from EL2, so
the EL2 vector replays the trap there and erets home with the
result. the kernel handoff hvc raises back to EL2 where
booting.rst wants it, the same vector slot dispatches PSCI hvc
from the kernel, boot handoff and semihosting by EC and function
id.
the payload load address was hardcoded 0x40200000, which is where
qemu placed our image, so the load overwrote the running
bootloader with kernel bytes mid flight. the load address is now
__image_copy_end plus 16MB, wherever the image actually runs.
receipt: run /init, tashaboot linux userspace reached, cores: 4,
busybox shell on a 4 cpu virt machine with initrd.
Diffstat (limited to 'arch/arm64/kernel/start.S')
| -rw-r--r-- | arch/arm64/kernel/start.S | 86 |
1 files changed, 77 insertions, 9 deletions
diff --git a/arch/arm64/kernel/start.S b/arch/arm64/kernel/start.S index 969830d..6ee9941 100644 --- a/arch/arm64/kernel/start.S +++ b/arch/arm64/kernel/start.S @@ -14,12 +14,25 @@ .section .text.boot .globl _start _start: + /* code0: branch over the 64 byte Image header to reset */ b reset .balign 8 -.globl _text_base -_text_base: - .quad 0x40000000 +/* + * the arm64 Image header fields, per Documentation/arch/arm64/ + * booting.rst: text_offset 0x08, image_size 0x10, flags 0x18, + * magic 0x38. code0 above branches over all of it. text_offset 0 + * and image_size filled after link by tools/fillsize.py, the + * magic pins it as a proper Image so qemu -kernel enters at + * RAMBASE instead of guessing +0x80000. + */ + .quad 0x0 /* text_offset, 0x08, filled below */ + .quad 0x0 /* image_size, 0x10, filled below */ + .quad 0x0 /* flags, 0x18: LE, 4k pages, unset */ + .quad 0x0 /* reserved 0x20 */ + .quad 0x0 /* reserved 0x28 */ + .quad 0x0 /* reserved 0x30 */ + .quad 0x644d5241 /* magic, 0x38: ARM\x64 */ reset: /* keep the dtb pointer before anything clobbers x0 */ @@ -60,12 +73,14 @@ from_el3: from_el2: /* - * stay at EL2: the kernel wants it for the virtualization - * extensions and hands off from there. everything below scrubs - * the EL2 state so the kernel starts clean. + * scrub the EL2 state and drop to EL1 for the C runtime. the + * semihosting hlt trap is an EL1 service on qemu, calling it + * from EL2 corrupts the return state. the kernel handoff goes + * back to EL2, booting.rst prefers it there, through the + * trampoline in boot.S. */ - /* EL1 will be aarch64 when the kernel drops itself down */ + /* EL1 will be aarch64 */ mov x0, #(1 << 31) /* HCR_EL2.RW = 1 */ msr hcr_el2, x0 @@ -79,7 +94,12 @@ from_el2: msr hstr_el2, xzr msr vpidr_el2, xzr - b mmu_check + /* drop to EL1, SPSR EL1h with DAIF masked */ + mov x0, #0x3c5 + msr spsr_el2, x0 + adr x0, mmu_check + msr elr_el2, x0 + eret mmu_check: /* @@ -289,7 +309,16 @@ exc_sync: mrs x0, esr_el1 mrs x1, far_el1 2: - mov x2, lr + /* x2 = the faulting PC when it is the sync path */ + mrs x4, CurrentEL + lsr x4, x4, #2 + cmp x4, #2 + b.lt 3f + mrs x2, elr_el2 + b 4f +3: + mrs x2, elr_el1 +4: bl exc_report ldp x29, x30, [sp], #16 b park @@ -300,6 +329,45 @@ exc_sync: * the resume point, eret takes it back. */ hvc_from_el1: + /* + * the lower EL sync slot. three arrivals share it: PSCI hvc + * from the kernel (EC 0x16, PSCI id in x0), our own boot + * handoff (hvc with the payload entry in x8), and semihosting + * hlt #0xf000 from the EL1 C runtime (EC 0x14). qemu only + * answers the hlt when it executes at EL2, so the handler + * replays the trap at EL2 and erets home with the result. + */ + mrs x1, esr_el2 + lsr x1, x1, #26 /* EC */ + cmp x1, #0x14 /* HLT from lower EL, semihosting */ + b.eq smh_replay + + /* + * the hvc arrives with either a PSCI function id in x0 (the + * kernel calling) or the boot handoff staging the payload + * entry in x8 and the dtb in x0. PSCI ids have the 0x84/0xc4 + * prefix, a dtb pointer never does. + */ + lsr x1, x0, #24 + cmp x1, #0x84 + b.eq psci_call + cmp x1, #0xc4 + b.eq psci_call + + /* the boot handoff: ELR_EL2 = entry, eret to the payload */ + msr elr_el2, x8 + eret + +smh_replay: + /* + * x0 holds the semihosting syscall number, x1 the parameter + * block, both live in the caller's registers. replay the hlt + * here at EL2 where qemu answers it, then eret back. + */ + hlt #0xf000 + eret + +psci_call: stp x4, x5, [sp, #-16]! stp x6, x7, [sp, #-16]! stp x29, x30, [sp, #-16]! |
