diff options
Diffstat (limited to 'arch/arm64/kernel')
| -rw-r--r-- | arch/arm64/kernel/boot.S | 23 | ||||
| -rw-r--r-- | arch/arm64/kernel/start.S | 168 | ||||
| -rw-r--r-- | arch/arm64/kernel/tashaboot.lds | 24 |
3 files changed, 203 insertions, 12 deletions
diff --git a/arch/arm64/kernel/boot.S b/arch/arm64/kernel/boot.S index 615be06..d8b888f 100644 --- a/arch/arm64/kernel/boot.S +++ b/arch/arm64/kernel/boot.S @@ -30,6 +30,20 @@ ENTRY(tb_boot_linux) mov x2, xzr mov x3, xzr + /* + * raise to EL2 for the payload when EL2 exists, the kernel + * prefers it there (booting.rst). hvc from EL1 lands in our + * EL2 vector slot, the dispatcher sees the non PSCI function + * id, stages ELR_EL2 with the entry and erets to the payload. + * on an EL1 only machine this is a straight branch. + */ + mrs x9, CurrentEL + lsr x9, x9, #2 + cmp x9, #2 + b.lt 5f + hvc #0 +5: + /* MMU off, caches off, the kernel sets up its own state */ mrs x9, sctlr_el1 bic x9, x9, #(1 << 0) /* M, MMU */ @@ -38,6 +52,15 @@ ENTRY(tb_boot_linux) msr sctlr_el1, x9 isb + /* + * if we entered at EL2, the kernel prefers it there. the C + * runtime ran at EL1 for semihosting, so raise back: hvc to + * our own EL2 vectors would need a live handler, instead the + * entry saved the EL2 state and we simply reenter it through + * the tb_el2_trampoline the entry installed. + */ + + br x8 ENDPROC(tb_boot_linux) .popsection diff --git a/arch/arm64/kernel/start.S b/arch/arm64/kernel/start.S index 705721f..6ee9941 100644 --- a/arch/arm64/kernel/start.S +++ b/arch/arm64/kernel/start.S @@ -14,12 +14,25 @@ .section .text.boot .globl _start _start: + /* code0: branch over the 64 byte Image header to reset */ b reset .balign 8 -.globl _text_base -_text_base: - .quad 0x40000000 +/* + * the arm64 Image header fields, per Documentation/arch/arm64/ + * booting.rst: text_offset 0x08, image_size 0x10, flags 0x18, + * magic 0x38. code0 above branches over all of it. text_offset 0 + * and image_size filled after link by tools/fillsize.py, the + * magic pins it as a proper Image so qemu -kernel enters at + * RAMBASE instead of guessing +0x80000. + */ + .quad 0x0 /* text_offset, 0x08, filled below */ + .quad 0x0 /* image_size, 0x10, filled below */ + .quad 0x0 /* flags, 0x18: LE, 4k pages, unset */ + .quad 0x0 /* reserved 0x20 */ + .quad 0x0 /* reserved 0x28 */ + .quad 0x0 /* reserved 0x30 */ + .quad 0x644d5241 /* magic, 0x38: ARM\x64 */ reset: /* keep the dtb pointer before anything clobbers x0 */ @@ -60,12 +73,14 @@ from_el3: from_el2: /* - * stay at EL2: the kernel wants it for the virtualization - * extensions and hands off from there. everything below scrubs - * the EL2 state so the kernel starts clean. + * scrub the EL2 state and drop to EL1 for the C runtime. the + * semihosting hlt trap is an EL1 service on qemu, calling it + * from EL2 corrupts the return state. the kernel handoff goes + * back to EL2, booting.rst prefers it there, through the + * trampoline in boot.S. */ - /* EL1 will be aarch64 when the kernel drops itself down */ + /* EL1 will be aarch64 */ mov x0, #(1 << 31) /* HCR_EL2.RW = 1 */ msr hcr_el2, x0 @@ -79,7 +94,12 @@ from_el2: msr hstr_el2, xzr msr vpidr_el2, xzr - b mmu_check + /* drop to EL1, SPSR EL1h with DAIF masked */ + mov x0, #0x3c5 + msr spsr_el2, x0 + adr x0, mmu_check + msr elr_el2, x0 + eret mmu_check: /* @@ -143,10 +163,15 @@ c_entry: 3: isb - /* stack for the bootloader, grows down from the image end */ - ldr x0, =__image_end + /* stack for the bootloader, its own region above the bss */ + ldr x0, =__stack_top mov sp, x0 + /* export the spin gate array address for the dtb patcher */ + adr x0, tb_spin_gates + adrp x1, tb_spin_gates_ptr + str x0, [x1, #:lo12:tb_spin_gates_ptr] + /* clear bss */ ldr x0, =__bss_start ldr x1, =__bss_end @@ -166,9 +191,44 @@ c_entry: bl tashaboot_main /* if main returns there is nothing sensible to do */ +/* + * the spin table pen, the Wait For Event mechanism from the manual + * (B2-144, D1-2255). each secondary watches its own gate, the + * cpu-release-addr the dtb names. WFE clears the event register and + * sleeps, the kernel writes the secondary entry to the gate, makes + * it visible, then SEV sets the event register on every PE. the load + * recheck after each wake covers a release that lands between the + * load and the WFE. entered with MMU and caches off, left the same. + */ +.globl park_ret +park_ret: park: + adr x0, tb_spin_gates + mrs x1, mpidr_el1 + and x1, x1, #0xff /* affinity 0, the core number */ + add x0, x0, x1, lsl #3 /* gate = gates + core * 8 */ + + /* diagnostic: stamp arrival, primary prints it later */ + adr x3, tb_pen_stamps + strb w1, [x3, x1] + sevl + wfe + sevl wfe - b park + +1: + ldr x2, [x0] + cbnz x2, 2f + wfe + b 1b +2: + mov x0, xzr /* secondaries enter with x0-x3 zero */ + mov x1, xzr + mov x2, xzr + mov x3, xzr + dsb sy + isb + br x2 /* * exception vectors, the armv8 layout: 16 slots, 128 bytes each, in @@ -218,6 +278,19 @@ vectors: .align 7 b exc_serr +.pushsection .data.tb_spin, "aw" +.align 3 +.globl tb_spin_gates +tb_spin_gates: + .quad 0, 0, 0, 0, 0, 0, 0, 0 +.globl tb_spin_gates_ptr +tb_spin_gates_ptr: + .quad 0 +.globl tb_pen_stamps +tb_pen_stamps: + .byte 0, 0, 0, 0, 0, 0, 0, 0 +.popsection + exc_sync: stp x29, x30, [sp, #-16]! mov x29, sp @@ -226,17 +299,88 @@ exc_sync: cmp x3, #2 b.lt 1f mrs x0, esr_el2 + mrs x2, elr_el2 + lsr x1, x0, #26 + cmp x1, #0x16 /* HVC from lower EL */ + b.eq hvc_from_el1 mrs x1, far_el2 b 2f 1: mrs x0, esr_el1 mrs x1, far_el1 2: - mov x2, lr + /* x2 = the faulting PC when it is the sync path */ + mrs x4, CurrentEL + lsr x4, x4, #2 + cmp x4, #2 + b.lt 3f + mrs x2, elr_el2 + b 4f +3: + mrs x2, elr_el1 +4: bl exc_report ldp x29, x30, [sp], #16 b park +/* + * HVC from EL1, the PSCI conduit. x0-x3 are the PSCI args in the + * caller registers, dispatch and return in x0. ELR_EL2 is already + * the resume point, eret takes it back. + */ +hvc_from_el1: + /* + * the lower EL sync slot. three arrivals share it: PSCI hvc + * from the kernel (EC 0x16, PSCI id in x0), our own boot + * handoff (hvc with the payload entry in x8), and semihosting + * hlt #0xf000 from the EL1 C runtime (EC 0x14). qemu only + * answers the hlt when it executes at EL2, so the handler + * replays the trap at EL2 and erets home with the result. + */ + mrs x1, esr_el2 + lsr x1, x1, #26 /* EC */ + cmp x1, #0x14 /* HLT from lower EL, semihosting */ + b.eq smh_replay + + /* + * the hvc arrives with either a PSCI function id in x0 (the + * kernel calling) or the boot handoff staging the payload + * entry in x8 and the dtb in x0. PSCI ids have the 0x84/0xc4 + * prefix, a dtb pointer never does. + */ + lsr x1, x0, #24 + cmp x1, #0x84 + b.eq psci_call + cmp x1, #0xc4 + b.eq psci_call + + /* the boot handoff: ELR_EL2 = entry, eret to the payload */ + msr elr_el2, x8 + eret + +smh_replay: + /* + * x0 holds the semihosting syscall number, x1 the parameter + * block, both live in the caller's registers. replay the hlt + * here at EL2 where qemu answers it, then eret back. + */ + hlt #0xf000 + eret + +psci_call: + stp x4, x5, [sp, #-16]! + stp x6, x7, [sp, #-16]! + stp x29, x30, [sp, #-16]! + mov x29, sp + + bl tb_psci_dispatch + + ldp x29, x30, [sp], #16 + ldp x6, x7, [sp], #16 + ldp x4, x5, [sp], #16 + ldp x29, x30, [sp], #16 + eret + exc_serr: stp x29, x30, [sp, #-16]! mov x29, sp diff --git a/arch/arm64/kernel/tashaboot.lds b/arch/arm64/kernel/tashaboot.lds index 4f8dfb1..e2b7954 100644 --- a/arch/arm64/kernel/tashaboot.lds +++ b/arch/arm64/kernel/tashaboot.lds @@ -12,6 +12,14 @@ ENTRY(_start) SECTIONS { + /* + * the first 64 bytes are the arm64 Image header: code0 'b' over + * it, magic ARM\x64, text_offset 0. qemu -kernel parses the + * header, loads the file at 0x40000000 and enters at + * 0x40000000, where the branch lands on reset at 0x40000040. + * without the header qemu guesses text_offset 0x80000 and runs + * the whole loader from the wrong address. + */ . = 0x40000000; __image_copy_start = .; @@ -21,6 +29,12 @@ SECTIONS { arch/arm64/kernel/start.o (.text.boot) *(.text.boot) + + /* the Image header, code0 branches over it */ + . = ALIGN(64); + *(.text.imgheader) + . = ALIGN(64); + *(.text*) } @@ -52,6 +66,16 @@ SECTIONS . = ALIGN(8); __bss_end = .; + /* + * the stack lives in its own region, clear of bss. page tables + * and buffers are bss objects, a stack sharing their address + * space grows down into them and the first deep call crushes + * whatever it meets. + */ + . = ALIGN(4096); + __stack_bottom = .; + . += 0x4000; + __stack_top = .; __image_copy_end = .; /DISCARD/ : { *(.dynsym) } |
