summaryrefslogtreecommitdiff
path: root/arch
diff options
context:
space:
mode:
authorBradley Morgan <brads@mainlining.org>2026-10-03 18:57:38 +0000
committerBradley Morgan <brads@mainlining.org>2026-10-03 18:57:38 +0000
commit00fc82f17cff9295387d516384ecdc443acb1b22 (patch)
tree798bef0185878286028fe0eaf7594616a66903e4 /arch
parent7e8514efb64774b6a635b87ae945e519a7c38551 (diff)
tashaboot: arm64 bringup, semihosting payload loader
consoles on the pl011 the dtb points at, walks the memory node and chosen, loads an arm64 kernel Image over semihosting and jumps to it with the documented register state. string functions, printf and libfdt are vendored from lk (lk2nd) and u-boot so the libc surface is the one those projects already proved. built and booted on qemu virt, payload exits clean through semihosting.
Diffstat (limited to 'arch')
-rw-r--r--arch/arm64/include/asm/linkage.h14
-rw-r--r--arch/arm64/include/asm/macro.h347
-rw-r--r--arch/arm64/kernel/boot.S35
-rw-r--r--arch/arm64/kernel/start.S94
-rw-r--r--arch/arm64/kernel/tashaboot.lds67
-rw-r--r--arch/arm64/lib/cache.S100
-rw-r--r--arch/arm64/lib/semihosting.S18
7 files changed, 675 insertions, 0 deletions
diff --git a/arch/arm64/include/asm/linkage.h b/arch/arm64/include/asm/linkage.h
new file mode 100644
index 0000000..b5b9706
--- /dev/null
+++ b/arch/arm64/include/asm/linkage.h
@@ -0,0 +1,14 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+#ifndef __ASM_LINKAGE_H
+#define __ASM_LINKAGE_H
+
+#define ALIGN .p2align 4
+#define ENTRY(name) \
+ .globl name; \
+ ALIGN; \
+ name:
+#define ENDPROC(name) \
+ .type name, %function; \
+ .size name, .-name
+
+#endif
diff --git a/arch/arm64/include/asm/macro.h b/arch/arm64/include/asm/macro.h
new file mode 100644
index 0000000..1a1edc9
--- /dev/null
+++ b/arch/arm64/include/asm/macro.h
@@ -0,0 +1,347 @@
+/* SPDX-License-Identifier: GPL-2.0+ */
+/*
+ * include/asm-arm/macro.h
+ *
+ * Copyright (C) 2009 Jean-Christophe PLAGNIOL-VILLARD <plagnioj@jcrosoft.com>
+ */
+
+#ifndef __ASM_ARM_MACRO_H__
+#define __ASM_ARM_MACRO_H__
+
+#ifdef CONFIG_ARM64
+#include <asm/system.h>
+#endif
+
+#ifdef __ASSEMBLY__
+
+/*
+ * These macros provide a convenient way to write 8, 16 and 32 bit data
+ * to any address.
+ * Registers r4 and r5 are used, any data in these registers are
+ * overwritten by the macros.
+ * The macros are valid for any ARM architecture, they do not implement
+ * any memory barriers so caution is recommended when using these when the
+ * caches are enabled or on a multi-core system.
+ */
+
+.macro write32, addr, data
+ ldr r4, =\addr
+ ldr r5, =\data
+ str r5, [r4]
+.endm
+
+.macro write16, addr, data
+ ldr r4, =\addr
+ ldrh r5, =\data
+ strh r5, [r4]
+.endm
+
+.macro write8, addr, data
+ ldr r4, =\addr
+ ldrb r5, =\data
+ strb r5, [r4]
+.endm
+
+/*
+ * This macro generates a loop that can be used for delays in the code.
+ * Register r4 is used, any data in this register is overwritten by the
+ * macro.
+ * The macro is valid for any ARM architeture. The actual time spent in the
+ * loop will vary from CPU to CPU though.
+ */
+
+.macro wait_timer, time
+ ldr r4, =\time
+1:
+ nop
+ subs r4, r4, #1
+ bcs 1b
+.endm
+
+#ifdef CONFIG_ARM64
+/*
+ * Register aliases.
+ */
+lr .req x30
+
+/*
+ * Branch according to exception level
+ */
+.macro switch_el, xreg, el3_label, el2_label, el1_label
+ mrs \xreg, CurrentEL
+ cmp \xreg, #0x8
+ b.gt \el3_label
+ b.eq \el2_label
+ b.lt \el1_label
+.endm
+
+/*
+ * Branch if we are not in the highest exception level
+ */
+.macro branch_if_not_highest_el, xreg, label
+ switch_el \xreg, 3f, 2f, 1f
+
+2: mrs \xreg, ID_AA64PFR0_EL1
+ and \xreg, \xreg, #(ID_AA64PFR0_EL1_EL3)
+ cbnz \xreg, \label
+ b 3f
+
+1: mrs \xreg, ID_AA64PFR0_EL1
+ and \xreg, \xreg, #(ID_AA64PFR0_EL1_EL3 | ID_AA64PFR0_EL1_EL2)
+ cbnz \xreg, \label
+
+3:
+.endm
+
+/*
+ * Branch if current processor is a Cortex-A57 core.
+ */
+.macro branch_if_a57_core, xreg, a57_label
+ mrs \xreg, midr_el1
+ lsr \xreg, \xreg, #4
+ and \xreg, \xreg, #0x00000FFF
+ cmp \xreg, #0xD07 /* Cortex-A57 MPCore processor. */
+ b.eq \a57_label
+.endm
+
+/*
+ * Branch if current processor is a Cortex-A53 core.
+ */
+.macro branch_if_a53_core, xreg, a53_label
+ mrs \xreg, midr_el1
+ lsr \xreg, \xreg, #4
+ and \xreg, \xreg, #0x00000FFF
+ cmp \xreg, #0xD03 /* Cortex-A53 MPCore processor. */
+ b.eq \a53_label
+.endm
+
+/*
+ * Branch if current processor is a slave,
+ * choose processor with all zero affinity value as the master.
+ */
+.macro branch_if_slave, xreg, slave_label
+#ifdef CONFIG_ARMV8_MULTIENTRY
+ mrs \xreg, mpidr_el1
+ and \xreg, \xreg, 0xffffffffff /* clear bits [63:40] */
+ and \xreg, \xreg, ~0x00ff000000 /* also clear bits [31:24] */
+ cbnz \xreg, \slave_label
+#endif
+.endm
+
+/*
+ * Branch if current processor is a master,
+ * choose processor with all zero affinity value as the master.
+ */
+.macro branch_if_master, xreg, master_label
+#ifdef CONFIG_ARMV8_MULTIENTRY
+ mrs \xreg, mpidr_el1
+ and \xreg, \xreg, 0xffffffffff /* clear bits [63:40] */
+ and \xreg, \xreg, ~0x00ff000000 /* also clear bits [31:24] */
+ cbz \xreg, \master_label
+#else
+ b \master_label
+#endif
+.endm
+
+/*
+ * Switch from EL3 to EL2 for ARMv8
+ * @ep: kernel entry point
+ * @flag: The execution state flag for lower exception
+ * level, ES_TO_AARCH64 or ES_TO_AARCH32
+ * @tmp: temporary register
+ *
+ * For loading 32-bit OS, x1 is machine nr and x2 is ftaddr.
+ * For loading 64-bit OS, x0 is physical address to the FDT blob.
+ * They will be passed to the guest.
+ */
+.macro armv8_switch_to_el2_m, ep, flag, tmp
+ msr cptr_el3, xzr /* Disable coprocessor traps to EL3 */
+ mov \tmp, #CPTR_EL2_RES1
+ msr cptr_el2, \tmp /* Disable coprocessor traps to EL2 */
+
+ /* Initialize Generic Timers */
+ msr cntvoff_el2, xzr
+
+ /* Initialize SCTLR_EL2
+ *
+ * setting RES1 bits (29,28,23,22,18,16,11,5,4) to 1
+ * and RES0 bits (31,30,27,26,24,21,20,17,15-13,10-6) +
+ * EE,WXN,I,SA,C,A,M to 0
+ */
+ ldr \tmp, =(SCTLR_EL2_RES1 | SCTLR_EL2_EE_LE |\
+ SCTLR_EL2_WXN_DIS | SCTLR_EL2_ICACHE_DIS |\
+ SCTLR_EL2_SA_DIS | SCTLR_EL2_DCACHE_DIS |\
+ SCTLR_EL2_ALIGN_DIS | SCTLR_EL2_MMU_DIS)
+ msr sctlr_el2, \tmp
+
+ mov \tmp, sp
+ msr sp_el2, \tmp /* Migrate SP */
+ mrs \tmp, vbar_el3
+ msr vbar_el2, \tmp /* Migrate VBAR */
+
+ /* Check switch to AArch64 EL2 or AArch32 Hypervisor mode */
+ cmp \flag, #ES_TO_AARCH32
+ b.eq 1f
+
+ /*
+ * The next lower exception level is AArch64, 64bit EL2 | HCE |
+ * RES1 (Bits[5:4]) | Non-secure EL0/EL1.
+ * and the SMD depends on requirements.
+ */
+#ifdef CONFIG_ARMV8_PSCI
+ ldr \tmp, =(SCR_EL3_RW_AARCH64 | SCR_EL3_HCE_EN |\
+ SCR_EL3_RES1 | SCR_EL3_NS_EN)
+#else
+ ldr \tmp, =(SCR_EL3_RW_AARCH64 | SCR_EL3_HCE_EN |\
+ SCR_EL3_SMD_DIS | SCR_EL3_RES1 |\
+ SCR_EL3_NS_EN)
+#endif
+
+#ifdef CONFIG_ARMV8_EA_EL3_FIRST
+ orr \tmp, \tmp, #SCR_EL3_EA_EN
+#endif
+ msr scr_el3, \tmp
+
+ /* Return to the EL2_SP2 mode from EL3 */
+ ldr \tmp, =(SPSR_EL_DEBUG_MASK | SPSR_EL_SERR_MASK |\
+ SPSR_EL_IRQ_MASK | SPSR_EL_FIQ_MASK |\
+ SPSR_EL_M_AARCH64 | SPSR_EL_M_EL2H)
+ msr spsr_el3, \tmp
+ msr elr_el3, \ep
+ eret
+
+1:
+ /*
+ * The next lower exception level is AArch32, 32bit EL2 | HCE |
+ * SMD | RES1 (Bits[5:4]) | Non-secure EL0/EL1.
+ */
+ ldr \tmp, =(SCR_EL3_RW_AARCH32 | SCR_EL3_HCE_EN |\
+ SCR_EL3_SMD_DIS | SCR_EL3_RES1 |\
+ SCR_EL3_NS_EN)
+ msr scr_el3, \tmp
+
+ /* Return to AArch32 Hypervisor mode */
+ ldr \tmp, =(SPSR_EL_END_LE | SPSR_EL_ASYN_MASK |\
+ SPSR_EL_IRQ_MASK | SPSR_EL_FIQ_MASK |\
+ SPSR_EL_T_A32 | SPSR_EL_M_AARCH32 |\
+ SPSR_EL_M_HYP)
+ msr spsr_el3, \tmp
+ msr elr_el3, \ep
+ eret
+.endm
+
+/*
+ * Switch from EL2 to EL1 for ARMv8
+ * @ep: kernel entry point
+ * @flag: The execution state flag for lower exception
+ * level, ES_TO_AARCH64 or ES_TO_AARCH32
+ * @tmp: temporary register
+ *
+ * For loading 32-bit OS, x1 is machine nr and x2 is ftaddr.
+ * For loading 64-bit OS, x0 is physical address to the FDT blob.
+ * They will be passed to the guest.
+ */
+.macro armv8_switch_to_el1_m, ep, flag, tmp, tmp2
+ /* Initialize Generic Timers */
+ mrs \tmp, cnthctl_el2
+ /* Enable EL1 access to timers */
+ orr \tmp, \tmp, #(CNTHCTL_EL2_EL1PCEN_EN |\
+ CNTHCTL_EL2_EL1PCTEN_EN)
+ msr cnthctl_el2, \tmp
+ msr cntvoff_el2, xzr
+
+ /* Initilize MPID/MPIDR registers */
+ mrs \tmp, midr_el1
+ msr vpidr_el2, \tmp
+ mrs \tmp, mpidr_el1
+ msr vmpidr_el2, \tmp
+
+ /* Disable coprocessor traps */
+ mov \tmp, #CPTR_EL2_RES1
+ msr cptr_el2, \tmp /* Disable coprocessor traps to EL2 */
+ msr hstr_el2, xzr /* Disable coprocessor traps to EL2 */
+ mov \tmp, #CPACR_EL1_FPEN_EN
+ msr cpacr_el1, \tmp /* Enable FP/SIMD at EL1 */
+
+ /* SCTLR_EL1 initialization
+ *
+ * setting RES1 bits (29,28,23,22,20,11) to 1
+ * and RES0 bits (31,30,27,21,17,13,10,6) +
+ * UCI,EE,EOE,WXN,nTWE,nTWI,UCT,DZE,I,UMA,SED,ITD,
+ * CP15BEN,SA0,SA,C,A,M to 0
+ */
+ ldr \tmp, =(SCTLR_EL1_RES1 | SCTLR_EL1_UCI_DIS |\
+ SCTLR_EL1_EE_LE | SCTLR_EL1_WXN_DIS |\
+ SCTLR_EL1_NTWE_DIS | SCTLR_EL1_NTWI_DIS |\
+ SCTLR_EL1_UCT_DIS | SCTLR_EL1_DZE_DIS |\
+ SCTLR_EL1_ICACHE_DIS | SCTLR_EL1_UMA_DIS |\
+ SCTLR_EL1_SED_EN | SCTLR_EL1_ITD_EN |\
+ SCTLR_EL1_CP15BEN_DIS | SCTLR_EL1_SA0_DIS |\
+ SCTLR_EL1_SA_DIS | SCTLR_EL1_DCACHE_DIS |\
+ SCTLR_EL1_ALIGN_DIS | SCTLR_EL1_MMU_DIS)
+ msr sctlr_el1, \tmp
+
+ mov \tmp, sp
+ msr sp_el1, \tmp /* Migrate SP */
+ mrs \tmp, vbar_el2
+ msr vbar_el1, \tmp /* Migrate VBAR */
+
+ /* Check switch to AArch64 EL1 or AArch32 Supervisor mode */
+ cmp \flag, #ES_TO_AARCH32
+ b.eq 1f
+
+ /* Initialize HCR_EL2 */
+ /* Only disable PAuth traps if PAuth is supported */
+ mrs \tmp, id_aa64isar1_el1
+ ldr \tmp2, =(ID_AA64ISAR1_EL1_GPI | ID_AA64ISAR1_EL1_GPA | \
+ ID_AA64ISAR1_EL1_API | ID_AA64ISAR1_EL1_APA)
+ tst \tmp, \tmp2
+ mov \tmp2, #(HCR_EL2_RW_AARCH64 | HCR_EL2_HCD_DIS)
+ orr \tmp, \tmp2, #(HCR_EL2_APK | HCR_EL2_API)
+ csel \tmp, \tmp2, \tmp, eq
+ msr hcr_el2, \tmp
+
+ /* Return to the EL1_SP1 mode from EL2 */
+ ldr \tmp, =(SPSR_EL_DEBUG_MASK | SPSR_EL_SERR_MASK |\
+ SPSR_EL_IRQ_MASK | SPSR_EL_FIQ_MASK |\
+ SPSR_EL_M_AARCH64 | SPSR_EL_M_EL1H)
+ msr spsr_el2, \tmp
+ msr elr_el2, \ep
+ eret
+
+1:
+ /* Initialize HCR_EL2 */
+ ldr \tmp, =(HCR_EL2_RW_AARCH32 | HCR_EL2_HCD_DIS)
+ msr hcr_el2, \tmp
+
+ /* Return to AArch32 Supervisor mode from EL2 */
+ ldr \tmp, =(SPSR_EL_END_LE | SPSR_EL_ASYN_MASK |\
+ SPSR_EL_IRQ_MASK | SPSR_EL_FIQ_MASK |\
+ SPSR_EL_T_A32 | SPSR_EL_M_AARCH32 |\
+ SPSR_EL_M_SVC)
+ msr spsr_el2, \tmp
+ msr elr_el2, \ep
+ eret
+.endm
+
+#if defined(CONFIG_GICV3)
+.macro gic_wait_for_interrupt_m xreg1
+0 : wfi
+ mrs \xreg1, ICC_IAR1_EL1
+ msr ICC_EOIR1_EL1, \xreg1
+ cbnz \xreg1, 0b
+.endm
+#elif defined(CONFIG_GICV2)
+.macro gic_wait_for_interrupt_m xreg1, wreg2
+0 : wfi
+ ldr \wreg2, [\xreg1, GICC_AIAR]
+ str \wreg2, [\xreg1, GICC_AEOIR]
+ and \wreg2, \wreg2, #0x3ff
+ cbnz \wreg2, 0b
+.endm
+#endif
+
+#endif /* CONFIG_ARM64 */
+
+#endif /* __ASSEMBLY__ */
+#endif /* __ASM_ARM_MACRO_H__ */
diff --git a/arch/arm64/kernel/boot.S b/arch/arm64/kernel/boot.S
new file mode 100644
index 0000000..8c656b1
--- /dev/null
+++ b/arch/arm64/kernel/boot.S
@@ -0,0 +1,35 @@
+/* SPDX-License-Identifier: GPL-2.0+ */
+/*
+ * boot.S - the final jump to the payload. x0 = dtb, x1 = x2 = x3 = 0,
+ * MMU and caches off, D cache flushed, I cache invalidated. that is
+ * the whole contract from Documentation/arch/arm64/booting.rst.
+ *
+ * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org>
+ */
+
+#include <asm/linkage.h>
+
+.pushsection .text.tb_boot_linux, "ax"
+ENTRY(tb_boot_linux)
+ /* ep in x0, dtb in x1, per the kernel boot protocol */
+ mov x8, x0
+ mov x0, x1
+ mov x1, xzr
+ mov x2, xzr
+ mov x3, xzr
+
+ /* clean the caches so the kernel reads the image from RAM */
+ bl tb_flush_dcache_all
+ bl tb_invalidate_icache_all
+
+ /* MMU off, caches off, the kernel sets up its own state */
+ mrs x9, sctlr_el1
+ bic x9, x9, #(1 << 0) /* M, MMU */
+ bic x9, x9, #(1 << 2) /* C, D-cache */
+ bic x9, x9, #(1 << 12) /* I, I-cache */
+ msr sctlr_el1, x9
+ isb
+
+ br x8
+ENDPROC(tb_boot_linux)
+.popsection
diff --git a/arch/arm64/kernel/start.S b/arch/arm64/kernel/start.S
new file mode 100644
index 0000000..776744c
--- /dev/null
+++ b/arch/arm64/kernel/start.S
@@ -0,0 +1,94 @@
+/* SPDX-License-Identifier: GPL-2.0+ */
+/*
+ * tashaboot arm64 entry. firmware drops us here, usually at EL2 on
+ * qemu virt. we park the secondaries, drop to EL1, set the stack and
+ * hand the dtb pointer to C.
+ *
+ * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org>
+ */
+
+#include <asm/macro.h>
+
+.section .text.boot
+.globl _start
+_start:
+ b reset
+
+ .balign 8
+.globl _text_base
+_text_base:
+ .quad 0x40000000
+
+reset:
+ /* keep the dtb pointer before anything clobbers x0 */
+ mov x19, x0
+
+ /* park secondary cores, they have nothing to do yet */
+ mrs x0, mpidr_el1
+ and x0, x0, #0xff
+ cbnz x0, park
+
+ /* figure out which EL we got */
+ mrs x0, CurrentEL
+ lsr x0, x0, #2
+ cmp x0, #3
+ b.eq from_el3
+ cmp x0, #2
+ b.eq from_el2
+ cmp x0, #1
+ b.eq el1_entry
+ b park
+
+from_el3:
+ /* EL3 is a dead end for us, drop through the EL2 path if we can */
+ b park
+
+from_el2:
+ /* EL1 must be aarch64 */
+ mov x0, #(1 << 31)
+ msr hcr_el2, x0
+
+ /* scrub the EL2 state so EL1 starts clean */
+ msr cptr_el2, xzr
+ msr hstr_el2, xzr
+ mrs x0, cnthctl_el2
+ orr x0, x0, #(3 << 0)
+ msr cnthctl_el2, x0
+ msr vpidr_el2, xzr
+
+ mov x0, #0x3c5
+ msr scr_el3, x0
+
+ /* no MMU, no caches at EL1 yet */
+ msr sctlr_el1, xzr
+ isb
+
+ /* SPSR for EL1h with interrupts masked */
+ mov x0, #0x3c5
+ msr spsr_el2, x0
+ adr x0, el1_entry
+ msr elr_el2, x0
+ eret
+
+el1_entry:
+ /* stack for the bootloader, grows down from the image end */
+ ldr x0, =__image_end
+ mov sp, x0
+
+ /* clear bss */
+ ldr x0, =__bss_start
+ ldr x1, =__bss_end
+1: cmp x0, x1
+ b.hs 2f
+ str xzr, [x0], #8
+ b 1b
+2:
+
+ /* dtb pointer into C arg 0 */
+ mov x0, x19
+ bl tashaboot_main
+
+ /* if main returns there is nothing sensible to do */
+park:
+ wfe
+ b park
diff --git a/arch/arm64/kernel/tashaboot.lds b/arch/arm64/kernel/tashaboot.lds
new file mode 100644
index 0000000..4f8dfb1
--- /dev/null
+++ b/arch/arm64/kernel/tashaboot.lds
@@ -0,0 +1,67 @@
+/* SPDX-License-Identifier: GPL-2.0+ */
+/*
+ * tashaboot arm64 memory layout. one segment, loaded at the bottom of
+ * RAM, right where qemu -kernel drops a raw image on virt.
+ *
+ * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org>
+ */
+
+OUTPUT_FORMAT("elf64-littleaarch64", "elf64-littleaarch64", "elf64-littleaarch64")
+OUTPUT_ARCH(aarch64)
+ENTRY(_start)
+
+SECTIONS
+{
+ . = 0x40000000;
+
+ __image_copy_start = .;
+ _text_start = .;
+
+ .text :
+ {
+ arch/arm64/kernel/start.o (.text.boot)
+ *(.text.boot)
+ *(.text*)
+ }
+
+ . = ALIGN(8);
+ __text_end = .;
+
+ .rodata :
+ {
+ *(SORT_BY_ALIGNMENT(.rodata*))
+ }
+
+ . = ALIGN(8);
+ __rodata_end = .;
+
+ .data :
+ {
+ *(.data*)
+ }
+
+ . = ALIGN(8);
+ __image_end = .;
+
+ __bss_start = .;
+ .bss :
+ {
+ *(.bss*)
+ *(COMMON)
+ }
+ . = ALIGN(8);
+ __bss_end = .;
+
+ __image_copy_end = .;
+
+ /DISCARD/ : { *(.dynsym) }
+ /DISCARD/ : { *(.dynstr*) }
+ /DISCARD/ : { *(.dynamic*) }
+ /DISCARD/ : { *(.plt*) }
+ /DISCARD/ : { *(.interp*) }
+ /DISCARD/ : { *(.gnu*) }
+ /DISCARD/ : { *(.ARM.attributes) }
+ /DISCARD/ : { *(.comment) }
+ /DISCARD/ : { *(.note*) }
+ /DISCARD/ : { *(.eh_frame*) }
+}
diff --git a/arch/arm64/lib/cache.S b/arch/arm64/lib/cache.S
new file mode 100644
index 0000000..d8ccea2
--- /dev/null
+++ b/arch/arm64/lib/cache.S
@@ -0,0 +1,100 @@
+/* SPDX-License-Identifier: GPL-2.0+ */
+/*
+ * cache.S - set/way cache maintenance, walked off CLIDR_EL1 the same
+ * way u-boot and the kernel's own __flush_dcache_all do it. needed
+ * before jumping to the payload so it starts from memory, not cache.
+ *
+ * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org>
+ */
+
+#include <asm/linkage.h>
+
+.pushsection .text.tb_dcache_level, "ax"
+ENTRY(tb_dcache_level)
+ lsl x12, x0, #1
+ msr csselr_el1, x12 /* select cache level */
+ isb /* sync change of ccsidr_el1 */
+ mrs x6, ccsidr_el1 /* read the new ccsidr_el1 */
+ ubfx x2, x6, #0, #3 /* x2 <- log2(cache line size)-4 */
+ ubfx x3, x6, #3, #10 /* x3 <- number of cache ways - 1 */
+ ubfx x4, x6, #13, #15 /* x4 <- number of cache sets - 1 */
+ add x2, x2, #4 /* x2 <- log2(cache line size) */
+ clz w5, w3 /* x5 <- bit position of #ways */
+ /* x12 <- cache level << 1 */
+ /* x2 <- line length offset */
+ /* x3 <- number of cache ways - 1 */
+ /* x4 <- number of cache sets - 1 */
+ /* x5 <- bit position of #ways */
+
+loop_set:
+ mov x6, x3 /* x6 <- working copy of #ways */
+loop_way:
+ lsl x7, x6, x5
+ orr x9, x12, x7 /* map way and level to cisw value */
+ lsl x7, x4, x2
+ orr x9, x9, x7 /* map set number to cisw value */
+ dc cisw, x9 /* clean & invalidate by set/way */
+ subs x6, x6, #1 /* decrement the way */
+ b.ge loop_way
+ subs x4, x4, #1 /* decrement the set */
+ b.ge loop_set
+
+ ret
+ENDPROC(tb_dcache_level)
+.popsection
+
+/*
+ * void tb_flush_dcache_all(void)
+ *
+ * clean & invalidate the whole D cache by set/way.
+ */
+.pushsection .text.tb_flush_dcache_all, "ax"
+ENTRY(tb_flush_dcache_all)
+ mov x1, x0
+ dsb sy
+ mrs x10, clidr_el1 /* read clidr_el1 */
+ ubfx x11, x10, #24, #3 /* x11 <- loc */
+ cbz x11, finished /* if loc is 0, exit */
+ mov x15, lr
+ mov x0, #0 /* start flush at cache level 0 */
+ /* x0 <- cache level */
+ /* x10 <- clidr_el1 */
+ /* x11 <- loc */
+ /* x15 <- return address */
+
+loop_level:
+ add x12, x0, x0, lsl #1 /* x12 <- tripled cache level */
+ lsr x12, x10, x12
+ and x12, x12, #7 /* x12 <- cache type */
+ cmp x12, #2
+ b.lt skip /* skip if no cache or icache */
+ bl tb_dcache_level /* flush this level */
+skip:
+ add x0, x0, #1 /* increment cache level */
+ cmp x11, x0
+ b.gt loop_level
+
+ mov x0, #0
+ msr csselr_el1, x0 /* restore csselr_el1 */
+ dsb sy
+ isb
+ mov lr, x15
+
+finished:
+ ret
+ENDPROC(tb_flush_dcache_all)
+.popsection
+
+/*
+ * void tb_invalidate_icache_all(void)
+ *
+ * I cache invalidation to PoU, one ic iallu covers the local core.
+ */
+.pushsection .text.tb_invalidate_icache_all, "ax"
+ENTRY(tb_invalidate_icache_all)
+ ic iallu
+ dsb sy
+ isb
+ ret
+ENDPROC(tb_invalidate_icache_all)
+.popsection
diff --git a/arch/arm64/lib/semihosting.S b/arch/arm64/lib/semihosting.S
new file mode 100644
index 0000000..6e3fc31
--- /dev/null
+++ b/arch/arm64/lib/semihosting.S
@@ -0,0 +1,18 @@
+/* SPDX-License-Identifier: GPL-2.0+ */
+/*
+ * semihosting.S - the trap instruction itself. qemu answers this when
+ * it is started with -semihosting, and nothing happens without it, so
+ * every caller has to cope with the no-debugger case.
+ *
+ * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org>
+ */
+
+#include <asm/linkage.h>
+
+.pushsection .text.smh_trap, "ax"
+/* long smh_trap(unsigned int sysnum, void *addr); */
+ENTRY(smh_trap)
+ hlt #0xf000
+ ret
+ENDPROC(smh_trap)
+.popsection