diff options
| author | Bradley Morgan <brads@mainlining.org> | 2026-10-04 10:42:10 +0000 |
|---|---|---|
| committer | Bradley Morgan <brads@mainlining.org> | 2026-10-04 10:46:46 +0000 |
| commit | 5ff54a875642962fc857cee00bde17f9a465f1fa (patch) | |
| tree | b265a12d538167c3bbb86b24118a7ad7a062fc74 /arch/arm64 | |
holy shit it's here, Tashaboot, based from arm arm, enjoy
reading this masterpiece
Signed-off-by: Bradley Morgan <brads@mainlining.org>
Diffstat (limited to 'arch/arm64')
| -rw-r--r-- | arch/arm64/include/asm/linkage.h | 14 | ||||
| -rw-r--r-- | arch/arm64/include/asm/macro.h | 347 | ||||
| -rw-r--r-- | arch/arm64/include/asm/mmu.h | 51 | ||||
| -rw-r--r-- | arch/arm64/include/asm/psci.h | 37 | ||||
| -rw-r--r-- | arch/arm64/kernel/boot.S | 66 | ||||
| -rw-r--r-- | arch/arm64/kernel/exceptions.c | 67 | ||||
| -rw-r--r-- | arch/arm64/kernel/halt.c | 18 | ||||
| -rw-r--r-- | arch/arm64/kernel/monitor.S | 132 | ||||
| -rw-r--r-- | arch/arm64/kernel/start.S | 456 | ||||
| -rw-r--r-- | arch/arm64/kernel/tashaboot.lds | 91 | ||||
| -rw-r--r-- | arch/arm64/lib/cache.S | 100 | ||||
| -rw-r--r-- | arch/arm64/lib/cache_va.c | 73 | ||||
| -rw-r--r-- | arch/arm64/lib/gic.c | 105 | ||||
| -rw-r--r-- | arch/arm64/lib/mmu.c | 205 | ||||
| -rw-r--r-- | arch/arm64/lib/psci.c | 118 | ||||
| -rw-r--r-- | arch/arm64/lib/semihosting.S | 18 | ||||
| -rw-r--r-- | arch/arm64/lib/system.c | 70 | ||||
| -rw-r--r-- | arch/arm64/lib/timer.c | 48 |
18 files changed, 2016 insertions, 0 deletions
diff --git a/arch/arm64/include/asm/linkage.h b/arch/arm64/include/asm/linkage.h new file mode 100644 index 0000000..b5b9706 --- /dev/null +++ b/arch/arm64/include/asm/linkage.h @@ -0,0 +1,14 @@ +/* SPDX-License-Identifier: GPL-2.0 */ +#ifndef __ASM_LINKAGE_H +#define __ASM_LINKAGE_H + +#define ALIGN .p2align 4 +#define ENTRY(name) \ + .globl name; \ + ALIGN; \ + name: +#define ENDPROC(name) \ + .type name, %function; \ + .size name, .-name + +#endif diff --git a/arch/arm64/include/asm/macro.h b/arch/arm64/include/asm/macro.h new file mode 100644 index 0000000..1a1edc9 --- /dev/null +++ b/arch/arm64/include/asm/macro.h @@ -0,0 +1,347 @@ +/* SPDX-License-Identifier: GPL-2.0+ */ +/* + * include/asm-arm/macro.h + * + * Copyright (C) 2009 Jean-Christophe PLAGNIOL-VILLARD <plagnioj@jcrosoft.com> + */ + +#ifndef __ASM_ARM_MACRO_H__ +#define __ASM_ARM_MACRO_H__ + +#ifdef CONFIG_ARM64 +#include <asm/system.h> +#endif + +#ifdef __ASSEMBLY__ + +/* + * These macros provide a convenient way to write 8, 16 and 32 bit data + * to any address. + * Registers r4 and r5 are used, any data in these registers are + * overwritten by the macros. + * The macros are valid for any ARM architecture, they do not implement + * any memory barriers so caution is recommended when using these when the + * caches are enabled or on a multi-core system. + */ + +.macro write32, addr, data + ldr r4, =\addr + ldr r5, =\data + str r5, [r4] +.endm + +.macro write16, addr, data + ldr r4, =\addr + ldrh r5, =\data + strh r5, [r4] +.endm + +.macro write8, addr, data + ldr r4, =\addr + ldrb r5, =\data + strb r5, [r4] +.endm + +/* + * This macro generates a loop that can be used for delays in the code. + * Register r4 is used, any data in this register is overwritten by the + * macro. + * The macro is valid for any ARM architeture. The actual time spent in the + * loop will vary from CPU to CPU though. + */ + +.macro wait_timer, time + ldr r4, =\time +1: + nop + subs r4, r4, #1 + bcs 1b +.endm + +#ifdef CONFIG_ARM64 +/* + * Register aliases. + */ +lr .req x30 + +/* + * Branch according to exception level + */ +.macro switch_el, xreg, el3_label, el2_label, el1_label + mrs \xreg, CurrentEL + cmp \xreg, #0x8 + b.gt \el3_label + b.eq \el2_label + b.lt \el1_label +.endm + +/* + * Branch if we are not in the highest exception level + */ +.macro branch_if_not_highest_el, xreg, label + switch_el \xreg, 3f, 2f, 1f + +2: mrs \xreg, ID_AA64PFR0_EL1 + and \xreg, \xreg, #(ID_AA64PFR0_EL1_EL3) + cbnz \xreg, \label + b 3f + +1: mrs \xreg, ID_AA64PFR0_EL1 + and \xreg, \xreg, #(ID_AA64PFR0_EL1_EL3 | ID_AA64PFR0_EL1_EL2) + cbnz \xreg, \label + +3: +.endm + +/* + * Branch if current processor is a Cortex-A57 core. + */ +.macro branch_if_a57_core, xreg, a57_label + mrs \xreg, midr_el1 + lsr \xreg, \xreg, #4 + and \xreg, \xreg, #0x00000FFF + cmp \xreg, #0xD07 /* Cortex-A57 MPCore processor. */ + b.eq \a57_label +.endm + +/* + * Branch if current processor is a Cortex-A53 core. + */ +.macro branch_if_a53_core, xreg, a53_label + mrs \xreg, midr_el1 + lsr \xreg, \xreg, #4 + and \xreg, \xreg, #0x00000FFF + cmp \xreg, #0xD03 /* Cortex-A53 MPCore processor. */ + b.eq \a53_label +.endm + +/* + * Branch if current processor is a slave, + * choose processor with all zero affinity value as the master. + */ +.macro branch_if_slave, xreg, slave_label +#ifdef CONFIG_ARMV8_MULTIENTRY + mrs \xreg, mpidr_el1 + and \xreg, \xreg, 0xffffffffff /* clear bits [63:40] */ + and \xreg, \xreg, ~0x00ff000000 /* also clear bits [31:24] */ + cbnz \xreg, \slave_label +#endif +.endm + +/* + * Branch if current processor is a master, + * choose processor with all zero affinity value as the master. + */ +.macro branch_if_master, xreg, master_label +#ifdef CONFIG_ARMV8_MULTIENTRY + mrs \xreg, mpidr_el1 + and \xreg, \xreg, 0xffffffffff /* clear bits [63:40] */ + and \xreg, \xreg, ~0x00ff000000 /* also clear bits [31:24] */ + cbz \xreg, \master_label +#else + b \master_label +#endif +.endm + +/* + * Switch from EL3 to EL2 for ARMv8 + * @ep: kernel entry point + * @flag: The execution state flag for lower exception + * level, ES_TO_AARCH64 or ES_TO_AARCH32 + * @tmp: temporary register + * + * For loading 32-bit OS, x1 is machine nr and x2 is ftaddr. + * For loading 64-bit OS, x0 is physical address to the FDT blob. + * They will be passed to the guest. + */ +.macro armv8_switch_to_el2_m, ep, flag, tmp + msr cptr_el3, xzr /* Disable coprocessor traps to EL3 */ + mov \tmp, #CPTR_EL2_RES1 + msr cptr_el2, \tmp /* Disable coprocessor traps to EL2 */ + + /* Initialize Generic Timers */ + msr cntvoff_el2, xzr + + /* Initialize SCTLR_EL2 + * + * setting RES1 bits (29,28,23,22,18,16,11,5,4) to 1 + * and RES0 bits (31,30,27,26,24,21,20,17,15-13,10-6) + + * EE,WXN,I,SA,C,A,M to 0 + */ + ldr \tmp, =(SCTLR_EL2_RES1 | SCTLR_EL2_EE_LE |\ + SCTLR_EL2_WXN_DIS | SCTLR_EL2_ICACHE_DIS |\ + SCTLR_EL2_SA_DIS | SCTLR_EL2_DCACHE_DIS |\ + SCTLR_EL2_ALIGN_DIS | SCTLR_EL2_MMU_DIS) + msr sctlr_el2, \tmp + + mov \tmp, sp + msr sp_el2, \tmp /* Migrate SP */ + mrs \tmp, vbar_el3 + msr vbar_el2, \tmp /* Migrate VBAR */ + + /* Check switch to AArch64 EL2 or AArch32 Hypervisor mode */ + cmp \flag, #ES_TO_AARCH32 + b.eq 1f + + /* + * The next lower exception level is AArch64, 64bit EL2 | HCE | + * RES1 (Bits[5:4]) | Non-secure EL0/EL1. + * and the SMD depends on requirements. + */ +#ifdef CONFIG_ARMV8_PSCI + ldr \tmp, =(SCR_EL3_RW_AARCH64 | SCR_EL3_HCE_EN |\ + SCR_EL3_RES1 | SCR_EL3_NS_EN) +#else + ldr \tmp, =(SCR_EL3_RW_AARCH64 | SCR_EL3_HCE_EN |\ + SCR_EL3_SMD_DIS | SCR_EL3_RES1 |\ + SCR_EL3_NS_EN) +#endif + +#ifdef CONFIG_ARMV8_EA_EL3_FIRST + orr \tmp, \tmp, #SCR_EL3_EA_EN +#endif + msr scr_el3, \tmp + + /* Return to the EL2_SP2 mode from EL3 */ + ldr \tmp, =(SPSR_EL_DEBUG_MASK | SPSR_EL_SERR_MASK |\ + SPSR_EL_IRQ_MASK | SPSR_EL_FIQ_MASK |\ + SPSR_EL_M_AARCH64 | SPSR_EL_M_EL2H) + msr spsr_el3, \tmp + msr elr_el3, \ep + eret + +1: + /* + * The next lower exception level is AArch32, 32bit EL2 | HCE | + * SMD | RES1 (Bits[5:4]) | Non-secure EL0/EL1. + */ + ldr \tmp, =(SCR_EL3_RW_AARCH32 | SCR_EL3_HCE_EN |\ + SCR_EL3_SMD_DIS | SCR_EL3_RES1 |\ + SCR_EL3_NS_EN) + msr scr_el3, \tmp + + /* Return to AArch32 Hypervisor mode */ + ldr \tmp, =(SPSR_EL_END_LE | SPSR_EL_ASYN_MASK |\ + SPSR_EL_IRQ_MASK | SPSR_EL_FIQ_MASK |\ + SPSR_EL_T_A32 | SPSR_EL_M_AARCH32 |\ + SPSR_EL_M_HYP) + msr spsr_el3, \tmp + msr elr_el3, \ep + eret +.endm + +/* + * Switch from EL2 to EL1 for ARMv8 + * @ep: kernel entry point + * @flag: The execution state flag for lower exception + * level, ES_TO_AARCH64 or ES_TO_AARCH32 + * @tmp: temporary register + * + * For loading 32-bit OS, x1 is machine nr and x2 is ftaddr. + * For loading 64-bit OS, x0 is physical address to the FDT blob. + * They will be passed to the guest. + */ +.macro armv8_switch_to_el1_m, ep, flag, tmp, tmp2 + /* Initialize Generic Timers */ + mrs \tmp, cnthctl_el2 + /* Enable EL1 access to timers */ + orr \tmp, \tmp, #(CNTHCTL_EL2_EL1PCEN_EN |\ + CNTHCTL_EL2_EL1PCTEN_EN) + msr cnthctl_el2, \tmp + msr cntvoff_el2, xzr + + /* Initilize MPID/MPIDR registers */ + mrs \tmp, midr_el1 + msr vpidr_el2, \tmp + mrs \tmp, mpidr_el1 + msr vmpidr_el2, \tmp + + /* Disable coprocessor traps */ + mov \tmp, #CPTR_EL2_RES1 + msr cptr_el2, \tmp /* Disable coprocessor traps to EL2 */ + msr hstr_el2, xzr /* Disable coprocessor traps to EL2 */ + mov \tmp, #CPACR_EL1_FPEN_EN + msr cpacr_el1, \tmp /* Enable FP/SIMD at EL1 */ + + /* SCTLR_EL1 initialization + * + * setting RES1 bits (29,28,23,22,20,11) to 1 + * and RES0 bits (31,30,27,21,17,13,10,6) + + * UCI,EE,EOE,WXN,nTWE,nTWI,UCT,DZE,I,UMA,SED,ITD, + * CP15BEN,SA0,SA,C,A,M to 0 + */ + ldr \tmp, =(SCTLR_EL1_RES1 | SCTLR_EL1_UCI_DIS |\ + SCTLR_EL1_EE_LE | SCTLR_EL1_WXN_DIS |\ + SCTLR_EL1_NTWE_DIS | SCTLR_EL1_NTWI_DIS |\ + SCTLR_EL1_UCT_DIS | SCTLR_EL1_DZE_DIS |\ + SCTLR_EL1_ICACHE_DIS | SCTLR_EL1_UMA_DIS |\ + SCTLR_EL1_SED_EN | SCTLR_EL1_ITD_EN |\ + SCTLR_EL1_CP15BEN_DIS | SCTLR_EL1_SA0_DIS |\ + SCTLR_EL1_SA_DIS | SCTLR_EL1_DCACHE_DIS |\ + SCTLR_EL1_ALIGN_DIS | SCTLR_EL1_MMU_DIS) + msr sctlr_el1, \tmp + + mov \tmp, sp + msr sp_el1, \tmp /* Migrate SP */ + mrs \tmp, vbar_el2 + msr vbar_el1, \tmp /* Migrate VBAR */ + + /* Check switch to AArch64 EL1 or AArch32 Supervisor mode */ + cmp \flag, #ES_TO_AARCH32 + b.eq 1f + + /* Initialize HCR_EL2 */ + /* Only disable PAuth traps if PAuth is supported */ + mrs \tmp, id_aa64isar1_el1 + ldr \tmp2, =(ID_AA64ISAR1_EL1_GPI | ID_AA64ISAR1_EL1_GPA | \ + ID_AA64ISAR1_EL1_API | ID_AA64ISAR1_EL1_APA) + tst \tmp, \tmp2 + mov \tmp2, #(HCR_EL2_RW_AARCH64 | HCR_EL2_HCD_DIS) + orr \tmp, \tmp2, #(HCR_EL2_APK | HCR_EL2_API) + csel \tmp, \tmp2, \tmp, eq + msr hcr_el2, \tmp + + /* Return to the EL1_SP1 mode from EL2 */ + ldr \tmp, =(SPSR_EL_DEBUG_MASK | SPSR_EL_SERR_MASK |\ + SPSR_EL_IRQ_MASK | SPSR_EL_FIQ_MASK |\ + SPSR_EL_M_AARCH64 | SPSR_EL_M_EL1H) + msr spsr_el2, \tmp + msr elr_el2, \ep + eret + +1: + /* Initialize HCR_EL2 */ + ldr \tmp, =(HCR_EL2_RW_AARCH32 | HCR_EL2_HCD_DIS) + msr hcr_el2, \tmp + + /* Return to AArch32 Supervisor mode from EL2 */ + ldr \tmp, =(SPSR_EL_END_LE | SPSR_EL_ASYN_MASK |\ + SPSR_EL_IRQ_MASK | SPSR_EL_FIQ_MASK |\ + SPSR_EL_T_A32 | SPSR_EL_M_AARCH32 |\ + SPSR_EL_M_SVC) + msr spsr_el2, \tmp + msr elr_el2, \ep + eret +.endm + +#if defined(CONFIG_GICV3) +.macro gic_wait_for_interrupt_m xreg1 +0 : wfi + mrs \xreg1, ICC_IAR1_EL1 + msr ICC_EOIR1_EL1, \xreg1 + cbnz \xreg1, 0b +.endm +#elif defined(CONFIG_GICV2) +.macro gic_wait_for_interrupt_m xreg1, wreg2 +0 : wfi + ldr \wreg2, [\xreg1, GICC_AIAR] + str \wreg2, [\xreg1, GICC_AEOIR] + and \wreg2, \wreg2, #0x3ff + cbnz \wreg2, 0b +.endm +#endif + +#endif /* CONFIG_ARM64 */ + +#endif /* __ASSEMBLY__ */ +#endif /* __ASM_ARM_MACRO_H__ */ diff --git a/arch/arm64/include/asm/mmu.h b/arch/arm64/include/asm/mmu.h new file mode 100644 index 0000000..342a2ff --- /dev/null +++ b/arch/arm64/include/asm/mmu.h @@ -0,0 +1,51 @@ +/* SPDX-License-Identifier: GPL-2.0 */ +#ifndef __ASM_MMU_H +#define __ASM_MMU_H + +/* + * VMSAv8-64 stage 1 translation at EL2. descriptor layouts and + * attribute fields per the ARM ARM (DDI 0487), block and table + * descriptors D5-2444, page descriptors D5-2447, stage 1 attribute + * fields D5-2451, MAIR region attributes D5-2476. + */ + +#include <stdint.h> + +/* descriptor bits[1:0]: 0b01 block (page at level 3), 0b11 table */ +#define TB_DESC_FAULT 0ULL +#define TB_DESC_BLOCK 1ULL +#define TB_DESC_TABLE 3ULL + +/* lower block/page attribute bits, D5-2451 */ +#define TB_DESC_AF (1ULL << 10) /* access flag, set by hand */ +#define TB_DESC_SH_IS (3ULL << 8) /* inner shareable */ +#define TB_DESC_XN (1ULL << 54) /* XN at EL2, no execute */ + +/* MAIR_ELx attribute indices used by the maps below */ +#define TB_ATTR_NORMAL 0 /* writeback, read allocate */ +#define TB_ATTR_DEVICE 1 /* device nGnRE */ + +/* + * TCR setup, 4KB granule. T0SZ 16 gives a 48-bit VA and the walk + * starts at level 0 (Address size configuration, D5-2399), which is + * what the three level table structure below assumes. a 39-bit VA + * (T0SZ 25) would start the walk at level 1 and misread the whole + * table. + */ +#define TB_TCR_T0SZ_48 16 +#define TB_TCR_SH0_IS (3ULL << 12) +#define TB_TCR_TG0_4K (0ULL << 14) +#define TB_TCR_IRGN0_WB (1ULL << 8) +#define TB_TCR_ORGN0_WB (1ULL << 10) +#define TB_TCR_IPS(x) ((uint64_t)(x) << 16) /* PA size from PARange */ + +/* the map itself, PA == VA everywhere, identity */ +#define TB_MAP_MMIO_BASE 0x00000000ULL +#define TB_MAP_MMIO_SIZE (1ULL << 30) /* low 1GB, devices live here */ +#define TB_MAP_RAM_BASE 0x40000000ULL +#define TB_MAP_RAM_SIZE (128ULL << 20) /* qemu virt default, 128MB */ + +int tb_mmu_enable(void); +void tb_mmu_disable(void); + +#endif /* __ASM_MMU_H */ diff --git a/arch/arm64/include/asm/psci.h b/arch/arm64/include/asm/psci.h new file mode 100644 index 0000000..d7ce0f3 --- /dev/null +++ b/arch/arm64/include/asm/psci.h @@ -0,0 +1,37 @@ +/* SPDX-License-Identifier: GPL-2.0 */ +#ifndef __ASM_PSCI_H +#define __ASM_PSCI_H + +#include <stdint.h> +/* + * PSCI 0.2 handler at EL2, the Power State Coordination Interface + * per DEN 0022. the payload calls it through the conduit the dtb + * names, hvc here, the call traps to EL2 and this dispatches. + */ + +/* standard function ids, DEN 0022 table 5-1 */ +#define PSCI_FN_VERSION 0x84000000 +#define PSCI_FN_CPU_OFF 0x84000002 +#define PSCI_FN_CPU_ON 0x84000003 +#define PSCI_FN_SYSTEM_OFF 0x84000008 +#define PSCI_FN_SYSTEM_RESET 0x84000009 + +/* version 0.2, major 0 minor 2 */ +#define PSCI_VERSION_0_2 0x00000002 + +/* error codes, DEN 0022 */ +#define PSCI_RET_SUCCESS 0 +#define PSCI_RET_NOT_SUPPORTED -1 +#define PSCI_RET_INVALID_PARAMS -2 +#define PSCI_RET_DENIED -3 +#define PSCI_RET_ALREADY_ON -4 +#define PSCI_RET_ON_PENDING -5 +#define PSCI_RET_INTERNAL_FAIL -6 +#define PSCI_RET_NOT_PRESENT -7 +#define PSCI_RET_DISABLED -8 + +/* the asm HVC vector calls this with the caller's x0-x3 in place */ +uint64_t tb_psci_dispatch(uint64_t fn, uint64_t x1, uint64_t x2, + uint64_t x3); + +#endif /* __ASM_PSCI_H */ diff --git a/arch/arm64/kernel/boot.S b/arch/arm64/kernel/boot.S new file mode 100644 index 0000000..d8b888f --- /dev/null +++ b/arch/arm64/kernel/boot.S @@ -0,0 +1,66 @@ +/* SPDX-License-Identifier: GPL-2.0+ */ +/* + * boot.S - the final jump to the payload. x0 = dtb, x1 = x2 = x3 = 0, + * MMU and caches off, D cache flushed, I cache invalidated. that is + * the whole contract from Documentation/arch/arm64/booting.rst. + * + * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org> + */ + +#include <asm/linkage.h> + +.pushsection .text.tb_boot_linux, "ax" +ENTRY(tb_boot_linux) + /* ep in x0, dtb in x1, per the kernel boot protocol */ + + /* + * cache maintenance first, register setup last. x0-x18 are + * caller saved per the AAPCS, the flush helpers are free to + * clobber them, so the args ride in x20/x21 across the calls. + */ + mov x20, x0 /* entry point */ + mov x21, x1 /* dtb */ + + bl tb_flush_dcache_all + bl tb_invalidate_icache_all + + mov x8, x20 + mov x0, x21 + mov x1, xzr + mov x2, xzr + mov x3, xzr + + /* + * raise to EL2 for the payload when EL2 exists, the kernel + * prefers it there (booting.rst). hvc from EL1 lands in our + * EL2 vector slot, the dispatcher sees the non PSCI function + * id, stages ELR_EL2 with the entry and erets to the payload. + * on an EL1 only machine this is a straight branch. + */ + mrs x9, CurrentEL + lsr x9, x9, #2 + cmp x9, #2 + b.lt 5f + hvc #0 +5: + + /* MMU off, caches off, the kernel sets up its own state */ + mrs x9, sctlr_el1 + bic x9, x9, #(1 << 0) /* M, MMU */ + bic x9, x9, #(1 << 2) /* C, D-cache */ + bic x9, x9, #(1 << 12) /* I, I-cache */ + msr sctlr_el1, x9 + isb + + /* + * if we entered at EL2, the kernel prefers it there. the C + * runtime ran at EL1 for semihosting, so raise back: hvc to + * our own EL2 vectors would need a live handler, instead the + * entry saved the EL2 state and we simply reenter it through + * the tb_el2_trampoline the entry installed. + */ + + + br x8 +ENDPROC(tb_boot_linux) +.popsection diff --git a/arch/arm64/kernel/exceptions.c b/arch/arm64/kernel/exceptions.c new file mode 100644 index 0000000..7b40690 --- /dev/null +++ b/arch/arm64/kernel/exceptions.c @@ -0,0 +1,67 @@ +/* SPDX-License-Identifier: GPL-2.0+ */ +/* + * exceptions.c - report an abort through the console before parking, + * so a firmware handoff bug says why it died instead of hanging quiet. + * ESR/FAR decode follows armv8 DDI 0487, the EC and ISS fields. + * + * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org> + */ + +#include <stdint.h> +#include <debug.h> + +struct exc_frame { + uint64_t esr; + uint64_t far; + uint64_t lr; +}; + +/* + * exception class from ESR, bits 31:26. the classes a bootloader can + * actually hit with any frequency. + */ +static const char *exc_class_str(uint64_t esr) +{ + switch (esr >> 26) { + case 0x04: return "data abort, lower EL"; + case 0x05: return "data abort, same EL"; + case 0x25: return "data abort, same EL"; + case 0x08: return "stack pointer misaligned"; + case 0x11: return "instruction abort, same EL"; + case 0x16: return "SError"; + case 0x1a: return "unhandled exception"; + case 0x22: return "pc alignment fault"; + case 0x24: return "unknown trap"; + case 0x26: return "same EL exception return"; + default: return "unknown EC"; + } +} + +/* + * far is only meaningful for the abort and alignment classes, note it + * for those and skip it otherwise so the report does not mislead. + */ +static int exc_far_valid(uint64_t esr) +{ + switch (esr >> 26) { + case 0x04: + case 0x05: + case 0x25: + case 0x11: + case 0x22: + return 1; + default: + return 0; + } +} + +void exc_report(uint64_t esr, uint64_t far, uint64_t lr) +{ + dprintf(CRITICAL, "tashaboot: exception %s\n", exc_class_str(esr)); + dprintf(CRITICAL, "esr %016llx lr %016llx\n", + (unsigned long long)esr, (unsigned long long)lr); + if (exc_far_valid(esr)) + dprintf(CRITICAL, "far %016llx\n", (unsigned long long)far); + + /* nothing recovers from an abort here, park after reporting */ +} diff --git a/arch/arm64/kernel/halt.c b/arch/arm64/kernel/halt.c new file mode 100644 index 0000000..d91d39a --- /dev/null +++ b/arch/arm64/kernel/halt.c @@ -0,0 +1,18 @@ +/* SPDX-License-Identifier: GPL-2.0+ */ +/* + * halt.c - stop the core, the ARM ARM's WFI loop. nothing recovers + * from a halt, the machine needs a reset. + * + * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org> + */ + +#include <debug.h> + +void platform_halt(void) +{ + dprintf(ALWAYS, "HALT: spinning forever...\n"); + + for (;;) { + asm volatile("wfi"); + } +} diff --git a/arch/arm64/kernel/monitor.S b/arch/arm64/kernel/monitor.S new file mode 100644 index 0000000..c6f5ec8 --- /dev/null +++ b/arch/arm64/kernel/monitor.S @@ -0,0 +1,132 @@ +/* + * monitor.S - the EL3 secure monitor, the resident layer real + * firmware ships. the loader drops to non-secure and never + * returns, but the kernel keeps calling into firmware: PSCI + * through the SMC conduit, and on hardware with the security + * extension the group routing of the interrupt controller is + * only writable from here. + * + * the entry path runs once per PE: configure EL3, install the + * monitor vectors, hand the next stage non-secure EL2 in the + * manual's boot state. SMCCC calls from the kernel trap into + * the SMC slot, the C dispatcher behind it is the same one the + * hvc path uses. + * + * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org> + */ + +/* + * the monitor stack. SP_EL3 needs memory no non-secure stage + * will touch, the region after the loader stack, sixteen + * bytes a call deep at most. + */ +.section .bss.el3stack, "aw", %nobits +.align 4 +.globl __el3_stack_bottom +__el3_stack_bottom: + .quad 0, 0, 0, 0 + .quad 0, 0, 0, 0 +.globl __el3_stack_top +__el3_stack_top: + +/* + * EL3 vectors, same sixteen slot layout every exception level + * uses. only the lower EL sync slot carries work, the SMC + * conduit, everything else parks. + */ +.balign 2048 +.globl tb_el3_vectors +tb_el3_vectors: + /* 0x000: current EL, SP_EL0, unused */ + .align 7 + b el3_park + .align 7 + b el3_park + .align 7 + b el3_park + .align 7 + b el3_park + + /* 0x200: current EL, SP_ELx, unused */ + .align 7 + b el3_park + .align 7 + b el3_park + .align 7 + b el3_park + .align 7 + b el3_park + + /* 0x400: lower EL, AArch64, the SMC conduit lives here */ + .align 7 + b el3_park + .align 7 + b el3_park + .align 7 + b el3_park + .align 7 + b el3_smc + + /* 0x600: lower EL, AArch32, unused */ + .align 7 + b el3_park + .align 7 + b el3_park + .align 7 + b el3_park + .align 7 + b el3_park + +/* + * one time per PE, from the reset path. x30 = the next stage + * entry in non-secure EL2, x0 = the dtb pointer. + */ +.globl tb_monitor_init +tb_monitor_init: + /* SP_EL3 on its own region */ + adr x1, __el3_stack_top + msr spsel, #0 + mov sp, x1 + msr spsel, #1 + + /* the monitor vectors */ + adr x1, tb_el3_vectors + msr vbar_el3, x1 + isb + + /* + * SMC as the conduit, SVE traps off, no interrupt routing + * into EL3: FIQ/IRQ stay whatever SCR_EL3.SCR left them, + * the kernel owns the world below. + */ + mrs x1, scr_el3 + bic x1, x1, #(1 << 2) /* SMD, SMC enabled */ + msr scr_el3, x1 + isb + + ret + +el3_park: + b el3_park + +/* + * the SMC trap from lower EL. the SMCCC calling convention is + * the SMC register set, function id in x0, arguments x1 to + * x3, results in x0 to x3. x17 and x18 are caller save in + * this convention, the dispatcher clobbers x0 to x18. + */ +el3_smc: + stp x29, x30, [sp, #-16]! + mov x29, sp + stp x19, x20, [sp, #-16]! + stp x21, x22, [sp, #-16]! + stp x23, x24, [sp, #-16]! + + bl tb_psci_dispatch + + ldp x23, x24, [sp], #16 + ldp x21, x22, [sp], #16 + ldp x19, x20, [sp], #16 + ldp x29, x30, [sp], #16 + + eret diff --git a/arch/arm64/kernel/start.S b/arch/arm64/kernel/start.S new file mode 100644 index 0000000..2a5e2ae --- /dev/null +++ b/arch/arm64/kernel/start.S @@ -0,0 +1,456 @@ +/* SPDX-License-Identifier: GPL-2.0+ */ +/* + * tashaboot arm64 entry. handles whatever EL the firmware left us in, + * EL3, EL2 or EL1, with the MMU either on or off, and arrives at a + * clean EL1 with the MMU off before calling C. + * + * the secondary cores park, spin table bringup is a later problem. + * + * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org> + */ + +#include <asm/macro.h> + +.section .text.boot +.globl _start +_start: + /* code0: branch over the 64 byte Image header to reset */ + b reset + + .balign 8 +/* + * the arm64 Image header fields, per Documentation/arch/arm64/ + * booting.rst: text_offset 0x08, image_size 0x10, flags 0x18, + * magic 0x38. code0 above branches over all of it. text_offset 0 + * and image_size filled after link by tools/fillsize.py, the + * magic pins it as a proper Image so qemu -kernel enters at + * RAMBASE instead of guessing +0x80000. + */ + .quad 0x0 /* text_offset, 0x08, filled below */ + .quad 0x0 /* image_size, 0x10, filled below */ + .quad 0x0 /* flags, 0x18: LE, 4k pages, unset */ + .quad 0x0 /* reserved 0x20 */ + .quad 0x0 /* reserved 0x28 */ + .quad 0x0 /* reserved 0x30 */ + .quad 0x644d5241 /* magic, 0x38: ARM\x64 */ + +reset: + /* keep the dtb pointer before anything clobbers x0 */ + mov x19, x0 + + /* + * park secondary cores, they have nothing to do yet. at + * EL3 they still get the monitor: a firmware call on any + * PE must land in a handler, a secondary with no EL3 + * vectors traps into nothing. + */ + mrs x0, mpidr_el1 + and x0, x0, #0xff + cbnz x0, secondary_boot + + mrs x0, CurrentEL + lsr x0, x0, #2 + cmp x0, #3 + b.eq from_el3 + cmp x0, #2 + b.eq from_el2 + cmp x0, #1 + b.eq mmu_check + b park + +secondary_boot: + mrs x0, CurrentEL + lsr x0, x0, #2 + cmp x0, #3 + b.ne park + /* + * the same security state as the primary: SCR_EL3.NS + * clear leaves a PE secure, and a secondary released + * into the kernel secure is the inconsistent mode boot + * the kernel warns about, its calls trap to EL3 as if + * they were firmware's own. + */ + mrs x0, scr_el3 + orr x0, x0, #1 + msr scr_el3, x0 + isb + bl tb_monitor_init + b park + +from_el3: + /* + * EL3 holds the security state, so the monitor lives here: + * vectors, its own stack, the SMC conduit. it is resident + * after this, the kernel's firmware calls trap into it. + */ + bl tb_monitor_init + + /* the kernel runs non-secure, drop to the EL2 it prefers */ + mrs x0, scr_el3 + orr x0, x0, #1 /* SCR_EL3.NS = 1, non-secure */ + msr scr_el3, x0 + isb + + mov x0, #0x3c9 /* EL2h, DAIF masked */ + msr spsr_el3, x0 + adr x0, from_el2 + msr elr_el3, x0 + eret + +from_el2: + /* + * scrub the EL2 state and drop to EL1 for the C runtime. the + * semihosting hlt trap is an EL1 service on qemu, calling it + * from EL2 corrupts the return state. the kernel handoff goes + * back to EL2, booting.rst prefers it there, through the + * trampoline in boot.S. + */ + + /* EL1 will be aarch64 */ + mov x0, #(1 << 31) /* HCR_EL2.RW = 1 */ + msr hcr_el2, x0 + + /* let EL1 reach the counter, booting.rst demands it */ + mrs x0, cnthctl_el2 + orr x0, x0, #(3 << 0) /* EL1PCTEN | EL1PCEN */ + msr cnthctl_el2, x0 + + /* no traps to EL2 behind EL1's back */ + msr cptr_el2, xzr + msr hstr_el2, xzr + msr vpidr_el2, xzr + + /* drop to EL1, SPSR EL1h with DAIF masked */ + mov x0, #0x3c5 + msr spsr_el2, x0 + adr x0, mmu_check + msr elr_el2, x0 + eret + +mmu_check: + /* + * whether the firmware left an MMU on: M bit, bit 0, of sctlr at + * the current EL. writing the register off would not fault, but + * the page tables it built are in its own memory, better to kill + * it here than trip over a stale mapping. + */ + mrs x0, CurrentEL + lsr x0, x0, #2 + cmp x0, #2 + b.lt mmu_el1 + mrs x0, sctlr_el2 + tbz x0, #0, c_entry + + mov x0, xzr + msr sctlr_el2, x0 + isb + tlbi alle2 + dsb sy + isb + b c_entry + +mmu_el1: + mrs x0, sctlr_el1 + tbz x0, #0, c_entry + + mov x0, xzr + msr sctlr_el1, x0 + isb + ic iallu + dsb sy + tlbi vmalle1 + dsb sy + isb + +c_entry: + /* + * program the counter frequency, the kernel reads CNTFRQ right + * away (booting.rst). qemu virt runs the system counter at + * 62.5 MHz. the register is RW only at the highest implemented EL. + */ + mrs x0, CurrentEL + lsr x0, x0, #2 + cmp x0, #2 + b.lt 1f + ldr x0, =62500000 + msr cntfrq_el0, x0 + isb +1: + /* our own vectors, so aborts print instead of vanishing */ + adr x0, vectors + mrs x1, CurrentEL + lsr x1, x1, #2 + cmp x1, #2 + b.lt 2f + msr vbar_el2, x0 + b 3f +2: + msr vbar_el1, x0 +3: + isb + + /* stack for the bootloader, its own region above the bss */ + ldr x0, =__stack_top + mov sp, x0 + + /* export the spin gate array address for the dtb patcher */ + adr x0, tb_spin_gates + adrp x1, tb_spin_gates_ptr + str x0, [x1, #:lo12:tb_spin_gates_ptr] + + /* clear bss */ + ldr x0, =__bss_start + ldr x1, =__bss_end +1: cmp x0, x1 + b.hs 2f + str xzr, [x0], #8 + b 1b +2: + + /* FP/SIMD access, some kernels assume it is on */ + mov x0, #(3 << 20) + msr cpacr_el1, x0 + isb + + /* dtb pointer into C arg 0 */ + mov x0, x19 + bl tashaboot_main + + /* if main returns there is nothing sensible to do */ +/* + * the spin table pen, the Wait For Event mechanism from the manual + * (B2-144, D1-2255). each secondary watches its own gate, the + * cpu-release-addr the dtb names. WFE clears the event register and + * sleeps, the kernel writes the secondary entry to the gate, makes + * it visible, then SEV sets the event register on every PE. the load + * recheck after each wake covers a release that lands between the + * load and the WFE. entered with MMU and caches off, left the same. + */ +.globl park_ret +park_ret: +park: + adr x0, tb_spin_gates + mrs x1, mpidr_el1 + and x1, x1, #0xff /* affinity 0, the core number */ + add x0, x0, x1, lsl #3 /* gate = gates + core * 8 */ + + /* diagnostic: stamp arrival, primary prints it later */ + adr x3, tb_pen_stamps + strb w1, [x3, x1] + sevl + wfe + sevl + wfe + +1: + ldr x2, [x0] + cbnz x2, 2f + wfe + b 1b +2: + /* interrupts masked at release, the manual's boot state */ + msr daifset, #0xf + /* + * every PE must read the same virtual counter. whatever + * ran before this loader could have left a per cpu offset + * in the virtual counter view, the kernel has no way to + * repair that itself. CNTVOFF_EL2 is writable at EL2 and + * the write holds for the EL1 virtual timer the kernel + * runs on. below EL2 it is out of reach, the reset value + * is the best a lower EL can do. + */ + mrs x4, CurrentEL + lsr x4, x4, #2 + cmp x4, #2 + b.lt 3f + msr cntvoff_el2, xzr + isb +3: + mov x0, xzr /* secondaries enter with x0-x3 zero */ + mov x1, xzr + mov x2, xzr + mov x3, xzr + dsb sy + isb + br x2 + +/* + * exception vectors, the armv8 layout: 16 slots, 128 bytes each, in + * the order the manual fixes. taken from EL1h the interesting slots + * are 0x200 sync and 0x380 SError, irq and fiq just park, the + * bootloader never enables interrupts on purpose. + */ + .balign 2048 +vectors: + /* 0x000: current EL, SP_EL0 */ + .align 7 + b exc_sync + .align 7 + b exc_park_irq + .align 7 + b exc_park_irq + .align 7 + b exc_serr + + /* 0x200: current EL, SP_ELx */ + .align 7 + b exc_sync + .align 7 + b exc_park_irq + .align 7 + b exc_park_irq + .align 7 + b exc_serr + + /* 0x400: lower EL, AArch64 */ + .align 7 + b exc_sync + .align 7 + b exc_park_irq + .align 7 + b exc_park_irq + .align 7 + b exc_serr + + /* 0x600: lower EL, AArch32 */ + .align 7 + b exc_sync + .align 7 + b exc_park_irq + .align 7 + b exc_park_irq + .align 7 + b exc_serr + +.pushsection .data.tb_spin, "aw" +.align 3 +.globl tb_spin_gates +tb_spin_gates: + .quad 0, 0, 0, 0, 0, 0, 0, 0 +.globl tb_spin_gates_ptr +tb_spin_gates_ptr: + .quad 0 +.globl tb_pen_stamps +tb_pen_stamps: + .byte 0, 0, 0, 0, 0, 0, 0, 0 +.popsection + +exc_sync: + stp x29, x30, [sp, #-16]! + mov x29, sp + mrs x3, CurrentEL + lsr x3, x3, #2 + cmp x3, #2 + b.lt 1f + mrs x0, esr_el2 + mrs x2, elr_el2 + lsr x1, x0, #26 + cmp x1, #0x16 /* HVC from lower EL */ + b.eq hvc_from_el1 + mrs x1, far_el2 + b 2f +1: + mrs x0, esr_el1 + mrs x1, far_el1 +2: + /* x2 = the faulting PC when it is the sync path */ + mrs x4, CurrentEL + lsr x4, x4, #2 + cmp x4, #2 + b.lt 3f + mrs x2, elr_el2 + b 4f +3: + mrs x2, elr_el1 +4: + bl exc_report + ldp x29, x30, [sp], #16 + b park + +/* + * HVC from EL1, the PSCI conduit. x0-x3 are the PSCI args in the + * caller registers, dispatch and return in x0. ELR_EL2 is already + * the resume point, eret takes it back. + */ +hvc_from_el1: + /* + * the lower EL sync slot. three arrivals share it: PSCI hvc + * from the kernel (EC 0x16, PSCI id in x0), our own boot + * handoff (hvc with the payload entry in x8), and semihosting + * hlt #0xf000 from the EL1 C runtime (EC 0x14). qemu only + * answers the hlt when it executes at EL2, so the handler + * replays the trap at EL2 and erets home with the result. + */ + mrs x1, esr_el2 + lsr x1, x1, #26 /* EC */ + cmp x1, #0x14 /* HLT from lower EL, semihosting */ + b.eq smh_replay + + /* + * the hvc arrives with either a PSCI function id in x0 (the + * kernel calling) or the boot handoff staging the payload + * entry in x8 and the dtb in x0. PSCI ids have the 0x84/0xc4 + * prefix, a dtb pointer never does. + */ + lsr x1, x0, #24 + cmp x1, #0x84 + b.eq psci_call + cmp x1, #0xc4 + b.eq psci_call + + /* the boot handoff: ELR_EL2 = entry, eret to the payload */ + msr elr_el2, x8 + eret + +smh_replay: + /* + * x0 holds the semihosting syscall number, x1 the parameter + * block, both live in the caller's registers. replay the hlt + * here at EL2 where qemu answers it, then eret back. + */ + hlt #0xf000 + eret + +psci_call: + stp x4, x5, [sp, #-16]! + stp x6, x7, [sp, #-16]! + stp x29, x30, [sp, #-16]! + mov x29, sp + + bl tb_psci_dispatch + + ldp x29, x30, [sp], #16 + ldp x6, x7, [sp], #16 + ldp x4, x5, [sp], #16 + ldp x29, x30, [sp], #16 + eret + +exc_serr: + stp x29, x30, [sp, #-16]! + mov x29, sp + mrs x3, CurrentEL + lsr x3, x3, #2 + cmp x3, #2 + b.lt 1f + mrs x0, esr_el2 + b 2f +1: + mrs x0, esr_el1 +2: + mov x1, #0 + mov x2, lr + bl exc_report + /* + * an SError while this loader runs means the machine is + * broken. handing the kernel a cpu that already lost is + * worse than stopping: report, then drive the reset domain + * the same way PSCI SYSTEM_RESET does. the reset call does + * not return, the park below is the fallback if a reset + * domain ignores the request. + */ + bl tb_system_reset + ldp x29, x30, [sp], #16 + b park + +exc_park_irq: + b park diff --git a/arch/arm64/kernel/tashaboot.lds b/arch/arm64/kernel/tashaboot.lds new file mode 100644 index 0000000..e2b7954 --- /dev/null +++ b/arch/arm64/kernel/tashaboot.lds @@ -0,0 +1,91 @@ +/* SPDX-License-Identifier: GPL-2.0+ */ +/* + * tashaboot arm64 memory layout. one segment, loaded at the bottom of + * RAM, right where qemu -kernel drops a raw image on virt. + * + * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org> + */ + +OUTPUT_FORMAT("elf64-littleaarch64", "elf64-littleaarch64", "elf64-littleaarch64") +OUTPUT_ARCH(aarch64) +ENTRY(_start) + +SECTIONS +{ + /* + * the first 64 bytes are the arm64 Image header: code0 'b' over + * it, magic ARM\x64, text_offset 0. qemu -kernel parses the + * header, loads the file at 0x40000000 and enters at + * 0x40000000, where the branch lands on reset at 0x40000040. + * without the header qemu guesses text_offset 0x80000 and runs + * the whole loader from the wrong address. + */ + . = 0x40000000; + + __image_copy_start = .; + _text_start = .; + + .text : + { + arch/arm64/kernel/start.o (.text.boot) + *(.text.boot) + + /* the Image header, code0 branches over it */ + . = ALIGN(64); + *(.text.imgheader) + . = ALIGN(64); + + *(.text*) + } + + . = ALIGN(8); + __text_end = .; + + .rodata : + { + *(SORT_BY_ALIGNMENT(.rodata*)) + } + + . = ALIGN(8); + __rodata_end = .; + + .data : + { + *(.data*) + } + + . = ALIGN(8); + __image_end = .; + + __bss_start = .; + .bss : + { + *(.bss*) + *(COMMON) + } + . = ALIGN(8); + __bss_end = .; + + /* + * the stack lives in its own region, clear of bss. page tables + * and buffers are bss objects, a stack sharing their address + * space grows down into them and the first deep call crushes + * whatever it meets. + */ + . = ALIGN(4096); + __stack_bottom = .; + . += 0x4000; + __stack_top = .; + __image_copy_end = .; + + /DISCARD/ : { *(.dynsym) } + /DISCARD/ : { *(.dynstr*) } + /DISCARD/ : { *(.dynamic*) } + /DISCARD/ : { *(.plt*) } + /DISCARD/ : { *(.interp*) } + /DISCARD/ : { *(.gnu*) } + /DISCARD/ : { *(.ARM.attributes) } + /DISCARD/ : { *(.comment) } + /DISCARD/ : { *(.note*) } + /DISCARD/ : { *(.eh_frame*) } +} diff --git a/arch/arm64/lib/cache.S b/arch/arm64/lib/cache.S new file mode 100644 index 0000000..d8ccea2 --- /dev/null +++ b/arch/arm64/lib/cache.S @@ -0,0 +1,100 @@ +/* SPDX-License-Identifier: GPL-2.0+ */ +/* + * cache.S - set/way cache maintenance, walked off CLIDR_EL1 the same + * way u-boot and the kernel's own __flush_dcache_all do it. needed + * before jumping to the payload so it starts from memory, not cache. + * + * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org> + */ + +#include <asm/linkage.h> + +.pushsection .text.tb_dcache_level, "ax" +ENTRY(tb_dcache_level) + lsl x12, x0, #1 + msr csselr_el1, x12 /* select cache level */ + isb /* sync change of ccsidr_el1 */ + mrs x6, ccsidr_el1 /* read the new ccsidr_el1 */ + ubfx x2, x6, #0, #3 /* x2 <- log2(cache line size)-4 */ + ubfx x3, x6, #3, #10 /* x3 <- number of cache ways - 1 */ + ubfx x4, x6, #13, #15 /* x4 <- number of cache sets - 1 */ + add x2, x2, #4 /* x2 <- log2(cache line size) */ + clz w5, w3 /* x5 <- bit position of #ways */ + /* x12 <- cache level << 1 */ + /* x2 <- line length offset */ + /* x3 <- number of cache ways - 1 */ + /* x4 <- number of cache sets - 1 */ + /* x5 <- bit position of #ways */ + +loop_set: + mov x6, x3 /* x6 <- working copy of #ways */ +loop_way: + lsl x7, x6, x5 + orr x9, x12, x7 /* map way and level to cisw value */ + lsl x7, x4, x2 + orr x9, x9, x7 /* map set number to cisw value */ + dc cisw, x9 /* clean & invalidate by set/way */ + subs x6, x6, #1 /* decrement the way */ + b.ge loop_way + subs x4, x4, #1 /* decrement the set */ + b.ge loop_set + + ret +ENDPROC(tb_dcache_level) +.popsection + +/* + * void tb_flush_dcache_all(void) + * + * clean & invalidate the whole D cache by set/way. + */ +.pushsection .text.tb_flush_dcache_all, "ax" +ENTRY(tb_flush_dcache_all) + mov x1, x0 + dsb sy + mrs x10, clidr_el1 /* read clidr_el1 */ + ubfx x11, x10, #24, #3 /* x11 <- loc */ + cbz x11, finished /* if loc is 0, exit */ + mov x15, lr + mov x0, #0 /* start flush at cache level 0 */ + /* x0 <- cache level */ + /* x10 <- clidr_el1 */ + /* x11 <- loc */ + /* x15 <- return address */ + +loop_level: + add x12, x0, x0, lsl #1 /* x12 <- tripled cache level */ + lsr x12, x10, x12 + and x12, x12, #7 /* x12 <- cache type */ + cmp x12, #2 + b.lt skip /* skip if no cache or icache */ + bl tb_dcache_level /* flush this level */ +skip: + add x0, x0, #1 /* increment cache level */ + cmp x11, x0 + b.gt loop_level + + mov x0, #0 + msr csselr_el1, x0 /* restore csselr_el1 */ + dsb sy + isb + mov lr, x15 + +finished: + ret +ENDPROC(tb_flush_dcache_all) +.popsection + +/* + * void tb_invalidate_icache_all(void) + * + * I cache invalidation to PoU, one ic iallu covers the local core. + */ +.pushsection .text.tb_invalidate_icache_all, "ax" +ENTRY(tb_invalidate_icache_all) + ic iallu + dsb sy + isb + ret +ENDPROC(tb_invalidate_icache_all) +.popsection diff --git a/arch/arm64/lib/cache_va.c b/arch/arm64/lib/cache_va.c new file mode 100644 index 0000000..1fd7804 --- /dev/null +++ b/arch/arm64/lib/cache_va.c @@ -0,0 +1,73 @@ +/* SPDX-License-Identifier: GPL-2.0+ */ +/* + * cache_va.c - cache maintenance by virtual address, the operations + * the manual prescribes for boot handoff: clean to point of + * coherency (dc cvac), invalidate (dc ivac), and clean and + * invalidate (dc civac), plus icache invalidate by VA to the point + * of unification (ic ivau). by VA beats by set and way when the + * address range is known, the manual's own guidance, set and way + * only for the full flush cases in cache.S. + * + * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org> + */ + +#include <stdint.h> +#include <sys/types.h> + +#define CACHE_LINE_SHIFT 6 /* 64 byte lines on cortex-a class */ +#define CACHE_LINE_SIZE (1 << CACHE_LINE_SHIFT) + +void tb_clean_dcache_range(uintptr_t start, size_t len) +{ + uintptr_t line = start & ~(uintptr_t)(CACHE_LINE_SIZE - 1); + uintptr_t end = start + len; + + while (line < end) { + asm volatile("dc cvac, %0" :: "r" (line) : "memory"); + line += CACHE_LINE_SIZE; + } + + asm volatile("dsb sy" ::: "memory"); +} + +void tb_inval_dcache_range(uintptr_t start, size_t len) +{ + uintptr_t line = start & ~(uintptr_t)(CACHE_LINE_SIZE - 1); + uintptr_t end = start + len; + + while (line < end) { + asm volatile("dc ivac, %0" :: "r" (line) : "memory"); + line += CACHE_LINE_SIZE; + } + + asm volatile("dsb sy" ::: "memory"); +} + +void tb_clean_inval_dcache_range(uintptr_t start, size_t len) +{ + uintptr_t line = start & ~(uintptr_t)(CACHE_LINE_SIZE - 1); + uintptr_t end = start + len; + + while (line < end) { + asm volatile("dc civac, %0" :: "r" (line) : "memory"); + line += CACHE_LINE_SIZE; + } + + asm volatile("dsb sy" ::: "memory"); +} + +void tb_inval_icache_range(uintptr_t start, size_t len) +{ + uintptr_t line = start & ~(uintptr_t)(CACHE_LINE_SIZE - 1); + uintptr_t end = start + len; + + while (line < end) { + asm volatile("ic ivau, %0" :: "r" (line) : "memory"); + line += CACHE_LINE_SIZE; + } + + asm volatile( + "dsb ish\n" + "isb\n" + ::: "memory"); +} diff --git a/arch/arm64/lib/gic.c b/arch/arm64/lib/gic.c new file mode 100644 index 0000000..647e987 --- /dev/null +++ b/arch/arm64/lib/gic.c @@ -0,0 +1,105 @@ +/* + * gic.c - the interrupt controller state a bootloader owns. the + * kernel programs the gic itself for the running system, but it + * trusts the state it inherits: on real hardware the secure + * world configures which interrupts are visible to non-secure, + * and a bootloader that leaves random enables or secure group + * bits set hands the kernel a half-configured distributor that + * can fire before the kernel's irqchip driver is up. + * + * this is the gicv2 sequence from the TRM, the same shape + * u-boot leaves the machine in: distributor off, every + * interrupt in the non-secure group, all per interrupt enables + * cleared, pending state cleared, cpu interfaces off. defined + * state, nothing firing, the kernel starts from zero. + * + * GICv1 shows the same register map minus the security + * extension registers, the writes below are harmless there. + * + * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org> + */ + +#include <string.h> +#include <stdint.h> +#include <boot.h> +#include <reg.h> + +/* distributor registers, offsets from the GICD base */ +#define GICD_CTLR 0x000 +#define GICD_TYPER 0x004 +#define GICD_ISENABLER(n) (0x100 + (n) * 4) +#define GICD_ICENABLER(n) (0x180 + (n) * 4) +#define GICD_ICPENDR(n) (0x280 + (n) * 4) +#define GICD_ICACTIVER(n) (0x380 + (n) * 4) + +/* cpu interface registers, offsets from the GICC base */ +#define GICC_CTLR 0x000 +#define GICC_PMR 0x004 + +/* GICD_CTLR bits */ +#define GICD_CTLR_ENABLE_GRP1 (1 << 0) +#define GICD_CTLR_ENABLE_GRP0 (1 << 1) + +/* GICC_CTLR bits */ +#define GICC_CTLR_ENABLE (1 << 0) + +#define GICD_TYPER_ITLINES_MASK 0x1f + +/* + * how many 32-irq lines the distributor carries, TYPER.ITLines + * holds count of (irqs / 32) - 1, clamped per the spec because + * the field is 5 bits and caps at 1020 irqs. + */ +static int gicd_irq_lines(uintptr_t gicd) +{ + uint32_t typer = readl(REG32(gicd + GICD_TYPER)); + + return ((typer & GICD_TYPER_ITLINES_MASK) + 1); +} + +/* + * leave the gic in the defined state the kernel expects. the + * addresses come from the devicetree the caller walked, qemu + * virt carries a gicv2 at 0x08000000 with the cpu interface at + * +0x10000. + */ +int tb_gic_init(uintptr_t gicd, uintptr_t gicc) +{ + int lines; + int n; + + if (!gicd || !gicc) + return -1; + + /* the distributor is off while it is reconfigured */ + writel(0, REG32(gicd + GICD_CTLR)); + writel(0, REG32(gicc + GICC_CTLR)); + + lines = gicd_irq_lines(gicd); + + /* + * the group routing is deliberately untouched. the group + * registers are the secure world's, a non-secure loader's + * writes are dropped on hardware that implements the + * security extension, and on emulators that accept them + * the timer's per cpu interrupts stop reaching the + * kernel. group config belongs to the EL3 monitor, this + * loader runs without one. + */ + + /* no per interrupt enables, nothing pending */ + for (n = 0; n < lines; n++) { + writel(0xffffffff, REG32(gicd + GICD_ICENABLER(n))); + writel(0xffffffff, REG32(gicd + GICD_ICPENDR(n))); + } + + /* + * the cpu interface stays off with the priority mask at + * the lowest priority, the kernel raises it when it + * brings its own irq handling up. off is the defined + * state, the enable is the kernel's decision to make. + */ + writel(0, REG32(gicc + GICC_PMR)); + + return 0; +} diff --git a/arch/arm64/lib/mmu.c b/arch/arm64/lib/mmu.c new file mode 100644 index 0000000..03eb355 --- /dev/null +++ b/arch/arm64/lib/mmu.c @@ -0,0 +1,205 @@ +/* SPDX-License-Identifier: GPL-2.0+ */ +/* + * mmu.c - VMSAv8-64 stage 1 identity map for EL2. + * + * One level 0 table plus the subtables for the low 1GB of MMIO and + * the RAM region. everything is identity mapped, the bootloader + * never needs a different VA view, it just needs caching rules that + * let the payload start from an architecture-defined state. + * + * The descriptor layouts are from the manual (DDI 0487), level 0/1/2 + * and level 3 formats at D5-2444 and D5-2447, attribute fields at + * D5-2451, MAIR at D5-2476. feature bits come from the ID registers, + * never hardcoded, the PA size from ID_AA64MMFR0_EL1.PARange per + * "Address size configuration" D5-2399. + * + * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org> + */ + +#include <asm/mmu.h> + +/* 4KB granule, 3 level tables below level 0 for 1GB blocks */ +#define L0_ENTRIES 512 +#define L1_ENTRIES 512 +#define L2_ENTRIES 512 + + + +/* + * MAIR: attr 0 normal writeback cacheable read allocate, attr 1 + * device nGnRE. encodings straight from D5-2476, B2-122 for the + * memory types. + */ +#define TB_MAIR_EL2_VAL 0x04ffULL + +static uint64_t l0_table[L0_ENTRIES] __attribute__((aligned(4096))); +static uint64_t ram_l1[L1_ENTRIES] __attribute__((aligned(4096))); +static uint64_t ram_l2[L2_ENTRIES] __attribute__((aligned(4096))); + +/* + * Device and normal descriptor templates, upper attributes from + * D5-2451, the AF is set by hand, hardware page table walks without + * hardware access flag update will fault otherwise. + */ +#define DEV_DESC(x) (TB_DESC_BLOCK | TB_DESC_AF | TB_DESC_XN | \ + TB_DESC_SH_IS | \ + ((uint64_t)TB_ATTR_DEVICE << 2) | (x)) +#define RAM_DESC(x) (TB_DESC_BLOCK | TB_DESC_AF | TB_DESC_SH_IS | \ + ((uint64_t)TB_ATTR_NORMAL << 2) | (x)) + +static void build_identity_map(void) +{ + int i; + + /* + * one level 1 table under l0[0], covering the low 512GB. the + * MMIO hole and RAM are both in it, device block at index 0 + * (0..1GB) and the RAM table at index 1 (1GB..2GB). + */ + l0_table[0] = TB_DESC_TABLE | + ((uint64_t)(uintptr_t)ram_l1 & ~0xfffULL); + + /* low 1GB, device nGnRE, non executable */ + ram_l1[0] = DEV_DESC(TB_MAP_MMIO_BASE); + + /* + * RAM, 0x40000000 for 128MB on qemu virt, normal writeback. + * the level 2 table splits the 1GB into 2MB blocks so the map + * can be carved later. + */ + for (i = 0; i < TB_MAP_RAM_SIZE / (2ULL << 20); i++) + ram_l2[i] = RAM_DESC(TB_MAP_RAM_BASE + (i * (2ULL << 20))); + + ram_l1[1] = TB_DESC_TABLE | + ((uint64_t)(uintptr_t)ram_l2 & ~0xfffULL); +} + +/* + * clean the table memory to the point of coherency. the tables were + * written with the dcache off, the page table walker reads them as + * memory the TCR walk attributes describe, and a dirty line sitting + * in the cache would never reach RAM. dc cvac is by cache line, walk + * every page of table memory. + */ +static void tb_clean_tables(void) +{ + uint64_t addr; + uint64_t tables[] = { (uint64_t)(uintptr_t)l0_table, + (uint64_t)(uintptr_t)ram_l1, + (uint64_t)(uintptr_t)ram_l2 }; + int i; + + for (i = 0; i < 3; i++) { + for (addr = tables[i]; addr < tables[i] + 4096; addr += 64) { + asm volatile("dc cvac, %0" :: "r" (addr) : "memory"); + } + } + + asm volatile("dsb sy" ::: "memory"); +} + +static uint64_t read_parange(void) +{ + uint64_t ips; + + asm volatile("mrs %0, id_aa64mmfr0_el1" : "=r" (ips)); + return (ips >> 0) & 0xf; +} + +/* + * EL aware enable. the EL1&0 regime registers at EL1, the EL2 regime + * registers at EL2, one code path per the manual, one translation + * regime per exception level (D1-2146). + */ +int tb_mmu_enable(void) +{ + uint64_t tcr, mair; + uint64_t el; + + build_identity_map(); + tb_clean_tables(); + + asm volatile("mrs %0, CurrentEL" : "=r" (el)); + el >>= 2; + + /* tcr value and PA size, D5-2399 address size configuration */ + tcr = TB_TCR_T0SZ_48 | TB_TCR_SH0_IS | TB_TCR_TG0_4K | + TB_TCR_IRGN0_WB | TB_TCR_ORGN0_WB | TB_TCR_IPS(read_parange()); + mair = TB_MAIR_EL2_VAL; + + if (el == 2) { + asm volatile( + "dsb sy\n" + "msr ttbr0_el2, %1\n" + "msr tcr_el2, %2\n" + "msr mair_el2, %3\n" + "isb\n" + "tlbi alle2\n" + "dsb sy\n" + "ic iallu\n" + "dsb sy\n" + "isb\n" + : "=r" (tcr) + : "r" (l0_table), "r" (tcr), "r" (mair) + : "memory"); + asm volatile( + "mrs x0, sctlr_el2\n" + "orr x0, x0, #1\n" + "msr sctlr_el2, x0\n" + "isb\n" + ::: "x0", "memory"); + } else { + asm volatile( + "dsb sy\n" + "msr ttbr0_el1, %1\n" + "msr tcr_el1, %2\n" + "msr mair_el1, %3\n" + "isb\n" + "tlbi vmalle1\n" + "dsb sy\n" + "ic iallu\n" + "dsb sy\n" + "isb\n" + : "=r" (tcr) + : "r" (l0_table), "r" (tcr), "r" (mair) + : "memory"); + asm volatile( + "mrs x0, sctlr_el1\n" + "orr x0, x0, #1\n" + "msr sctlr_el1, x0\n" + "isb\n" + ::: "x0", "memory"); + } + + return 0; +} + +void tb_mmu_disable(void) +{ + uint64_t el; + + asm volatile("mrs %0, CurrentEL" : "=r" (el)); + el >>= 2; + + if (el == 2) { + asm volatile( + "mrs x0, sctlr_el2\n" + "bic x0, x0, #1\n" + "msr sctlr_el2, x0\n" + "dsb sy\n" + "tlbi alle2\n" + "dsb sy\n" + "isb\n" + ::: "x0", "memory"); + } else { + asm volatile( + "mrs x0, sctlr_el1\n" + "bic x0, x0, #1\n" + "msr sctlr_el1, x0\n" + "dsb sy\n" + "tlbi vmalle1\n" + "dsb sy\n" + "isb\n" + ::: "x0", "memory"); + } +} diff --git a/arch/arm64/lib/psci.c b/arch/arm64/lib/psci.c new file mode 100644 index 0000000..ad5f461 --- /dev/null +++ b/arch/arm64/lib/psci.c @@ -0,0 +1,118 @@ +/* SPDX-License-Identifier: GPL-2.0+ */ +/* + * psci.c - PSCI 0.2 at EL2, the calls a payload makes to control + * cores and the system, per DEN 0022. the call arrives as an HVC + * trap at EL2 (EC 0x16), x0 holds the function id, x1 to x3 the + * arguments, the return value goes back in x0 and eret resumes the + * caller at EL1. + * + * CPU_ON writes the spin gate of the target core and SEVs, the pen + * from start.S does the release. CPU_OFF parks the calling core. + * SYSTEM_OFF and SYSTEM_RESET drive the architecture reset domain. + * + * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org> + */ + +#include <asm/psci.h> +#include <debug.h> + +extern void tb_system_reset(void); +extern void tb_system_off(void); + +/* the gates and stamps from start.S, one per possible core */ +extern unsigned long tb_spin_gates[8]; +extern unsigned char tb_pen_stamps[8]; + +static uint64_t psci_cpu_on(uint64_t target, uint64_t entry, + uint64_t ctx) +{ + unsigned long mpidr; + int cpu; + + /* affinity 0 only, our gate array indexes cores 0..7 */ + if (target > 7) + return PSCI_RET_INVALID_PARAMS; + + asm volatile("mrs %0, mpidr_el1" : "=r" (mpidr)); + if ((mpidr & 0xff) == target) + return PSCI_RET_ALREADY_ON; + + cpu = (int)target; + + /* + * the pen saves no context, CPU_ON per DEN 0022 passes an + * entry and a context id. the pen enters with x0 = ctx, the + * kernel secondary entry takes x0 as its context pointer. + * the gate holds the entry, the stamp array the ctx. + */ + tb_spin_gates[cpu] = entry; + tb_pen_stamps[cpu] = (unsigned char)(ctx & 0xff); + + /* make the gate write visible before the wake, D1-2255 */ + asm volatile("dsb sy"); + asm volatile("sev"); + + dprintf(ALWAYS, "psci: cpu_on %llu -> %llx\n", + (unsigned long long)target, + (unsigned long long)entry); + + return PSCI_RET_SUCCESS; +} + +extern void park_ret(void); + +static uint64_t psci_cpu_off(void) +{ + unsigned long mpidr; + int cpu; + + asm volatile("mrs %0, mpidr_el1" : "=r" (mpidr)); + cpu = (int)(mpidr & 0xff); + + if (cpu > 7) + return PSCI_RET_NOT_SUPPORTED; + + /* clear our own gate and go back to the pen */ + tb_spin_gates[cpu] = 0; + + dprintf(ALWAYS, "psci: cpu_off %d\n", cpu); + + asm volatile( + "dsb sy\n" + "b park_ret\n" + ); + + return PSCI_RET_INTERNAL_FAIL; /* not reached */ +} + +/* + * the asm vector calls this with the caller x0-x3 still in place, + * function id in x0, arguments in x1-x3, the return lands in x0. + */ +uint64_t tb_psci_dispatch(uint64_t fn, uint64_t x1, uint64_t x2, + uint64_t x3) +{ + switch (fn) { + case PSCI_FN_VERSION: + return PSCI_VERSION_0_2; + + case PSCI_FN_CPU_ON: + return psci_cpu_on(x1, x2, x3); + + case PSCI_FN_CPU_OFF: + return psci_cpu_off(); + + case PSCI_FN_SYSTEM_OFF: + dprintf(ALWAYS, "psci: system off\n"); + tb_system_off(); + return PSCI_RET_SUCCESS; + + case PSCI_FN_SYSTEM_RESET: + dprintf(ALWAYS, "psci: system reset\n"); + tb_system_reset(); + return PSCI_RET_INTERNAL_FAIL; /* not reached */ + + default: + return PSCI_RET_NOT_SUPPORTED; + } +} diff --git a/arch/arm64/lib/semihosting.S b/arch/arm64/lib/semihosting.S new file mode 100644 index 0000000..6e3fc31 --- /dev/null +++ b/arch/arm64/lib/semihosting.S @@ -0,0 +1,18 @@ +/* SPDX-License-Identifier: GPL-2.0+ */ +/* + * semihosting.S - the trap instruction itself. qemu answers this when + * it is started with -semihosting, and nothing happens without it, so + * every caller has to cope with the no-debugger case. + * + * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org> + */ + +#include <asm/linkage.h> + +.pushsection .text.smh_trap, "ax" +/* long smh_trap(unsigned int sysnum, void *addr); */ +ENTRY(smh_trap) + hlt #0xf000 + ret +ENDPROC(smh_trap) +.popsection diff --git a/arch/arm64/lib/system.c b/arch/arm64/lib/system.c new file mode 100644 index 0000000..25af2bf --- /dev/null +++ b/arch/arm64/lib/system.c @@ -0,0 +1,70 @@ +/* SPDX-License-Identifier: GPL-2.0+ */ +/* + * system.c - system power control, the PSCI SYSTEM_OFF and + * SYSTEM_RESET backends. off parks the core in WFI forever, the + * manual's low power entry (D1-2255). reset drives the PE reset + * domain: RMR_EL2 reset request with the system reset bit, RR bit 1, + * followed by a barrier pair so the request retires before anything + * else observes the core. + * + * On real hardware a SoC also needs a watchdog or PMIC write for a + * full board reset, that is board territory, the arch part is this. + * + * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org> + */ + +#include <debug.h> +#include <stdint.h> + +void tb_system_off(void) +{ + dprintf(ALWAYS, "system off\n"); + + for (;;) { + asm volatile("wfi"); + } +} + +void tb_system_reset(void) +{ + uint64_t rmr; + uint64_t el; + register uint64_t r0 asm("x0") = 0x84000008; + register uint64_t r1 asm("x1") = 0; + register uint64_t r2 asm("x2") = 0; + register uint64_t r3 asm("x3") = 0; + + dprintf(ALWAYS, "system reset\n"); + + /* + * the firmware conduit first, PSCI SYSTEM_RESET through + * the machine's own monitor. this is the only legal way + * up from EL1: RMR_EL1 is undefined on hardware that + * implements a higher exception level, the access traps + * and the machine never resets. + */ + asm volatile("smc #0" + : "+r"(r0), "+r"(r1), "+r"(r2), "+r"(r3)); + + /* + * no monitor answered, or it refused. ask the reset + * domain directly from a level that may write it. + */ + asm volatile("mrs %0, CurrentEL" : "=r" (el)); + el >>= 2; + + if (el >= 2) { + asm volatile("mrs %0, rmr_el2" : "=r" (rmr)); + rmr |= (1 << 1); /* RR, request reset */ + asm volatile( + "msr rmr_el2, %0\n" + "dsb sy\n" + "isb\n" + :: "r" (rmr)); + } + + /* nothing worked, park */ + for (;;) { + asm volatile("wfi"); + } +} diff --git a/arch/arm64/lib/timer.c b/arch/arm64/lib/timer.c new file mode 100644 index 0000000..605cd18 --- /dev/null +++ b/arch/arm64/lib/timer.c @@ -0,0 +1,48 @@ +/* SPDX-License-Identifier: GPL-2.0+ */ +/* + * timer.c - generic timer delays, the system counter from D10. the + * counter is a fixed frequency free running counter, CNTFRQ_EL0 + * carries the frequency, CNTVCT_EL0 the 64 bit count. delays are a + * busy wait on the counter, no interrupts needed, microsecond and + * millisecond granularity. + * + * CNTVCT_EL0 is the virtual counter view; at EL2 with no offset + * configured it is the physical count. the read is not speculative + * and needs an isb to serialize against subsequent counter reads + * per the counter access rules. + * + * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org> + */ + +#include <stdint.h> + +static uint64_t read_cntfrq(void) +{ + uint64_t v; + + asm volatile("mrs %0, cntfrq_el0" : "=r" (v)); + return v; +} + +static uint64_t read_counter(void) +{ + uint64_t v; + + asm volatile("isb\nmrs %0, cntvct_el0" : "=r" (v)); + return v; +} + +void tb_udelay(uint32_t us) +{ + uint64_t freq = read_cntfrq(); + uint64_t start = read_counter(); + uint64_t ticks = (uint64_t)us * freq / 1000000ULL; + + while (read_counter() - start < ticks) + ; +} + +void tb_mdelay(uint32_t ms) +{ + tb_udelay(ms * 1000); +} |
