summaryrefslogtreecommitdiff
path: root/arch/arm64/lib
diff options
context:
space:
mode:
authorBradley Morgan <brads@mainlining.org>2026-10-04 10:42:10 +0000
committerBradley Morgan <brads@mainlining.org>2026-10-04 10:46:46 +0000
commit5ff54a875642962fc857cee00bde17f9a465f1fa (patch)
treeb265a12d538167c3bbb86b24118a7ad7a062fc74 /arch/arm64/lib
tashaboot: arm64 bootloaderHEADmain
holy shit it's here, Tashaboot, based from arm arm, enjoy reading this masterpiece Signed-off-by: Bradley Morgan <brads@mainlining.org>
Diffstat (limited to 'arch/arm64/lib')
-rw-r--r--arch/arm64/lib/cache.S100
-rw-r--r--arch/arm64/lib/cache_va.c73
-rw-r--r--arch/arm64/lib/gic.c105
-rw-r--r--arch/arm64/lib/mmu.c205
-rw-r--r--arch/arm64/lib/psci.c118
-rw-r--r--arch/arm64/lib/semihosting.S18
-rw-r--r--arch/arm64/lib/system.c70
-rw-r--r--arch/arm64/lib/timer.c48
8 files changed, 737 insertions, 0 deletions
diff --git a/arch/arm64/lib/cache.S b/arch/arm64/lib/cache.S
new file mode 100644
index 0000000..d8ccea2
--- /dev/null
+++ b/arch/arm64/lib/cache.S
@@ -0,0 +1,100 @@
+/* SPDX-License-Identifier: GPL-2.0+ */
+/*
+ * cache.S - set/way cache maintenance, walked off CLIDR_EL1 the same
+ * way u-boot and the kernel's own __flush_dcache_all do it. needed
+ * before jumping to the payload so it starts from memory, not cache.
+ *
+ * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org>
+ */
+
+#include <asm/linkage.h>
+
+.pushsection .text.tb_dcache_level, "ax"
+ENTRY(tb_dcache_level)
+ lsl x12, x0, #1
+ msr csselr_el1, x12 /* select cache level */
+ isb /* sync change of ccsidr_el1 */
+ mrs x6, ccsidr_el1 /* read the new ccsidr_el1 */
+ ubfx x2, x6, #0, #3 /* x2 <- log2(cache line size)-4 */
+ ubfx x3, x6, #3, #10 /* x3 <- number of cache ways - 1 */
+ ubfx x4, x6, #13, #15 /* x4 <- number of cache sets - 1 */
+ add x2, x2, #4 /* x2 <- log2(cache line size) */
+ clz w5, w3 /* x5 <- bit position of #ways */
+ /* x12 <- cache level << 1 */
+ /* x2 <- line length offset */
+ /* x3 <- number of cache ways - 1 */
+ /* x4 <- number of cache sets - 1 */
+ /* x5 <- bit position of #ways */
+
+loop_set:
+ mov x6, x3 /* x6 <- working copy of #ways */
+loop_way:
+ lsl x7, x6, x5
+ orr x9, x12, x7 /* map way and level to cisw value */
+ lsl x7, x4, x2
+ orr x9, x9, x7 /* map set number to cisw value */
+ dc cisw, x9 /* clean & invalidate by set/way */
+ subs x6, x6, #1 /* decrement the way */
+ b.ge loop_way
+ subs x4, x4, #1 /* decrement the set */
+ b.ge loop_set
+
+ ret
+ENDPROC(tb_dcache_level)
+.popsection
+
+/*
+ * void tb_flush_dcache_all(void)
+ *
+ * clean & invalidate the whole D cache by set/way.
+ */
+.pushsection .text.tb_flush_dcache_all, "ax"
+ENTRY(tb_flush_dcache_all)
+ mov x1, x0
+ dsb sy
+ mrs x10, clidr_el1 /* read clidr_el1 */
+ ubfx x11, x10, #24, #3 /* x11 <- loc */
+ cbz x11, finished /* if loc is 0, exit */
+ mov x15, lr
+ mov x0, #0 /* start flush at cache level 0 */
+ /* x0 <- cache level */
+ /* x10 <- clidr_el1 */
+ /* x11 <- loc */
+ /* x15 <- return address */
+
+loop_level:
+ add x12, x0, x0, lsl #1 /* x12 <- tripled cache level */
+ lsr x12, x10, x12
+ and x12, x12, #7 /* x12 <- cache type */
+ cmp x12, #2
+ b.lt skip /* skip if no cache or icache */
+ bl tb_dcache_level /* flush this level */
+skip:
+ add x0, x0, #1 /* increment cache level */
+ cmp x11, x0
+ b.gt loop_level
+
+ mov x0, #0
+ msr csselr_el1, x0 /* restore csselr_el1 */
+ dsb sy
+ isb
+ mov lr, x15
+
+finished:
+ ret
+ENDPROC(tb_flush_dcache_all)
+.popsection
+
+/*
+ * void tb_invalidate_icache_all(void)
+ *
+ * I cache invalidation to PoU, one ic iallu covers the local core.
+ */
+.pushsection .text.tb_invalidate_icache_all, "ax"
+ENTRY(tb_invalidate_icache_all)
+ ic iallu
+ dsb sy
+ isb
+ ret
+ENDPROC(tb_invalidate_icache_all)
+.popsection
diff --git a/arch/arm64/lib/cache_va.c b/arch/arm64/lib/cache_va.c
new file mode 100644
index 0000000..1fd7804
--- /dev/null
+++ b/arch/arm64/lib/cache_va.c
@@ -0,0 +1,73 @@
+/* SPDX-License-Identifier: GPL-2.0+ */
+/*
+ * cache_va.c - cache maintenance by virtual address, the operations
+ * the manual prescribes for boot handoff: clean to point of
+ * coherency (dc cvac), invalidate (dc ivac), and clean and
+ * invalidate (dc civac), plus icache invalidate by VA to the point
+ * of unification (ic ivau). by VA beats by set and way when the
+ * address range is known, the manual's own guidance, set and way
+ * only for the full flush cases in cache.S.
+ *
+ * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org>
+ */
+
+#include <stdint.h>
+#include <sys/types.h>
+
+#define CACHE_LINE_SHIFT 6 /* 64 byte lines on cortex-a class */
+#define CACHE_LINE_SIZE (1 << CACHE_LINE_SHIFT)
+
+void tb_clean_dcache_range(uintptr_t start, size_t len)
+{
+ uintptr_t line = start & ~(uintptr_t)(CACHE_LINE_SIZE - 1);
+ uintptr_t end = start + len;
+
+ while (line < end) {
+ asm volatile("dc cvac, %0" :: "r" (line) : "memory");
+ line += CACHE_LINE_SIZE;
+ }
+
+ asm volatile("dsb sy" ::: "memory");
+}
+
+void tb_inval_dcache_range(uintptr_t start, size_t len)
+{
+ uintptr_t line = start & ~(uintptr_t)(CACHE_LINE_SIZE - 1);
+ uintptr_t end = start + len;
+
+ while (line < end) {
+ asm volatile("dc ivac, %0" :: "r" (line) : "memory");
+ line += CACHE_LINE_SIZE;
+ }
+
+ asm volatile("dsb sy" ::: "memory");
+}
+
+void tb_clean_inval_dcache_range(uintptr_t start, size_t len)
+{
+ uintptr_t line = start & ~(uintptr_t)(CACHE_LINE_SIZE - 1);
+ uintptr_t end = start + len;
+
+ while (line < end) {
+ asm volatile("dc civac, %0" :: "r" (line) : "memory");
+ line += CACHE_LINE_SIZE;
+ }
+
+ asm volatile("dsb sy" ::: "memory");
+}
+
+void tb_inval_icache_range(uintptr_t start, size_t len)
+{
+ uintptr_t line = start & ~(uintptr_t)(CACHE_LINE_SIZE - 1);
+ uintptr_t end = start + len;
+
+ while (line < end) {
+ asm volatile("ic ivau, %0" :: "r" (line) : "memory");
+ line += CACHE_LINE_SIZE;
+ }
+
+ asm volatile(
+ "dsb ish\n"
+ "isb\n"
+ ::: "memory");
+}
diff --git a/arch/arm64/lib/gic.c b/arch/arm64/lib/gic.c
new file mode 100644
index 0000000..647e987
--- /dev/null
+++ b/arch/arm64/lib/gic.c
@@ -0,0 +1,105 @@
+/*
+ * gic.c - the interrupt controller state a bootloader owns. the
+ * kernel programs the gic itself for the running system, but it
+ * trusts the state it inherits: on real hardware the secure
+ * world configures which interrupts are visible to non-secure,
+ * and a bootloader that leaves random enables or secure group
+ * bits set hands the kernel a half-configured distributor that
+ * can fire before the kernel's irqchip driver is up.
+ *
+ * this is the gicv2 sequence from the TRM, the same shape
+ * u-boot leaves the machine in: distributor off, every
+ * interrupt in the non-secure group, all per interrupt enables
+ * cleared, pending state cleared, cpu interfaces off. defined
+ * state, nothing firing, the kernel starts from zero.
+ *
+ * GICv1 shows the same register map minus the security
+ * extension registers, the writes below are harmless there.
+ *
+ * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org>
+ */
+
+#include <string.h>
+#include <stdint.h>
+#include <boot.h>
+#include <reg.h>
+
+/* distributor registers, offsets from the GICD base */
+#define GICD_CTLR 0x000
+#define GICD_TYPER 0x004
+#define GICD_ISENABLER(n) (0x100 + (n) * 4)
+#define GICD_ICENABLER(n) (0x180 + (n) * 4)
+#define GICD_ICPENDR(n) (0x280 + (n) * 4)
+#define GICD_ICACTIVER(n) (0x380 + (n) * 4)
+
+/* cpu interface registers, offsets from the GICC base */
+#define GICC_CTLR 0x000
+#define GICC_PMR 0x004
+
+/* GICD_CTLR bits */
+#define GICD_CTLR_ENABLE_GRP1 (1 << 0)
+#define GICD_CTLR_ENABLE_GRP0 (1 << 1)
+
+/* GICC_CTLR bits */
+#define GICC_CTLR_ENABLE (1 << 0)
+
+#define GICD_TYPER_ITLINES_MASK 0x1f
+
+/*
+ * how many 32-irq lines the distributor carries, TYPER.ITLines
+ * holds count of (irqs / 32) - 1, clamped per the spec because
+ * the field is 5 bits and caps at 1020 irqs.
+ */
+static int gicd_irq_lines(uintptr_t gicd)
+{
+ uint32_t typer = readl(REG32(gicd + GICD_TYPER));
+
+ return ((typer & GICD_TYPER_ITLINES_MASK) + 1);
+}
+
+/*
+ * leave the gic in the defined state the kernel expects. the
+ * addresses come from the devicetree the caller walked, qemu
+ * virt carries a gicv2 at 0x08000000 with the cpu interface at
+ * +0x10000.
+ */
+int tb_gic_init(uintptr_t gicd, uintptr_t gicc)
+{
+ int lines;
+ int n;
+
+ if (!gicd || !gicc)
+ return -1;
+
+ /* the distributor is off while it is reconfigured */
+ writel(0, REG32(gicd + GICD_CTLR));
+ writel(0, REG32(gicc + GICC_CTLR));
+
+ lines = gicd_irq_lines(gicd);
+
+ /*
+ * the group routing is deliberately untouched. the group
+ * registers are the secure world's, a non-secure loader's
+ * writes are dropped on hardware that implements the
+ * security extension, and on emulators that accept them
+ * the timer's per cpu interrupts stop reaching the
+ * kernel. group config belongs to the EL3 monitor, this
+ * loader runs without one.
+ */
+
+ /* no per interrupt enables, nothing pending */
+ for (n = 0; n < lines; n++) {
+ writel(0xffffffff, REG32(gicd + GICD_ICENABLER(n)));
+ writel(0xffffffff, REG32(gicd + GICD_ICPENDR(n)));
+ }
+
+ /*
+ * the cpu interface stays off with the priority mask at
+ * the lowest priority, the kernel raises it when it
+ * brings its own irq handling up. off is the defined
+ * state, the enable is the kernel's decision to make.
+ */
+ writel(0, REG32(gicc + GICC_PMR));
+
+ return 0;
+}
diff --git a/arch/arm64/lib/mmu.c b/arch/arm64/lib/mmu.c
new file mode 100644
index 0000000..03eb355
--- /dev/null
+++ b/arch/arm64/lib/mmu.c
@@ -0,0 +1,205 @@
+/* SPDX-License-Identifier: GPL-2.0+ */
+/*
+ * mmu.c - VMSAv8-64 stage 1 identity map for EL2.
+ *
+ * One level 0 table plus the subtables for the low 1GB of MMIO and
+ * the RAM region. everything is identity mapped, the bootloader
+ * never needs a different VA view, it just needs caching rules that
+ * let the payload start from an architecture-defined state.
+ *
+ * The descriptor layouts are from the manual (DDI 0487), level 0/1/2
+ * and level 3 formats at D5-2444 and D5-2447, attribute fields at
+ * D5-2451, MAIR at D5-2476. feature bits come from the ID registers,
+ * never hardcoded, the PA size from ID_AA64MMFR0_EL1.PARange per
+ * "Address size configuration" D5-2399.
+ *
+ * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org>
+ */
+
+#include <asm/mmu.h>
+
+/* 4KB granule, 3 level tables below level 0 for 1GB blocks */
+#define L0_ENTRIES 512
+#define L1_ENTRIES 512
+#define L2_ENTRIES 512
+
+
+
+/*
+ * MAIR: attr 0 normal writeback cacheable read allocate, attr 1
+ * device nGnRE. encodings straight from D5-2476, B2-122 for the
+ * memory types.
+ */
+#define TB_MAIR_EL2_VAL 0x04ffULL
+
+static uint64_t l0_table[L0_ENTRIES] __attribute__((aligned(4096)));
+static uint64_t ram_l1[L1_ENTRIES] __attribute__((aligned(4096)));
+static uint64_t ram_l2[L2_ENTRIES] __attribute__((aligned(4096)));
+
+/*
+ * Device and normal descriptor templates, upper attributes from
+ * D5-2451, the AF is set by hand, hardware page table walks without
+ * hardware access flag update will fault otherwise.
+ */
+#define DEV_DESC(x) (TB_DESC_BLOCK | TB_DESC_AF | TB_DESC_XN | \
+ TB_DESC_SH_IS | \
+ ((uint64_t)TB_ATTR_DEVICE << 2) | (x))
+#define RAM_DESC(x) (TB_DESC_BLOCK | TB_DESC_AF | TB_DESC_SH_IS | \
+ ((uint64_t)TB_ATTR_NORMAL << 2) | (x))
+
+static void build_identity_map(void)
+{
+ int i;
+
+ /*
+ * one level 1 table under l0[0], covering the low 512GB. the
+ * MMIO hole and RAM are both in it, device block at index 0
+ * (0..1GB) and the RAM table at index 1 (1GB..2GB).
+ */
+ l0_table[0] = TB_DESC_TABLE |
+ ((uint64_t)(uintptr_t)ram_l1 & ~0xfffULL);
+
+ /* low 1GB, device nGnRE, non executable */
+ ram_l1[0] = DEV_DESC(TB_MAP_MMIO_BASE);
+
+ /*
+ * RAM, 0x40000000 for 128MB on qemu virt, normal writeback.
+ * the level 2 table splits the 1GB into 2MB blocks so the map
+ * can be carved later.
+ */
+ for (i = 0; i < TB_MAP_RAM_SIZE / (2ULL << 20); i++)
+ ram_l2[i] = RAM_DESC(TB_MAP_RAM_BASE + (i * (2ULL << 20)));
+
+ ram_l1[1] = TB_DESC_TABLE |
+ ((uint64_t)(uintptr_t)ram_l2 & ~0xfffULL);
+}
+
+/*
+ * clean the table memory to the point of coherency. the tables were
+ * written with the dcache off, the page table walker reads them as
+ * memory the TCR walk attributes describe, and a dirty line sitting
+ * in the cache would never reach RAM. dc cvac is by cache line, walk
+ * every page of table memory.
+ */
+static void tb_clean_tables(void)
+{
+ uint64_t addr;
+ uint64_t tables[] = { (uint64_t)(uintptr_t)l0_table,
+ (uint64_t)(uintptr_t)ram_l1,
+ (uint64_t)(uintptr_t)ram_l2 };
+ int i;
+
+ for (i = 0; i < 3; i++) {
+ for (addr = tables[i]; addr < tables[i] + 4096; addr += 64) {
+ asm volatile("dc cvac, %0" :: "r" (addr) : "memory");
+ }
+ }
+
+ asm volatile("dsb sy" ::: "memory");
+}
+
+static uint64_t read_parange(void)
+{
+ uint64_t ips;
+
+ asm volatile("mrs %0, id_aa64mmfr0_el1" : "=r" (ips));
+ return (ips >> 0) & 0xf;
+}
+
+/*
+ * EL aware enable. the EL1&0 regime registers at EL1, the EL2 regime
+ * registers at EL2, one code path per the manual, one translation
+ * regime per exception level (D1-2146).
+ */
+int tb_mmu_enable(void)
+{
+ uint64_t tcr, mair;
+ uint64_t el;
+
+ build_identity_map();
+ tb_clean_tables();
+
+ asm volatile("mrs %0, CurrentEL" : "=r" (el));
+ el >>= 2;
+
+ /* tcr value and PA size, D5-2399 address size configuration */
+ tcr = TB_TCR_T0SZ_48 | TB_TCR_SH0_IS | TB_TCR_TG0_4K |
+ TB_TCR_IRGN0_WB | TB_TCR_ORGN0_WB | TB_TCR_IPS(read_parange());
+ mair = TB_MAIR_EL2_VAL;
+
+ if (el == 2) {
+ asm volatile(
+ "dsb sy\n"
+ "msr ttbr0_el2, %1\n"
+ "msr tcr_el2, %2\n"
+ "msr mair_el2, %3\n"
+ "isb\n"
+ "tlbi alle2\n"
+ "dsb sy\n"
+ "ic iallu\n"
+ "dsb sy\n"
+ "isb\n"
+ : "=r" (tcr)
+ : "r" (l0_table), "r" (tcr), "r" (mair)
+ : "memory");
+ asm volatile(
+ "mrs x0, sctlr_el2\n"
+ "orr x0, x0, #1\n"
+ "msr sctlr_el2, x0\n"
+ "isb\n"
+ ::: "x0", "memory");
+ } else {
+ asm volatile(
+ "dsb sy\n"
+ "msr ttbr0_el1, %1\n"
+ "msr tcr_el1, %2\n"
+ "msr mair_el1, %3\n"
+ "isb\n"
+ "tlbi vmalle1\n"
+ "dsb sy\n"
+ "ic iallu\n"
+ "dsb sy\n"
+ "isb\n"
+ : "=r" (tcr)
+ : "r" (l0_table), "r" (tcr), "r" (mair)
+ : "memory");
+ asm volatile(
+ "mrs x0, sctlr_el1\n"
+ "orr x0, x0, #1\n"
+ "msr sctlr_el1, x0\n"
+ "isb\n"
+ ::: "x0", "memory");
+ }
+
+ return 0;
+}
+
+void tb_mmu_disable(void)
+{
+ uint64_t el;
+
+ asm volatile("mrs %0, CurrentEL" : "=r" (el));
+ el >>= 2;
+
+ if (el == 2) {
+ asm volatile(
+ "mrs x0, sctlr_el2\n"
+ "bic x0, x0, #1\n"
+ "msr sctlr_el2, x0\n"
+ "dsb sy\n"
+ "tlbi alle2\n"
+ "dsb sy\n"
+ "isb\n"
+ ::: "x0", "memory");
+ } else {
+ asm volatile(
+ "mrs x0, sctlr_el1\n"
+ "bic x0, x0, #1\n"
+ "msr sctlr_el1, x0\n"
+ "dsb sy\n"
+ "tlbi vmalle1\n"
+ "dsb sy\n"
+ "isb\n"
+ ::: "x0", "memory");
+ }
+}
diff --git a/arch/arm64/lib/psci.c b/arch/arm64/lib/psci.c
new file mode 100644
index 0000000..ad5f461
--- /dev/null
+++ b/arch/arm64/lib/psci.c
@@ -0,0 +1,118 @@
+/* SPDX-License-Identifier: GPL-2.0+ */
+/*
+ * psci.c - PSCI 0.2 at EL2, the calls a payload makes to control
+ * cores and the system, per DEN 0022. the call arrives as an HVC
+ * trap at EL2 (EC 0x16), x0 holds the function id, x1 to x3 the
+ * arguments, the return value goes back in x0 and eret resumes the
+ * caller at EL1.
+ *
+ * CPU_ON writes the spin gate of the target core and SEVs, the pen
+ * from start.S does the release. CPU_OFF parks the calling core.
+ * SYSTEM_OFF and SYSTEM_RESET drive the architecture reset domain.
+ *
+ * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org>
+ */
+
+#include <asm/psci.h>
+#include <debug.h>
+
+extern void tb_system_reset(void);
+extern void tb_system_off(void);
+
+/* the gates and stamps from start.S, one per possible core */
+extern unsigned long tb_spin_gates[8];
+extern unsigned char tb_pen_stamps[8];
+
+static uint64_t psci_cpu_on(uint64_t target, uint64_t entry,
+ uint64_t ctx)
+{
+ unsigned long mpidr;
+ int cpu;
+
+ /* affinity 0 only, our gate array indexes cores 0..7 */
+ if (target > 7)
+ return PSCI_RET_INVALID_PARAMS;
+
+ asm volatile("mrs %0, mpidr_el1" : "=r" (mpidr));
+ if ((mpidr & 0xff) == target)
+ return PSCI_RET_ALREADY_ON;
+
+ cpu = (int)target;
+
+ /*
+ * the pen saves no context, CPU_ON per DEN 0022 passes an
+ * entry and a context id. the pen enters with x0 = ctx, the
+ * kernel secondary entry takes x0 as its context pointer.
+ * the gate holds the entry, the stamp array the ctx.
+ */
+ tb_spin_gates[cpu] = entry;
+ tb_pen_stamps[cpu] = (unsigned char)(ctx & 0xff);
+
+ /* make the gate write visible before the wake, D1-2255 */
+ asm volatile("dsb sy");
+ asm volatile("sev");
+
+ dprintf(ALWAYS, "psci: cpu_on %llu -> %llx\n",
+ (unsigned long long)target,
+ (unsigned long long)entry);
+
+ return PSCI_RET_SUCCESS;
+}
+
+extern void park_ret(void);
+
+static uint64_t psci_cpu_off(void)
+{
+ unsigned long mpidr;
+ int cpu;
+
+ asm volatile("mrs %0, mpidr_el1" : "=r" (mpidr));
+ cpu = (int)(mpidr & 0xff);
+
+ if (cpu > 7)
+ return PSCI_RET_NOT_SUPPORTED;
+
+ /* clear our own gate and go back to the pen */
+ tb_spin_gates[cpu] = 0;
+
+ dprintf(ALWAYS, "psci: cpu_off %d\n", cpu);
+
+ asm volatile(
+ "dsb sy\n"
+ "b park_ret\n"
+ );
+
+ return PSCI_RET_INTERNAL_FAIL; /* not reached */
+}
+
+/*
+ * the asm vector calls this with the caller x0-x3 still in place,
+ * function id in x0, arguments in x1-x3, the return lands in x0.
+ */
+uint64_t tb_psci_dispatch(uint64_t fn, uint64_t x1, uint64_t x2,
+ uint64_t x3)
+{
+ switch (fn) {
+ case PSCI_FN_VERSION:
+ return PSCI_VERSION_0_2;
+
+ case PSCI_FN_CPU_ON:
+ return psci_cpu_on(x1, x2, x3);
+
+ case PSCI_FN_CPU_OFF:
+ return psci_cpu_off();
+
+ case PSCI_FN_SYSTEM_OFF:
+ dprintf(ALWAYS, "psci: system off\n");
+ tb_system_off();
+ return PSCI_RET_SUCCESS;
+
+ case PSCI_FN_SYSTEM_RESET:
+ dprintf(ALWAYS, "psci: system reset\n");
+ tb_system_reset();
+ return PSCI_RET_INTERNAL_FAIL; /* not reached */
+
+ default:
+ return PSCI_RET_NOT_SUPPORTED;
+ }
+}
diff --git a/arch/arm64/lib/semihosting.S b/arch/arm64/lib/semihosting.S
new file mode 100644
index 0000000..6e3fc31
--- /dev/null
+++ b/arch/arm64/lib/semihosting.S
@@ -0,0 +1,18 @@
+/* SPDX-License-Identifier: GPL-2.0+ */
+/*
+ * semihosting.S - the trap instruction itself. qemu answers this when
+ * it is started with -semihosting, and nothing happens without it, so
+ * every caller has to cope with the no-debugger case.
+ *
+ * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org>
+ */
+
+#include <asm/linkage.h>
+
+.pushsection .text.smh_trap, "ax"
+/* long smh_trap(unsigned int sysnum, void *addr); */
+ENTRY(smh_trap)
+ hlt #0xf000
+ ret
+ENDPROC(smh_trap)
+.popsection
diff --git a/arch/arm64/lib/system.c b/arch/arm64/lib/system.c
new file mode 100644
index 0000000..25af2bf
--- /dev/null
+++ b/arch/arm64/lib/system.c
@@ -0,0 +1,70 @@
+/* SPDX-License-Identifier: GPL-2.0+ */
+/*
+ * system.c - system power control, the PSCI SYSTEM_OFF and
+ * SYSTEM_RESET backends. off parks the core in WFI forever, the
+ * manual's low power entry (D1-2255). reset drives the PE reset
+ * domain: RMR_EL2 reset request with the system reset bit, RR bit 1,
+ * followed by a barrier pair so the request retires before anything
+ * else observes the core.
+ *
+ * On real hardware a SoC also needs a watchdog or PMIC write for a
+ * full board reset, that is board territory, the arch part is this.
+ *
+ * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org>
+ */
+
+#include <debug.h>
+#include <stdint.h>
+
+void tb_system_off(void)
+{
+ dprintf(ALWAYS, "system off\n");
+
+ for (;;) {
+ asm volatile("wfi");
+ }
+}
+
+void tb_system_reset(void)
+{
+ uint64_t rmr;
+ uint64_t el;
+ register uint64_t r0 asm("x0") = 0x84000008;
+ register uint64_t r1 asm("x1") = 0;
+ register uint64_t r2 asm("x2") = 0;
+ register uint64_t r3 asm("x3") = 0;
+
+ dprintf(ALWAYS, "system reset\n");
+
+ /*
+ * the firmware conduit first, PSCI SYSTEM_RESET through
+ * the machine's own monitor. this is the only legal way
+ * up from EL1: RMR_EL1 is undefined on hardware that
+ * implements a higher exception level, the access traps
+ * and the machine never resets.
+ */
+ asm volatile("smc #0"
+ : "+r"(r0), "+r"(r1), "+r"(r2), "+r"(r3));
+
+ /*
+ * no monitor answered, or it refused. ask the reset
+ * domain directly from a level that may write it.
+ */
+ asm volatile("mrs %0, CurrentEL" : "=r" (el));
+ el >>= 2;
+
+ if (el >= 2) {
+ asm volatile("mrs %0, rmr_el2" : "=r" (rmr));
+ rmr |= (1 << 1); /* RR, request reset */
+ asm volatile(
+ "msr rmr_el2, %0\n"
+ "dsb sy\n"
+ "isb\n"
+ :: "r" (rmr));
+ }
+
+ /* nothing worked, park */
+ for (;;) {
+ asm volatile("wfi");
+ }
+}
diff --git a/arch/arm64/lib/timer.c b/arch/arm64/lib/timer.c
new file mode 100644
index 0000000..605cd18
--- /dev/null
+++ b/arch/arm64/lib/timer.c
@@ -0,0 +1,48 @@
+/* SPDX-License-Identifier: GPL-2.0+ */
+/*
+ * timer.c - generic timer delays, the system counter from D10. the
+ * counter is a fixed frequency free running counter, CNTFRQ_EL0
+ * carries the frequency, CNTVCT_EL0 the 64 bit count. delays are a
+ * busy wait on the counter, no interrupts needed, microsecond and
+ * millisecond granularity.
+ *
+ * CNTVCT_EL0 is the virtual counter view; at EL2 with no offset
+ * configured it is the physical count. the read is not speculative
+ * and needs an isb to serialize against subsequent counter reads
+ * per the counter access rules.
+ *
+ * Copyright (C) 2026 Bradley Morgan <brads@mainlining.org>
+ */
+
+#include <stdint.h>
+
+static uint64_t read_cntfrq(void)
+{
+ uint64_t v;
+
+ asm volatile("mrs %0, cntfrq_el0" : "=r" (v));
+ return v;
+}
+
+static uint64_t read_counter(void)
+{
+ uint64_t v;
+
+ asm volatile("isb\nmrs %0, cntvct_el0" : "=r" (v));
+ return v;
+}
+
+void tb_udelay(uint32_t us)
+{
+ uint64_t freq = read_cntfrq();
+ uint64_t start = read_counter();
+ uint64_t ticks = (uint64_t)us * freq / 1000000ULL;
+
+ while (read_counter() - start < ticks)
+ ;
+}
+
+void tb_mdelay(uint32_t ms)
+{
+ tb_udelay(ms * 1000);
+}