//! ESP32-P4 peripherals, modelled at comptime. //! //! There is no HAL here and no generated 20k-line register header: a `Reg` is a typed pointer to //! an MMIO word, and a peripheral is a struct of them. Everything is `inline`, so `gpio.setHigh(20)` //! compiles to the single `sw` instruction it should be, and a wrong bit index is a compile error //! rather than a silent write. //! //! Addresses are from ESP-IDF v6.0.2 `components/soc/esp32p4/register/hw_ver1/soc/` - the pre-v3 //! header set, which is the one that matches this silicon (rev v1.3) - except the GPIO matrix //! signal index, which lives in `components/soc/esp32p4/include/soc/gpio_sig_map.h`. const std = @import("std"); /// A 32-bit memory-mapped register. pub fn Reg(comptime addr: usize) type { return struct { pub const address = addr; const ptr: *volatile u32 = @ptrFromInt(addr); pub inline fn read() u32 { return ptr.*; } pub inline fn write(value: u32) void { ptr.* = value; } pub inline fn set(mask: u32) void { ptr.* = ptr.* | mask; } pub inline fn clear(mask: u32) void { ptr.* = ptr.* & ~mask; } /// Read-modify-write a bitfield: `modify(.{ .shift = 12, .width = 3 }, 5)`. pub inline fn modify(comptime field: Field, value: u32) void { const mask: u32 = ((@as(u32, 1) << field.width) - 1) << field.shift; ptr.* = (ptr.* & ~mask) | ((value << field.shift) & mask); } }; } pub const Field = struct { shift: u5, width: u5 }; /// A register with a named layout: pass a packed struct whose bit width is 32 and the accessors /// become typed, so a pad is configured by naming fields instead of shifting bits. Read-modify- /// write stays explicit - `var v = reg.read(); v.mcu_sel = 1; reg.write(v);` - because that is one /// load and one store, and hiding it behind a partial-update type buys nothing here. pub fn Typed(comptime T: type, comptime addr: usize) type { comptime std.debug.assert(@bitSizeOf(T) == 32); return struct { pub const address = addr; const ptr: *volatile T = @ptrFromInt(addr); pub inline fn read() T { return ptr.*; } pub inline fn write(value: T) void { ptr.* = value; } /// Apply `f` to the current value and write the result back. pub inline fn modify(comptime f: fn (T) T) void { ptr.* = f(ptr.*); } }; } /// An array of identical registers. The index type is narrowed to the array's real range, so an /// out-of-range access is a compile error in every optimize mode - an `assert` would have been /// compiled out under ReleaseSmall, which is this project's default. pub fn RegArray(comptime base: usize, comptime stride: usize, comptime count: usize) type { return struct { pub const Index = std.math.IntFittingRange(0, count - 1); /// comptime, because `IntFittingRange` rounds up to a whole width: for a 57-entry array the /// index type is u6, which would happily accept 57..63. Every caller here passes a comptime /// pin anyway, so this costs nothing and makes the bound real in all optimize modes. pub inline fn at(comptime index: Index) *volatile u32 { comptime std.debug.assert(index < count); return @ptrFromInt(base + @as(usize, index) * stride); } }; } const hp_periph1 = 0x500C0000; /// GPIO and the IO MUX now live in the HAL, which builds them out of ESP-IDF's own register macros /// (`hal/gpio.zig`) instead of the hand-transcribed addresses that used to be here. The /// transcription is exactly the kind of thing that goes quietly wrong: this file's matrix constant /// said 256 with a comment warning that the S3's is 128, and the first hand-written replacement in /// the HAL used 128 anyway. It now comes from `SIG_GPIO_OUT_IDX` in IDF's `gpio_sig_map.h`. pub const gpio = @import("hal").gpio; /// Mask ROM routines. These are the only "library" a bare image links against: the addresses come /// from `components/esp_rom/esp32p4/ld/esp32p4.rom.ld` and the linker script re-declares them. pub const rom = struct { pub extern fn ets_printf(fmt: [*:0]const u8, ...) c_int; pub extern fn ets_delay_us(us: u32) void; /// Tell the ROM the CPU's new frequency, in MHz. `ets_delay_us` and everything else built on /// `g_ticks_per_us` busy-waits by a cycle count derived from it, so a clock change without this /// makes every ROM delay wrong by exactly the ratio. `esp32p4.rom.ld:32`, 0x4fc00044. pub extern fn ets_update_cpu_frequency(mhz: u32) void; /// Invalidate the caches. `map` selects which, from `rom/cache.h:228-236`: /// L1 ICache0 = 1, ICache1 = 2, L1 DCache = 0x10, L2 = 0x20; `cache_all` is all four. /// /// This is not optional on a large image, and finding that out took a while. The second-stage /// bootloader leaves cache lines behind that do not match the mapping it finally installs, so an /// application can read its own `.rodata` and get someone else's bytes - deterministically, /// which is what makes it look like anything other than a cache. Measured: a load at 0x40035a1c /// returned `93 85 85 0f`, and after touching 512 KiB to evict the line the same load returned /// `3c ee 08 40`, which is what the image holds there. `_start` calls this before it clears /// `.bss`, so nothing of ours is in flight when the lines are dropped. pub extern fn Cache_Invalidate_All(map: u32) c_int; /// Every cache this chip has: L1 ICache0 | L1 ICache1 | L1 DCache | L2. pub const cache_all: u32 = 0x33; pub inline fn print(comptime fmt: [*:0]const u8, args: anytype) void { _ = @call(.auto, ets_printf, .{fmt} ++ args); } }; /// Evict every flash-mapped cache line the bootloader left behind, by reading more flash than the /// caches can hold. Call it before the first byte of `.rodata` is touched. /// /// This is a workaround for a real defect in the hand-over, not a tidiness measure. The second-stage /// bootloader leaves lines cached against a mapping it then replaces, so an application reads its /// own `.rodata` and gets its own `.text` back - **deterministically**, which is exactly what makes /// it look like anything other than a cache. Measured on this die: a load at `0x40035A1C` returned /// `93 85 85 0f` before eviction and `3c ee 08 40` after, and the second is what the image holds /// there. A string literal read before this runs is machine code, so a firmware whose first act is /// to print a marker prints garbage and looks like it never booted at all. /// /// Everything cheaper was tried first and every one of them said the hardware was fine, which is /// why the list is here rather than being rediscovered: /// /// * The MMU table is correct. Entries 0..9 read `0x1001`..`0x100a` - the valid bit plus physical /// page N+1 - which is exactly what the image builder's single flash-to-vaddr anchor requires, /// and 10..11 are unmapped as they should be. Read off the die through /// `SPI_MEM_C_MMU_ITEM_INDEX_REG`, not inferred. /// * The flash is correct. `zig build flash` verifies an MD5 of what the ROM stored, and the /// image matches the ELF byte for byte at the addresses that misread. /// * The page size is not in question: hardwired to 64 KiB on this chip /// (`hal/esp32p4/mmu_ll.h:126-130` returns `MMU_PAGE_64KB` and the setter asserts it). /// * Not fragmentation of the mapping either: the bad bytes arrive in one contiguous run of /// >= 192 B, not in 64-byte lines, and they are identical across three resets and two /// reflashes - determinism is what kept this looking like anything but a cache. /// /// 512 KiB is four times the 128 KiB the L2 measured at (`examples/memprobe.zig` found real RAM /// stopping at `0x4FFA0000`, the cache taking the rest), with the L1s smaller still. /// /// `rom.Cache_Invalidate_All` is the instrument that ought to do this and does not: called from an /// image the ROM did not launch, it faults inside the ROM with its argument stranded in `a2`. /// Capacity eviction needs no preconditions, which is the whole reason it is what ships. 512 KiB at /// a 64-byte stride is 8,192 loads, once per boot. pub fn flushFlashCache() void { var sink: u32 = 0; var p: u32 = 0x4000_0000; while (p < 0x4008_0000) : (p += 64) { sink +%= @as(*volatile u32, @ptrFromInt(p)).*; } // Consumed through a volatile store, or the optimiser drops the whole loop as dead. @as(*volatile u32, &flush_sink).* = sink; } var flush_sink: u32 = 0; /// Busy-wait for a number of CPU cycles, using the cycle counter rather than the mask ROM. Useful /// when an image must not depend on ROM entry points at all, and for delays shorter than the ROM's /// microsecond granularity. pub inline fn delayCycles(n: u64) void { const start = cycles(); while (cycles() - start < n) {} } /// Cycle counter: CSR 0xC00/0xC80, i.e. `cycle`/`cycleh` - the unprivileged shadows of mcycle, and /// what ESP-IDF itself reads on this part (`rv_utils.h`: `RV_READ_CSR(cycle)`, because /// SOC_CPU_HAS_CSR_PC is not defined for the P4). /// /// Read high-low-high: two separate CSR reads can straddle a wrap of the low word, which would /// otherwise report a value 2^32 too large roughly every 47 seconds at 90 MHz. pub inline fn cycles() u64 { while (true) { var hi0: u32 = undefined; var lo: u32 = undefined; var hi1: u32 = undefined; asm volatile ("csrr %[r], 0xC80" : [r] "=r" (hi0), ); asm volatile ("csrr %[r], 0xC00" : [r] "=r" (lo), ); asm volatile ("csrr %[r], 0xC80" : [r] "=r" (hi1), ); if (hi0 == hi1) return (@as(u64, hi0) << 32) | lo; } }