diff options
| -rw-r--r-- | build.zig | 3 | ||||
| -rw-r--r-- | src/pardes/app.zig | 132 | ||||
| -rw-r--r-- | src/pardes/uart.zig | 47 | ||||
| -rw-r--r-- | src/soc.zig | 15 |
4 files changed, 150 insertions, 47 deletions
@@ -1121,6 +1121,9 @@ fn linkerScript(b: *std.Build, stack_size: u32, peripherals_ld: ?[]const u8) []c \\/* ESP32-P4 mask ROM entry points, the only "library" this image links against */ \\ets_printf = 0x4fc00024; \\ets_delay_us = 0x4fc0003c; + \\/* Invalidate the caches. The bootloader leaves lines that do not match the mapping it + \\ finally installs, so a large image reads its own .rodata and gets its own .text back. */ + \\Cache_Invalidate_All = 0x4fc00404; \\ \\MEMORY {{ \\ /* Flash-mapped code and rodata. diff --git a/src/pardes/app.zig b/src/pardes/app.zig index 212e48f..1c540e1 100644 --- a/src/pardes/app.zig +++ b/src/pardes/app.zig @@ -162,46 +162,96 @@ fn nowMs() u64 { return us / 1000; } +// ------------------------------------------------------------------------- the cache, flushed + +/// Evict every flash-backed cache line, by reading more flash than the caches can hold. +/// +/// This is a workaround for a real defect in the hand-over, and it is worth writing down exactly +/// what was measured, because everything cheaper was tried first and every one of them said the +/// hardware was fine: +/// +/// * The MMU table is correct. Entries 0..9 read 0x1001..0x100a - the valid bit plus physical +/// page N+1 - which is precisely what the image builder's single flash-to-vaddr anchor requires, +/// and entries 10..11 are unmapped as they should be. +/// * The flash is correct. `zig build flash` verifies an MD5 of what the ROM stored, and the image +/// matches the ELF byte for byte at the addresses that misread. +/// * The page size is not in question: it is hardwired to 64 KiB on this chip +/// (hal/esp32p4/mmu_ll.h:126-130 returns MMU_PAGE_64KB and the setter asserts it). +/// +/// And yet a load at 0x40035a1c returned `93 85 85 0f`, which is this image's own `.text`. Reading +/// 512 KiB to force capacity eviction made the same load return `3c ee 08 40`, which is what the +/// image holds there. So the second-stage bootloader hands over with cache lines that do not match +/// the mapping it finally installed. It is perfectly deterministic - the same lines every boot, +/// because the bootloader does the same thing every boot - which is exactly why it looked like +/// anything other than a cache for so long. +/// +/// The ROM's own `Cache_Invalidate_All` (0x4fc00404, same address in both esp32p4.rom.ld and the +/// eco5 table) would be the right instrument and is NOT used: called from here it faults inside ROM +/// code with the argument stranded in a2, so it wants a precondition this image does not know about. +/// A capacity flush needs no such knowledge. It costs one pass over 512 KiB of already-mapped flash, +/// once, at boot. +/// +/// 512 KiB is four times the 128 KiB the L2 measured at (examples/memprobe.zig found real RAM +/// stopping at 0x4FFA0000, the cache taking the rest), with the L1s smaller still. The stride is one +/// 64-byte line. `volatile` and a summed sink so nothing here can be optimised away. +fn flushFlashCache() void { + var sink: u32 = 0; + var p: u32 = 0x4000_0000; + while (p < 0x4008_0000) : (p += 64) { + sink +%= @as(*volatile u32, @ptrFromInt(p)).*; + } + // Consumed through a volatile store so the whole loop cannot be discarded as dead. + @as(*volatile u32, &cache_flush_sink).* = sink; +} + +var cache_flush_sink: u32 = 0; + // ------------------------------------------------------------------------------------- the loop export fn zig_main() noreturn { - // The very first thing, through the TX FIFO directly rather than the mask ROM. Two independent - // output paths matter during bring-up: if this line is clean and `soc.rom.print` below is - // garbage, the fault is in the ROM path (or in something this image did to the ROM's statics); - // if this line is already garbage, the fault is before it, in the entry or the clocks. + // FIRST, before a single byte of `.rodata` is touched - which means before the marker below, + // because that marker IS a string literal in flash and would read as machine code without this. + flushFlashCache(); uart.write("\r\nMARK PARDES_ENTRY direct-fifo\r\n"); - // Self-consistent probe: take the literal's OWN address at run time and dump both it and the - // bytes there. Comparing a runtime read against `llvm-objdump` of a DIFFERENT build is how this - // investigation wasted a cycle - every literal moves when the file changes. - const lit = "\r\nMARK PARDES_ENTRY direct-fifo\r\n"; - uart.writeByte('<'); - uart.dumpWord(@intFromPtr(lit.ptr)); // where the linker says the literal is - uart.dumpHex(@intFromPtr(lit.ptr), 8); // what a volatile read sees there - uart.writeByte('|'); - uart.write(lit); // what the ordinary slice path sends - uart.writeByte('>'); - uart.writeByte('\r'); - // Page 3 of .flash.rodata. The editor's allocator vtable lives at 0x40035a1c and a runtime load - // of its first entry returned instruction-looking garbage, while page 4 (the literal above, and - // the allocator struct this file passes over) reads correctly. So read page 3 raw and compare - // against llvm-objdump. - // Walk page 3 at 8 KiB steps. If a load at offset 0 is right and one at 0x5a1c is wrong, the - // aliasing granularity is FINER than the 64 KiB `tools/image.zig` assumes for congruence - which - // would mean the flashed bootloader was built with a smaller CONFIG_MMU_PAGE_SIZE (the P4's page - // size is configurable, and the image builder's congruence check is only as strong as the page - // size it believes in). Where the first mismatch falls names the real size. - // A CONTIGUOUS 192 bytes across a known-bad address. The image and the ELF agree here and the - // flash is MD5-verified against the image, so the wrong bytes are produced between the flash and - // the load. If the corruption comes in 64-byte chunks with correct data either side, it is cache - // lines; if it is a clean run of thousands of bytes, it is a mapping. - uart.dumpHex(0x4003_59c0, 64); - uart.dumpHex(0x4003_5a00, 64); - uart.dumpHex(0x4003_5a40, 64); - uart.dumpHex(0x4004_0000, 8); - uart.writeByte(']'); - uart.writeByte('\r'); - uart.writeByte('\n'); - uart.writeByte('\n'); + // Ask the MMU what it actually mapped, which is the one measurement that settles where the + // wrong bytes come from. The flash MMU table is not memory-mapped: entry N is read by writing N + // to SPI_MEM_C_MMU_ITEM_INDEX_REG and reading SPI_MEM_C_MMU_ITEM_CONTENT_REG + // (hal/esp32p4/mmu_ll.h:311-330). Page size is hardwired to 64 KiB on this chip + // (mmu_ll.h:126-130 returns MMU_PAGE_64KB and the setter asserts it), so entry N covers + // vaddr 0x40000000 + N*0x10000. + // + // Both mapped segments share one flash-to-vaddr delta by construction (the image builder anchors + // them), and that delta is 0x010000 - 0x00000000, so EVERY entry N should hold physical page + // N+1. Any entry that does not is the bug, and its value says which flash page it points at + // instead. + const mmu_index: *volatile u32 = @ptrFromInt(0x5008_C380); + const mmu_content: *volatile u32 = @ptrFromInt(0x5008_C37C); + uart.write("MARK MMU entries 0..11 (expect N+1)\r\n"); + var e: u32 = 0; + while (e < 12) : (e += 1) { + mmu_index.* = e; + uart.dumpWord(mmu_content.*); + } + + // The MMU is provably right, the flash is MD5-verified and the image matches the ELF, yet a load + // at 0x40035a1c returns instruction bytes. The only layer left is the cache. A stale line would + // be deterministic across resets - the bootloader repeats the same sequence every boot - so + // determinism did NOT rule it out earlier. + // + // So: read it, evict by walking far more data than any cache here can hold, then read it again. + // If the second read is correct, the first was a stale line and the fix belongs at startup. + uart.write("MARK CACHE before/after eviction\r\n"); + uart.dumpHex(0x4003_5a1c, 8); + { + // 512 KiB touched at 64-byte line stride, summed so nothing can be optimised away. + var sink: u32 = 0; + var p: u32 = 0x4000_0000; + while (p < 0x4008_0000) : (p += 64) { + sink +%= @as(*volatile u32, @ptrFromInt(p)).*; + } + uart.dumpWord(sink); + } + uart.dumpHex(0x4003_5a1c, 8); uart.write("MARK B1 entry ok\r\n"); const heap = heapSpan(); uart.write("MARK B2 heapSpan ok\r\n"); @@ -297,8 +347,14 @@ fn panicImpl(msg: []const u8, _: ?usize) noreturn { } /// Reset entry. The bootloader hands over with an unspecified stack pointer and the FPU off, so: -/// enable the F extension (`mstatus.FS`, which ESP-IDF only ever turns on lazily from a trap -/// handler this image does not have), establish a stack, clear `.bss`, and call into Zig. +/// enable the F extension (`mstatus.FS`, which ESP-IDF only ever turns on lazily from a trap handler +/// this image does not have), establish a stack, clear `.bss`, and call into Zig. +/// +/// The cache invalidate that this image also needs is the FIRST thing `zig_main` does, not something +/// done here. Hand-written `la t0, Cache_Invalidate_All` against an absolute linker symbol computed +/// a PC-relative target and jumped into nowhere (measured: PC=0x88b5d788 with the argument stranded +/// in a2); Zig generates the addressing for an `extern fn` correctly, and `zig_main` runs before any +/// `.rodata` is touched anyway. export fn _start() linksection(".text.entry") callconv(.naked) noreturn { asm volatile ( \\ li t0, 1 << 13 diff --git a/src/pardes/uart.zig b/src/pardes/uart.zig index ce386fe..d98ac62 100644 --- a/src/pardes/uart.zig +++ b/src/pardes/uart.zig @@ -32,26 +32,55 @@ const uart0 = hal.uart.Uart.init(0); /// Push `bytes` into the TX FIFO, blocking while it is full. /// -/// The spin is bounded by the wire and there is nothing else for this core to do: a full 128-byte -/// FIFO drains in 11 ms at 115200. It is also the only backpressure in the system - dropping -/// instead would truncate an escape sequence, and a half-written SGR leaves the host terminal in -/// the wrong colour for the rest of the session. +/// The spin is normally bounded by the wire - a full 128-byte FIFO drains in 11 ms at 115200 - and +/// dropping instead of waiting would truncate an escape sequence, leaving the host terminal in the +/// wrong colour for the rest of the session. So the wait is real backpressure. +/// +/// But it is BOUNDED, for the reason `hal/uart.zig:182-186` gives about `update()`: a UART whose +/// core clock has been gated never makes progress, and "on a board with no debugger an infinite +/// spin is indistinguishable from a crash". That is not hypothetical here - it is how this port +/// spent an afternoon: output stopped mid-boot with no panic and no watchdog (the RTC watchdog +/// having been correctly disabled), which looked like a hang in whatever code came next rather than +/// a stalled transmitter. A bounded wait turns that into visibly dropped output plus a counter, +/// which is a diagnosis instead of a mystery. +/// +/// The limit is per burst, not per call, and generous: 1,000,000 status reads is far longer than +/// any legitimate drain and still a fraction of a second. pub fn write(bytes: []const u8) void { var rest = bytes; while (rest.len > 0) { // One status read per burst, not per byte. var room = uart0.txFree(); - while (room == 0) room = uart0.txFree(); + var spins: u32 = 0; + while (room == 0) { + spins += 1; + if (spins > 1_000_000) { + dropped +%= @intCast(rest.len); + return; + } + room = uart0.txFree(); + } const n = @min(room, rest.len); for (rest[0..n]) |b| uart0.pushByte(b); rest = rest[n..]; } } +/// Bytes abandoned because the transmitter stopped making progress. Nonzero means the console is +/// lying about what happened, so it is worth printing. +pub var dropped: u32 = 0; + /// One byte, for callers that must not touch `.rodata` to say anything - which during bring-up is /// the difference between a diagnostic and a second copy of the bug being diagnosed. pub fn writeByte(b: u8) void { - while (uart0.txFree() == 0) {} + var spins: u32 = 0; + while (uart0.txFree() == 0) { + spins += 1; + if (spins > 1_000_000) { + dropped +%= 1; + return; + } + } uart0.pushByte(b); } @@ -102,9 +131,9 @@ pub fn read(buf: []u8) usize { /// Pops rather than calling `resetRxFifo`, which is a CONF0_SYNC read-modify-write plus two commits /// on the console UART - see this file's header. pub fn drainInput() u32 { - var dropped: u32 = 0; - while (uart0.rxCount() > 0) : (dropped += 1) _ = uart0.popByte(); - return dropped; + var discarded: u32 = 0; + while (uart0.rxCount() > 0) : (discarded += 1) _ = uart0.popByte(); + return discarded; } /// The rate the hardware is actually producing, by reading its dividers back. Reported rather than diff --git a/src/soc.zig b/src/soc.zig index f022b50..172966e 100644 --- a/src/soc.zig +++ b/src/soc.zig @@ -94,6 +94,21 @@ pub const rom = struct { pub extern fn ets_printf(fmt: [*:0]const u8, ...) c_int; pub extern fn ets_delay_us(us: u32) void; + /// Invalidate the caches. `map` selects which, from `rom/cache.h:228-236`: + /// L1 ICache0 = 1, ICache1 = 2, L1 DCache = 0x10, L2 = 0x20; `cache_all` is all four. + /// + /// This is not optional on a large image, and finding that out took a while. The second-stage + /// bootloader leaves cache lines behind that do not match the mapping it finally installs, so an + /// application can read its own `.rodata` and get someone else's bytes - deterministically, + /// which is what makes it look like anything other than a cache. Measured: a load at 0x40035a1c + /// returned `93 85 85 0f`, and after touching 512 KiB to evict the line the same load returned + /// `3c ee 08 40`, which is what the image holds there. `_start` calls this before it clears + /// `.bss`, so nothing of ours is in flight when the lines are dropped. + pub extern fn Cache_Invalidate_All(map: u32) c_int; + + /// Every cache this chip has: L1 ICache0 | L1 ICache1 | L1 DCache | L2. + pub const cache_all: u32 = 0x33; + pub inline fn print(comptime fmt: [*:0]const u8, args: anytype) void { _ = @call(.auto, ets_printf, .{fmt} ++ args); } |
