diff options
Diffstat (limited to 'src/pardes/app.zig')
| -rw-r--r-- | src/pardes/app.zig | 132 |
1 files changed, 94 insertions, 38 deletions
diff --git a/src/pardes/app.zig b/src/pardes/app.zig index 212e48f..1c540e1 100644 --- a/src/pardes/app.zig +++ b/src/pardes/app.zig @@ -162,46 +162,96 @@ fn nowMs() u64 { return us / 1000; } +// ------------------------------------------------------------------------- the cache, flushed + +/// Evict every flash-backed cache line, by reading more flash than the caches can hold. +/// +/// This is a workaround for a real defect in the hand-over, and it is worth writing down exactly +/// what was measured, because everything cheaper was tried first and every one of them said the +/// hardware was fine: +/// +/// * The MMU table is correct. Entries 0..9 read 0x1001..0x100a - the valid bit plus physical +/// page N+1 - which is precisely what the image builder's single flash-to-vaddr anchor requires, +/// and entries 10..11 are unmapped as they should be. +/// * The flash is correct. `zig build flash` verifies an MD5 of what the ROM stored, and the image +/// matches the ELF byte for byte at the addresses that misread. +/// * The page size is not in question: it is hardwired to 64 KiB on this chip +/// (hal/esp32p4/mmu_ll.h:126-130 returns MMU_PAGE_64KB and the setter asserts it). +/// +/// And yet a load at 0x40035a1c returned `93 85 85 0f`, which is this image's own `.text`. Reading +/// 512 KiB to force capacity eviction made the same load return `3c ee 08 40`, which is what the +/// image holds there. So the second-stage bootloader hands over with cache lines that do not match +/// the mapping it finally installed. It is perfectly deterministic - the same lines every boot, +/// because the bootloader does the same thing every boot - which is exactly why it looked like +/// anything other than a cache for so long. +/// +/// The ROM's own `Cache_Invalidate_All` (0x4fc00404, same address in both esp32p4.rom.ld and the +/// eco5 table) would be the right instrument and is NOT used: called from here it faults inside ROM +/// code with the argument stranded in a2, so it wants a precondition this image does not know about. +/// A capacity flush needs no such knowledge. It costs one pass over 512 KiB of already-mapped flash, +/// once, at boot. +/// +/// 512 KiB is four times the 128 KiB the L2 measured at (examples/memprobe.zig found real RAM +/// stopping at 0x4FFA0000, the cache taking the rest), with the L1s smaller still. The stride is one +/// 64-byte line. `volatile` and a summed sink so nothing here can be optimised away. +fn flushFlashCache() void { + var sink: u32 = 0; + var p: u32 = 0x4000_0000; + while (p < 0x4008_0000) : (p += 64) { + sink +%= @as(*volatile u32, @ptrFromInt(p)).*; + } + // Consumed through a volatile store so the whole loop cannot be discarded as dead. + @as(*volatile u32, &cache_flush_sink).* = sink; +} + +var cache_flush_sink: u32 = 0; + // ------------------------------------------------------------------------------------- the loop export fn zig_main() noreturn { - // The very first thing, through the TX FIFO directly rather than the mask ROM. Two independent - // output paths matter during bring-up: if this line is clean and `soc.rom.print` below is - // garbage, the fault is in the ROM path (or in something this image did to the ROM's statics); - // if this line is already garbage, the fault is before it, in the entry or the clocks. + // FIRST, before a single byte of `.rodata` is touched - which means before the marker below, + // because that marker IS a string literal in flash and would read as machine code without this. + flushFlashCache(); uart.write("\r\nMARK PARDES_ENTRY direct-fifo\r\n"); - // Self-consistent probe: take the literal's OWN address at run time and dump both it and the - // bytes there. Comparing a runtime read against `llvm-objdump` of a DIFFERENT build is how this - // investigation wasted a cycle - every literal moves when the file changes. - const lit = "\r\nMARK PARDES_ENTRY direct-fifo\r\n"; - uart.writeByte('<'); - uart.dumpWord(@intFromPtr(lit.ptr)); // where the linker says the literal is - uart.dumpHex(@intFromPtr(lit.ptr), 8); // what a volatile read sees there - uart.writeByte('|'); - uart.write(lit); // what the ordinary slice path sends - uart.writeByte('>'); - uart.writeByte('\r'); - // Page 3 of .flash.rodata. The editor's allocator vtable lives at 0x40035a1c and a runtime load - // of its first entry returned instruction-looking garbage, while page 4 (the literal above, and - // the allocator struct this file passes over) reads correctly. So read page 3 raw and compare - // against llvm-objdump. - // Walk page 3 at 8 KiB steps. If a load at offset 0 is right and one at 0x5a1c is wrong, the - // aliasing granularity is FINER than the 64 KiB `tools/image.zig` assumes for congruence - which - // would mean the flashed bootloader was built with a smaller CONFIG_MMU_PAGE_SIZE (the P4's page - // size is configurable, and the image builder's congruence check is only as strong as the page - // size it believes in). Where the first mismatch falls names the real size. - // A CONTIGUOUS 192 bytes across a known-bad address. The image and the ELF agree here and the - // flash is MD5-verified against the image, so the wrong bytes are produced between the flash and - // the load. If the corruption comes in 64-byte chunks with correct data either side, it is cache - // lines; if it is a clean run of thousands of bytes, it is a mapping. - uart.dumpHex(0x4003_59c0, 64); - uart.dumpHex(0x4003_5a00, 64); - uart.dumpHex(0x4003_5a40, 64); - uart.dumpHex(0x4004_0000, 8); - uart.writeByte(']'); - uart.writeByte('\r'); - uart.writeByte('\n'); - uart.writeByte('\n'); + // Ask the MMU what it actually mapped, which is the one measurement that settles where the + // wrong bytes come from. The flash MMU table is not memory-mapped: entry N is read by writing N + // to SPI_MEM_C_MMU_ITEM_INDEX_REG and reading SPI_MEM_C_MMU_ITEM_CONTENT_REG + // (hal/esp32p4/mmu_ll.h:311-330). Page size is hardwired to 64 KiB on this chip + // (mmu_ll.h:126-130 returns MMU_PAGE_64KB and the setter asserts it), so entry N covers + // vaddr 0x40000000 + N*0x10000. + // + // Both mapped segments share one flash-to-vaddr delta by construction (the image builder anchors + // them), and that delta is 0x010000 - 0x00000000, so EVERY entry N should hold physical page + // N+1. Any entry that does not is the bug, and its value says which flash page it points at + // instead. + const mmu_index: *volatile u32 = @ptrFromInt(0x5008_C380); + const mmu_content: *volatile u32 = @ptrFromInt(0x5008_C37C); + uart.write("MARK MMU entries 0..11 (expect N+1)\r\n"); + var e: u32 = 0; + while (e < 12) : (e += 1) { + mmu_index.* = e; + uart.dumpWord(mmu_content.*); + } + + // The MMU is provably right, the flash is MD5-verified and the image matches the ELF, yet a load + // at 0x40035a1c returns instruction bytes. The only layer left is the cache. A stale line would + // be deterministic across resets - the bootloader repeats the same sequence every boot - so + // determinism did NOT rule it out earlier. + // + // So: read it, evict by walking far more data than any cache here can hold, then read it again. + // If the second read is correct, the first was a stale line and the fix belongs at startup. + uart.write("MARK CACHE before/after eviction\r\n"); + uart.dumpHex(0x4003_5a1c, 8); + { + // 512 KiB touched at 64-byte line stride, summed so nothing can be optimised away. + var sink: u32 = 0; + var p: u32 = 0x4000_0000; + while (p < 0x4008_0000) : (p += 64) { + sink +%= @as(*volatile u32, @ptrFromInt(p)).*; + } + uart.dumpWord(sink); + } + uart.dumpHex(0x4003_5a1c, 8); uart.write("MARK B1 entry ok\r\n"); const heap = heapSpan(); uart.write("MARK B2 heapSpan ok\r\n"); @@ -297,8 +347,14 @@ fn panicImpl(msg: []const u8, _: ?usize) noreturn { } /// Reset entry. The bootloader hands over with an unspecified stack pointer and the FPU off, so: -/// enable the F extension (`mstatus.FS`, which ESP-IDF only ever turns on lazily from a trap -/// handler this image does not have), establish a stack, clear `.bss`, and call into Zig. +/// enable the F extension (`mstatus.FS`, which ESP-IDF only ever turns on lazily from a trap handler +/// this image does not have), establish a stack, clear `.bss`, and call into Zig. +/// +/// The cache invalidate that this image also needs is the FIRST thing `zig_main` does, not something +/// done here. Hand-written `la t0, Cache_Invalidate_All` against an absolute linker symbol computed +/// a PC-relative target and jumped into nowhere (measured: PC=0x88b5d788 with the argument stranded +/// in a2); Zig generates the addressing for an `extern fn` correctly, and `zig_main` runs before any +/// `.rodata` is touched anyway. export fn _start() linksection(".text.entry") callconv(.naked) noreturn { asm volatile ( \\ li t0, 1 << 13 |
