From c2129d652fc2ab929fceb89e37430170075d2627 Mon Sep 17 00:00:00 2001 From: Gabriel Schneider Date: Tue, 25 Aug 2026 15:03:58 -0300 Subject: Bound the transmit spin; retire the probes that have answered Housekeeping on the instrumentation, plus one fix that stands on its own. uart.write and uart.writeByte no longer spin forever waiting for TX FIFO space. hal/uart.zig:182-186 already made this argument about update() - "on a board with no debugger an infinite spin is indistinguishable from a crash" - and this port demonstrated it: the console going quiet mid-boot read as a hang in whatever code came next, for hours, when a stalled transmitter would have looked identical. The wait is bounded per burst and abandoned bytes are counted in `uart.dropped`, so a lying console is at least a countable one. The bound is deliberately generous: 1,000,000 status reads against an 11 ms drain at 115200. Removed, because each has answered its question and the answers are recorded in comments where they matter: * the MMU table dump - the table is CORRECT, entries 0..9 holding 0x1001..0x100a, exactly the valid bit plus physical page N+1 that the image builder's anchor requires. That is now stated in flushFlashCache's doc comment rather than re-measured every boot. * the before/after eviction read - it established that the same load returns 93 85 85 0f before a capacity flush and 3c ee 08 40 after, which is what the image holds there. The flush itself stays; the proof of why it is needed is in the comment. * the allocator and vtable pointer dumps in src/p4.zig - they showed the struct crosses the seam intact, with its function pointers landing in .flash.text. * the bisect early return - it showed that execution does not come back from pardes_p4_init at all. What is left in place, on purpose: the I1..I10 markers inside pardes_p4_init and the B1..B9 markers in the firmware. The port does not work yet and they are how the next person finds out where it stops. Where it stops: the console goes silent immediately after the editor's pardes.allocators.init and never resumes. It is not the transmit spin - lowering the bound to 20,000 and watching for 30 seconds changed nothing - and it is not a trap the ROM can report, because no Guru Meditation is printed. Execution does not return from pardes_p4_init even when that function is made to return immediately after the marker that does print. So the CPU is lost inside a call whose only work is handing a string to a function pointer, which points at the firmware's own writeOut. That needs an instrument this setup does not have: JTAG, or a GPIO-based tracer that does not depend on the UART at all. Everything cheaper has been tried and is recorded above. --- src/pardes/app.zig | 41 ++--------------------------------------- 1 file changed, 2 insertions(+), 39 deletions(-) (limited to 'src/pardes/app.zig') diff --git a/src/pardes/app.zig b/src/pardes/app.zig index 1c540e1..cb0851e 100644 --- a/src/pardes/app.zig +++ b/src/pardes/app.zig @@ -36,6 +36,7 @@ const std = @import("std"); const soc = @import("soc"); +const config = @import("config"); const hal = @import("hal"); const heapmod = @import("heap"); const uart = @import("uart.zig"); @@ -213,45 +214,6 @@ export fn zig_main() noreturn { // because that marker IS a string literal in flash and would read as machine code without this. flushFlashCache(); uart.write("\r\nMARK PARDES_ENTRY direct-fifo\r\n"); - // Ask the MMU what it actually mapped, which is the one measurement that settles where the - // wrong bytes come from. The flash MMU table is not memory-mapped: entry N is read by writing N - // to SPI_MEM_C_MMU_ITEM_INDEX_REG and reading SPI_MEM_C_MMU_ITEM_CONTENT_REG - // (hal/esp32p4/mmu_ll.h:311-330). Page size is hardwired to 64 KiB on this chip - // (mmu_ll.h:126-130 returns MMU_PAGE_64KB and the setter asserts it), so entry N covers - // vaddr 0x40000000 + N*0x10000. - // - // Both mapped segments share one flash-to-vaddr delta by construction (the image builder anchors - // them), and that delta is 0x010000 - 0x00000000, so EVERY entry N should hold physical page - // N+1. Any entry that does not is the bug, and its value says which flash page it points at - // instead. - const mmu_index: *volatile u32 = @ptrFromInt(0x5008_C380); - const mmu_content: *volatile u32 = @ptrFromInt(0x5008_C37C); - uart.write("MARK MMU entries 0..11 (expect N+1)\r\n"); - var e: u32 = 0; - while (e < 12) : (e += 1) { - mmu_index.* = e; - uart.dumpWord(mmu_content.*); - } - - // The MMU is provably right, the flash is MD5-verified and the image matches the ELF, yet a load - // at 0x40035a1c returns instruction bytes. The only layer left is the cache. A stale line would - // be deterministic across resets - the bootloader repeats the same sequence every boot - so - // determinism did NOT rule it out earlier. - // - // So: read it, evict by walking far more data than any cache here can hold, then read it again. - // If the second read is correct, the first was a stale line and the fix belongs at startup. - uart.write("MARK CACHE before/after eviction\r\n"); - uart.dumpHex(0x4003_5a1c, 8); - { - // 512 KiB touched at 64-byte line stride, summed so nothing can be optimised away. - var sink: u32 = 0; - var p: u32 = 0x4000_0000; - while (p < 0x4008_0000) : (p += 64) { - sink +%= @as(*volatile u32, @ptrFromInt(p)).*; - } - uart.dumpWord(sink); - } - uart.dumpHex(0x4003_5a1c, 8); uart.write("MARK B1 entry ok\r\n"); const heap = heapSpan(); uart.write("MARK B2 heapSpan ok\r\n"); @@ -282,6 +244,7 @@ export fn zig_main() noreturn { const rc = pardes_p4_init(&editor_allocator, writeOut, null, 80, 24); uart.write("MARK B10 pardes_p4_init returned\r\n"); + if (rc != 0) { soc.rom.print("MARK PARDES_INIT_FAIL rc=%u\r\n", .{rc}); const s = gpa_heap.stats(); -- cgit v1.3