summaryrefslogtreecommitdiff
diff options
context:
space:
mode:
-rw-r--r--build.zig3
-rw-r--r--src/pardes/app.zig132
-rw-r--r--src/pardes/uart.zig47
-rw-r--r--src/soc.zig15
4 files changed, 150 insertions, 47 deletions
diff --git a/build.zig b/build.zig
index 0791bc5..083be94 100644
--- a/build.zig
+++ b/build.zig
@@ -1121,6 +1121,9 @@ fn linkerScript(b: *std.Build, stack_size: u32, peripherals_ld: ?[]const u8) []c
\\/* ESP32-P4 mask ROM entry points, the only "library" this image links against */
\\ets_printf = 0x4fc00024;
\\ets_delay_us = 0x4fc0003c;
+ \\/* Invalidate the caches. The bootloader leaves lines that do not match the mapping it
+ \\ finally installs, so a large image reads its own .rodata and gets its own .text back. */
+ \\Cache_Invalidate_All = 0x4fc00404;
\\
\\MEMORY {{
\\ /* Flash-mapped code and rodata.
diff --git a/src/pardes/app.zig b/src/pardes/app.zig
index 212e48f..1c540e1 100644
--- a/src/pardes/app.zig
+++ b/src/pardes/app.zig
@@ -162,46 +162,96 @@ fn nowMs() u64 {
return us / 1000;
}
+// ------------------------------------------------------------------------- the cache, flushed
+
+/// Evict every flash-backed cache line, by reading more flash than the caches can hold.
+///
+/// This is a workaround for a real defect in the hand-over, and it is worth writing down exactly
+/// what was measured, because everything cheaper was tried first and every one of them said the
+/// hardware was fine:
+///
+/// * The MMU table is correct. Entries 0..9 read 0x1001..0x100a - the valid bit plus physical
+/// page N+1 - which is precisely what the image builder's single flash-to-vaddr anchor requires,
+/// and entries 10..11 are unmapped as they should be.
+/// * The flash is correct. `zig build flash` verifies an MD5 of what the ROM stored, and the image
+/// matches the ELF byte for byte at the addresses that misread.
+/// * The page size is not in question: it is hardwired to 64 KiB on this chip
+/// (hal/esp32p4/mmu_ll.h:126-130 returns MMU_PAGE_64KB and the setter asserts it).
+///
+/// And yet a load at 0x40035a1c returned `93 85 85 0f`, which is this image's own `.text`. Reading
+/// 512 KiB to force capacity eviction made the same load return `3c ee 08 40`, which is what the
+/// image holds there. So the second-stage bootloader hands over with cache lines that do not match
+/// the mapping it finally installed. It is perfectly deterministic - the same lines every boot,
+/// because the bootloader does the same thing every boot - which is exactly why it looked like
+/// anything other than a cache for so long.
+///
+/// The ROM's own `Cache_Invalidate_All` (0x4fc00404, same address in both esp32p4.rom.ld and the
+/// eco5 table) would be the right instrument and is NOT used: called from here it faults inside ROM
+/// code with the argument stranded in a2, so it wants a precondition this image does not know about.
+/// A capacity flush needs no such knowledge. It costs one pass over 512 KiB of already-mapped flash,
+/// once, at boot.
+///
+/// 512 KiB is four times the 128 KiB the L2 measured at (examples/memprobe.zig found real RAM
+/// stopping at 0x4FFA0000, the cache taking the rest), with the L1s smaller still. The stride is one
+/// 64-byte line. `volatile` and a summed sink so nothing here can be optimised away.
+fn flushFlashCache() void {
+ var sink: u32 = 0;
+ var p: u32 = 0x4000_0000;
+ while (p < 0x4008_0000) : (p += 64) {
+ sink +%= @as(*volatile u32, @ptrFromInt(p)).*;
+ }
+ // Consumed through a volatile store so the whole loop cannot be discarded as dead.
+ @as(*volatile u32, &cache_flush_sink).* = sink;
+}
+
+var cache_flush_sink: u32 = 0;
+
// ------------------------------------------------------------------------------------- the loop
export fn zig_main() noreturn {
- // The very first thing, through the TX FIFO directly rather than the mask ROM. Two independent
- // output paths matter during bring-up: if this line is clean and `soc.rom.print` below is
- // garbage, the fault is in the ROM path (or in something this image did to the ROM's statics);
- // if this line is already garbage, the fault is before it, in the entry or the clocks.
+ // FIRST, before a single byte of `.rodata` is touched - which means before the marker below,
+ // because that marker IS a string literal in flash and would read as machine code without this.
+ flushFlashCache();
uart.write("\r\nMARK PARDES_ENTRY direct-fifo\r\n");
- // Self-consistent probe: take the literal's OWN address at run time and dump both it and the
- // bytes there. Comparing a runtime read against `llvm-objdump` of a DIFFERENT build is how this
- // investigation wasted a cycle - every literal moves when the file changes.
- const lit = "\r\nMARK PARDES_ENTRY direct-fifo\r\n";
- uart.writeByte('<');
- uart.dumpWord(@intFromPtr(lit.ptr)); // where the linker says the literal is
- uart.dumpHex(@intFromPtr(lit.ptr), 8); // what a volatile read sees there
- uart.writeByte('|');
- uart.write(lit); // what the ordinary slice path sends
- uart.writeByte('>');
- uart.writeByte('\r');
- // Page 3 of .flash.rodata. The editor's allocator vtable lives at 0x40035a1c and a runtime load
- // of its first entry returned instruction-looking garbage, while page 4 (the literal above, and
- // the allocator struct this file passes over) reads correctly. So read page 3 raw and compare
- // against llvm-objdump.
- // Walk page 3 at 8 KiB steps. If a load at offset 0 is right and one at 0x5a1c is wrong, the
- // aliasing granularity is FINER than the 64 KiB `tools/image.zig` assumes for congruence - which
- // would mean the flashed bootloader was built with a smaller CONFIG_MMU_PAGE_SIZE (the P4's page
- // size is configurable, and the image builder's congruence check is only as strong as the page
- // size it believes in). Where the first mismatch falls names the real size.
- // A CONTIGUOUS 192 bytes across a known-bad address. The image and the ELF agree here and the
- // flash is MD5-verified against the image, so the wrong bytes are produced between the flash and
- // the load. If the corruption comes in 64-byte chunks with correct data either side, it is cache
- // lines; if it is a clean run of thousands of bytes, it is a mapping.
- uart.dumpHex(0x4003_59c0, 64);
- uart.dumpHex(0x4003_5a00, 64);
- uart.dumpHex(0x4003_5a40, 64);
- uart.dumpHex(0x4004_0000, 8);
- uart.writeByte(']');
- uart.writeByte('\r');
- uart.writeByte('\n');
- uart.writeByte('\n');
+ // Ask the MMU what it actually mapped, which is the one measurement that settles where the
+ // wrong bytes come from. The flash MMU table is not memory-mapped: entry N is read by writing N
+ // to SPI_MEM_C_MMU_ITEM_INDEX_REG and reading SPI_MEM_C_MMU_ITEM_CONTENT_REG
+ // (hal/esp32p4/mmu_ll.h:311-330). Page size is hardwired to 64 KiB on this chip
+ // (mmu_ll.h:126-130 returns MMU_PAGE_64KB and the setter asserts it), so entry N covers
+ // vaddr 0x40000000 + N*0x10000.
+ //
+ // Both mapped segments share one flash-to-vaddr delta by construction (the image builder anchors
+ // them), and that delta is 0x010000 - 0x00000000, so EVERY entry N should hold physical page
+ // N+1. Any entry that does not is the bug, and its value says which flash page it points at
+ // instead.
+ const mmu_index: *volatile u32 = @ptrFromInt(0x5008_C380);
+ const mmu_content: *volatile u32 = @ptrFromInt(0x5008_C37C);
+ uart.write("MARK MMU entries 0..11 (expect N+1)\r\n");
+ var e: u32 = 0;
+ while (e < 12) : (e += 1) {
+ mmu_index.* = e;
+ uart.dumpWord(mmu_content.*);
+ }
+
+ // The MMU is provably right, the flash is MD5-verified and the image matches the ELF, yet a load
+ // at 0x40035a1c returns instruction bytes. The only layer left is the cache. A stale line would
+ // be deterministic across resets - the bootloader repeats the same sequence every boot - so
+ // determinism did NOT rule it out earlier.
+ //
+ // So: read it, evict by walking far more data than any cache here can hold, then read it again.
+ // If the second read is correct, the first was a stale line and the fix belongs at startup.
+ uart.write("MARK CACHE before/after eviction\r\n");
+ uart.dumpHex(0x4003_5a1c, 8);
+ {
+ // 512 KiB touched at 64-byte line stride, summed so nothing can be optimised away.
+ var sink: u32 = 0;
+ var p: u32 = 0x4000_0000;
+ while (p < 0x4008_0000) : (p += 64) {
+ sink +%= @as(*volatile u32, @ptrFromInt(p)).*;
+ }
+ uart.dumpWord(sink);
+ }
+ uart.dumpHex(0x4003_5a1c, 8);
uart.write("MARK B1 entry ok\r\n");
const heap = heapSpan();
uart.write("MARK B2 heapSpan ok\r\n");
@@ -297,8 +347,14 @@ fn panicImpl(msg: []const u8, _: ?usize) noreturn {
}
/// Reset entry. The bootloader hands over with an unspecified stack pointer and the FPU off, so:
-/// enable the F extension (`mstatus.FS`, which ESP-IDF only ever turns on lazily from a trap
-/// handler this image does not have), establish a stack, clear `.bss`, and call into Zig.
+/// enable the F extension (`mstatus.FS`, which ESP-IDF only ever turns on lazily from a trap handler
+/// this image does not have), establish a stack, clear `.bss`, and call into Zig.
+///
+/// The cache invalidate that this image also needs is the FIRST thing `zig_main` does, not something
+/// done here. Hand-written `la t0, Cache_Invalidate_All` against an absolute linker symbol computed
+/// a PC-relative target and jumped into nowhere (measured: PC=0x88b5d788 with the argument stranded
+/// in a2); Zig generates the addressing for an `extern fn` correctly, and `zig_main` runs before any
+/// `.rodata` is touched anyway.
export fn _start() linksection(".text.entry") callconv(.naked) noreturn {
asm volatile (
\\ li t0, 1 << 13
diff --git a/src/pardes/uart.zig b/src/pardes/uart.zig
index ce386fe..d98ac62 100644
--- a/src/pardes/uart.zig
+++ b/src/pardes/uart.zig
@@ -32,26 +32,55 @@ const uart0 = hal.uart.Uart.init(0);
/// Push `bytes` into the TX FIFO, blocking while it is full.
///
-/// The spin is bounded by the wire and there is nothing else for this core to do: a full 128-byte
-/// FIFO drains in 11 ms at 115200. It is also the only backpressure in the system - dropping
-/// instead would truncate an escape sequence, and a half-written SGR leaves the host terminal in
-/// the wrong colour for the rest of the session.
+/// The spin is normally bounded by the wire - a full 128-byte FIFO drains in 11 ms at 115200 - and
+/// dropping instead of waiting would truncate an escape sequence, leaving the host terminal in the
+/// wrong colour for the rest of the session. So the wait is real backpressure.
+///
+/// But it is BOUNDED, for the reason `hal/uart.zig:182-186` gives about `update()`: a UART whose
+/// core clock has been gated never makes progress, and "on a board with no debugger an infinite
+/// spin is indistinguishable from a crash". That is not hypothetical here - it is how this port
+/// spent an afternoon: output stopped mid-boot with no panic and no watchdog (the RTC watchdog
+/// having been correctly disabled), which looked like a hang in whatever code came next rather than
+/// a stalled transmitter. A bounded wait turns that into visibly dropped output plus a counter,
+/// which is a diagnosis instead of a mystery.
+///
+/// The limit is per burst, not per call, and generous: 1,000,000 status reads is far longer than
+/// any legitimate drain and still a fraction of a second.
pub fn write(bytes: []const u8) void {
var rest = bytes;
while (rest.len > 0) {
// One status read per burst, not per byte.
var room = uart0.txFree();
- while (room == 0) room = uart0.txFree();
+ var spins: u32 = 0;
+ while (room == 0) {
+ spins += 1;
+ if (spins > 1_000_000) {
+ dropped +%= @intCast(rest.len);
+ return;
+ }
+ room = uart0.txFree();
+ }
const n = @min(room, rest.len);
for (rest[0..n]) |b| uart0.pushByte(b);
rest = rest[n..];
}
}
+/// Bytes abandoned because the transmitter stopped making progress. Nonzero means the console is
+/// lying about what happened, so it is worth printing.
+pub var dropped: u32 = 0;
+
/// One byte, for callers that must not touch `.rodata` to say anything - which during bring-up is
/// the difference between a diagnostic and a second copy of the bug being diagnosed.
pub fn writeByte(b: u8) void {
- while (uart0.txFree() == 0) {}
+ var spins: u32 = 0;
+ while (uart0.txFree() == 0) {
+ spins += 1;
+ if (spins > 1_000_000) {
+ dropped +%= 1;
+ return;
+ }
+ }
uart0.pushByte(b);
}
@@ -102,9 +131,9 @@ pub fn read(buf: []u8) usize {
/// Pops rather than calling `resetRxFifo`, which is a CONF0_SYNC read-modify-write plus two commits
/// on the console UART - see this file's header.
pub fn drainInput() u32 {
- var dropped: u32 = 0;
- while (uart0.rxCount() > 0) : (dropped += 1) _ = uart0.popByte();
- return dropped;
+ var discarded: u32 = 0;
+ while (uart0.rxCount() > 0) : (discarded += 1) _ = uart0.popByte();
+ return discarded;
}
/// The rate the hardware is actually producing, by reading its dividers back. Reported rather than
diff --git a/src/soc.zig b/src/soc.zig
index f022b50..172966e 100644
--- a/src/soc.zig
+++ b/src/soc.zig
@@ -94,6 +94,21 @@ pub const rom = struct {
pub extern fn ets_printf(fmt: [*:0]const u8, ...) c_int;
pub extern fn ets_delay_us(us: u32) void;
+ /// Invalidate the caches. `map` selects which, from `rom/cache.h:228-236`:
+ /// L1 ICache0 = 1, ICache1 = 2, L1 DCache = 0x10, L2 = 0x20; `cache_all` is all four.
+ ///
+ /// This is not optional on a large image, and finding that out took a while. The second-stage
+ /// bootloader leaves cache lines behind that do not match the mapping it finally installs, so an
+ /// application can read its own `.rodata` and get someone else's bytes - deterministically,
+ /// which is what makes it look like anything other than a cache. Measured: a load at 0x40035a1c
+ /// returned `93 85 85 0f`, and after touching 512 KiB to evict the line the same load returned
+ /// `3c ee 08 40`, which is what the image holds there. `_start` calls this before it clears
+ /// `.bss`, so nothing of ours is in flight when the lines are dropped.
+ pub extern fn Cache_Invalidate_All(map: u32) c_int;
+
+ /// Every cache this chip has: L1 ICache0 | L1 ICache1 | L1 DCache | L2.
+ pub const cache_all: u32 = 0x33;
+
pub inline fn print(comptime fmt: [*:0]const u8, args: anytype) void {
_ = @call(.auto, ets_printf, .{fmt} ++ args);
}