//! pardes's own test suite for the ESP32-P4, which runs ON THE DIE. //! //! WHY THIS EXISTS. Every serious bug this port has produced was invisible to a host test, and two //! of them were invisible for months. `std.mem.eql` compares a byte at a time on this target and //! several times faster on the host, so the firmware's largest read was three times slower than it //! needed to be and nothing on a laptop could tell. A lone ESC resolves to the Escape key, which is //! right when a kernel hands over a whole escape sequence and wrong when a 115200 line hands over //! one byte every 87 microseconds. A full transmit FIFO stopped anything draining the receiver, and //! the FIFO depth is a hardware number. None of those is a logic error you can reason your way to //! from a host: they are properties of THIS chip, THIS clock and THIS wire. //! //! So the checks below are chosen by one rule: a check belongs here only if the die can answer it //! and a host cannot. Anything that is pure logic - the ring's wrap-around, the mouse coalescer's //! ordering - already has a deterministic host test in `zig build unit-test`, which is faster, needs //! no hardware, and is where such a thing belongs. Duplicating those here would make this suite //! longer and no stronger. //! //! ## Why the suite is in the editor's repository //! //! It was written in the `05-zig-p4` toolchain repository as `examples/selftest.zig`, and it was //! never an example of anything. Re-read the paragraph above: every claim it makes is a claim about //! THIS program on this board. The word-at-a-time `std.mem.eql` it measures against is the reason //! the editor's frame diff compares rows a `u32` at a time; the lone-ESC decode and the //! transmit-FIFO backpressure are `input_rescue.zig`, a sibling file in this directory; the heap //! span and the allocator churn are what decide how large a grid this board can drive; and the //! clock check is the divisor under every `-Dprof` cycle count the editor reports. A suite that //! specific to one application belongs beside it, so that changing a source and changing the check //! which defends it are one diff in one repository rather than two commits in two trees. //! //! What stayed behind in the toolchain is the half that is about the CHIP: the image builder's rules, //! the register layer's field arithmetic, the console bridge's escape matcher. Those still run under //! `zig build test` over there and need no board. //! //! ## What it expects from the toolchain package //! //! Four of its modules and no more: `soc` (mask-ROM printf, the cycle counter), `hal` (UART0, the //! systimer, clock/reset), `config` (this build's `cpu_mhz`) and `heap` (the coalescing allocator). //! The sibling `05-zig-p4` build supplies these modules and the firmware image tooling. //! //! `input_rescue` is the fifth import and is NOT that package's: it is the sibling file in this //! directory, handed over as a named MODULE rather than imported as a path. Deliberately so - the //! `pub` on `FakePort`'s methods below is what lets that module reach them by duck typing across the //! boundary. The toolchain's `selftest` step wires this root and that module together. //! //! Not a `zig test` binary, deliberately. Zig's test runner wants an OS, and `std.testing.allocator` //! is a debug allocator over the page allocator, which on freestanding is either a compile error or //! a lie. A hand-rolled harness is thirty lines and answers to nobody. //! //! From `../05-zig-p4`, `zig build selftest` flashes and runs this suite on hardware. const std = @import("std"); const soc = @import("soc"); const hal = @import("hal"); const config = @import("config"); const heapmod = @import("heap"); const input_rescue = @import("input_rescue"); /// Reset entry. Identical in shape to `src/main.zig`'s and for the same reasons: the bootloader hands /// over with an unspecified stack pointer and the FPU off, so set `mstatus.FS`, establish a stack, /// clear `.bss`, and jump. /// /// Leaving this out is what made the first draft of this file unbuildable, and the failure said /// nothing useful: `-fentry=_start` found no such symbol, `--gc-sections` then discarded every /// function as unreachable, and the image builder reported `NotTwoMappedSegments` because /// `.flash.text` had nothing left in it. `zig build layout` now prints `entry 0x0` for exactly that, /// which is the same diagnosis in one line. export fn _start() linksection(".text.entry") callconv(.naked) noreturn { asm volatile ( \\ li t0, 1 << 13 \\ csrs mstatus, t0 \\ la sp, __stack_top \\ mv fp, sp \\ la t0, __bss_start \\ la t1, __bss_end \\ bgeu t0, t1, 2f \\1: \\ sw zero, 0(t0) \\ addi t0, t0, 4 \\ bltu t0, t1, 1b \\2: \\ j zig_main ); } pub const panic = std.debug.FullPanic(struct { fn call(msg: []const u8, first_trace_addr: ?usize) noreturn { // `msg` is a slice and carries no terminator - std's own panics are formatted into a buffer - // so it goes out with a length rather than through a `%s` that would read past the end of it. soc.rom.print("\r\nMARK SELFTEST_PANIC at 0x%08x: ", .{@as(u32, @truncate(first_trace_addr orelse 0))}); hal.uart.Uart.init(0).write(msg); // A panic is a FAILED RUN, and the host is watching for the summary line. Without this the // run looks like a board that never answered, which is a different diagnosis entirely. soc.rom.print("\r\nMARK SELFTEST DONE pass=%u fail=%u\r\n", .{ passed, failed + 1 }); while (true) {} } }.call); var passed: u32 = 0; var failed: u32 = 0; /// One claim about the silicon. Printed either way: a suite that only speaks up when it fails gives /// no way to tell "all good" from "never ran", and on a board that difference matters. fn check(name: [*:0]const u8, ok: bool) void { if (ok) { passed += 1; soc.rom.print("MARK SELFTEST ok %s\r\n", .{name}); } else { failed += 1; soc.rom.print("MARK SELFTEST FAIL %s\r\n", .{name}); } } /// Word-at-a-time equality, the same shape the editor's frame diff uses. fn sameBytesWordwise(a: []const u8, b: []const u8) bool { if (a.len != b.len) return false; if ((@intFromPtr(a.ptr) | @intFromPtr(b.ptr)) & 3 == 0) { const n = a.len / 4; const wa: [*]align(4) const u32 = @ptrCast(@alignCast(a.ptr)); const wb: [*]align(4) const u32 = @ptrCast(@alignCast(b.ptr)); for (wa[0..n], wb[0..n]) |x, y| { if (x != y) return false; } return std.mem.eql(u8, a[n * 4 ..], b[n * 4 ..]); } return std.mem.eql(u8, a, b); } /// A UART with a receive FIFO that loses whatever arrives into a full one, as the hardware does. /// `pub` on the methods because `input_rescue` is a separate module here and reaches them by duck /// typing across it. const FakePort = struct { tx_cap: u32, tx_used: u32 = 0, ticks: u32 = 0, sent: u32 = 0, incoming: []const u8, delivered: usize = 0, rx: [8]u8 = undefined, rx_head: usize = 0, rx_len: usize = 0, lost: u32 = 0, fn tick(p: *FakePort) void { p.ticks += 1; if (p.ticks % 4 == 0 and p.tx_used > 0) p.tx_used -= 1; if (p.delivered < p.incoming.len) { const byte = p.incoming[p.delivered]; p.delivered += 1; if (p.rx_len == p.rx.len) { p.lost += 1; } else { p.rx[(p.rx_head + p.rx_len) % p.rx.len] = byte; p.rx_len += 1; } } } pub fn txFree(p: *FakePort) u32 { p.tick(); return p.tx_cap - p.tx_used; } pub fn pushByte(p: *FakePort, _: u8) void { p.sent += 1; p.tx_used += 1; } pub fn rxCount(p: *FakePort) u32 { return @intCast(p.rx_len); } pub fn popByte(p: *FakePort) u8 { const byte = p.rx[p.rx_head]; p.rx_head = (p.rx_head + 1) % p.rx.len; p.rx_len -= 1; return byte; } }; /// Backing store for the heap checks. Static, because the point is to exercise the allocator on real /// L2MEM rather than to find out where a stack array happens to land. var heap_area: [64 * 1024]u8 align(16) = undefined; export fn zig_main() noreturn { // THE CLOCK RAISE, performed here rather than assumed, which is what turns the frequency check // at the end into a test of `setCpuFreq` instead of a tautology. The first version of this file // read `config.cpu_mhz` and compared the die against it without ever setting it, so // `-Dcpu-mhz=360` failed with `khz=90001 want=360000` - the check was right and the expectation // was wrong. Same call, and the same order, as `src/esp32p4/app.zig`. if (config.cpu_mhz != 90) hal.clkrst.setCpuFreq(switch (config.cpu_mhz) { 360 => .mhz360, else => .mhz90, }); hal.systimer.init(); soc.rom.print("\r\nMARK SELFTEST_START cpu_mhz=%u\r\n", .{@as(u32, config.cpu_mhz)}); // ------------------------------------------------ 1. the memory model the memory words assume // // `Peek`, `Poke` and `Hexdump` reach the bus through `*allowzero volatile` pointers and refuse an // unaligned word. Both halves are claims about this core, and neither is checkable on a host. { const cell: *volatile u32 = @ptrCast(@alignCast(&heap_area[0])); cell.* = 0xdeadbeef; check("an aligned word round-trips through a volatile pointer", cell.* == 0xdeadbeef); // Every byte offset in a word, readable and writable: this is what `Hexdump` does, and it is // why `Hexdump` needs no alignment while `Peek` does. var all_offsets_ok = true; for (0..4) |i| { const at: *volatile u8 = @ptrCast(&heap_area[16 + i]); at.* = @intCast(0xa0 + i); if (at.* != 0xa0 + @as(u8, @intCast(i))) all_offsets_ok = false; } check("a byte at every offset in a word round-trips", all_offsets_ok); // THE VOLATILE PROMISE. Two reads of a running counter must be two reads. Were the optimiser // allowed to fold them, `Peek` would print one value twice for a register that had changed, // which is the one thing a memory word must never do. const first = hal.systimer.micros(.unit0) orelse 0; var spin: u32 = 0; while (spin < 4000) : (spin += 1) asm volatile ("" ::: .{ .memory = true }); const second = hal.systimer.micros(.unit0) orelse 0; check("two reads of a live counter are two reads", second != first); check("and that counter runs forwards", second > first); } // --------------------------------------- 2. word-wise equality, on THIS instruction set // // The editor's frame diff compares rows a `u32` at a time because `std.mem.eql` compares a byte // at a time here: 223 us against 66 for the same answer. "The same answer" is the part that has // to hold on the target rather than on the host, so it is checked against `std.mem.eql` itself, // at every difference position, at both alignments, and at lengths that are not multiples of 4. { var a: [64]u8 align(4) = undefined; var b: [64]u8 align(4) = undefined; for (&a, 0..) |*slot, i| slot.* = @intCast(i); @memcpy(&b, &a); var agree = true; for (0..a.len) |len| { if (sameBytesWordwise(a[0..len], b[0..len]) != std.mem.eql(u8, a[0..len], b[0..len])) agree = false; } check("aligned equality agrees with std.mem.eql at every length", agree); agree = true; for (0..a.len) |i| { b[i] ^= 0xff; if (sameBytesWordwise(&a, &b) != std.mem.eql(u8, &a, &b)) agree = false; if (sameBytesWordwise(a[0..33], b[0..33]) != std.mem.eql(u8, a[0..33], b[0..33])) agree = false; b[i] ^= 0xff; } check("a difference at any position is found, as std.mem.eql finds it", agree); agree = true; // UNALIGNED, which is why `sameBytes` tests alignment at RUNTIME: `Cell` is all u8 fields, so // whether a row starts on a word boundary belongs to the allocator and not to the type. for (1..4) |off| { const ua = a[off..]; const ub = b[off..]; if (sameBytesWordwise(ua, ub) != std.mem.eql(u8, ua, ub)) agree = false; b[off + 5] ^= 0xff; if (sameBytesWordwise(ua, ub) != std.mem.eql(u8, ua, ub)) agree = false; b[off + 5] ^= 0xff; } check("unaligned spans fall back and still agree", agree); } // ------------------------------------------------- 3. the rescue, on the real codegen // // The logic has a host test. What that cannot say is whether it behaves the same compiled for // this core at this optimisation level, which is a question only a board answers. { const typed = "the quick brown fox jumps over the lazy dog, and then some more besides"; var port: FakePort = .{ .tx_cap = 2, .incoming = typed }; var ring: input_rescue.Ring = .{}; const frame = "\x1b[1;1H" ++ "x" ** 300; const abandoned = input_rescue.pump(&port, &ring, frame, 1_000_000); check("the frame went out whole", abandoned == 0 and port.sent == frame.len); check("the receive FIFO never overflowed", port.lost == 0); check("the ring dropped nothing", ring.dropped == 0); var got: [128]u8 = undefined; var n = ring.pop(&got); while (port.rxCount() > 0 and n < got.len) : (n += 1) got[n] = port.popByte(); check("every rescued byte, in order", n == typed.len and std.mem.eql(u8, got[0..n], typed)); } // --------------------------------------------------- 4. the allocator on real L2MEM // // The editor's whole geometry ceiling is an allocator question, and this is the allocator, on the // memory it actually runs in rather than on a host's malloc. { var h = heapmod.Heap.init(heap_area[0..]); const gpa = h.allocator(); const one = gpa.alloc(u32, 256) catch null; check("a modest allocation succeeds", one != null); if (one) |slice| { check("and is aligned for its element", @intFromPtr(slice.ptr) % @alignOf(u32) == 0); for (slice, 0..) |*slot, i| slot.* = @intCast(i * 7); var intact = true; for (slice, 0..) |slot, i| { if (slot != i * 7) intact = false; } check("and holds what was written to it", intact); gpa.free(slice); } // FREE THEN REUSE. A heap that cannot hand the same bytes back is a heap that runs out, which // on this board is the difference between a 40x12 grid and an 80x24 one. var churn_ok = true; for (0..64) |_| { const block = gpa.alloc(u8, 1024) catch { churn_ok = false; break; }; gpa.free(block); } check("a kilobyte can be taken and returned repeatedly", churn_ok); // AND IT REFUSES CLEANLY. An allocator that returns garbage instead of an error when it is // out is the failure mode that cost an afternoon during bring-up. const absurd = gpa.alloc(u8, heap_area.len * 4); check("an impossible allocation returns an error", absurd == error.OutOfMemory); } // ------------------------------------------ 5. the clock, which everything divides by // // Every cycle count this firmware reports is divided by the configured frequency somewhere, and // the systimer is clocked from the crystal rather than from the CPU - which is exactly what makes // it a reference the CPU cannot flatter. The 90-to-360 MHz raise rested on this comparison. { const t0 = hal.systimer.micros(.unit0) orelse 0; const c0 = soc.cycles(); while ((hal.systimer.micros(.unit0) orelse 0) -% t0 < 20_000) {} const us = (hal.systimer.micros(.unit0) orelse 0) -% t0; const cy = soc.cycles() - c0; const khz: u32 = if (us > 0) @intCast(cy * 1000 / us) else 0; const want: u32 = @as(u32, config.cpu_mhz) * 1000; // Two percent: far wider than either clock's error, far narrower than the 4x a wrong divider // would produce. const slack = want / 50; soc.rom.print("MARK SELFTEST_CLOCK khz=%u want=%u\r\n", .{ khz, want }); check("the cycle counter and the systimer agree on the CPU frequency", khz > want - slack and khz < want + slack); } soc.rom.print("MARK SELFTEST DONE pass=%u fail=%u\r\n", .{ passed, failed }); while (true) {} }