diff options
| -rw-r--r-- | .gitignore | 7 | ||||
| -rw-r--r-- | README.md | 77 | ||||
| -rw-r--r-- | build.zig | 208 | ||||
| -rw-r--r-- | examples/echo.zig | 146 | ||||
| -rw-r--r-- | examples/heapcheck.zig | 240 | ||||
| -rw-r--r-- | examples/memprobe.zig | 215 | ||||
| -rw-r--r-- | examples/minimal.zig | 2 | ||||
| -rw-r--r-- | src/pardes/app.zig | 318 | ||||
| -rw-r--r-- | src/pardes/uart.zig | 115 | ||||
| -rw-r--r-- | tools/console.zig | 239 | ||||
| -rw-r--r-- | tools/image.zig | 13 | ||||
| -rw-r--r-- | tools/image_test.zig | 3 | ||||
| -rw-r--r-- | tools/serial.zig | 24 |
13 files changed, 1587 insertions, 20 deletions
@@ -7,6 +7,13 @@ # ELF under -Delf). Both are derived from tracked sources. /zig-out/ +# Third-party package closure. Nothing here is ours and nothing here belongs in +# this history: one `-Dpardes` build materialised 2.6 GB across 106 packages and +# 42,736 files - SDL, wayland, glslang, mupdf, wasmtime, 25+ tree-sitter +# grammars - and jj tracked every one under 1 MiB while merely warning about the +# rest. Kept as a rule rather than a memory, because a fetch is one command away. +/zig-pkg/ + # Scope captures: the .json IS the measurement and stays tracked. The .png is a # 1.1 MB screenshot of that same JSON and is referenced by nothing. /captures/*.png @@ -5,6 +5,8 @@ zig build # compile, link, and emit a flashable image zig build flash # ...then write it to the chip and run it zig build run # flash, then print the console (ordered; `flash monitor` is not) zig build monitor # reset the board and print its console +zig build console # attach an interactive terminal: keystrokes in, screen out (Ctrl-] detaches) +zig build interact # flash, then attach that terminal (ordered, like `run`) zig build reset # just pulse the reset line zig build size # where every byte of the image went zig build test # host tests: image builder, and the register layer's field arithmetic @@ -65,11 +67,14 @@ Everything is a `b.option`, so `zig build -h` lists them all. | `-Dflash-size=<enum>` | `16MB` | fitted flash, written into the image header | | `-Dmin-rev`/`-Dmax-rev` | `100`/`199` | silicon revision window. The pre-v3 P4 needs 100..199 | | `-Ddescriptor=<enum>` | `minimal` | 184-byte descriptor, or `full` for the 256-byte one `esptool image-info` can parse | -| `-Dstack=<u32>` | `8192` | stack size; the generated linker script follows | +| `-Dstack=<u32>` | `8192` | stack size; the generated linker script follows. `32768` under `-Dpardes` | | `-Dverify=<bool>` | `true` | ask the ROM for an MD5 of what it stored and compare | | `-Delf=<bool>` | `false` | also install the ELF | | `-Doptimize=<mode>` | `ReleaseSmall` | firmware default, not Debug (Debug costs ~780 B here) | | `-Dseconds=<u32>` | `5` | how long `monitor` listens | +| `-Dconsole-baud=<enum>` | `b115200` | the interactive console's rate: what the bootloader leaves UART0 at. Distinct from `-Dbaud`, which the ROM loader auto-detects | +| `-Dpardes=<bool>` | `false` | build the pardes editor as the application. Needs the object below | +| `-Dpardes-obj=<path>` | `../02-pardes-code/zig-out/pardes-p4.o` | the editor, compiled freestanding by its own build and linked here | ## Layout @@ -79,16 +84,82 @@ tools/image.zig ELF -> ESP image. Header, segments, congruence filler, ch tools/image_test.zig 8 host tests, one per rule the ROM bootloader enforces tools/rom.zig SLIP framing + the ROM loader protocol. No software stub tools/serial.zig termios2 raw mode, arbitrary baud, DTR/RTS reset dance +tools/console.zig the interactive bridge: raw stdin <-> UART, and the window-size handshake src/soc.zig comptime register model: GPIO, IOMUX, mask-ROM entry points, cycle counter src/appdesc.zig esp_app_desc_t, linked as its own object so it cannot be optimised away src/main.zig demo: prints what it can prove, then blinks +src/pardes/ the pardes editor as firmware: entry, heap, UART, and the C ABI it links to examples/minimal.zig the floor: 432 B, blinks and nothing else +examples/echo.zig UART0 duplex echo: the proof that receive works on the die +examples/memprobe.zig what RAM this board actually has, measured rather than assumed +examples/heapcheck.zig the allocator under an editor's workload, on the die ``` +## What RAM this board has + +Measured by `examples/memprobe.zig`, on the die, because it cannot be read off ESP-IDF's linker +fragments. The relevant one (`esp_system/ld/esp32p4/memory.ld.in:18-33`) is parameterised on +`CONFIG_CACHE_L2_CACHE_SIZE`, and that Kconfig's own help text says the size is set "on application +startup" — by an application this is not. + +``` +0x4FF03000..0x4FF3F000 240 KiB RAM .data/.bss/.stack live at the bottom of this +0x4FF3F000..0x4FF40000 4 KiB ROM the mask ROM's .data/.bss; ets_printf needs it +0x4FF40000..0x4FFA0000 384 KiB RAM handed over whole as __heap_start..__heap_end +0x4FFA0000..0x4FFC0000 128 KiB cache NOT memory: the L2 cache lives here +0x48000000 PSRAM 32 MB fitted, untrained; touching it hangs the core +``` + +That last RAM line cost a bug worth repeating, because the first version of the probe reported the +whole upper 512 KiB as usable. It wrote a pattern to a page and read it straight back, one page at a +time — and a store followed immediately by a load of the *same* address returns the stored value +whether the backing store is real memory, an address mirror, or merely a dirty cache line. Writing +every page before reading any page separates the three, and the top 128 KiB then failed. ESP-IDF's +own arithmetic agrees exactly: `SRAM_HIGH_SIZE = 0x80000 - CONFIG_CACHE_L2_CACHE_SIZE`, with the +128 KiB default from `esp_system/port/soc/esp32p4/Kconfig.cache:5,19`. The wrong number had already +been committed to the linker script, where it handed 128 KiB of live L2 cache to an allocator. + +PSRAM stays untrained deliberately. ESP-IDF's ESP32-P4 implementation runs past a thousand lines — +MPLL, MSPI clocking, pin drive and DQS, CS timing, mode registers, a connectivity check, and a +whole timing-calibration subsystem — and the mask ROM exposes only MMU mapping +(`Cache_PSRAM_MMU_Init`, `Cache_PSRAM_MMU_Set`), no device init. The probe reads `0x48000000` on +purpose and hangs there, which is why it prints its cursor before every access rather than after. + +**The RTC watchdog is armed when the bootloader hands over**, and it expects the application to take +it over. Nothing here did, so every example in this repo had been resetting on a ten-second cycle, +invisibly, for as long as no run lasted eight seconds. `examples/heapcheck.zig`'s 20,000 allocations +is the first run that did. `hal.rwdt.disable()` is the fix and `hal.rwdt.armed()` is worth printing +at startup. + ## What adversarial review found -Five reviewers went at this in two rounds; sixteen findings were applied. The ones worth knowing -about, because each is a trap the next person will hit too: +Five reviewers went at this in two rounds; sixteen findings were applied. Two more rounds went at +the memory map and the editor port and found the three below first. The ones worth knowing about, +because each is a trap the next person will hit too: + +* **A linker symbol declared as an object gives the optimiser a zero-sized object, and it will drop + your stores.** `extern const __heap_start: anyopaque` plus `@intFromPtr`/`@ptrFromInt` looks like + the obvious way to reach a region the linker script defines. The pointer it produces carries + provenance for zero bytes, so the ordinary (non-volatile) store the allocator makes through it is + dead code the backend may remove — and did. The first block header read back as + `size=2988759312 next=0x14284684` instead of `{393216, 0xFFFFFFFF}`, the free-list walk followed + garbage, and with asserts compiled out in `ReleaseSmall` that is a silent hang with no console + output after `MARK HEAP_INIT`. `@extern([*]u8, .{ .name = "__heap_start" })` has no size to lose. + Note that `examples/memprobe.zig` could not have caught this: it writes through a `volatile` + pointer, which the optimiser must leave alone. The two files disagreed about whether the same + address worked, which is what made it findable. + +* **A one-page memory probe cannot tell RAM from a mirror or a cache line.** See the section above: + writing and reading the same address back-to-back succeeds in all three cases, and the pattern + being address-derived does not help because the alias is written *and* read through the alias. + This one had already shipped a wrong 512 KiB into the linker script. + +* **Ignoring `POLL.HUP`/`ERR`/`NVAL` is a hot spin, not a no-op.** `tools/console.zig` polls stdin + and the port and acted only on `POLL.IN`. Unplug the CH340 mid-session and `revents` carries + `HUP|ERR|NVAL` forever: `poll` returns immediately with a non-zero count, neither branch matches, + and the loop burns a core with the terminal still in raw mode and `ISIG` off — so Ctrl-C cannot + even end it. `std.posix.poll` cannot report it as an error either, because a dead descriptor is + delivered in `revents` rather than errno (`std/posix.zig:1007-1017` maps `INVAL` to `unreachable`). * **The ROM's status byte is at `data[len-4]`, not `data[len-2]`.** The four-byte trailer is a ROM-versus-stub difference (esptool `loader.py:653-655`). Reading the wrong byte made *every* ROM @@ -3,6 +3,7 @@ //! zig build compile, link, and emit a flashable image //! zig build flash the above, then write it to the chip over the serial port //! zig build monitor open the console +//! zig build console attach an interactive terminal: keystrokes in, screen out //! zig build size print where every byte of the image went //! //! There is no CMake, no ninja, no idf.py, no esptool and no external linker: Zig's own LLD does @@ -14,6 +15,7 @@ const std = @import("std"); const image = @import("tools/image.zig"); const serial = @import("tools/serial.zig"); const rom = @import("tools/rom.zig"); +const console = @import("tools/console.zig"); pub fn build(b: *std.Build) void { // ---------------------------------------------------------------- board and target knobs @@ -32,7 +34,12 @@ pub fn build(b: *std.Build) void { const min_rev = b.option(u16, "min-rev", "minimum silicon revision, major*100+minor (default 100)") orelse 100; const max_rev = b.option(u16, "max-rev", "maximum silicon revision (default 199)") orelse 199; const descriptor = b.option(DescriptorKind, "descriptor", "app descriptor: minimal (184 B) or full (256 B)") orelse .minimal; - const stack_size = b.option(u32, "stack", "stack size in bytes (default 8192)") orelse 8192; + // `-Dpardes` swaps in the editor as the application. It is a distinct option rather than just + // `-Dapp=src/pardes/app.zig` because it also resolves the lazy `pardes` dependency and raises + // the default stack: the core recurses through layout and 8 KiB is not enough for it. + const pardes_app = b.option(bool, "pardes", "build the pardes editor as the application (needs ../02-pardes-code)") orelse false; + const stack_size = b.option(u32, "stack", "stack size in bytes (default 8192, or 32768 under -Dpardes)") orelse + @as(u32, if (pardes_app) 32768 else 8192); // ReleaseSmall by default: this is firmware, and `standardOptimizeOption` would otherwise // hand out Debug builds - which for this target means panic machinery and formatting code // linked into a 500-byte image. @@ -150,7 +157,10 @@ pub fn build(b: *std.Build) void { }), }); - const app_source = b.option([]const u8, "app", "root source file (default src/main.zig)") orelse "src/main.zig"; + // The app root still comes from `-Dapp`, so pointing that at a different shell over the same + // module stays possible. + const app_source = b.option([]const u8, "app", "root source file (default src/main.zig, or src/pardes/app.zig under -Dpardes)") orelse + if (pardes_app) "src/pardes/app.zig" else "src/main.zig"; const app = b.addExecutable(.{ .name = "app", .root_module = b.createModule(.{ @@ -212,6 +222,53 @@ pub fn build(b: *std.Build) void { }); app.root_module.addImport("io", io_mod); + // The general-purpose allocator, its own module for exactly the reason given above for `io`: a + // module's imports cannot escape its root directory, so neither src/pardes/ nor examples/ can + // reach src/net/heap.zig as a file. Pointed at the existing file rather than copied - `Heap` is + // a coalescing free-list over one caller-supplied span and has nothing to do with the radio; it + // lives under src/net/ only because ESP-Hosted needed it first. The file has zero `export`s, so + // compiling it into two modules cannot collide. + // + // Added unconditionally, like `io`: an application that never imports it costs nothing, because + // an unreferenced module emits no code. + const heap_mod = b.createModule(.{ + .root_source_file = b.path("src/net/heap.zig"), + .target = target, + .optimize = optimize, + .single_threaded = true, + }); + app.root_module.addImport("heap", heap_mod); + + if (pardes_app) { + // The editor arrives as a linked OBJECT, not as a package dependency, and that is a + // measurement rather than a preference. + // + // The obvious design was `build.zig.zon` with a path dependency on ../02-pardes-code, and + // `dep.module("pardes_p4")`. It was written, and it broke EVERY build in this repo - + // `zig build`, every example, the oracle - because merely DECLARING it nests pardes's + // ~30-package graph under this one. Two failures, both from just the declaration: + // + // * std/Build.zig:2091 evaluates `mem.eql(u8, decl.name, pkg_hash)` over the whole + // dependency table at comptime, and the enlarged table exceeds the 1000 backwards + // branch quota. It is reached from ghostty's own build (SharedDeps.zig:874 calls + // `b.lazyImport`), which is not lazy in pardes's manifest and so is always compiled. + // * pardes's package cache holds seven tree_sitter versions, and the stale ones use + // `Compile.addCSourceFile`/`linkLibrary`, removed in Zig 0.16. Nesting made them + // reachable and their build.zig files failed to compile. + // + // Neither is fixable from this side, and both would come back the next time the editor + // gained a dependency. So the seam is a file instead: pardes's own build emits one + // freestanding object exporting a small C ABI, and this links it. The consequences are all + // improvements - this repo keeps having no manifest and no dependencies, the editor's + // renderer stays next to the vaxis it needs, and the boundary is bytes in / bytes out. + const obj = b.option([]const u8, "pardes-obj", "path to pardes's p4 object (default ../02-pardes-code/zig-out/pardes-p4.o)") orelse + "../02-pardes-code/zig-out/pardes-p4.o"; + app.root_module.addObjectFile(if (std.fs.path.isAbsolute(obj)) + .{ .cwd_relative = obj } + else + b.path(obj)); + } + if (hosted) { // The Zig half: the port table, the libc surface and the IP stack. // @@ -291,6 +348,25 @@ pub fn build(b: *std.Build) void { run_mon.step.dependOn(&flash.step); b.step("run", "flash the image, then print its console output").dependOn(&run_mon.step); + // The interactive counterpart of `monitor`. `monitor` prints for N seconds and sends nothing, + // which is right for an application that only reports; an application the human drives needs + // the keyboard on the wire. `-Dconsole-baud` is separate from `-Dbaud` because they are + // genuinely different rates: the flasher's rate is negotiated by the ROM loader's SYNC + // auto-detect, while the console's is whatever the running firmware programmed into UART0. + const console_baud = b.option( + serial.Baud, + "console-baud", + "interactive console baud (default 115200, the rate the bootloader leaves UART0 at)", + ) orelse .b115200; + const con = ConsoleStep.create(b, port_path, console_baud); + b.step("console", "attach an interactive terminal to the running application").dependOn(&con.step); + + // Ordered, for the same reason `run` is: an unordered `flash console` lets the console reset + // the board out from under the writer. + const run_con = ConsoleStep.create(b, port_path, console_baud); + run_con.step.dependOn(&flash.step); + b.step("interact", "flash the image, then attach an interactive terminal").dependOn(&run_con.step); + const reset = ResetStep.create(b, port_path); b.step("reset", "reset the board and let the flashed application run").dependOn(&reset.step); @@ -1055,27 +1131,94 @@ fn linkerScript(b: *std.Build, stack_size: u32, peripherals_ld: ?[]const u8) []c \\ that -Dhosted links ESP-Hosted's transport and RPC layers, and the generated protobuf \\ descriptors alone are 33,000 lines of .rodata. One window overflowed by 14,744 bytes. \\ - \\ 1 MiB now. It is a *region*, not a reservation: the image contains only the sections - \\ actually emitted, so a blink app is still ~1.7 KB. The ceiling that matters is the - \\ factory partition, 0x177000 = 1.5 MiB, and this stays comfortably inside it. Segments - \\ still come in the two the loader wants; they simply span more than one MMU page now, - \\ which the bootloader maps without complaint. */ - \\ flash (rx) : ORIGIN = 0x40000020, LENGTH = 0xFFFE0 - \\ l2mem (rw) : ORIGIN = 0x4FF00000, LENGTH = 0x20000 + \\ 1.5 MiB now, less 32 KiB of slack. It is a *region*, not a reservation: the image + \\ contains only the sections actually emitted, so a blink app is still ~1.7 KB. The + \\ ceiling that matters is the factory partition - `zig build monitor` reads it off the + \\ bootloader's own table as `factory 00010000 00177000` - and the region is deliberately + \\ set just under it so that an application which outgrows the partition fails at the + \\ LINK, with a section-overflow naming the section, rather than at the flash write or + \\ (worse) at boot. Segments still come in the two the loader wants; they simply span + \\ more than one MMU page now, which the bootloader maps without complaint. */ + \\ flash (rx) : ORIGIN = 0x40000020, LENGTH = 0x170000 + \\ + \\ /* .data/.bss/.stack. This was 0x20000 for as long as nothing needed more. + \\ + \\ The die was asked (examples/memprobe.zig) rather than the datasheet, because the + \\ relevant ESP-IDF fragment (esp_system/ld/esp32p4/memory.ld.in:18-33) is parameterised + \\ on CONFIG_CACHE_L2_CACHE_SIZE, whose Kconfig help says the size is set "on application + \\ startup" - by an application this is not. Measured, on this rev v1.3 die: + \\ + \\ 0x4FF03000..0x4FF3F000 240 KiB RAM + \\ 0x4FF3F000..0x4FF40000 4 KiB the mask ROM's .data/.bss, never written + \\ 0x4FF40000..0x4FFA0000 384 KiB RAM + \\ 0x4FFA0000..0x4FFC0000 128 KiB NOT memory - the L2 cache lives here + \\ + \\ That last line cost a bug worth recording. The first version of the probe wrote a + \\ pattern and read it back one page at a time, and reported the whole upper 512 KiB as + \\ RAM - because a store followed immediately by a load of the SAME address returns the + \\ stored value whether the backing store is real, an address mirror, or merely a dirty + \\ cache line. Writing every page before reading any page separates the three, and the + \\ top 128 KiB then failed. It is the L2 cache, and ESP-IDF's own arithmetic agrees + \\ exactly: SRAM_HIGH_SIZE = 0x80000 - CONFIG_CACHE_L2_CACHE_SIZE, with the Kconfig + \\ default of 128 KiB (esp_system/port/soc/esp32p4/Kconfig.cache:5,19). Handing those + \\ 128 KiB to an allocator hung the heap on its first free-list walk. + \\ + \\ This region stops at 0x4FF3F000 because `ets_printf` - the only console this image + \\ has - reads the ROM statics that begin at 0x4FF3FBA4, and losing them loses the + \\ ability to report having lost them. */ + \\ l2mem (rw) : ORIGIN = 0x4FF00000, LENGTH = 0x3F000 + \\ + \\ /* The upper RAM, 384 KiB and contiguous, free once the second-stage bootloader has + \\ jumped away. Nothing is *linked* here: it carries no section, and the two symbols + \\ below hand it to the application as one span for a run-time heap. Kept out of `l2mem` + \\ because the ROM's statics sit between the two, and stops at 0x4FFA0000 because the + \\ L2 cache is above that - see the measurement in `l2mem`'s comment. */ + \\ l2high (rw) : ORIGIN = 0x4FF40000, LENGTH = 0x60000 \\}} \\ + \\/* The heap span, from the linker rather than from a constant in Zig, so the region above is + \\ the single place these addresses are written down. */ + \\__heap_start = ORIGIN(l2high); + \\__heap_end = ORIGIN(l2high) + LENGTH(l2high); + \\ \\SECTIONS {{ + \\ /* Two flash-mapped segments, rodata then text, and the split is NOT optional: the + \\ second-stage bootloader asserts it. On a chip whose D and I external vaddr ranges are + \\ shared - which the P4's are (soc.h:146-149, both 0x40000000..0x44000000) - ESP-IDF + \\ takes the `SOC_MMU_DI_VADDR_SHARED` branch of `unpack_load_app` + \\ (bootloader_utility.c:805-851). That branch does not classify segments as D or I at + \\ all; it collects them positionally into rom_addr[2] and ends with + \\ + \\ assert(rom_index == 2); + \\ + \\ so one mapped segment aborts the boot with + \\ `Assert failed in unpack_load_app, bootloader_utility.c:842 (rom_index == 2)`, and a + \\ third trips `assert(rom_index < 2)` inside the loop. Measured, by trying it. */ \\ .flash.rodata : ALIGN(16) {{ \\ KEEP(*(.rodata.appdesc)) /* the loader reads esp_app_desc_t at image offset 0x20 */ \\ *(.rodata .rodata.* .srodata .srodata.*) - \\ /* Leave exactly the 8 bytes the image builder needs for the next segment's header - \\ before the 64-byte boundary that .flash.text starts on. ALIGN(64) alone does not - \\ guarantee any hole: for one rodata length in eight the gap is 0 or 4 bytes and the - \\ build dies with MappedSegmentsTooClose. */ - \\ . = ALIGN(. + 8, 64) - 8; + \\ /* Leave the 8 bytes the image builder needs for the next segment's header before the + \\ boundary .flash.text starts on. ALIGN alone does not guarantee any hole: for one + \\ rodata length in eight the gap is 0 or 4 bytes and the build dies with + \\ MappedSegmentsTooClose. */ + \\ . = ALIGN(. + 8, 0x10000) - 8; \\ }} > flash \\ - \\ .flash.text : ALIGN(64) {{ + \\ /* ALIGN(0x10000), one whole MMU page, and NOT the 64 bytes this used to be. + \\ + \\ Two mapped segments may share a 64 KiB MMU page only if they also share a flash page, + \\ which the image builder's anchor guarantees - and that reasoning holds right up until + \\ an application is large enough for rodata to END inside the page where text BEGINS. + \\ Measured on the die with a 578 KB image: a volatile read of a string literal at + \\ 0x4004a1d1 returned `37 09 fa 4f`, which disassembles to `lui s2, 0x4ffa0` - this + \\ image's own .flash.text. Every literal in that shared page read as code, so the first + \\ thing the firmware tried to print was machine code. + \\ + \\ Aligning text to a page boundary makes the segments page-disjoint, so no MMU entry is + \\ ever claimed by both. It costs up to 64 KiB of image padding, which against a 1.5 MiB + \\ partition is not worth reasoning about - the packing trick this project opened with + \\ only mattered when the alternative was 64 KiB of zeros in a 1 KB image. */ + \\ .flash.text : ALIGN(0x10000) {{ \\ *(.text.entry) \\ *(.text .text.*) \\ }} > flash @@ -1348,6 +1491,41 @@ const MonitorStep = struct { } }; +/// `monitor`, but the wire runs both ways. See tools/console.zig for the size handshake, which is +/// the only part of this that is not a straight byte copy. +/// +/// Takes the same `port_lock` as every other step that opens the port, so `zig build flash console` +/// cannot have the console pull the board out of download mode mid-write. It then HOLDS that lock +/// for the whole interactive session, which is correct and worth saying out loud: the session ends +/// when the user detaches, so any other port step named on the same command line waits for the +/// human rather than racing them. +const ConsoleStep = struct { + step: std.Build.Step, + port: []const u8, + baud: serial.Baud, + + fn create(b: *std.Build, port: []const u8, baud: serial.Baud) *ConsoleStep { + const self = b.allocator.create(ConsoleStep) catch @panic("OOM"); + self.* = .{ + .step = std.Build.Step.init(.{ .id = .custom, .name = "console", .owner = b, .makeFn = make }), + .port = port, + .baud = baud, + }; + return self; + } + + fn make(step: *std.Build.Step, _: std.Build.Step.MakeOptions) anyerror!void { + const self: *ConsoleStep = @fieldParentPtr("step", step); + const b = step.owner; + + port_lock.lockUncancelable(b.graph.io); + defer port_lock.unlock(b.graph.io); + + console.attach(self.port, self.baud, .{}) catch |err| + return step.fail("console on {s}: {s}", .{ self.port, @errorName(err) }); + } +}; + /// Pulse the reset line and leave. One ioctl pair, but it is the difference between "did my app /// hang or did I forget to reset it" during development. const ResetStep = struct { diff --git a/examples/echo.zig b/examples/echo.zig new file mode 100644 index 0000000..6f17c98 --- /dev/null +++ b/examples/echo.zig @@ -0,0 +1,146 @@ +//! UART0 as a duplex byte pipe, which is the one thing this toolchain had never done. +//! +//! Everything else here talks to the host through `soc.rom.print`, a mask-ROM `ets_printf`. That is +//! one-way and it is slow: the ROM formats, then pushes a byte at a time and spins on the FIFO. An +//! editor rendering a screen needs the other direction and needs the fast path, so this example +//! exists to prove three things on the die before anything larger depends on them: +//! +//! 1. **RX works at all.** `hal/uart.zig` has had `rxCount`/`popByte` since the differential +//! suite needed them, but that suite runs on UART1 in internal loopback - no byte has ever +//! arrived from the outside world on UART0. +//! 2. **UART0 can be driven without reconfiguring it.** The header of `hal/uart.zig` is blunt +//! about the hazard: resetting UART0 clears UART_CLKDIV, the console turns to garbage +//! mid-sentence and the board dies on a watchdog reset. So this touches no configuration +//! register - the second-stage bootloader already set the divider, the format and the pad +//! routing, and this code only reads and writes the FIFO. +//! 3. **What the wire rate actually is.** Printed by asking the hardware +//! (`Uart.baudrate`), not by assuming the 115200 the host tooling opens with. +//! +//! Protocol, so the host side has something unambiguous to assert on: +//! +//! any byte -> echoed back verbatim +//! CR (0x0d) -> echoed as CRLF, so a human sees lines +//! '!' -> also emit `bulk_len` bytes of a counted pattern and report the cycles it took +//! Ctrl-D -> print the byte/frame counters +//! +//! The echo is verbatim rather than uppercased or otherwise transformed on purpose: a transform +//! that happens to be idempotent hides a duplicated byte, and a duplicated byte is exactly the +//! failure a FIFO-polling loop produces when `rxCount` is misread. + +const std = @import("std"); +const soc = @import("soc"); +const hal = @import("hal"); + +/// UART0. Instance 0 because that is the pad pair the CH340 is wired to and the one the ROM +/// configured; nothing here may reset it. +const con = hal.uart.Uart.init(0); + +/// The bulk burst `!` emits. 4 KiB is ~35 ms of wire time at 115200 and ~4.5 ms at 921600, so the +/// difference between the two is obvious to the naked eye on the host. +const bulk_len = 4096; + +/// The clock the UART's baud generator is dividing. XTAL is the reset default and what the ROM +/// leaves selected; `Uart.clockSource` is read below rather than assumed, so a bootloader that +/// switched to PLL_F80M shows up as a wrong rate instead of a silent 2x error. +fn sourceHz() u32 { + return con.clockSource().nominalHz(); +} + +/// Block until the TX FIFO has room, then push. The spin is bounded by the wire: at 115200 a full +/// 128-byte FIFO drains in 11 ms, and there is nothing else for this core to do. +/// +/// `txFree` and not "is the FIFO empty": pushing whenever there is a single free slot keeps the +/// transmitter fed, which is what makes this ~10x the throughput of the ROM's per-byte printf. +fn put(byte: u8) void { + while (con.txFree() == 0) {} + con.pushByte(byte); +} + +fn puts(bytes: []const u8) void { + for (bytes) |b| put(b); +} + +export fn zig_main() noreturn { + // Deliberately the ROM path for the banner: if the direct-FIFO writes below are wrong, the + // banner still arrives and says so. Mixing the two is safe because both end up in the same + // FIFO and this is the only writer. + soc.rom.print("\r\nMARK ECHO_BOOT uart0 duplex echo\r\n", .{}); + soc.rom.print("MARK ECHO_BAUD hw=%u src=%u Hz\r\n", .{ con.baudrate(sourceHz()), sourceHz() }); + + // Not `resetRxFifo`: that is a CONF0_SYNC read-modify-write plus two commits on the console + // UART, and this file's whole premise is that UART0's configuration is untouchable. Draining by + // popping has the same effect on the FIFO and touches only offset 0x000. + var dropped: u32 = 0; + while (con.rxCount() > 0) : (dropped += 1) _ = con.popByte(); + soc.rom.print("MARK ECHO_DRAIN dropped=%u stale bytes\r\n", .{dropped}); + puts("MARK ECHO_FIFO direct-fifo tx works\r\n"); + puts("type; '!' bulk, ctrl-D stats\r\n"); + + var rx_total: u32 = 0; + var bursts: u32 = 0; + while (true) { + if (con.rxCount() == 0) continue; + const byte = con.popByte(); + rx_total += 1; + + switch (byte) { + '\r' => puts("\r\n"), + 0x04 => { + var buf: [96]u8 = undefined; + const line = std.fmt.bufPrint( + &buf, + "\r\nMARK ECHO_STATS rx={d} bursts={d} baud={d}\r\n", + .{ rx_total, bursts, con.baudrate(sourceHz()) }, + ) catch "\r\nMARK ECHO_STATS fmt failed\r\n"; + puts(line); + }, + '!' => { + bursts += 1; + put(byte); + const t0 = soc.cycles(); + // A counted pattern, not a constant: a run of identical bytes cannot reveal a + // dropped or reordered one, and the host asserts on the sequence. + var i: u32 = 0; + while (i < bulk_len) : (i += 1) put('0' + @as(u8, @intCast(i % 10))); + const cycles = soc.cycles() - t0; + var buf: [96]u8 = undefined; + const line = std.fmt.bufPrint( + &buf, + "\r\nMARK ECHO_BULK {d} bytes in {d} cycles\r\n", + .{ bulk_len, cycles }, + ) catch "\r\nMARK ECHO_BULK fmt failed\r\n"; + puts(line); + }, + else => put(byte), + } + } +} + +/// Reset entry. Identical in shape to `src/main.zig`'s and for the same reasons - the bootloader +/// hands over with an unspecified stack pointer and the FPU off - but `std.fmt.bufPrint` above is +/// the reason the FPU bit matters here too: its float formatting path is reachable from a generic +/// `bufPrint` instantiation even when no argument is a float. +export fn _start() linksection(".text.entry") callconv(.naked) noreturn { + asm volatile ( + \\ li t0, 1 << 13 + \\ csrs mstatus, t0 + \\ la sp, __stack_top + \\ mv fp, sp + \\ la t0, __bss_start + \\ la t1, __bss_end + \\ bgeu t0, t1, 2f + \\1: + \\ sw zero, 0(t0) + \\ addi t0, t0, 4 + \\ bltu t0, t1, 1b + \\2: + \\ j zig_main + ); +} + +pub const panic = std.debug.FullPanic(struct { + fn call(msg: []const u8, _: ?usize) noreturn { + soc.rom.print("MARK ECHO_PANIC %s\r\n", .{msg.ptr}); + while (true) {} + } +}.call); diff --git a/examples/heapcheck.zig b/examples/heapcheck.zig new file mode 100644 index 0000000..9d61f71 --- /dev/null +++ b/examples/heapcheck.zig @@ -0,0 +1,240 @@ +//! Does `src/net/heap.zig` survive an editor's allocation pattern in 512 KiB? +//! +//! `Heap` was written for ESP-Hosted and sized by it: its own header says the workload it is +//! "sized for is the one the parent measured - `mempool.c` recycling fixed-size buffers - which is +//! exactly the pattern that keeps a coalescing free list short and its first fit O(1) in practice". +//! An editor is not that workload. pardes allocates panes, text buffers, cell grids and per-frame +//! scratch in a dozen different sizes and frees them in an order nobody chose. Two properties that +//! were free under fixed-size recycling stop being free: +//! +//! * **First fit fragments.** Varied sizes leave gaps too small for the next request, and the +//! symptom is not a clean failure - it is `largest_free` collapsing while `free` stays healthy, +//! so the heap reports plenty of room and cannot satisfy a grid reallocation. +//! * **First fit is O(n) in the free list.** One long-lived allocation in the middle of the arena +//! splits it permanently, and every subsequent walk pays for it. +//! +//! So this measures both, on the die, before the editor depends on it. Each phase prints a line +//! whether it passed or not: a number is the deliverable, not a verdict. +//! +//! The interesting column is `largest/free`. At 1.00 the free space is one block and the heap is +//! pristine; as it falls, that fraction is the largest single allocation still possible. An editor +//! that cannot get one contiguous cell grid is dead regardless of how many bytes are notionally +//! free. + +const std = @import("std"); +const soc = @import("soc"); +const hal = @import("hal"); +const heapmod = @import("heap"); + +/// The upper RAM chunk, from the linker script - the same span the firmware gets. +/// +/// Reached with `@extern`, NOT with `extern const __heap_start: anyopaque` plus +/// `@intFromPtr`/`@ptrFromInt`. That spelling cost real debugging on this board. Declaring a linker +/// symbol as an `anyopaque` OBJECT gives the optimiser a zero-sized object to reason about, so a +/// pointer derived from its address carries provenance for zero bytes - and the ordinary +/// (non-volatile) store `Heap.init` makes through it was simply dropped. The symptom was the block +/// header reading back as `size=2988759312 next=0x14284684` instead of `{393216, 0xFFFFFFFF}`, after +/// which the first free-list walk followed garbage and never terminated. With asserts compiled out +/// in ReleaseSmall that is a silent hang, and `examples/memprobe.zig` could not see it because it +/// writes through a `volatile` pointer, which the optimiser must not touch. +/// +/// `@extern` with a `[*]u8` result has no size to lose. +const heap_start = @extern([*]align(heapmod.Heap.granule) u8, .{ .name = "__heap_start" }); +const heap_end = @extern([*]align(heapmod.Heap.granule) u8, .{ .name = "__heap_end" }); + +var gpa_heap: heapmod.Heap = undefined; + +fn span() []align(heapmod.Heap.granule) u8 { + return heap_start[0 .. @intFromPtr(heap_end) - @intFromPtr(heap_start)]; +} + +fn report(tag: [*:0]const u8) void { + const s = gpa_heap.stats(); + // Split across two calls, and no `%%`: the mask ROM's printf is size-optimised and this file's + // first version passed it six varargs plus a literal `%%`, which hung. Four is known to work + // (examples/memprobe.zig's range lines), so this stays inside what has been demonstrated. + soc.rom.print("MARK HEAP %s total=%u free=%u blocks=%u\r\n", .{ + tag, s.total, s.free, s.free_blocks, + }); + // `largest` is the number that matters under fragmentation: it is the largest single allocation + // still possible, whatever `free` claims. + soc.rom.print("MARK HEAP %s largest=%u\r\n", .{ tag, s.largest_free }); +} + +/// A deterministic LCG, so a bad run is reproducible. Numerical Recipes' constants. +var rng_state: u32 = 0x1234_5678; +fn rand() u32 { + rng_state = rng_state *% 1664525 +% 1013904223; + return rng_state; +} + +/// How many live pointers the phases below track. 512 slots at an average of ~512 B is ~256 KiB, +/// half the arena, which is enough to fragment it without trivially exhausting it. +const slots = 512; +var live: [slots][]u8 = undefined; +var live_len: usize = 0; + +export fn zig_main() noreturn { + const s = span(); + // The bootloader leaves the RTC watchdog running and expects the application to take it over. + // A bare image never did, so every demo in this repo has been resetting on a ten-second cycle + // (README.md:289-293) - invisible until a run lasted longer than eight seconds. The churn phase + // below does 20,000 allocations, so this is the first example here that would have hit it: the + // symptom was HEAP_BOOT printed twice and nothing after. + const was_armed = hal.rwdt.disable(); + soc.rom.print("MARK HEAP_RWDT was_armed=%u now_armed=%u\r\n", .{ + @as(u32, @intFromBool(was_armed)), @as(u32, @intFromBool(hal.rwdt.armed())), + }); + soc.rom.print("\r\nMARK HEAP_BOOT 0x%08x..0x%08x %u KiB\r\n", .{ + @as(u32, @intFromPtr(s.ptr)), + @as(u32, @intFromPtr(s.ptr)) + @as(u32, @intCast(s.len)), + @as(u32, @intCast(s.len / 1024)), + }); + gpa_heap = heapmod.Heap.init(s); + soc.rom.print("MARK HEAP_INIT done\r\n", .{}); + // Read the block header straight back through a volatile pointer. `stats()` walks the free list + // starting here, so if these two words are not {len, 0xFFFFFFFF} the walk follows garbage and + // never terminates - and with asserts compiled out in ReleaseSmall that is a silent hang. + { + const hdr: *volatile [2]u32 = @ptrFromInt(0x4FF4_0000); + soc.rom.print("MARK HEAP_HDR size=%u next=0x%08x\r\n", .{ hdr[0], hdr[1] }); + } + const a = gpa_heap.allocator(); + soc.rom.print("MARK HEAP_VTABLE done\r\n", .{}); + const probe_stats = gpa_heap.stats(); + soc.rom.print("MARK HEAP_STATS total=%u\r\n", .{probe_stats.total}); + report("fresh"); + + // ---- phase 1: how much of the arena is actually reachable, and what does the header cost? + // Allocate one block at a time until refusal, to separate "512 KiB of RAM" from "512 KiB of + // usable allocations". Each block costs an 8-byte header, so the answer is strictly less. + { + var n: u32 = 0; + var total: usize = 0; + while (n < slots) { + const blk = a.alloc(u8, 512) catch break; + live[n] = blk; + total += blk.len; + n += 1; + } + live_len = n; + soc.rom.print("MARK HEAP_FILL %u blocks of 512 B = %u B payload\r\n", .{ n, @as(u32, @intCast(total)) }); + report("filled"); + for (live[0..live_len]) |blk| a.free(blk); + live_len = 0; + // Coalescing is the whole design: after freeing everything the arena must be ONE block + // again. If it is not, `insert`'s merge is wrong and every later number is meaningless. + report("emptied"); + } + + // ---- phase 2: the fragmentation case. Fill with alternating sizes, free every other block. + // This is the adversarial pattern for first fit: the holes are all the smaller size, and the + // next larger request has to walk past every one of them. + { + var n: usize = 0; + while (n < slots) : (n += 1) { + const size: usize = if (n % 2 == 0) 192 else 320; + live[n] = a.alloc(u8, size) catch break; + } + live_len = n; + var i: usize = 0; + while (i < live_len) : (i += 2) a.free(live[i]); + report("holed"); + + // Now ask for something that fits in no single hole and see what the heap does. + if (a.alloc(u8, 4096)) |big| { + soc.rom.print("MARK HEAP_BIG 4096 B satisfied after holing\r\n", .{}); + a.free(big); + } else |_| { + soc.rom.print("MARK HEAP_BIG 4096 B REFUSED after holing\r\n", .{}); + } + i = 1; + while (i < live_len) : (i += 2) a.free(live[i]); + live_len = 0; + report("unholed"); + } + + // ---- phase 3: the O(n) first-fit walk, measured rather than argued. + // Build a long free list, then time one allocation that has to traverse it. `soc.cycles()` is + // the cycle counter, so this is in real CPU cycles. + { + var n: usize = 0; + while (n < slots) : (n += 1) live[n] = a.alloc(u8, 256) catch break; + live_len = n; + // Free every other block to make `free_blocks` large, then measure a request that no hole + // can satisfy, which is the worst case: the full walk. + var i: usize = 0; + while (i < live_len) : (i += 2) a.free(live[i]); + const before = gpa_heap.stats().free_blocks; + + const t0 = soc.cycles(); + const probe = a.alloc(u8, 1024) catch null; + const cycles = soc.cycles() - t0; + soc.rom.print("MARK HEAP_WALK %u free blocks, alloc took %u cycles\r\n", .{ before, @as(u32, @intCast(cycles)) }); + if (probe) |p| a.free(p); + + i = 1; + while (i < live_len) : (i += 2) a.free(live[i]); + live_len = 0; + report("after-walk"); + } + + // ---- phase 4: a long random churn, which is the closest thing here to a real session. + // Random sizes, random free order, and a running count of refusals. A heap that fragments + // itself to death shows up as refusals climbing while `free` stays large. + { + var refusals: u32 = 0; + var ops: u32 = 0; + live_len = 0; + while (ops < 20000) : (ops += 1) { + const keep = live_len < slots and (live_len < 64 or rand() % 100 < 55); + if (keep) { + // 24 B to ~6 KiB: the spread pardes actually shows, from a small string to a + // reallocated row buffer. + const size = 24 + (rand() % 6000); + if (a.alloc(u8, size)) |blk| { + live[live_len] = blk; + live_len += 1; + } else |_| refusals += 1; + } else if (live_len > 0) { + const victim = rand() % @as(u32, @intCast(live_len)); + a.free(live[victim]); + live[victim] = live[live_len - 1]; + live_len -= 1; + } + } + soc.rom.print("MARK HEAP_CHURN %u ops, %u live, %u refusals\r\n", .{ ops, @as(u32, @intCast(live_len)), refusals }); + report("churned"); + for (live[0..live_len]) |blk| a.free(blk); + live_len = 0; + report("drained"); + } + + soc.rom.print("MARK HEAP_DONE\r\n", .{}); + while (true) {} +} + +export fn _start() linksection(".text.entry") callconv(.naked) noreturn { + asm volatile ( + \\ li t0, 1 << 13 + \\ csrs mstatus, t0 + \\ la sp, __stack_top + \\ mv fp, sp + \\ la t0, __bss_start + \\ la t1, __bss_end + \\ bgeu t0, t1, 2f + \\1: + \\ sw zero, 0(t0) + \\ addi t0, t0, 4 + \\ bltu t0, t1, 1b + \\2: + \\ j zig_main + ); +} + +pub const panic = std.debug.FullPanic(struct { + fn call(msg: []const u8, _: ?usize) noreturn { + soc.rom.print("MARK HEAP_PANIC %s\r\n", .{msg.ptr}); + while (true) {} + } +}.call); diff --git a/examples/memprobe.zig b/examples/memprobe.zig new file mode 100644 index 0000000..1ce8a0b --- /dev/null +++ b/examples/memprobe.zig @@ -0,0 +1,215 @@ +//! What RAM does this board actually have, and where? +//! +//! The linker script maps one 128 KiB window at 0x4FF00000 and has never needed more. Hosting a +//! real application needs an answer with more than one digit in it, and the answer cannot be read +//! off ESP-IDF's linker fragments, because the fragment that matters +//! (`esp_system/ld/esp32p4/memory.ld.in:18-33`) is parameterised on two things this image does not +//! have: +//! +//! * `CONFIG_ESP32P4_SELECTS_REV_LESS_V3` - true for this rev v1.3 die, which selects a SPLIT +//! layout: a low region 0x4FF00000..0x4FF2BBD0 and a high region from 0x4FF40000, with the +//! mask ROM's own .data/.bss in between at 0x4FF3FBA4..0x4FF40000. +//! * `CONFIG_CACHE_L2_CACHE_SIZE` - the L2 cache is carved out of the SAME 768 KiB array, from +//! the TOP, so `SRAM_HIGH_SIZE = 0x80000 - cache_size`. Its Kconfig default is 128 KiB, but the +//! help text says "to be set on application startup" - the APPLICATION sets it, and this +//! application does not. So the live size is whatever the ROM and the second-stage bootloader +//! left behind, which is exactly the sort of thing that has to be measured. +//! +//! So: probe. For every 4 KiB page in the array, save the first word, write a value derived from +//! the page's own address, read it back, and restore. An address-derived pattern is the point - a +//! constant cannot distinguish real memory from an alias, and aliasing is the specific failure mode +//! of poking at a region the cache controller owns. A page that reads back what it was given is +//! RAM; anything else is reported with what it actually returned. +//! +//! Two pages are never touched: the one holding this image's own .data/.bss/stack, and the mask +//! ROM's reserved window - `soc.rom.print` is the only way this program can report anything, and +//! corrupting the ROM's statics would take the console down with it. +//! +//! A page that is neither RAM nor decoded may raise a bus fault, and this image has no trap +//! handler, so a fault is a silent hang. That is why the scan prints its cursor as it goes: if this +//! stops, the last address printed is the one that killed it, which is itself the result. + +const std = @import("std"); +const soc = @import("soc"); + +/// The whole L2MEM array, per `soc/esp32p4/include/soc/soc.h:161-164` +/// (SOC_DRAM_LOW 0x4ff00000, SOC_DRAM_HIGH 0x4ffc0000). +const l2mem_low: u32 = 0x4FF0_0000; +const l2mem_high: u32 = 0x4FFC_0000; + +/// The mask ROM's .data/.bss, from `bootloader.memory.ld.in:13-16`. Not reclaimable while anything +/// still calls into the ROM, and `soc.rom.print` does. +const rom_data_low: u32 = 0x4FF3_FBA4; +const rom_data_high: u32 = 0x4FF4_0000; + +/// Where PSRAM appears once a driver has trained it (`soc.h:151-153`). Nothing here trains it, so +/// this is expected to fail; it is probed anyway because the cost is four instructions and the +/// alternative is assuming. +const psram_base: u32 = 0x4800_0000; + +const page: u32 = 0x1000; + +/// This image's own footprint, from the linker script's symbols. `.data` starts at the region base +/// and `__stack_top` is the last thing in it, so [l2mem_low, __stack_top) is off limits. +extern const __stack_top: anyopaque; + +fn selfEnd() u32 { + return @intFromPtr(&__stack_top); +} + +/// The heap span the linker script hands over (`build.zig`'s MEMORY block defines both from +/// `l2high`). Referenced here so the report states the same numbers the linker will give the real +/// application, rather than a second copy of them written down in Zig. +extern const __heap_start: anyopaque; +extern const __heap_end: anyopaque; + +/// The value page `addr` must return if it is real, distinct memory. +inline fn pattern(addr: u32) u32 { + // Not `addr` itself: an address bus stuck high would return something that looks plausible. + // XOR with a constant that has bits set where an address never does. + return addr ^ 0xA5A5_0F0F; +} + +/// One saved word per page, so the array can be written whole and read back whole. +const max_pages = (l2mem_high - l2mem_low) / page; +var saved: [max_pages]u32 = @splat(0); + +/// Is this page one the program refuses to touch? +fn skipped(addr: u32) bool { + return (addr < selfEnd()) or (addr + page > rom_data_low and addr < rom_data_high); +} + +/// WHY THIS IS TWO PASSES, and not a save/write/read/restore per page. +/// +/// The per-page version is what this file did first, and it cannot distinguish the three things it +/// most needs to: a store immediately followed by a load of the SAME address returns the stored +/// value under real distinct SRAM, under an address mirror, and under a dirty line in any cache +/// covering L2MEM. The pattern being address-derived does not help, because the alias is written and +/// read through the alias. Cache residency is not excluded by scan length either: 192 pages touch +/// one cache line each, ~12 KiB in total, which fits in any plausible L1 and so is never evicted. +/// +/// Writing every page before reading any page fixes both. If 0x4FF80000 mirrors 0x4FF00000, the +/// later write lands on the earlier page and the read pass sees the WRONG pattern at one of them. +/// The distance between the two passes is 192 pages of traffic, which no L1 holds. +/// +/// This matters more than a tidier loop: `__heap_end` hands the upper span straight to an allocator, +/// so a mirror reported as RAM is silent heap corruption. +fn writePass() void { + var addr = l2mem_low; + while (addr < l2mem_high) : (addr += page) { + if (skipped(addr)) continue; + const p: *volatile u32 = @ptrFromInt(addr); + saved[(addr - l2mem_low) / page] = p.*; + p.* = pattern(addr); + } +} + +/// Read every page back, then put the original word back. Restoring in the same pass is safe: the +/// comparison for this page is already done, and a mirror has by now already been detected at +/// whichever of the two aliases was read second. +fn readPass(addr: u32) Kind { + if (skipped(addr)) return .skipped; + const p: *volatile u32 = @ptrFromInt(addr); + const got = p.*; + p.* = saved[(addr - l2mem_low) / page]; + return if (got == pattern(addr)) .ram else .dead; +} + +/// What a page turned out to be. Three outcomes, not two: a page this program refuses to write is +/// neither RAM nor dead, and folding "skipped" into "dead" is what made the first run of this +/// report `dead 0x00000000..0x4ff02000`, a range that does not exist. +const Kind = enum { + ram, + dead, + skipped, + + fn label(k: Kind) [*:0]const u8 { + return switch (k) { + .ram => "ram", + .dead => "dead", + .skipped => "skipped", + }; + } +}; + +/// Report a maximal run of pages that all behaved the same way. +fn flush(kind: Kind, start: u32, end: u32) void { + if (end <= start) return; + soc.rom.print("MARK MEM_RANGE %s 0x%08x..0x%08x %u KiB\r\n", .{ + kind.label(), start, end, (end - start) / 1024, + }); +} + +export fn zig_main() noreturn { + soc.rom.print("\r\nMARK MEM_BOOT probing L2MEM 0x%08x..0x%08x\r\n", .{ l2mem_low, l2mem_high }); + soc.rom.print("MARK MEM_SELF image occupies 0x%08x..0x%08x\r\n", .{ l2mem_low, selfEnd() }); + soc.rom.print("MARK MEM_ROMRSV rom .data 0x%08x..0x%08x (never written)\r\n", .{ rom_data_low, rom_data_high }); + soc.rom.print("MARK MEM_HEAP linker gives 0x%08x..0x%08x %u KiB\r\n", .{ + @as(u32, @intFromPtr(&__heap_start)), + @as(u32, @intFromPtr(&__heap_end)), + (@as(u32, @intFromPtr(&__heap_end)) - @as(u32, @intFromPtr(&__heap_start))) / 1024, + }); + + // Write every page first, read every page second. See `writePass` for why one pass cannot + // answer this question at all. + soc.rom.print("MARK MEM_PASS write\r\n", .{}); + writePass(); + soc.rom.print("MARK MEM_PASS read\r\n", .{}); + + // Runs are coalesced so the output is a map rather than 192 lines. Every page belongs to + // exactly one run, and every run is printed, so the ranges tile the array with no gaps - which + // is the property that makes the report checkable. + var run: Kind = .skipped; + var run_start: u32 = l2mem_low; + var addr: u32 = l2mem_low; + while (addr < l2mem_high) : (addr += page) { + const kind = readPass(addr); + if (kind != run) { + flush(run, run_start, addr); + run = kind; + run_start = addr; + } + } + flush(run, run_start, l2mem_high); + + // PSRAM, untrained. The question here is only "does the bus answer at all", not "is it + // distinct", so a single write-read-restore is the right shape - and it is expected to fault. + // The line is printed BEFORE the access so a hang is unambiguous. + soc.rom.print("MARK MEM_PSRAM probing 0x%08x (untrained, may hang)\r\n", .{psram_base}); + const pp: *volatile u32 = @ptrFromInt(psram_base); + const ps_saved = pp.*; + pp.* = pattern(psram_base); + const ps = pp.*; + pp.* = ps_saved; + soc.rom.print("MARK MEM_PSRAM read 0x%08x expect 0x%08x %s\r\n", .{ + ps, pattern(psram_base), if (ps == pattern(psram_base)) "answers".ptr else "absent".ptr, + }); + + soc.rom.print("MARK MEM_DONE\r\n", .{}); + while (true) {} +} + +export fn _start() linksection(".text.entry") callconv(.naked) noreturn { + asm volatile ( + \\ li t0, 1 << 13 + \\ csrs mstatus, t0 + \\ la sp, __stack_top + \\ mv fp, sp + \\ la t0, __bss_start + \\ la t1, __bss_end + \\ bgeu t0, t1, 2f + \\1: + \\ sw zero, 0(t0) + \\ addi t0, t0, 4 + \\ bltu t0, t1, 1b + \\2: + \\ j zig_main + ); +} + +pub const panic = std.debug.FullPanic(struct { + fn call(msg: []const u8, _: ?usize) noreturn { + soc.rom.print("MARK MEM_PANIC %s\r\n", .{msg.ptr}); + while (true) {} + } +}.call); diff --git a/examples/minimal.zig b/examples/minimal.zig index a70f0ef..bf93de8 100644 --- a/examples/minimal.zig +++ b/examples/minimal.zig @@ -15,7 +15,7 @@ pub const panic = std.debug.FullPanic(struct { const led: u6 = @intCast(config.led_pin); export fn zig_main() noreturn { - soc.gpio.configureOutput(led); + soc.gpio.configureOutput(led, .{}); while (true) { soc.gpio.setHigh(led); soc.rom.ets_delay_us(100_000); diff --git a/src/pardes/app.zig b/src/pardes/app.zig new file mode 100644 index 0000000..212e48f --- /dev/null +++ b/src/pardes/app.zig @@ -0,0 +1,318 @@ +//! pardes, as ESP32-P4 firmware. +//! +//! There is no operating system under this. `_start` is the reset entry the second-stage bootloader +//! jumps to, and this file is the entire platform: a heap, a millisecond clock, and UART0. +//! +//! ## Where the editor is +//! +//! Not in this package. `../02-pardes-code` compiles its core for riscv32-freestanding and emits +//! ONE object exporting the six C functions declared below; `-Dpardes` links it. The seam is a file +//! rather than a package dependency for a reason recorded at length in `build.zig`: declaring the +//! editor as a `build.zig.zon` path dependency nested its ~30-package graph under this one and +//! broke every build in this repo, including the ones that have nothing to do with it. +//! +//! The seam is deliberately **bytes in, bytes out**. Everything that needs to know what a cell is - +//! vaxis, the ANSI encoder, the input parser, the capability handshake - lives on the far side, +//! next to the vaxis it is built against. What crosses is a byte stream in each direction, which is +//! exactly what a serial line is, so this file has no opinion about terminals at all. +//! +//! ## Where the memory is +//! +//! Measured on this die by `examples/memprobe.zig`, not read off a datasheet: +//! +//! 0x4FF02000..0x4FF3F000 244 KiB .data/.bss/.stack live at the bottom of this +//! 0x4FF3F000..0x4FF40000 4 KiB mask ROM .data/.bss - untouchable, ets_printf needs it +//! 0x4FF40000..0x4FFC0000 512 KiB handed to the editor as its entire heap +//! +//! The 512 KiB arrives as `__heap_start`/`__heap_end` from the generated linker script, so those +//! addresses are written down in exactly one place. The editor owns that span outright: it is +//! passed in at init and this file never allocates from it. +//! +//! PSRAM is not used. The board has 32 MB fitted and it would make all of this comfortable, but +//! ESP-IDF's own ESP32-P4 implementation runs past a thousand lines - MPLL, MSPI clocking, pin +//! drive and DQS, CS timing, mode registers, a connectivity check, and an entire timing-calibration +//! subsystem - and the mask ROM offers only MMU mapping, no device init. Touching it untrained +//! faults and hangs the core, which `examples/memprobe.zig` demonstrates on purpose. + +const std = @import("std"); +const soc = @import("soc"); +const hal = @import("hal"); +const heapmod = @import("heap"); +const uart = @import("uart.zig"); + +// ------------------------------------------------------------------------------------- the ABI +// Seven functions, all `callconv(.c)`, all implemented in the linked object. This is the complete +// interface between this board and the editor, and it is deliberately bytes-and-memory only: the +// editor never learns what a UART is, and this file never learns what a cell is. + +/// How the editor emits bytes. Called with finished runs of ANSI, many times per frame. +const WriteFn = *const fn (ctx: ?*anyopaque, ptr: [*]const u8, len: usize) callconv(.c) void; + +/// This board's allocator, handed across as plain function pointers. `log2_align` is a log2 value, +/// which is exactly how `std.mem.Alignment` represents itself, so neither side needs a conversion +/// table. +/// +/// The memory belongs to THIS side: only the firmware knows that the heap is the 384 KiB at +/// 0x4FF40000, that the 128 KiB above it is L2 cache, and that PSRAM is untrained. The editor gets +/// an allocator, not an address range. +const Allocator = extern struct { + ctx: ?*anyopaque, + alloc: *const fn (ctx: ?*anyopaque, len: usize, log2_align: u8) callconv(.c) ?[*]u8, + resize: *const fn (ctx: ?*anyopaque, ptr: [*]u8, len: usize, log2_align: u8, new_len: usize) callconv(.c) bool, + free: *const fn (ctx: ?*anyopaque, ptr: [*]u8, len: usize, log2_align: u8) callconv(.c) void, +}; + +/// The one number both sides must agree on. Linkers do not type-check C symbols, so a signature +/// that drifts on one side of this seam links cleanly and then corrupts the stack; checking this +/// before calling anything else turns that into a refusal to boot. +const abi_version: u32 = 1; +extern fn pardes_p4_abi_version() callconv(.c) u32; + +/// Hand over the allocator and the output sink, and state the initial window size. Returns 0, or a +/// small non-zero code this file can only report. +extern fn pardes_p4_init( + alloc: *const Allocator, + write: WriteFn, + ctx: ?*anyopaque, + cols: u16, + rows: u16, +) callconv(.c) u32; + +/// Raw bytes off the wire: keystrokes, capability-query replies, and the host bridge's in-band +/// resize reports. The editor parses all three; this file distinguishes none of them. +extern fn pardes_p4_input(ptr: [*]const u8, len: usize) callconv(.c) void; + +/// Advance time. Separate from `input` because animations and timeouts must progress on a wire +/// where nothing is arriving. +extern fn pardes_p4_tick(now_ms: u64) callconv(.c) void; + +/// Emit one frame through the write callback. Returns 0 or an error code. +extern fn pardes_p4_render() callconv(.c) u32; + +/// Is there anything to draw - a dirty surface or a running animation? Asked every iteration so a +/// quiet editor costs no bytes on a 115200-baud link. +extern fn pardes_p4_wants_frame() callconv(.c) bool; + +/// Has the user asked to leave? There is nowhere to go, so this only stops the loop. +extern fn pardes_p4_quit() callconv(.c) bool; + +// ------------------------------------------------------------------------------------ the sink + +/// The write callback handed to `pardes_p4_init`. No context is needed - there is one UART. +fn writeOut(_: ?*anyopaque, ptr: [*]const u8, len: usize) callconv(.c) void { + uart.write(ptr[0..len]); +} + +// ------------------------------------------------------------------------------------- the heap + +/// The span the linker script hands over, from `l2high`'s ORIGIN and LENGTH. +/// +/// Reached with `@extern`, NOT with `extern const __heap_start: anyopaque` plus +/// `@intFromPtr`/`@ptrFromInt`. That spelling was here first and it was silently wrong: declaring a +/// linker symbol as an `anyopaque` OBJECT gives the optimiser a zero-sized object, so a pointer +/// derived from its address carries provenance for zero bytes, and ordinary (non-volatile) stores +/// through it are dead code it may drop. `examples/heapcheck.zig` caught it on the die - the +/// allocator's first block header read back as `size=2988759312 next=0xffffffff`-not, and the free +/// list walk never terminated. A `[*]u8` from `@extern` has no size to lose. +const heap_start = @extern([*]align(heapmod.Heap.granule) u8, .{ .name = "__heap_start" }); +const heap_end = @extern([*]align(heapmod.Heap.granule) u8, .{ .name = "__heap_end" }); + +fn heapSpan() []align(heapmod.Heap.granule) u8 { + return heap_start[0 .. @intFromPtr(heap_end) - @intFromPtr(heap_start)]; +} + +/// The one heap. A K&R coalescing free list over that span, validated on this die by +/// `examples/heapcheck.zig`: 512 blocks fill and free back to a single 393,216-byte block, a holed +/// arena still satisfies a 4 KiB request, and 20,000 random operations drain back to one block. +var gpa_heap: heapmod.Heap = undefined; + +// The four C forwarders the editor is handed. `log2_align` round-trips through +// `std.mem.Alignment`, whose representation IS the log2 value. + +fn cAlloc(_: ?*anyopaque, len: usize, log2_align: u8) callconv(.c) ?[*]u8 { + const a = gpa_heap.allocator(); + return a.vtable.alloc(a.ptr, len, @enumFromInt(log2_align), @returnAddress()); +} + +fn cResize(_: ?*anyopaque, ptr: [*]u8, len: usize, log2_align: u8, new_len: usize) callconv(.c) bool { + const a = gpa_heap.allocator(); + return a.vtable.resize(a.ptr, ptr[0..len], @enumFromInt(log2_align), new_len, @returnAddress()); +} + +fn cFree(_: ?*anyopaque, ptr: [*]u8, len: usize, log2_align: u8) callconv(.c) void { + const a = gpa_heap.allocator(); + a.vtable.free(a.ptr, ptr[0..len], @enumFromInt(log2_align), @returnAddress()); +} + +const editor_allocator: Allocator = .{ + .ctx = null, + .alloc = cAlloc, + .resize = cResize, + .free = cFree, +}; + +// ------------------------------------------------------------------------------------ the clock + +/// Milliseconds since boot, off the systimer - a 16 MHz counter (`hal/systimer.zig:31`), which is +/// the cheapest trustworthy clock on this chip. `read` returns null if the unit is not running, in +/// which case time simply does not advance and the editor stops animating; that is a better failure +/// than a clock that jumps. +fn nowMs() u64 { + const us = hal.systimer.micros(.unit0) orelse return 0; + return us / 1000; +} + +// ------------------------------------------------------------------------------------- the loop + +export fn zig_main() noreturn { + // The very first thing, through the TX FIFO directly rather than the mask ROM. Two independent + // output paths matter during bring-up: if this line is clean and `soc.rom.print` below is + // garbage, the fault is in the ROM path (or in something this image did to the ROM's statics); + // if this line is already garbage, the fault is before it, in the entry or the clocks. + uart.write("\r\nMARK PARDES_ENTRY direct-fifo\r\n"); + // Self-consistent probe: take the literal's OWN address at run time and dump both it and the + // bytes there. Comparing a runtime read against `llvm-objdump` of a DIFFERENT build is how this + // investigation wasted a cycle - every literal moves when the file changes. + const lit = "\r\nMARK PARDES_ENTRY direct-fifo\r\n"; + uart.writeByte('<'); + uart.dumpWord(@intFromPtr(lit.ptr)); // where the linker says the literal is + uart.dumpHex(@intFromPtr(lit.ptr), 8); // what a volatile read sees there + uart.writeByte('|'); + uart.write(lit); // what the ordinary slice path sends + uart.writeByte('>'); + uart.writeByte('\r'); + // Page 3 of .flash.rodata. The editor's allocator vtable lives at 0x40035a1c and a runtime load + // of its first entry returned instruction-looking garbage, while page 4 (the literal above, and + // the allocator struct this file passes over) reads correctly. So read page 3 raw and compare + // against llvm-objdump. + // Walk page 3 at 8 KiB steps. If a load at offset 0 is right and one at 0x5a1c is wrong, the + // aliasing granularity is FINER than the 64 KiB `tools/image.zig` assumes for congruence - which + // would mean the flashed bootloader was built with a smaller CONFIG_MMU_PAGE_SIZE (the P4's page + // size is configurable, and the image builder's congruence check is only as strong as the page + // size it believes in). Where the first mismatch falls names the real size. + // A CONTIGUOUS 192 bytes across a known-bad address. The image and the ELF agree here and the + // flash is MD5-verified against the image, so the wrong bytes are produced between the flash and + // the load. If the corruption comes in 64-byte chunks with correct data either side, it is cache + // lines; if it is a clean run of thousands of bytes, it is a mapping. + uart.dumpHex(0x4003_59c0, 64); + uart.dumpHex(0x4003_5a00, 64); + uart.dumpHex(0x4003_5a40, 64); + uart.dumpHex(0x4004_0000, 8); + uart.writeByte(']'); + uart.writeByte('\r'); + uart.writeByte('\n'); + uart.writeByte('\n'); + uart.write("MARK B1 entry ok\r\n"); + const heap = heapSpan(); + uart.write("MARK B2 heapSpan ok\r\n"); + soc.rom.print("\r\nMARK B3 rom.print heap 0x%08x..0x%08x %u KiB\r\n", .{ + @as(u32, @intFromPtr(heap.ptr)), + @as(u32, @intFromPtr(heap.ptr)) + @as(u32, @intCast(heap.len)), + @as(u32, @intCast(heap.len / 1024)), + }); + uart.write("MARK B4 rom.print returned\r\n"); + + const rwdt_was_armed = hal.rwdt.disable(); + uart.write("MARK B5 rwdt ok\r\n"); + hal.systimer.init(); + uart.write("MARK B6 systimer ok\r\n"); + _ = rwdt_was_armed; + + const their_abi = pardes_p4_abi_version(); + uart.write("MARK B7 abi call returned\r\n"); + if (their_abi != abi_version) { + uart.write("MARK PARDES_ABI_MISMATCH\r\n"); + while (true) {} + } + + gpa_heap = heapmod.Heap.init(heap); + uart.write("MARK B8 heap init ok\r\n"); + _ = uart.drainInput(); + uart.write("MARK B9 drain ok, calling pardes_p4_init\r\n"); + + const rc = pardes_p4_init(&editor_allocator, writeOut, null, 80, 24); + uart.write("MARK B10 pardes_p4_init returned\r\n"); + if (rc != 0) { + soc.rom.print("MARK PARDES_INIT_FAIL rc=%u\r\n", .{rc}); + const s = gpa_heap.stats(); + soc.rom.print("MARK PARDES_HEAP free=%u largest=%u blocks=%u\r\n", .{ + s.free, s.largest_free, s.free_blocks, + }); + while (true) {} + } + soc.rom.print("MARK PARDES_READY\r\n", .{}); + + var in: [256]u8 = undefined; + while (!pardes_p4_quit()) { + const n = uart.read(&in); + if (n > 0) pardes_p4_input(&in, n); + + pardes_p4_tick(nowMs()); + + // Only when there is something to show. On a link this slow an unconditional repaint per + // iteration would saturate the wire and starve input. + if (pardes_p4_wants_frame()) { + const err = pardes_p4_render(); + if (err != 0) soc.rom.print("MARK PARDES_RENDER_FAIL rc=%u\r\n", .{err}); + } + } + + soc.rom.print("\r\nMARK PARDES_QUIT\r\n", .{}); + while (true) {} +} + +// --------------------------------------------------------------------------- the root's own duties + +/// `page_size_min`/`max`: the board has no MMU and no pages, but std derives allocator alignment +/// from these. 4 KiB is the ESP32-P4's cache and DMA granularity. +/// +/// `logFn` is not cosmetic. std's default log implementation reaches `std.debug_io`, which +/// instantiates `std.Io.Threaded` - a thread pool, `getrandom`, `IOV_MAX`, `mremap` - none of which +/// exist here, and one `log.warn` from anywhere is enough to drag all of it into the image. +pub const std_options: std.Options = .{ + .page_size_min = 4096, + .page_size_max = 4096, + .logFn = logFn, +}; + +fn logFn( + comptime level: std.log.Level, + comptime scope: @EnumLiteral(), + comptime fmt: []const u8, + args: anytype, +) void { + var buf: [256]u8 = undefined; + const line = std.fmt.bufPrint(&buf, "\r\n[" ++ level.asText() ++ "/" ++ @tagName(scope) ++ "] " ++ fmt ++ "\r\n", args) catch + "\r\n[log overflow]\r\n"; + uart.write(line); +} + +pub const panic = std.debug.FullPanic(panicImpl); + +fn panicImpl(msg: []const u8, _: ?usize) noreturn { + // The ROM path deliberately: a panic may BE the console writer failing, and `ets_printf` shares + // nothing with `uart.write` except the FIFO itself. + soc.rom.print("\r\nMARK PARDES_PANIC %s\r\n", .{msg.ptr}); + while (true) {} +} + +/// Reset entry. The bootloader hands over with an unspecified stack pointer and the FPU off, so: +/// enable the F extension (`mstatus.FS`, which ESP-IDF only ever turns on lazily from a trap +/// handler this image does not have), establish a stack, clear `.bss`, and call into Zig. +export fn _start() linksection(".text.entry") callconv(.naked) noreturn { + asm volatile ( + \\ li t0, 1 << 13 + \\ csrs mstatus, t0 + \\ la sp, __stack_top + \\ mv fp, sp + \\ la t0, __bss_start + \\ la t1, __bss_end + \\ bgeu t0, t1, 2f + \\1: + \\ sw zero, 0(t0) + \\ addi t0, t0, 4 + \\ bltu t0, t1, 1b + \\2: + \\ j zig_main + ); +} diff --git a/src/pardes/uart.zig b/src/pardes/uart.zig new file mode 100644 index 0000000..ce386fe --- /dev/null +++ b/src/pardes/uart.zig @@ -0,0 +1,115 @@ +//! UART0 as the editor's terminal: bytes out, bytes in, and nothing else. +//! +//! This is the whole of the firmware's I/O. There is no framebuffer and no keyboard; the board +//! emits ANSI and consumes ANSI, and the terminal emulator on the far end of the CH340 does the +//! rest of the work - including answering the editor's own capability queries, which travel down +//! this wire like any other bytes. +//! +//! Deliberately not a `std.Io.Writer`. The ANSI encoding lives on the other side of the C ABI, next +//! to the vaxis that produces it (see `src/pardes/app.zig` for why the seam is there and not +//! elsewhere), so what crosses into this file is already a finished run of bytes. A writer here +//! would be a second buffer in front of one that already exists. +//! +//! Two decisions worth stating, because both are measurements rather than preferences. +//! +//! **Batched FIFO access.** The naive push is `while (txFree() == 0) {}` then `pushByte`, once per +//! byte: one MMIO read per byte at best, many while the FIFO is full. Reading `txFree` once and +//! then pushing that many cuts the status reads by up to the FIFO depth (128, `hal/uart.zig:52`). +//! At 115200 the wire costs ~86 us per byte and dwarfs either version, so today this is merely +//! free - and it stops being free the moment the divider is raised. +//! +//! **UART0's configuration is never touched.** Not the divider, not the format, not the pad +//! routing, and above all not `reset()`. The second-stage bootloader configured this block, and +//! `hal/uart.zig:195-211` records what happens if it is reset: UART_CLKDIV returns to its power-on +//! value, the console turns to garbage mid-sentence, and the board takes a watchdog reset with +//! nothing readable left to explain it. Everything here touches FIFO offset 0x000 and the status +//! register, and nothing else. + +const hal = @import("hal"); + +/// UART0: the instance the CH340 is wired to, and the one the ROM and bootloader configured. +const uart0 = hal.uart.Uart.init(0); + +/// Push `bytes` into the TX FIFO, blocking while it is full. +/// +/// The spin is bounded by the wire and there is nothing else for this core to do: a full 128-byte +/// FIFO drains in 11 ms at 115200. It is also the only backpressure in the system - dropping +/// instead would truncate an escape sequence, and a half-written SGR leaves the host terminal in +/// the wrong colour for the rest of the session. +pub fn write(bytes: []const u8) void { + var rest = bytes; + while (rest.len > 0) { + // One status read per burst, not per byte. + var room = uart0.txFree(); + while (room == 0) room = uart0.txFree(); + const n = @min(room, rest.len); + for (rest[0..n]) |b| uart0.pushByte(b); + rest = rest[n..]; + } +} + +/// One byte, for callers that must not touch `.rodata` to say anything - which during bring-up is +/// the difference between a diagnostic and a second copy of the bug being diagnosed. +pub fn writeByte(b: u8) void { + while (uart0.txFree() == 0) {} + uart0.pushByte(b); +} + +/// Emit `n` bytes read from `addr` as two hex digits each, computing the digits arithmetically so +/// nothing here reads a lookup table. Used to answer "does a load from this address return what the +/// linker put there", which is not a question a string literal can be trusted to ask. +pub fn dumpHex(addr: u32, n: u32) void { + const p: [*]const volatile u8 = @ptrFromInt(addr); + var i: u32 = 0; + while (i < n) : (i += 1) { + const byte = p[i]; + for ([2]u8{ byte >> 4, byte & 0xf }) |nib| { + writeByte(if (nib < 10) '0' + nib else 'a' + (nib - 10)); + } + } + writeByte('\r'); + writeByte('\n'); +} + +/// A u32 as eight hex digits, reading no memory at all. +pub fn dumpWord(v: u32) void { + var shift: u5 = 28; + while (true) { + const nib: u8 = @intCast((v >> shift) & 0xf); + writeByte(if (nib < 10) '0' + nib else 'a' + (nib - 10)); + if (shift == 0) break; + shift -= 4; + } + writeByte('\r'); + writeByte('\n'); +} + +/// Move whatever the host has sent into `buf`, without waiting. Returns the count. +/// +/// Non-blocking on purpose: the loop has a frame to render and a core to pump, and the editor must +/// not stall on a keystroke that may never come. `rxCount` is read once per call and the FIFO +/// drained to that mark, so a fast typist or a pasted buffer cannot hold the loop here. +pub fn read(buf: []u8) usize { + const waiting = @min(uart0.rxCount(), buf.len); + for (buf[0..waiting]) |*slot| slot.* = uart0.popByte(); + return waiting; +} + +/// Discard anything already received, returning how much. Used once at startup: the host-side +/// bridge injects a window-size report before this program exists, and the bootloader's chatter has +/// already been echoed at the host. Neither is user input. +/// +/// Pops rather than calling `resetRxFifo`, which is a CONF0_SYNC read-modify-write plus two commits +/// on the console UART - see this file's header. +pub fn drainInput() u32 { + var dropped: u32 = 0; + while (uart0.rxCount() > 0) : (dropped += 1) _ = uart0.popByte(); + return dropped; +} + +/// The rate the hardware is actually producing, by reading its dividers back. Reported rather than +/// assumed: the host has to be opened at the same rate, and a mismatch shows up as garbage on the +/// screen rather than as an error anyone can act on. +pub fn baudrate() u32 { + return uart0.baudrate(uart0.clockSource().nominalHz()); +} diff --git a/tools/console.zig b/tools/console.zig new file mode 100644 index 0000000..2e924e3 --- /dev/null +++ b/tools/console.zig @@ -0,0 +1,239 @@ +//! An interactive terminal on the far side of the serial port. +//! +//! `MonitorStep` prints the board's output for N seconds and never sends anything. That is the +//! right tool for a program that only reports. It is the wrong tool for a program the human is +//! supposed to *use*, which needs the host's keystrokes on the wire and the host's terminal out of +//! the way. So this is the other half: raw-mode stdin forwarded to the UART, UART forwarded to +//! stdout, until the escape byte. +//! +//! The division of labour is the interesting part. The board runs the application and emits ANSI; +//! the host terminal emulator (ghostty here) does the actual terminal work - fonts, scrollback, +//! selection - and answers the application's capability queries itself. This process is a wire, and +//! deliberately almost transparent: a `\x1b[?1049h` from the board reaches ghostty, ghostty's reply +//! to a `CSI c` reaches the board, and neither end needs to know there are 3 metres of USB cable +//! and a CH340 in between. +//! +//! Almost transparent, because of one thing a wire cannot pass through: **size**. A terminal +//! application learns its window size from `ioctl(TIOCGWINSZ)`, and firmware has no ioctl. DEC mode +//! 2048 ("in-band resize") solves half of it - a terminal that has been sent `\x1b[?2048h` reports +//! every subsequent resize as `CSI 48 ; rows ; cols ; ypix ; xpix t`, and those reports flow down +//! the wire like any other bytes. The half it does not solve is the *first* size, because a change +//! notification is only sent on a change. This process owns the real tty, so it is the only party +//! that can answer, and it injects that same sequence itself: +//! +//! * once at attach, for a board that never asks; +//! * on SIGWINCH, because the host's window changed and mode 2048 may not be enabled; +//! * and on seeing `\x1b[?2048h` come *back* from the board - which is the exact moment the +//! application has declared itself ready to understand one. That is the deterministic trigger; +//! the other two are belt and braces. +//! +//! Injecting a resize report the host terminal would also have sent is harmless: it carries the +//! size as data, so a duplicate is idempotent, and `vaxis`'s parser (`src/Parser.zig:528-551`) +//! treats both identically because they are byte-for-byte the same sequence. + +const std = @import("std"); +const posix = std.posix; +const linux = std.os.linux; +const serial = @import("serial.zig"); + +/// Ctrl-] , telnet's escape and not a key any full-screen application binds. Ctrl-C, Ctrl-Q and +/// Ctrl-Z all had to be rejected: an editor wants every one of them, and a bridge that swallowed +/// them would be lying about being transparent. +pub const escape_byte: u8 = 0x1d; + +/// Set by the SIGWINCH handler, read by the loop. `volatile` rather than atomic because a signal +/// handler on the same thread is not a concurrent writer - it is an interruption - and this only +/// needs the compiler to stop caching the load. +var winch_pending: bool = false; + +/// The handler takes `posix.SIG`, not an int: std's `Sigaction.handler_fn` is +/// `*align(1) const fn (SIG) callconv(.c) void` (std/os/linux.zig:6026). +fn onWinch(_: posix.SIG) callconv(.c) void { + @as(*volatile bool, &winch_pending).* = true; +} + +const Winsize = extern struct { row: u16, col: u16, xpixel: u16, ypixel: u16 }; +const TIOCGWINSZ = 0x5413; + +fn windowSize() Winsize { + var ws: Winsize = .{ .row = 24, .col = 80, .xpixel = 0, .ypixel = 0 }; + // A failure here is not fatal: 80x24 is a defensible terminal, and the alternative is refusing + // to attach because the size could not be read. + _ = linux.ioctl(0, TIOCGWINSZ, @intFromPtr(&ws)); + if (ws.row == 0) ws.row = 24; + if (ws.col == 0) ws.col = 80; + return ws; +} + +/// The in-band resize report, in the form `vaxis` parses: `CSI 48 ; rows ; cols ; ypix ; xpix t`. +/// Note the pixel fields are height-then-width, which is the opposite order from the +/// `struct winsize` they come out of - `Parser.zig:536-539` reads height first. +fn sendWinsize(port: *serial.Port, ws: Winsize) void { + var buf: [64]u8 = undefined; + const seq = std.fmt.bufPrint(&buf, "\x1b[48;{d};{d};{d};{d}t", .{ + ws.row, ws.col, ws.ypixel, ws.xpixel, + }) catch return; + port.write(seq) catch {}; +} + +/// Matches `\x1b[?2048h` in the board's output stream one byte at a time, because the sequence can +/// be split across reads. Returns true on the byte that completes it. +const ModeWatch = struct { + const want = "\x1b[?2048h"; + at: usize = 0, + + fn feed(m: *ModeWatch, byte: u8) bool { + if (byte == want[m.at]) { + m.at += 1; + if (m.at == want.len) { + m.at = 0; + return true; + } + } else { + // Restart, and allow this byte to be a fresh start - otherwise "\x1b\x1b[?2048h" is + // missed. + m.at = if (byte == want[0]) 1 else 0; + } + return false; + } +}; + +pub const Options = struct { + /// Pulse reset so the application starts from boot with the console already attached. Without + /// it, attaching to a board that has been running for a while shows a screen mid-session with + /// no redraw until something changes. + reset: bool = true, + /// Print the escape-key hint. Suppressed for scripted runs, whose output is being asserted on. + banner: bool = true, +}; + +/// Forward bytes both ways until the escape byte arrives on stdin. +/// +/// Returns normally on escape; the terminal is always restored, including on error, because the +/// alternative is handing the user back a shell with no echo. +pub fn attach(port_path: []const u8, baud: serial.Baud, opts: Options) !void { + var port = try serial.Port.open(port_path, baud); + defer port.close(); + + // No `isatty`: `tcgetattr` answers the same question with the same syscall this needs anyway, + // and a null here means "stdin is a pipe" - which is a supported way to run this, for scripted + // sessions whose input is a file. + const saved: ?posix.termios = posix.tcgetattr(0) catch null; + if (saved) |prev| { + var raw = prev; + // The same raw mode `serial.Port.open` builds for the port, for the same reason: every byte + // the user types has to reach the board unmodified, including the ones the line discipline + // would otherwise interpret. ISIG off is what lets Ctrl-C reach the application instead of + // killing this process. + raw.lflag.ICANON = false; + raw.lflag.ECHO = false; + raw.lflag.ISIG = false; + raw.lflag.IEXTEN = false; + raw.iflag.IXON = false; + raw.iflag.ICRNL = false; + raw.iflag.INLCR = false; + raw.iflag.BRKINT = false; + raw.oflag.OPOST = false; + try posix.tcsetattr(0, .FLUSH, raw); + } + defer if (saved) |prev| posix.tcsetattr(0, .FLUSH, prev) catch {}; + + // Installed after raw mode so a resize during setup cannot be missed-but-flagged. + posix.sigaction(posix.SIG.WINCH, &.{ + .handler = .{ .handler = onWinch }, + .mask = posix.sigemptyset(), + .flags = 0, + }, null); + + // stdin/stdout as `std.Io.File`, because std 0.16 has no `posix.read`/`posix.write` any more - + // byte traffic goes through std.Io. The `io` is borrowed from the port, which already holds the + // single-threaded instance `serial.Port.open` created. + const io = port.io; + const stdin: std.Io.File = .{ .handle = 0, .flags = .{ .nonblocking = false } }; + const stdout: std.Io.File = .{ .handle = 1, .flags = .{ .nonblocking = false } }; + + if (opts.banner) { + var hint: [96]u8 = undefined; + const line = std.fmt.bufPrint(&hint, "[zig-p4 console @ {d} baud - Ctrl-] to detach]\r\n", .{ + baud.rate(), + }) catch "[zig-p4 console - Ctrl-] to detach]\r\n"; + stdout.writeStreamingAll(io, line) catch {}; + } + + if (opts.reset) try port.resetToRun(.{}); + sendWinsize(&port, windowSize()); + + var watch: ModeWatch = .{}; + var from_board: [4096]u8 = undefined; + var from_user: [256]u8 = undefined; + + while (true) { + if (@as(*volatile bool, &winch_pending).*) { + @as(*volatile bool, &winch_pending).* = false; + sendWinsize(&port, windowSize()); + } + + var pfd = [_]posix.pollfd{ + .{ .fd = 0, .events = posix.POLL.IN, .revents = 0 }, + .{ .fd = port.file.handle, .events = posix.POLL.IN, .revents = 0 }, + }; + // A bounded wait rather than an infinite one so a SIGWINCH that lands between the check + // above and the poll below is still serviced promptly; poll reports the signal itself as + // an interrupt, which is handled as "go round again". + const ready = posix.poll(&pfd, 200) catch continue; + if (ready == 0) continue; + + // POLL.IN is not the only thing poll reports, and ignoring the rest is a hot spin, not a + // no-op: unplug the CH340 mid-session and the port's revents carries HUP|ERR|NVAL forever. + // poll then returns immediately with a non-zero count, neither branch below matches because + // both test POLL.IN, and the loop burns a core with stdin still in raw mode and ISIG off. + // `std.posix.poll` cannot surface it as an error either - it maps INVAL to `unreachable` + // (std/posix.zig:1007-1017) because a dead descriptor is reported in `revents`, not errno. + const gone = posix.POLL.HUP | posix.POLL.ERR | posix.POLL.NVAL; + if (pfd[1].revents & gone != 0) return error.PortDisconnected; + // stdin dying is ordinary: a pipe ran out, or the terminal closed. Detach quietly. + if (pfd[0].revents & gone != 0) return; + + if (pfd[1].revents & posix.POLL.IN != 0) { + const n = port.read(&from_board) catch 0; + if (n > 0) { + stdout.writeStreamingAll(io, from_board[0..n]) catch {}; + for (from_board[0..n]) |b| { + if (watch.feed(b)) sendWinsize(&port, windowSize()); + } + } + } + + if (pfd[0].revents & posix.POLL.IN != 0) { + const n = stdin.readStreaming(io, &.{&from_user}) catch 0; + if (n == 0) return; // stdin closed: a pipe ran out, so detach + if (std.mem.indexOfScalar(u8, from_user[0..n], escape_byte)) |cut| { + // Everything before the escape still belongs to the board. + if (cut > 0) port.write(from_user[0..cut]) catch {}; + if (opts.banner) stdout.writeStreamingAll(io, "\r\n[detached]\r\n") catch {}; + return; + } + port.write(from_user[0..n]) catch {}; + } + } +} + +test "ModeWatch completes only on the full sequence" { + var m: ModeWatch = .{}; + for ("\x1b[?2048") |b| try std.testing.expect(!m.feed(b)); + try std.testing.expect(m.feed('h')); +} + +test "ModeWatch resynchronises on a false start" { + var m: ModeWatch = .{}; + // A prefix that dies, then the real thing immediately after: the naive reset-to-zero misses + // this because the byte that broke the match is itself the next match's ESC. + for ("\x1b[?20") |b| try std.testing.expect(!m.feed(b)); + for ("\x1b[?2048") |b| try std.testing.expect(!m.feed(b)); + try std.testing.expect(m.feed('h')); +} + +test "ModeWatch ignores unrelated traffic" { + var m: ModeWatch = .{}; + for ("hello \x1b[?1049h world \x1b[0m") |b| try std.testing.expect(!m.feed(b)); +} diff --git a/tools/image.zig b/tools/image.zig index dea843f..ba60064 100644 --- a/tools/image.zig +++ b/tools/image.zig @@ -13,7 +13,10 @@ //! ships as a 1.4 KB image instead of a 66 KB one. Verified on ESP32-P4 rev v1.3 silicon. //! //! Rules the loader enforces, each learned by flashing a deliberately broken image at the board: -//! * exactly two segments must land in the mapped range (bootloader_utility.c:842) +//! * exactly two segments must land in the mapped range: on a chip with shared D/I external +//! vaddr (the P4, soc.h:146-149) the loader collects them positionally and asserts +//! rom_index == 2 (bootloader_utility.c:805-851); one segment aborts the boot and a third +//! trips an assert inside the loop (bootloader_utility.c:842) //! * every segment length must be a multiple of 4 (esp_image_format.c:857) //! * image offset 0x20 begins an esp_app_desc_t, and min/max_efuse_blk_rev_full are read from //! it whether or not it is really a descriptor (esp_image_format.c:796-806) @@ -116,6 +119,14 @@ pub const Layout = struct { } off += seg_header_len + s.len; } + // EXACTLY two, and the bootloader is what says so. On a chip whose D/I external vaddr ranges + // are shared - the P4's are (soc.h:146-149) - ESP-IDF takes the SOC_MMU_DI_VADDR_SHARED + // branch of `unpack_load_app` (bootloader_utility.c:805-851), which does not classify + // segments as D or I at all: it collects them positionally into rom_addr[2] and ends with + // `assert(rom_index == 2)`. One mapped segment aborts the boot with + // "Assert failed in unpack_load_app, bootloader_utility.c:842 (rom_index == 2)" - measured, + // by shipping one - and a third trips `assert(rom_index < 2)` inside the loop. Enforcing it + // here turns a boot-time abort into a build-time error. if (mapped != 2) return error.NotTwoMappedSegments; // One MMU entry per vaddr page: any two mapped segments in the same vaddr page must come diff --git a/tools/image_test.zig b/tools/image_test.zig index 9f2cbc0..c13adf7 100644 --- a/tools/image_test.zig +++ b/tools/image_test.zig @@ -120,6 +120,8 @@ test "an image with one mapped segment is rejected before it can brick a board" var layout = try image.fromElf(gpa, elf, .{}); defer layout.deinit(gpa); + // Not a guess: shipping a one-segment image aborted the boot with + // "Assert failed in unpack_load_app, bootloader_utility.c:842 (rom_index == 2)". try testing.expectError(error.NotTwoMappedSegments, layout.validate(.{})); } @@ -275,6 +277,7 @@ fn checkBytes(bytes: []const u8, flash_offset: u32) !void { } off += 8 + len; } + // Exactly two: the SOC_MMU_DI_VADDR_SHARED branch asserts rom_index == 2. try testing.expectEqual(@as(usize, 2), mapped); // bootloader_utility.c:842 // Two mapped segments sharing a vaddr page must share the flash page: one MMU entry each. diff --git a/tools/serial.zig b/tools/serial.zig index c13942c..da225fd 100644 --- a/tools/serial.zig +++ b/tools/serial.zig @@ -43,6 +43,10 @@ const Termios2 = extern struct { /// _IOR('T', 0x2A, struct termios2) and _IOW('T', 0x2B, struct termios2) for a 44-byte struct. const TCGETS2: u32 = 0x802C542A; const TCSETS2: u32 = 0x402C542B; + /// _IOW('T', 0x2C, ...): TCSETS2's draining sibling. Changing the divisor while bytes are + /// still in the kernel's output queue sends the tail of the old line at the new rate, which on + /// a mid-session baud switch corrupts exactly the handshake line the switch was announced by. + const TCSETSW2: u32 = 0x402C542C; /// CBAUD escape meaning "take the rate from ispeed/ospeed" (asm-generic/termbits.h). const BOTHER: u32 = 0o010000; @@ -110,6 +114,26 @@ pub const Port = struct { return .{ .file = file, .io = io, .saved = saved }; } + /// Re-rate an already-open port, leaving the raw-mode flags and the exclusive claim alone. + /// + /// This exists because the console and the flasher want different rates on the same wire. The + /// ROM loader auto-detects the host's rate from SYNC's 0x55 pattern, so `open` can simply pick + /// one; a running application cannot, because its UART divider was programmed by the + /// second-stage bootloader and the only way to change it is for the firmware to reprogram its + /// own divider and for the host to follow. That makes the switch a two-sided handshake, and + /// `TCSETSW2` is the half that has to be ordered: the firmware announces the change on the old + /// rate, and the announcement must have physically left the wire before either side moves. + pub fn setBaud(p: *Port, baud: Baud) !void { + var t: Termios2 = undefined; + if (@as(isize, @bitCast(linux.ioctl(p.file.handle, Termios2.TCGETS2, @intFromPtr(&t)))) < 0) + return error.NotATerminal; + t.cflag = (t.cflag & ~Termios2.CBAUD) | Termios2.BOTHER; + t.ispeed = baud.rate(); + t.ospeed = baud.rate(); + if (@as(isize, @bitCast(linux.ioctl(p.file.handle, Termios2.TCSETSW2, @intFromPtr(&t)))) < 0) + return error.SetAttrFailed; + } + pub fn close(p: *Port) void { _ = linux.ioctl(p.file.handle, Termios2.TCSETS2, @intFromPtr(&p.saved)); p.file.close(p.io); |
