//! The ESP32-P4 firmware shell: pardes as one freestanding object, bytes in and bytes out. //! //! `zig build -Dplatform=esp32p4` emits this file as a single object exporting the C //! ABI below; the sibling `05-zig-p4` toolchain links it beside `_start`, its generated linker script, //! and its UART driver. Nothing here knows what a UART is. //! //! **Why an object and not a module.** The obvious arrangement was for zig-p4 to declare this //! package in its `build.zig.zon` and import `pardes_p4`. That was built, and it broke every build //! in that repo: nesting this package's ~30-package graph under one whose own claim is "host //! dependencies: Zig, that is the whole list" made `std/Build.zig:2091` exceed its 1000-branch //! comptime quota (through ghostty's `SharedDeps.zig:874` `lazyImport`), dragged in seven cached //! tree-sitter versions whose `build.zig` uses APIs removed in 0.16, and materialised 2.6 GB across //! 42,736 files into that repo's working copy. A linked object has none of those properties and one //! extra virtue: the seam is bytes, so neither side can accidentally depend on the other's types. //! //! **Where the terminal is.** On the host. The board writes ANSI and reads ANSI; the terminal //! emulator at the far end of the serial line does the font rendering, and answers this program's //! own capability queries. That is why `vaxis` works here unmodified: `Vaxis.render`, //! `queryTerminalSend` and `enableDetectedFeatures` all take a bare `*std.Io.Writer` //! (`Vaxis.zig:375,278,329`), so the transport is a parameter. `vaxis.Tty` and `vaxis.Loop` are //! termios/ioctl/SIGWINCH bound and are not used. //! //! **Where the memory is.** Not here either. The firmware measured its own RAM (240 KiB low, //! 384 KiB high, and a 128 KiB region that turned out to be L2 cache) and owns the allocator; this //! file receives four function pointers and rebuilds a `std.mem.Allocator` from them. Everything //! the editor allocates comes from there. //! //! **Window size** arrives as DEC mode 2048 in-band resize reports, parsed by `vaxis.Parser` like //! any other input. Firmware has no `TIOCGWINSZ`, so the host-side bridge synthesises the first one. const std = @import("std"); const builtin = @import("builtin"); const pardes = @import("pardes.zig"); const vaxis = @import("vaxis"); // ------------------------------------------------------------------ what a freestanding root owes // // These are ROOT-module declarations: std reads them off whichever file is the compilation root, and // as of the build change that emits this file as the object, that is this file. They are not // ceremony - each one was discovered by the build failing without it. /// The board has no MMU and no pages, but std derives allocator alignment from these two. 4 KiB is /// the ESP32-P4's cache and DMA granularity. Without them: "riscv32-freestanding has unknown /// page_size_min" from std/heap.zig:48. /// /// `logFn` is the load-bearing one. std's DEFAULT log implementation reaches `std.debug_io`, which /// instantiates `std.Io.Threaded` - a thread pool, `getrandom`, `IOV_MAX`, `mremap` - none of which /// exist on this target, and ONE `log.warn` anywhere in the core or in vaxis is enough to drag the /// whole thing in and fail the build with "no member named 'getrandom'". pub const std_options: std.Options = .{ .page_size_min = 4096, .page_size_max = 4096, .logFn = logFn, }; /// Logs go out the same byte sink as the frames, which is the only sink there is. Truncated rather /// than allocated: a log line is never worth an allocation on a 384 KiB heap, and a logger that can /// fail on OOM is a logger that disappears exactly when it is needed. fn logFn( comptime level: std.log.Level, comptime scope: @EnumLiteral(), comptime fmt: []const u8, args: anytype, ) void { if (out_ctx == null and @intFromPtr(out_write) == 0) return; var buf: [256]u8 = undefined; const line = std.fmt.bufPrint( &buf, "\r\n[" ++ level.asText() ++ "/" ++ @tagName(scope) ++ "] " ++ fmt ++ "\r\n", args, ) catch "\r\n[log truncated]\r\n"; out_write(out_ctx, line.ptr, line.len); } pub const panic = std.debug.FullPanic(panicImpl); /// A panic here cannot unwind and has nowhere to go, so it reports through the write callback and /// stops. `@trap` and not a spin: the firmware's own panic handler prints through the mask ROM, /// which shares nothing with this path but the FIFO, so a trap leaves that diagnostic route intact. fn panicImpl(msg: []const u8, _: ?usize) noreturn { const prefix = "\r\nMARK PARDES_CORE_PANIC "; out_write(out_ctx, prefix.ptr, prefix.len); out_write(out_ctx, msg.ptr, msg.len); out_write(out_ctx, "\r\n", 2); @trap(); } // ---------------------------------------------------------------------------------- the C ABI // // Deliberately tiny, and versioned. Linkers do not type-check C symbols, so a signature that drifts // on one side of this seam links cleanly and then corrupts the stack. `pardes_esp32p4_abi_version` is the // cheapest possible defence: the firmware calls it first and refuses to continue on a mismatch. /// Bumped whenever any signature below changes, including a type. /// 2 added `GpioFn` to `pardes_esp32p4_init`. A firmware built against 1 passes five arguments where six /// are read, which is exactly the silent-corruption case this counter exists to turn into a message. const abi_version: u32 = 2; export fn pardes_esp32p4_abi_version() callconv(.c) u32 { return abi_version; } /// The firmware's allocator, as C function pointers. `alignment` is a log2 value, matching /// `std.mem.Alignment`'s own representation, so no translation table is needed. /// /// `remap` is absent on purpose: this allocator cannot move a block without copying it, so /// `std.mem.Allocator`'s remap is implemented locally as "resize in place, or fail" and the caller's /// own alloc/copy/free path handles the rest. pub const Allocator = extern struct { ctx: ?*anyopaque, alloc: *const fn (ctx: ?*anyopaque, len: usize, log2_align: u8) callconv(.c) ?[*]u8, resize: *const fn (ctx: ?*anyopaque, ptr: [*]u8, len: usize, log2_align: u8, new_len: usize) callconv(.c) bool, free: *const fn (ctx: ?*anyopaque, ptr: [*]u8, len: usize, log2_align: u8) callconv(.c) void, }; /// How finished runs of ANSI leave this object. pub const WriteFn = *const fn (ctx: ?*anyopaque, ptr: [*]const u8, len: usize) callconv(.c) void; /// Flip one pad and report the level before and after; false if the firmware declines. OPTIONAL on /// the wire, so a host with no pads (or one that has not implemented them yet) passes null and the /// `Gpio` word answers "no pads" instead of the object having to know which firmwares exist. /// /// The board's side, not the editor's, because a correct toggle is the IO MUX, the GPIO matrix, the /// pad's own bits and the output enable - four register files behind a per-pin table that the /// firmware already has and checks against ESP-IDF. See `Host.VTable.gpio_toggle`. pub const GpioFn = *const fn (ctx: ?*anyopaque, pin: u16, was: *u8, now: *u8) callconv(.c) bool; // ------------------------------------------------------------------- the allocator, rebuilt // One `std.mem.Allocator` whose vtable forwards to the four pointers above. The indirection is the // price of the seam and it is paid once per allocation, which on a first-fit heap is already the // cheap part (measured on the die: 8,229 cycles for one allocation across 257 free blocks). var host_alloc: Allocator = undefined; fn hostAlloc(_: *anyopaque, len: usize, alignment: std.mem.Alignment, _: usize) ?[*]u8 { return host_alloc.alloc(host_alloc.ctx, len, @intFromEnum(alignment)); } fn hostResize(_: *anyopaque, mem: []u8, alignment: std.mem.Alignment, new_len: usize, _: usize) bool { return host_alloc.resize(host_alloc.ctx, mem.ptr, mem.len, @intFromEnum(alignment), new_len); } fn hostRemap(_: *anyopaque, mem: []u8, alignment: std.mem.Alignment, new_len: usize, _: usize) ?[*]u8 { return if (host_alloc.resize(host_alloc.ctx, mem.ptr, mem.len, @intFromEnum(alignment), new_len)) mem.ptr else null; } fn hostFree(_: *anyopaque, mem: []u8, alignment: std.mem.Alignment, _: usize) void { host_alloc.free(host_alloc.ctx, mem.ptr, mem.len, @intFromEnum(alignment)); } const host_vtable: std.mem.Allocator.VTable = .{ .alloc = hostAlloc, .resize = hostResize, .remap = hostRemap, .free = hostFree, }; /// `ptr` is never dereferenced - the four forwarders read the file-scope `host_alloc` - but /// `std.mem.Allocator` requires a non-null context, so it points at the record itself. fn gpa() std.mem.Allocator { return .{ .ptr = @ptrCast(&host_alloc), .vtable = &host_vtable }; } // ------------------------------------------------------------------------------- the ANSI sink // A `std.Io.Writer` over the firmware's write callback. Buffered, because vaxis emits a frame as a // long run of small writes - cursor move, SGR run, grapheme, repeat - and an unbuffered writer would // make a C call per fragment. var out_write: WriteFn = undefined; var out_ctx: ?*anyopaque = null; var host_gpio: ?GpioFn = null; var out_buf: [8192]u8 = undefined; var out: std.Io.Writer = undefined; fn drain(w: *std.Io.Writer, data: []const []const u8, splat: usize) std.Io.Writer.Error!usize { // The shape std documents at Io/Writer.zig:46-63: buffer first, then every slice of `data`, with // the LAST slice repeated `splat` times, and the count returned excluding the buffered bytes. if (w.end > 0) { out_write(out_ctx, w.buffer.ptr, w.end); w.end = 0; } const head = data[0 .. data.len - 1]; const pattern = data[head.len]; var written: usize = 0; for (head) |bytes| { if (bytes.len > 0) out_write(out_ctx, bytes.ptr, bytes.len); written += bytes.len; } var i: usize = 0; while (i < splat) : (i += 1) { if (pattern.len > 0) out_write(out_ctx, pattern.ptr, pattern.len); } return written + pattern.len * splat; } // ------------------------------------------------------------------------------------ the state var core: ?*pardes.Pardes = null; var vx: vaxis.Vaxis = undefined; var parser: vaxis.Parser = .{}; /// vaxis wants an environment map. There is no environment; an empty one is the honest answer and /// the only thing vaxis reads it for is TERM-derived heuristics, which the capability queries /// supersede. var env_map: std.process.Environ.Map = undefined; /// Input that arrived mid-sequence. An escape sequence can be split across UART reads, and the /// parser reports "incomplete" by consuming nothing, so the tail has to survive until more arrives. var in_buf: [1024]u8 = undefined; var in_len: usize = 0; /// Bracketed paste: between the markers, keys are DATA and never commands. var paste_buf: std.ArrayListUnmanaged(u8) = .empty; var in_paste: bool = false; /// Set by anything that could change the screen; cleared by a render. The firmware asks before /// rendering, because on a 115200-baud link an unconditional repaint per loop saturates the wire and /// starves input. var dirty: bool = true; /// The largest grid this board can render, and the reason it is not just the host's terminal size. /// /// A cell is paid for TWICE now, not four times: pardes keeps its `Surface` and this shell keeps a /// shadow copy of it to diff against. vaxis used to keep a `Screen` and an `InternalScreen` as well, /// and with `direct_emit` neither is ever read - the emitter diffs against the Surface and writes /// the wire itself - so `init` sizes vaxis to a single cell and those two grids cost nothing. /// /// Set with `-Desp32p4-cols` / `-Desp32p4-rows`, because the ceiling is a measurement rather than a constant /// and it moves for two independent reasons: the 384 KiB heap, and the round trip, which grows with /// the cell count because every frame walks the whole grid. See the geometry table in /// `05-zig-p4/experiments/report.typ` for both curves. /// /// Raising these further is what PSRAM would buy: this board has 32 MB fitted and untrained. pub const max_cols: u16 = @import("pardes_config").esp32p4_cols; pub const max_rows: u16 = @import("pardes_config").esp32p4_rows; var cur_winsize: vaxis.Winsize = .{ .rows = max_rows, .cols = max_cols, .x_pixel = 0, .y_pixel = 0 }; /// How big vaxis's own grids need to be. /// /// ONE CELL under `direct_emit`, because neither of them is ever read: vaxis keeps a `Screen` and an /// `InternalScreen`, and the emitter diffs the Surface against its own shadow and writes the escapes /// itself. Those two grids were the largest single claim on a 384 KiB heap and the reason the board /// was held to 40x12 - the comment above `max_cols` used to say a cell was paid for four times over, /// and this is what took it down to two. vaxis is still doing the work only it can do here: entering /// the alternate screen, asking the terminal what it is, and parsing everything that comes back. fn vaxisSize() vaxis.Winsize { return if (direct_emit) .{ .rows = 1, .cols = 1, .x_pixel = 0, .y_pixel = 0 } else cur_winsize; } // -------------------------------------------------------------------------------------- exports /// Hand over the allocator and the output sink, state the initial window size, and bring the editor /// up. Returns 0, or a small non-zero code the firmware can only report. export fn pardes_esp32p4_init( alloc: *const Allocator, write: WriteFn, gpio: ?GpioFn, ctx: ?*anyopaque, cols: u16, rows: u16, ) callconv(.c) u32 { host_alloc = alloc.*; out_write = write; host_gpio = gpio; out_ctx = ctx; out = .{ .vtable = &.{ .drain = drain }, .buffer = &out_buf }; const a = gpa(); env_map = .{ .array_hash_map = .empty, .allocator = a }; // Clamped, so a firmware asking for more than the heap affords still starts. See `max_cols`. cur_winsize = .{ .rows = @min(rows, max_rows), .cols = @min(cols, max_cols), .x_pixel = 0, .y_pixel = 0, }; const allocs = pardes.memory.init(a); // `std.Io.failing` and not a real Io: every path in the core that would perform I/O is behind // the Host vtable, and the ones that are not are the ones this platform does not have. pardes.image.start(std.Io.failing, allocs.image); pardes.syntax.start(allocs.tree_sitter); vx = vaxis.init(std.Io.failing, a, &env_map, .{}) catch |err| return errCode(err); vx.resize(a, &out, vaxisSize()) catch |err| return errCode(err); // Ask the terminal what it is. Both halves are pure byte writers, which is the whole reason this // works over a serial line: the replies arrive as ordinary input and are parsed like any key. vx.enterAltScreen(&out) catch |err| return errCode(err); vx.queryTerminalSend(&out) catch |err| return errCode(err); // MOUSE REPORTING, spelled out here rather than taken from `vx.setMouseMode`. // // vaxis enables `1002;1003;1004;1006`, and 1003 is ANY-MOTION tracking: the terminal reports // every cell the pointer crosses with no button held. On a 115200 line that is unaffordable - // one sweep across this grid is dozens of reports of ~15 bytes each, and each one arrives as // input that the editor must parse while it is trying to paint. Worse, it arrives whether or not // anybody wants it, so moving the mouse over the window would starve typing. // // 1002 reports presses, releases and motion WHILE A BUTTON IS HELD, which is exactly the set a // click and a drag-select need. 1004 is focus in/out, which `apply` already handles. 1006 is the // SGR encoding: unlike the original X10 form it is not limited to column 223, which a grid this // small does not need today but costs nothing to have and cannot be added later without the // terminal disagreeing with the editor about where the pointer is. out.writeAll("\x1b[?1002;1004;1006h") catch |err| return errCode(err); out.flush() catch |err| return errCode(err); // The CLAMPED geometry, because the core and vaxis must agree on the grid and vaxis was just // sized to `cur_winsize`. core = pardes.Pardes.init(allocs.pardes, .{ .cols = cur_winsize.cols, .rows = cur_winsize.rows, .frame_allocator = allocs.frame, .image_allocator = allocs.image, .tree_sitter_allocator = allocs.tree_sitter, }) catch |err| return errCode(err); dirty = true; return 0; } /// Raw bytes off the wire: keystrokes, capability replies, and in-band resize reports. All three are /// the same kind of thing to `vaxis.Parser`, and this function does not distinguish them. export fn pardes_esp32p4_input(ptr: [*]const u8, len: usize) callconv(.c) void { const c = core orelse return; // Append, dropping the oldest on overflow: a full buffer means the parser is stuck on a // malformed sequence, and keeping the tail is what lets it resynchronise. const room = in_buf.len - in_len; const take = @min(room, len); if (take < len) { in_len = 0; @memcpy(in_buf[0..@min(len, in_buf.len)], ptr[0..@min(len, in_buf.len)]); in_len = @min(len, in_buf.len); } else { @memcpy(in_buf[in_len..][0..take], ptr[0..take]); in_len += take; } drainInput(c, false); } /// Parse what has accumulated, applying every event it yields. /// /// THE LONE ESCAPE IS AMBIGUOUS, and on this transport it is ambiguous constantly. `vaxis.Parser` /// resolves a buffer containing nothing but `0x1b` as the Escape KEY - deliberately, and correctly /// for a real terminal, where the kernel hands over a whole escape sequence in one read so a solitary /// ESC really does mean the key. A 115200 serial line hands over one byte at a time: 87 us apart, /// which is an eternity to this loop. So the first byte of EVERY escape sequence arrived alone and /// was resolved as Escape, and the rest arrived as ordinary keys. /// /// That is not a mouse bug, though it is why the mouse did not work: a click report came through as /// ten key presses - Escape, `[`, `<`, `0`, ... - and the `0` among them is "go to column zero" in /// normal mode, which is exactly where the cursor kept landing. Arrow keys, function keys and the /// host's in-band resize reports were all being shredded the same way. /// /// Longer partial sequences were never affected: the CSI scanner returns `n == 0` for "no final byte /// yet", and the loop below keeps those bytes. Only the one-byte case needed an answer, because it is /// the only one the parser answers wrongly instead of declining. /// /// `force` is how a real Escape keypress still works: `pardes_esp32p4_tick` calls with it set once the /// hold has lasted longer than any serial line would take to deliver the next byte. fn drainInput(c: *pardes.Pardes, force: bool) void { var off: usize = 0; while (off < in_len) { if (!force and in_len - off == 1 and in_buf[off] == 0x1b) break; const res = parser.parse(in_buf[off..in_len], gpa()) catch break; if (res.n == 0) break; // incomplete: wait for more bytes off += res.n; if (res.event) |ev| apply(c, ev); } // Keep whatever was not consumed: the tail of a split escape sequence. if (off > 0) { std.mem.copyForwards(u8, in_buf[0 .. in_len - off], in_buf[off..in_len]); in_len -= off; } // Start or clear the hold. `esc_held_at` is only ever set for a buffer that is exactly one ESC, // so a partial CSI - which the parser already declines - does not start a timer it does not need. if (in_len == 1 and in_buf[0] == 0x1b) { if (esc_held_at == null) esc_held_at = last_now_ms; } else esc_held_at = null; } /// One parsed vaxis event applied to the core. Mirrors the tty shell's `apply` /// (`src/tty/tty.zig:926-985`), minus everything that needs an OS. fn apply(c: *pardes.Pardes, ev: vaxis.Event) void { switch (ev) { .key_press => |key| if (in_paste) { // Between the brackets a key is DATA, never a command. vaxis gives control bytes no // text at all, so a line break inside a paste arrives as a bare CR (Key.enter) or, from // a terminal that does not translate them, as ctrl+j. const text = key.text orelse ""; const cp = mapKey(effCp(key)); const bytes: []const u8 = if (text.len > 0) text else if (cp == pardes.Key.tab) "\t" else if (cp == pardes.Key.enter or (key.mods.ctrl and cp == 'j')) "\n" else ""; if (bytes.len > 0) paste_buf.appendSlice(gpa(), bytes) catch {}; } else { c.update(.{ .key = .{ .cp = mapKey(effCp(key)), .text = key.text orelse "", .ctrl = key.mods.ctrl, .alt = key.mods.alt, .shift = key.mods.shift, } }); dirty = true; }, .paste_start => { paste_buf.clearRetainingCapacity(); in_paste = true; }, .paste_end => { in_paste = false; if (paste_buf.items.len > 0) { c.update(.{ .paste = paste_buf.items }); dirty = true; } paste_buf.clearRetainingCapacity(); }, // OSC 52. The bytes are the parser's, allocated from our own allocator, so they are freed // here rather than leaked - the core copies whatever it keeps. .paste => |text| { c.update(.{ .paste = text }); gpa().free(text); dirty = true; }, .mouse => |m| { const button: ?pardes.Mouse.Button = switch (m.button) { .left => .left, .middle => .middle, .right => .right, .wheel_up => .wheel_up, .wheel_down => .wheel_down, .wheel_left => .wheel_left, .wheel_right => .wheel_right, .none => .none, else => null, }; if (button) |b| { c.update(.{ .mouse = .{ .button = b, .kind = switch (m.type) { .press => .press, .release => .release, .motion => .motion, .drag => .drag, }, .col = @intCast(m.col), .row = @intCast(m.row), .ctrl = m.mods.ctrl, } }); dirty = true; } }, // The only way this platform learns its size, and the one place a 384 KiB heap shows through // to the user. Two things happen here that the tty shell does not need. // // CLAMPED, because the grids do not fit an arbitrary terminal: vaxis keeps a `Screen` and an // `InternalScreen`, pardes keeps its own `Surface` and `previous_cells`, so every cell is // paid for four times. Measured on the die - 40x12 initialises with room to spare, 80x24 // exhausts the heap and `Pardes.init` returns OutOfMemory with 9,128 bytes left. The host's // terminal is normally larger than the board can render, so the editor takes a corner of it // instead of refusing to start. // // ATOMIC, because `Vaxis.resize` deinits both screens BEFORE allocating the replacements // (Vaxis.zig:194-206), so a failed resize leaves vaxis with freed screens and renders // nothing at all. That is exactly how this was found: the host bridge injects a size report // on attach, the 80x24 it reported could not be allocated, and an editor that had just drawn // its interface went silent. A failure now puts the previous geometry back. .winsize => |ws| { const want: vaxis.Winsize = .{ .rows = @min(ws.rows, max_rows), .cols = @min(ws.cols, max_cols), .x_pixel = ws.x_pixel, .y_pixel = ws.y_pixel, }; if (want.cols == cur_winsize.cols and want.rows == cur_winsize.rows) return; const previous = cur_winsize; // vaxis is only resized when it is the thing doing the rendering. Under `direct_emit` its // grids are a single cell and stay that way - see `vaxisSize` - so there is nothing here // to reallocate, which also means a resize can no longer fail for want of two grids. if (!direct_emit) { vx.resize(gpa(), &out, want) catch { vx.resize(gpa(), &out, previous) catch {}; return; }; } cur_winsize = want; c.update(.{ .resize = .{ .cols = want.cols, .rows = want.rows } }); dirty = true; }, // A TTY cannot report a pointer leaving its grid, so losing focus is the only reliable // pointer-leave signal there is. .focus_out => { c.update(.pointer_leave); dirty = true; }, .focus_in, .mouse_leave => {}, // Capability replies. vaxis's own Loop sets these fields directly (`Loop.zig:377-403`); // with no Loop, this is where they land. DA1 is the terminator: every terminal answers it // last, so it is the signal that the whole handshake is in and the detected features can be // switched on. .cap_kitty_keyboard => vx.caps.kitty_keyboard = true, .cap_kitty_graphics => vx.caps.kitty_graphics = true, .cap_rgb => vx.caps.rgb = true, .cap_unicode => { vx.caps.unicode = .unicode; vx.screen.width_method = .unicode; }, .cap_sgr_pixels => vx.caps.sgr_pixels = true, .cap_color_scheme_updates => vx.caps.color_scheme_updates = true, .cap_multi_cursor => vx.caps.multi_cursor = true, .cap_da1 => { vx.enableDetectedFeatures(&out) catch {}; out.flush() catch {}; dirty = true; }, .color_report, .color_scheme => {}, .key_release => {}, } } /// The effective codepoint the way vaxis's own `Key.matches` sees it: a single-character `text` /// wins, because the terminal has already resolved shift; otherwise the shifted codepoint. fn effCp(key: vaxis.Key) u21 { if (key.text) |t| { const view = std.unicode.Utf8View.init(t) catch return key.codepoint; var it = view.iterator(); if (it.nextCodepoint()) |cp| { if (it.nextCodepoint() == null) return cp; } } return key.shifted_codepoint orelse key.codepoint; } /// vaxis functional-key codepoints -> core constants. The ASCII ones already coincide, so /// enter/tab/escape/backspace pass straight through. fn mapKey(cp: u21) u21 { return switch (cp) { vaxis.Key.up => pardes.Key.up, vaxis.Key.down => pardes.Key.down, vaxis.Key.left => pardes.Key.left, vaxis.Key.right => pardes.Key.right, vaxis.Key.home => pardes.Key.home, vaxis.Key.end => pardes.Key.end, vaxis.Key.page_up => pardes.Key.page_up, vaxis.Key.page_down => pardes.Key.page_down, vaxis.Key.delete => pardes.Key.delete, else => cp, }; } export fn pardes_esp32p4_tick(now_ms: u64) callconv(.c) void { const c = core orelse return; last_now_ms = now_ms; // The held Escape, released. Anything still waiting after this long is a key the human pressed, // not the head of a sequence: the next byte of a real sequence is 87 us behind on this line, and // even a slow terminal emulator answers a query in well under a millisecond. Ten is generous by // two orders of magnitude and imperceptible to the person pressing it - the same trade every // terminal editor makes for the same reason. if (esc_held_at) |at| { if (now_ms -% at >= esc_hold_ms) { drainInput(c, true); dirty = true; } } // The core steps its own animations by this clock (`Host.now`). c.advance(now_ms * std.time.ns_per_ms); if (c.needs_frame) dirty = true; } /// How long a lone ESC waits for a second byte before it counts as the Escape key. const esc_hold_ms = 10; /// The last timestamp `pardes_esp32p4_tick` was given, so `drainInput` can date a hold without needing a /// clock of its own - there is no clock on this side of the ABI. var last_now_ms: u64 = 0; /// When the buffer became a lone ESC, or null when it is not holding one. var esc_held_at: ?u64 = null; export fn pardes_esp32p4_wants_frame() callconv(.c) bool { const c = core orelse return false; return dirty or c.needs_frame; } export fn pardes_esp32p4_render() callconv(.c) u32 { const c = core orelse return 0; c.pump(.{ .ctx = null, .vtable = &pardes_host }) catch |err| return errCode(err); dirty = false; return 0; } export fn pardes_esp32p4_quit() callconv(.c) bool { const c = core orelse return true; return c.quit; } // ------------------------------------------------------------------------------------ the host const pardes_host: pardes.Host.VTable = .{ .now = boardNow, .present = present, .gpio_toggle = gpioToggle }; fn boardNow(_: ?*anyopaque) u64 { return last_now_ms * std.time.ns_per_ms; } /// The `Gpio` word's one seam to the board. Nothing here knows what a pad is; it forwards, and /// answers false when the firmware brought none, which is what puts "gpio: no pads" on the message /// row rather than a trap. fn gpioToggle(_: ?*anyopaque, pin: u16, was: *u8, now: *u8) bool { const f = host_gpio orelse return false; return f(out_ctx, pin, was, now); } /// The canonical surface -> the wire. Same shape as the tty shell's (`src/tty/tty.zig:1096`) minus /// the panel compositor and the kitty image path: neither has a reason to exist on a board with no /// pixels. Where the tty shell hands every cell to vaxis and lets it diff, this diffs against the /// Surface itself and can then emit the ANSI directly - see `direct_emit`. fn present(_: ?*anyopaque, surface: *const pardes.Surface) void { const t0 = cycles(); const win = vx.window(); const n = @as(usize, surface.cols) * @as(usize, surface.rows); // THE SHADOW GRID. Copying all 480 cells into vaxis every frame cost 6.75 ms on the die - 57% // of a keystroke, and it was paid whether or not anything changed: a second render with nothing // new measured the same as the first. vaxis already diffs its own grid against the terminal, but // it can only do that AFTER being told every cell, and being told is the expensive part // (`writeCell` builds a vaxis `Cell`, which carries an always-null image placement). // // So keep the previous Surface and tell vaxis only what moved. `Cell.visuallyEqual` is the // right comparison and already exists for the panel compositor's benefit: it ignores scratch // bytes past `len` and treats any two default cells as equal, so it cannot manufacture a write. // // STATIC, and that is not a micro-optimisation - it is a bug fix. The first version allocated // this from the editor's heap, and on a board whose 384 KiB is already nearly spoken for that // was enough to make `vx.resize` fail: a resize then hit its OOM path, restored the previous // geometry and returned, so the screen was never repainted. Measured as a resize emitting 80 // bytes where it had emitted 1,392. The grid is bounded by `max_cols` x `max_rows` at comptime, // so it belongs in `.bss` where it cannot compete with anything. const full = !shadow_grid or prev_cols != surface.cols or prev_rows != surface.rows; emit_bytes = 0; if (full) { prev_cols = surface.cols; prev_rows = surface.rows; if (direct_emit) { // Reset first: a `2J` while a non-default background is active fills the screen with it. emitRaw("\x1b[0m\x1b[2J") catch return; emit_style = .{}; emit_col = -1; } else win.clear(); } const usable = shadow_grid and n <= prev_cells.len; var y: u16 = 0; while (y < surface.rows) : (y += 1) { const row0 = @as(usize, y) * @as(usize, surface.cols); const src = surface.cells[row0..][0..surface.cols]; // A ROW AT A TIME FIRST. `Surface.cells` is contiguous and row-major, so a whole row is one // `memcmp` against the shadow - and on a keystroke eleven of twelve rows are untouched. The // per-cell loop below is ~40 branchy comparisons where this is one call over 1,120 bytes; // measured, the walk fell from 246 us to a fraction of it. Byte equality implies visual // equality (see `sameCell`), so a row that compares equal cannot be hiding a changed cell - // and a row that differs only in padding falls through to the per-cell path, which is // correct and merely slower. if (usable and !full) { const shadow = prev_cells[row0..][0..surface.cols]; if (sameBytes(std.mem.sliceAsBytes(src), std.mem.sliceAsBytes(shadow))) continue; } var x: u16 = 0; while (x < surface.cols) : (x += 1) { const cell = &src[x]; const idx = row0 + @as(usize, x); if (usable) { if (!full and sameCell(cell, &prev_cells[idx])) continue; prev_cells[idx] = cell.*; } else if (cell.default) continue; writeOne(win, x, y, cell, surface.cols) catch return; } } if (direct_emit) { // BOTH branches have to reach the packet boundary, and the second one is easy to forget: // measured, a frame that only hid the cursor was 6 bytes and cost 4014 us at 640 characters // against 3863 at 320, because 6 bytes never fills a packet and waited out the bridge's // timer. Hiding an already-hidden cursor is as idempotent as positioning it twice. if (surface.cursor) |cur| { cup(cur.y, cur.x) catch return; emitRaw("\x1b[?25h") catch return; emit_col = -1; while (emit_bytes < emit_min_frame) cup(cur.y, cur.x) catch return; } else { while (emit_bytes < emit_min_frame) emitRaw("\x1b[?25l") catch return; } } else if (surface.cursor) |cur| { win.showCursor(cur.x, cur.y); } else win.hideCursor(); const t1 = cycles(); // vaxis diffs against its own shadow grid, so this writes only what changed - which is what // makes an editor usable at 11.9 KB/s. With `direct_emit` that diff has already happened, one // stage earlier and against the Surface itself, so there is nothing left here to do. if (!direct_emit) vx.render(&out) catch return; const t2 = cycles(); out.flush() catch return; const t3 = cycles(); prof_copy_cy = t1 -% t0; prof_render_cy = t2 -% t1; prof_flush_cy = t3 -% t2; } /// One cell to the wire, either through vaxis or straight out. inline fn writeOne(win: vaxis.Window, x: u16, y: u16, cell: *const pardes.Cell, cols: u16) !void { if (!direct_emit) { // Changed TO default. `win.clear()` is what used to blank these, and it is not run on an // incremental frame, so say it explicitly. if (cell.default) return win.writeCell(x, y, .{ .char = .{ .grapheme = " " }, .style = .{} }); return win.writeCell(x, y, .{ .char = .{ .grapheme = cell.grapheme() }, .style = vaxisStyle(cell.style), }); } if (emit_row != y or emit_col != x) { try cup(y, x); emit_row = y; emit_col = @intCast(x); } const style: pardes.CellStyle = if (cell.default) .{} else cell.style; if (!std.meta.eql(emit_style, style)) { try emitStyle(style); emit_style = style; } try emitRaw(if (cell.default) " " else cell.grapheme()); // Where the terminal's cursor now is. A single printable ASCII byte advanced it exactly one // column; anything else - a wide glyph, a cluster, the spacer cell pardes writes after a wide // one - is not worth predicting, so give up and let the next cell emit an absolute CUP. The last // column is given up on too, because whether the cursor rests on it or has wrapped past it // depends on the terminal's deferred-wrap behaviour, and the two disagree by a whole row. if (x + 1 < cols and cell.len == 1 and cell.text[0] >= 0x20 and cell.text[0] < 0x7f) { emit_col += 1; } else emit_col = -1; } /// A style as an absolute SGR, always opening with a reset. /// /// Absolute rather than a delta from whatever is currently on, and that is what keeps it short /// enough to be worth having: no per-attribute off-codes, no state to keep beyond the last style /// emitted, and a frame that gets cut off cannot leave a later cell wearing an earlier one's colour. /// It costs a few bytes on a style change, against the ~9 of CUP a changed cell is paying anyway. fn emitStyle(s: pardes.CellStyle) !void { try emitRaw("\x1b[0"); if (s.bold) try emitRaw(";1"); if (s.dim) try emitRaw(";2"); if (s.italic) try emitRaw(";3"); if (s.blink) try emitRaw(";5"); if (s.reverse) try emitRaw(";7"); if (s.invisible) try emitRaw(";8"); if (s.strikethrough) try emitRaw(";9"); try emitRaw(switch (s.ul) { .off => "", .single => ";4", .double => ";4:2", .curly => ";4:3", .dotted => ";4:4", .dashed => ";4:5", }); try emitColor(s.fg, 30); try emitColor(s.bg, 40); try emitRaw("m"); } /// `base` is 30 for a foreground and 40 for a background, which is the only thing separating the two /// in every form SGR has for a colour: 30-37 against 40-47, 90-97 against 100-107, 38 against 48. fn emitColor(c: pardes.Color, comptime base: u16) !void { var b: [20]u8 = undefined; var i: usize = 0; switch (c) { // Already said by the reset this SGR opens with. .default => return, .index => |n| { b[i] = ';'; i += 1; if (n < 8) { i += dec(b[i..], base + n); } else if (n < 16) { i += dec(b[i..], base + 60 + (n - 8)); } else { i += dec(b[i..], base + 8); i += lit(b[i..], ";5;"); i += dec(b[i..], n); } }, .rgb => |v| { b[i] = ';'; i += 1; i += dec(b[i..], base + 8); i += lit(b[i..], ";2;"); for (v, 0..) |component, k| { if (k != 0) { b[i] = ';'; i += 1; } i += dec(b[i..], component); } }, } try emitRaw(b[0..i]); } /// Absolute cursor positioning, hand-rolled rather than through `out.print`. /// /// Not for elegance: this is the single most frequent sequence the emitter produces, at least one per /// changed run, and `std.fmt` brings a whole format-string interpreter to write at most two digits. /// The grid is bounded by `max_cols` x `max_rows`, so nothing here can exceed three. fn cup(row: u16, col: u16) !void { var b: [12]u8 = undefined; var i: usize = lit(&b, "\x1b["); i += dec(b[i..], row + 1); b[i] = ';'; i += 1; i += dec(b[i..], col + 1); b[i] = 'H'; i += 1; try emitRaw(b[0..i]); } /// Decimal, least significant digit first into a scratch buffer and then reversed. Five digits is /// every `u16`, so there is no fallback to `std.fmt` and no value this cannot write. fn dec(buf: []u8, v: u16) usize { var digits: [5]u8 = undefined; var n: usize = 0; var rest = v; while (true) { digits[n] = '0' + @as(u8, @intCast(rest % 10)); n += 1; rest /= 10; if (rest == 0) break; } for (0..n) |k| buf[k] = digits[n - 1 - k]; return n; } inline fn lit(buf: []u8, comptime s: []const u8) usize { @memcpy(buf[0..s.len], s); return s.len; } /// Every direct-emit byte goes through here, because the count is what the padding below needs. inline fn emitRaw(bytes: []const u8) !void { emit_bytes += bytes.len; try out.writeAll(bytes); } /// A/B switch for the emitter above, on the same terms as `shadow_grid`: false routes every cell back /// through vaxis, which is the reference. vaxis's own diff measured 631 us of a 4.37 ms keystroke and /// all of it was redundant - `present` has already worked out which cells moved, so vaxis was being /// told the answer and then computing it again from scratch. const direct_emit = true; /// THE FRAME HAS A MINIMUM SIZE, and it is the USB bridge's, not the terminal's. /// /// The board is wired to the host through a CH340, and 32 is not a guess: it is `wMaxPacketSize` of /// endpoint 0x82, the bulk IN, as the device itself reports it - a full-speed 0x0020. The bridge /// forwards a packet when the packet is FULL, so a frame shorter than that sits there until an /// internal timer gives up on more, which is worth about a millisecond - a quarter of the budget. /// /// Measured, at the same board cost and with the screen byte-identical: a 21-byte frame round-trips /// in 4817 us and the same frame padded to 49 bytes in 3814 us. MORE BYTES, ARRIVING SOONER. It also /// explains why routing through vaxis looked competitive - its frames are 81 bytes, so they fill a /// packet by accident and never wait. /// /// So pad to the packet boundary. The filler is repeated absolute cursor positioning: idempotent, /// already the sequence the emitter ends on, and it cannot alter a cell. This is the same bargain as /// an Ethernet runt frame - the medium has a minimum and the sender pays it - and it is a real /// trade, not free: the wasted bytes are wire time that delays a LATER frame, so it is only worth it /// while the frame is small, which is exactly when it applies. const emit_min_frame: usize = 32; /// Bytes emitted this frame, for `emit_min_frame`. var emit_bytes: usize = 0; /// What the terminal is currently wearing and where its cursor is, so that a run of changed cells in /// one row costs one CUP and one SGR rather than one of each per cell. `emit_col` is signed because /// -1 means "no longer known" - see `writeOne`. var emit_style: pardes.CellStyle = .{}; var emit_row: u16 = 0; var emit_col: i32 = -1; /// A/B switch, kept because this optimisation is exactly the kind that can be right about latency /// and wrong about the screen. With it false, `present` behaves as it did before the shadow grid - /// clear and write every cell - which is the reference any measurement of it should be compared /// against, and the way to tell a rendering bug from a rendering difference. const shadow_grid = true; /// The previous Surface, cell for cell, sized for the largest grid this board can drive. In `.bss` /// rather than on the heap: see `present`. `prev_cols`/`prev_rows` being zero on the first frame is /// what makes that frame a full one. var prev_cells: [@as(usize, max_cols) * @as(usize, max_rows)]pardes.Cell = if (shadow_grid) @splat(.{}) else undefined; var prev_cols: u16 = 0; var prev_rows: u16 = 0; /// Cell equality for the shadow grid, as bytes. /// /// `Cell.visuallyEqual` is the semantically exact answer and it is too slow to ask 480 times a /// frame: `std.meta.eql` on a `CellStyle` recurses through a colour union and eight booleans, and /// the walk measured 1.45 ms - about 270 cycles per comparison of a ~28-byte struct. /// /// Byte equality IMPLIES visual equality, so this can never claim two different cells are the same. /// It can miss an equality - scratch bytes past `len`, or padding - and the only cost of that is one /// redundant `writeCell` that vaxis then diffs away. Defaults are still handled by meaning rather /// than by bytes, because an unpainted cell's text and style are whatever the last frame left there. inline fn sameCell(a: *const pardes.Cell, b: *const pardes.Cell) bool { if (a.default or b.default) return a.default and b.default; return sameBytes(std.mem.asBytes(a), std.mem.asBytes(b)); } /// Exact byte equality, a word at a time when both spans are aligned for it. /// /// This comparison is the firmware's largest read by a wide margin - two 13 KB streams every frame - /// and it measured 3.2 cycles per byte, about four times what word-wide loads should need, which is /// what a byte-at-a-time loop looks like. The answer is identical either way: this is still exact /// byte equality, so it keeps the property the whole diff rests on, that byte equality implies /// visual equality. /// /// The alignment test is a RUNTIME one because `Cell` has alignment 1 - it is all `u8` fields - so /// whether a row begins on a word boundary is a property of whoever allocated the Surface and not of /// the type. A row is 40 cells of 26 bytes, which is divisible by four, so if the base is aligned /// every row is. When it is not, the byte loop is still here. inline fn sameBytes(a: []const u8, b: []const u8) bool { if (a.len != b.len) return false; if ((@intFromPtr(a.ptr) | @intFromPtr(b.ptr)) & 3 == 0) { const n = a.len / 4; const wa: [*]align(4) const u32 = @ptrCast(@alignCast(a.ptr)); const wb: [*]align(4) const u32 = @ptrCast(@alignCast(b.ptr)); for (wa[0..n], wb[0..n]) |x, y| { if (x != y) return false; } return std.mem.eql(u8, a[n * 4 ..], b[n * 4 ..]); } return std.mem.eql(u8, a, b); } // ------------------------------------------------------------------ where a frame's time goes // // A frame has three stages and they want different fixes, so the firmware is given all three rather // than one total. Measured on the die, a render costs ~11 ms whether or not anything changed, which // says the cost is the unconditional walk and not the edit - but "the walk" is two walks, the copy // into vaxis's grid and vaxis's own diff, and only one of them is ours to change. // // Two CSR reads per stage. `cycle` is the unprivileged counter, read high-low-high because two // 32-bit halves can straddle a wrap. var prof_copy_cy: u64 = 0; var prof_render_cy: u64 = 0; var prof_flush_cy: u64 = 0; inline fn cycles() u64 { if (builtin.cpu.arch != .riscv32) return 0; while (true) { const hi0 = asm volatile ("csrr %[o], cycleh" : [o] "=r" (-> u32), ); const lo = asm volatile ("csrr %[o], cycle" : [o] "=r" (-> u32), ); const hi1 = asm volatile ("csrr %[o], cycleh" : [o] "=r" (-> u32), ); if (hi0 == hi1) return (@as(u64, hi0) << 32) | lo; } } /// The last frame's three stages, in cycles. Zero on any platform without the CSR. export fn pardes_esp32p4_frame_prof(copy: *u64, render: *u64, flush: *u64) callconv(.c) void { copy.* = prof_copy_cy; render.* = prof_render_cy; flush.* = prof_flush_cy; } fn vaxisStyle(s: pardes.CellStyle) vaxis.Style { return .{ .fg = vaxisColor(s.fg), .bg = vaxisColor(s.bg), .bold = s.bold, .dim = s.dim, .italic = s.italic, .blink = s.blink, .reverse = s.reverse, .invisible = s.invisible, .strikethrough = s.strikethrough, .ul_style = switch (s.ul) { .off => .off, .single => .single, .double => .double, .curly => .curly, .dotted => .dotted, .dashed => .dashed, }, }; } fn vaxisColor(c: pardes.Color) vaxis.Color { return switch (c) { .default => .default, .index => |i| .{ .index = i }, .rgb => |rgb| .{ .rgb = rgb }, }; } /// Errors cross the ABI as small non-zero integers. `@intFromError` is not stable across builds, so /// it is not used: the firmware only reports the number, and a stable-looking value that silently /// changed meaning would be worse than an opaque one. fn errCode(err: anyerror) u32 { return switch (err) { error.OutOfMemory => 1, error.WriteFailed => 2, else => 255, }; }