summaryrefslogtreecommitdiff
path: root/src/p4.zig
diff options
context:
space:
mode:
Diffstat (limited to 'src/p4.zig')
-rw-r--r--src/p4.zig1030
1 files changed, 0 insertions, 1030 deletions
diff --git a/src/p4.zig b/src/p4.zig
deleted file mode 100644
index 8e46b2ff..00000000
--- a/src/p4.zig
+++ /dev/null
@@ -1,1030 +0,0 @@
-//! The ESP32-P4 firmware shell: pardes as one freestanding object, bytes in and bytes out.
-//!
-//! This is the fourth platform, and the only one that is not an executable. `zig build
-//! -Dplatform=p4 -Dtarget=riscv32-freestanding` emits this file as a single object exporting the C
-//! ABI below; the `zig-p4` package links it beside its own `_start`, its generated linker script,
-//! and its UART driver. Nothing here knows what a UART is.
-//!
-//! **Why an object and not a module.** The obvious arrangement was for zig-p4 to declare this
-//! package in its `build.zig.zon` and import `pardes_p4`. That was built, and it broke every build
-//! in that repo: nesting this package's ~30-package graph under one whose own claim is "host
-//! dependencies: Zig, that is the whole list" made `std/Build.zig:2091` exceed its 1000-branch
-//! comptime quota (through ghostty's `SharedDeps.zig:874` `lazyImport`), dragged in seven cached
-//! tree-sitter versions whose `build.zig` uses APIs removed in 0.16, and materialised 2.6 GB across
-//! 42,736 files into that repo's working copy. A linked object has none of those properties and one
-//! extra virtue: the seam is bytes, so neither side can accidentally depend on the other's types.
-//!
-//! **Where the terminal is.** On the host. The board writes ANSI and reads ANSI; the terminal
-//! emulator at the far end of the serial line does the font rendering, and answers this program's
-//! own capability queries. That is why `vaxis` works here unmodified: `Vaxis.render`,
-//! `queryTerminalSend` and `enableDetectedFeatures` all take a bare `*std.Io.Writer`
-//! (`Vaxis.zig:375,278,329`), so the transport is a parameter. `vaxis.Tty` and `vaxis.Loop` are
-//! termios/ioctl/SIGWINCH bound and are not used.
-//!
-//! **Where the memory is.** Not here either. The firmware measured its own RAM (240 KiB low,
-//! 384 KiB high, and a 128 KiB region that turned out to be L2 cache) and owns the allocator; this
-//! file receives four function pointers and rebuilds a `std.mem.Allocator` from them. Everything
-//! the editor allocates comes from there.
-//!
-//! **Window size** arrives as DEC mode 2048 in-band resize reports, parsed by `vaxis.Parser` like
-//! any other input. Firmware has no `TIOCGWINSZ`, so the host-side bridge synthesises the first one.
-
-const std = @import("std");
-const builtin = @import("builtin");
-const pardes = @import("pardes.zig");
-const vaxis = @import("vaxis");
-
-// ------------------------------------------------------------------ what a freestanding root owes
-//
-// These are ROOT-module declarations: std reads them off whichever file is the compilation root, and
-// as of the build change that emits this file as the object, that is this file. They are not
-// ceremony - each one was discovered by the build failing without it.
-
-/// The board has no MMU and no pages, but std derives allocator alignment from these two. 4 KiB is
-/// the ESP32-P4's cache and DMA granularity. Without them: "riscv32-freestanding has unknown
-/// page_size_min" from std/heap.zig:48.
-///
-/// `logFn` is the load-bearing one. std's DEFAULT log implementation reaches `std.debug_io`, which
-/// instantiates `std.Io.Threaded` - a thread pool, `getrandom`, `IOV_MAX`, `mremap` - none of which
-/// exist on this target, and ONE `log.warn` anywhere in the core or in vaxis is enough to drag the
-/// whole thing in and fail the build with "no member named 'getrandom'".
-pub const std_options: std.Options = .{
- .page_size_min = 4096,
- .page_size_max = 4096,
- .logFn = logFn,
-};
-
-/// Logs go out the same byte sink as the frames, which is the only sink there is. Truncated rather
-/// than allocated: a log line is never worth an allocation on a 384 KiB heap, and a logger that can
-/// fail on OOM is a logger that disappears exactly when it is needed.
-fn logFn(
- comptime level: std.log.Level,
- comptime scope: @EnumLiteral(),
- comptime fmt: []const u8,
- args: anytype,
-) void {
- if (out_ctx == null and @intFromPtr(out_write) == 0) return;
- var buf: [256]u8 = undefined;
- const line = std.fmt.bufPrint(
- &buf,
- "\r\n[" ++ level.asText() ++ "/" ++ @tagName(scope) ++ "] " ++ fmt ++ "\r\n",
- args,
- ) catch "\r\n[log truncated]\r\n";
- out_write(out_ctx, line.ptr, line.len);
-}
-
-pub const panic = std.debug.FullPanic(panicImpl);
-
-/// A panic here cannot unwind and has nowhere to go, so it reports through the write callback and
-/// stops. `@trap` and not a spin: the firmware's own panic handler prints through the mask ROM,
-/// which shares nothing with this path but the FIFO, so a trap leaves that diagnostic route intact.
-fn panicImpl(msg: []const u8, _: ?usize) noreturn {
- const prefix = "\r\nMARK PARDES_CORE_PANIC ";
- out_write(out_ctx, prefix.ptr, prefix.len);
- out_write(out_ctx, msg.ptr, msg.len);
- out_write(out_ctx, "\r\n", 2);
- @trap();
-}
-
-// ---------------------------------------------------------------------------------- the C ABI
-//
-// Deliberately tiny, and versioned. Linkers do not type-check C symbols, so a signature that drifts
-// on one side of this seam links cleanly and then corrupts the stack. `pardes_p4_abi_version` is the
-// cheapest possible defence: the firmware calls it first and refuses to continue on a mismatch.
-
-/// Bumped whenever any signature below changes, including a type.
-/// 2 added `GpioFn` to `pardes_p4_init`. A firmware built against 1 passes five arguments where six
-/// are read, which is exactly the silent-corruption case this counter exists to turn into a message.
-const abi_version: u32 = 2;
-
-export fn pardes_p4_abi_version() callconv(.c) u32 {
- return abi_version;
-}
-
-/// The firmware's allocator, as C function pointers. `alignment` is a log2 value, matching
-/// `std.mem.Alignment`'s own representation, so no translation table is needed.
-///
-/// `remap` is absent on purpose: this allocator cannot move a block without copying it, so
-/// `std.mem.Allocator`'s remap is implemented locally as "resize in place, or fail" and the caller's
-/// own alloc/copy/free path handles the rest.
-pub const Allocator = extern struct {
- ctx: ?*anyopaque,
- alloc: *const fn (ctx: ?*anyopaque, len: usize, log2_align: u8) callconv(.c) ?[*]u8,
- resize: *const fn (ctx: ?*anyopaque, ptr: [*]u8, len: usize, log2_align: u8, new_len: usize) callconv(.c) bool,
- free: *const fn (ctx: ?*anyopaque, ptr: [*]u8, len: usize, log2_align: u8) callconv(.c) void,
-};
-
-/// How finished runs of ANSI leave this object.
-pub const WriteFn = *const fn (ctx: ?*anyopaque, ptr: [*]const u8, len: usize) callconv(.c) void;
-
-/// Flip one pad and report the level before and after; false if the firmware declines. OPTIONAL on
-/// the wire, so a host with no pads (or one that has not implemented them yet) passes null and the
-/// `Gpio` word answers "no pads" instead of the object having to know which firmwares exist.
-///
-/// The board's side, not the editor's, because a correct toggle is the IO MUX, the GPIO matrix, the
-/// pad's own bits and the output enable - four register files behind a per-pin table that the
-/// firmware already has and checks against ESP-IDF. See `Host.VTable.pull_gpio_toggle`.
-pub const GpioFn = *const fn (ctx: ?*anyopaque, pin: u16, was: *u8, now: *u8) callconv(.c) bool;
-
-// ------------------------------------------------------------------- the allocator, rebuilt
-// One `std.mem.Allocator` whose vtable forwards to the four pointers above. The indirection is the
-// price of the seam and it is paid once per allocation, which on a first-fit heap is already the
-// cheap part (measured on the die: 8,229 cycles for one allocation across 257 free blocks).
-
-var host_alloc: Allocator = undefined;
-
-fn hostAlloc(_: *anyopaque, len: usize, alignment: std.mem.Alignment, _: usize) ?[*]u8 {
- return host_alloc.alloc(host_alloc.ctx, len, @intFromEnum(alignment));
-}
-
-fn hostResize(_: *anyopaque, mem: []u8, alignment: std.mem.Alignment, new_len: usize, _: usize) bool {
- return host_alloc.resize(host_alloc.ctx, mem.ptr, mem.len, @intFromEnum(alignment), new_len);
-}
-
-fn hostRemap(_: *anyopaque, mem: []u8, alignment: std.mem.Alignment, new_len: usize, _: usize) ?[*]u8 {
- return if (host_alloc.resize(host_alloc.ctx, mem.ptr, mem.len, @intFromEnum(alignment), new_len)) mem.ptr else null;
-}
-
-fn hostFree(_: *anyopaque, mem: []u8, alignment: std.mem.Alignment, _: usize) void {
- host_alloc.free(host_alloc.ctx, mem.ptr, mem.len, @intFromEnum(alignment));
-}
-
-const host_vtable: std.mem.Allocator.VTable = .{
- .alloc = hostAlloc,
- .resize = hostResize,
- .remap = hostRemap,
- .free = hostFree,
-};
-
-/// `ptr` is never dereferenced - the four forwarders read the file-scope `host_alloc` - but
-/// `std.mem.Allocator` requires a non-null context, so it points at the record itself.
-fn gpa() std.mem.Allocator {
- return .{ .ptr = @ptrCast(&host_alloc), .vtable = &host_vtable };
-}
-
-// ------------------------------------------------------------------------------- the ANSI sink
-// A `std.Io.Writer` over the firmware's write callback. Buffered, because vaxis emits a frame as a
-// long run of small writes - cursor move, SGR run, grapheme, repeat - and an unbuffered writer would
-// make a C call per fragment.
-
-var out_write: WriteFn = undefined;
-var out_ctx: ?*anyopaque = null;
-var host_gpio: ?GpioFn = null;
-var out_buf: [8192]u8 = undefined;
-var out: std.Io.Writer = undefined;
-
-fn drain(w: *std.Io.Writer, data: []const []const u8, splat: usize) std.Io.Writer.Error!usize {
- // The shape std documents at Io/Writer.zig:46-63: buffer first, then every slice of `data`, with
- // the LAST slice repeated `splat` times, and the count returned excluding the buffered bytes.
- if (w.end > 0) {
- out_write(out_ctx, w.buffer.ptr, w.end);
- w.end = 0;
- }
- const head = data[0 .. data.len - 1];
- const pattern = data[head.len];
- var written: usize = 0;
- for (head) |bytes| {
- if (bytes.len > 0) out_write(out_ctx, bytes.ptr, bytes.len);
- written += bytes.len;
- }
- var i: usize = 0;
- while (i < splat) : (i += 1) {
- if (pattern.len > 0) out_write(out_ctx, pattern.ptr, pattern.len);
- }
- return written + pattern.len * splat;
-}
-
-// ------------------------------------------------------------------------------------ the state
-
-var core: ?*pardes.Pardes = null;
-var vx: vaxis.Vaxis = undefined;
-var parser: vaxis.Parser = .{};
-
-/// vaxis wants an environment map. There is no environment; an empty one is the honest answer and
-/// the only thing vaxis reads it for is TERM-derived heuristics, which the capability queries
-/// supersede.
-var env_map: std.process.Environ.Map = undefined;
-
-/// Input that arrived mid-sequence. An escape sequence can be split across UART reads, and the
-/// parser reports "incomplete" by consuming nothing, so the tail has to survive until more arrives.
-var in_buf: [1024]u8 = undefined;
-var in_len: usize = 0;
-
-/// Bracketed paste: between the markers, keys are DATA and never commands.
-var paste_buf: std.ArrayListUnmanaged(u8) = .empty;
-var in_paste: bool = false;
-
-/// Set by anything that could change the screen; cleared by a render. The firmware asks before
-/// rendering, because on a 115200-baud link an unconditional repaint per loop saturates the wire and
-/// starves input.
-var dirty: bool = true;
-
-/// The largest grid this board can render, and the reason it is not just the host's terminal size.
-///
-/// A cell is paid for TWICE now, not four times: pardes keeps its `Surface` and this shell keeps a
-/// shadow copy of it to diff against. vaxis used to keep a `Screen` and an `InternalScreen` as well,
-/// and with `direct_emit` neither is ever read - the emitter diffs against the Surface and writes
-/// the wire itself - so `init` sizes vaxis to a single cell and those two grids cost nothing.
-///
-/// Set with `-Dp4-cols` / `-Dp4-rows`, because the ceiling is a measurement rather than a constant
-/// and it moves for two independent reasons: the 384 KiB heap, and the round trip, which grows with
-/// the cell count because every frame walks the whole grid. See the geometry table in
-/// `05-zig-p4/experiments/report.typ` for both curves.
-///
-/// Raising these further is what PSRAM would buy: this board has 32 MB fitted and untrained.
-pub const max_cols: u16 = @import("pardes_config").p4_cols;
-pub const max_rows: u16 = @import("pardes_config").p4_rows;
-
-var cur_winsize: vaxis.Winsize = .{ .rows = max_rows, .cols = max_cols, .x_pixel = 0, .y_pixel = 0 };
-
-/// How big vaxis's own grids need to be.
-///
-/// ONE CELL under `direct_emit`, because neither of them is ever read: vaxis keeps a `Screen` and an
-/// `InternalScreen`, and the emitter diffs the Surface against its own shadow and writes the escapes
-/// itself. Those two grids were the largest single claim on a 384 KiB heap and the reason the board
-/// was held to 40x12 - the comment above `max_cols` used to say a cell was paid for four times over,
-/// and this is what took it down to two. vaxis is still doing the work only it can do here: entering
-/// the alternate screen, asking the terminal what it is, and parsing everything that comes back.
-fn vaxisSize() vaxis.Winsize {
- return if (direct_emit)
- .{ .rows = 1, .cols = 1, .x_pixel = 0, .y_pixel = 0 }
- else
- cur_winsize;
-}
-
-// -------------------------------------------------------------------------------------- exports
-
-/// Hand over the allocator and the output sink, state the initial window size, and bring the editor
-/// up. Returns 0, or a small non-zero code the firmware can only report.
-export fn pardes_p4_init(
- alloc: *const Allocator,
- write: WriteFn,
- gpio: ?GpioFn,
- ctx: ?*anyopaque,
- cols: u16,
- rows: u16,
-) callconv(.c) u32 {
- host_alloc = alloc.*;
- out_write = write;
- host_gpio = gpio;
- out_ctx = ctx;
- out = .{ .vtable = &.{ .drain = drain }, .buffer = &out_buf };
-
- const a = gpa();
- env_map = .{ .array_hash_map = .empty, .allocator = a };
- // Clamped, so a firmware asking for more than the heap affords still starts. See `max_cols`.
- cur_winsize = .{
- .rows = @min(rows, max_rows),
- .cols = @min(cols, max_cols),
- .x_pixel = 0,
- .y_pixel = 0,
- };
-
- const allocs = pardes.allocators.init(a);
- // `std.Io.failing` and not a real Io: every path in the core that would perform I/O is behind
- // the Host vtable, and the ones that are not are the ones this platform does not have.
- pardes.image.start(std.Io.failing, allocs.image);
- pardes.syntax.start(allocs.tree_sitter);
-
- vx = vaxis.init(std.Io.failing, a, &env_map, .{}) catch |err| return errCode(err);
- vx.resize(a, &out, vaxisSize()) catch |err| return errCode(err);
-
- // Ask the terminal what it is. Both halves are pure byte writers, which is the whole reason this
- // works over a serial line: the replies arrive as ordinary input and are parsed like any key.
- vx.enterAltScreen(&out) catch |err| return errCode(err);
- vx.queryTerminalSend(&out) catch |err| return errCode(err);
-
- // MOUSE REPORTING, spelled out here rather than taken from `vx.setMouseMode`.
- //
- // vaxis enables `1002;1003;1004;1006`, and 1003 is ANY-MOTION tracking: the terminal reports
- // every cell the pointer crosses with no button held. On a 115200 line that is unaffordable -
- // one sweep across this grid is dozens of reports of ~15 bytes each, and each one arrives as
- // input that the editor must parse while it is trying to paint. Worse, it arrives whether or not
- // anybody wants it, so moving the mouse over the window would starve typing.
- //
- // 1002 reports presses, releases and motion WHILE A BUTTON IS HELD, which is exactly the set a
- // click and a drag-select need. 1004 is focus in/out, which `apply` already handles. 1006 is the
- // SGR encoding: unlike the original X10 form it is not limited to column 223, which a grid this
- // small does not need today but costs nothing to have and cannot be added later without the
- // terminal disagreeing with the editor about where the pointer is.
- out.writeAll("\x1b[?1002;1004;1006h") catch |err| return errCode(err);
- out.flush() catch |err| return errCode(err);
-
- // The CLAMPED geometry, because the core and vaxis must agree on the grid and vaxis was just
- // sized to `cur_winsize`.
- core = pardes.Pardes.init(allocs.pardes, .{
- .cols = cur_winsize.cols,
- .rows = cur_winsize.rows,
- .frame_allocator = allocs.frame,
- .image_allocator = allocs.image,
- .tree_sitter_allocator = allocs.tree_sitter,
- }) catch |err| return errCode(err);
-
- dirty = true;
- return 0;
-}
-
-/// Raw bytes off the wire: keystrokes, capability replies, and in-band resize reports. All three are
-/// the same kind of thing to `vaxis.Parser`, and this function does not distinguish them.
-export fn pardes_p4_input(ptr: [*]const u8, len: usize) callconv(.c) void {
- const c = core orelse return;
-
- // Append, dropping the oldest on overflow: a full buffer means the parser is stuck on a
- // malformed sequence, and keeping the tail is what lets it resynchronise.
- const room = in_buf.len - in_len;
- const take = @min(room, len);
- if (take < len) {
- in_len = 0;
- @memcpy(in_buf[0..@min(len, in_buf.len)], ptr[0..@min(len, in_buf.len)]);
- in_len = @min(len, in_buf.len);
- } else {
- @memcpy(in_buf[in_len..][0..take], ptr[0..take]);
- in_len += take;
- }
-
- drainInput(c, false);
-}
-
-/// Parse what has accumulated, applying every event it yields.
-///
-/// THE LONE ESCAPE IS AMBIGUOUS, and on this transport it is ambiguous constantly. `vaxis.Parser`
-/// resolves a buffer containing nothing but `0x1b` as the Escape KEY - deliberately, and correctly
-/// for a real terminal, where the kernel hands over a whole escape sequence in one read so a solitary
-/// ESC really does mean the key. A 115200 serial line hands over one byte at a time: 87 us apart,
-/// which is an eternity to this loop. So the first byte of EVERY escape sequence arrived alone and
-/// was resolved as Escape, and the rest arrived as ordinary keys.
-///
-/// That is not a mouse bug, though it is why the mouse did not work: a click report came through as
-/// ten key presses - Escape, `[`, `<`, `0`, ... - and the `0` among them is "go to column zero" in
-/// normal mode, which is exactly where the cursor kept landing. Arrow keys, function keys and the
-/// host's in-band resize reports were all being shredded the same way.
-///
-/// Longer partial sequences were never affected: the CSI scanner returns `n == 0` for "no final byte
-/// yet", and the loop below keeps those bytes. Only the one-byte case needed an answer, because it is
-/// the only one the parser answers wrongly instead of declining.
-///
-/// `force` is how a real Escape keypress still works: `pardes_p4_tick` calls with it set once the
-/// hold has lasted longer than any serial line would take to deliver the next byte.
-fn drainInput(c: *pardes.Pardes, force: bool) void {
- var off: usize = 0;
- while (off < in_len) {
- if (!force and in_len - off == 1 and in_buf[off] == 0x1b) break;
- const res = parser.parse(in_buf[off..in_len], gpa()) catch break;
- if (res.n == 0) break; // incomplete: wait for more bytes
- off += res.n;
- if (res.event) |ev| apply(c, ev);
- }
- // Keep whatever was not consumed: the tail of a split escape sequence.
- if (off > 0) {
- std.mem.copyForwards(u8, in_buf[0 .. in_len - off], in_buf[off..in_len]);
- in_len -= off;
- }
- // Start or clear the hold. `esc_held_at` is only ever set for a buffer that is exactly one ESC,
- // so a partial CSI - which the parser already declines - does not start a timer it does not need.
- if (in_len == 1 and in_buf[0] == 0x1b) {
- if (esc_held_at == null) esc_held_at = last_now_ms;
- } else esc_held_at = null;
-}
-
-/// One parsed vaxis event applied to the core. Mirrors the tty shell's `apply`
-/// (`src/tty/tty.zig:926-985`), minus everything that needs an OS.
-fn apply(c: *pardes.Pardes, ev: vaxis.Event) void {
- switch (ev) {
- .key_press => |key| if (in_paste) {
- // Between the brackets a key is DATA, never a command. vaxis gives control bytes no
- // text at all, so a line break inside a paste arrives as a bare CR (Key.enter) or, from
- // a terminal that does not translate them, as ctrl+j.
- const text = key.text orelse "";
- const cp = mapKey(effCp(key));
- const bytes: []const u8 = if (text.len > 0)
- text
- else if (cp == pardes.Key.tab)
- "\t"
- else if (cp == pardes.Key.enter or (key.mods.ctrl and cp == 'j'))
- "\n"
- else
- "";
- if (bytes.len > 0) paste_buf.appendSlice(gpa(), bytes) catch {};
- } else {
- c.update(.{ .key = .{
- .cp = mapKey(effCp(key)),
- .text = key.text orelse "",
- .ctrl = key.mods.ctrl,
- .alt = key.mods.alt,
- .shift = key.mods.shift,
- } });
- dirty = true;
- },
- .paste_start => {
- paste_buf.clearRetainingCapacity();
- in_paste = true;
- },
- .paste_end => {
- in_paste = false;
- if (paste_buf.items.len > 0) {
- c.update(.{ .paste = paste_buf.items });
- dirty = true;
- }
- paste_buf.clearRetainingCapacity();
- },
- // OSC 52. The bytes are the parser's, allocated from our own allocator, so they are freed
- // here rather than leaked - the core copies whatever it keeps.
- .paste => |text| {
- c.update(.{ .paste = text });
- gpa().free(text);
- dirty = true;
- },
- .mouse => |m| {
- const button: ?pardes.Mouse.Button = switch (m.button) {
- .left => .left,
- .middle => .middle,
- .right => .right,
- .wheel_up => .wheel_up,
- .wheel_down => .wheel_down,
- .wheel_left => .wheel_left,
- .wheel_right => .wheel_right,
- .none => .none,
- else => null,
- };
- if (button) |b| {
- c.update(.{ .mouse = .{
- .button = b,
- .kind = switch (m.type) {
- .press => .press,
- .release => .release,
- .motion => .motion,
- .drag => .drag,
- },
- .col = @intCast(m.col),
- .row = @intCast(m.row),
- .ctrl = m.mods.ctrl,
- } });
- dirty = true;
- }
- },
- // The only way this platform learns its size, and the one place a 384 KiB heap shows through
- // to the user. Two things happen here that the tty shell does not need.
- //
- // CLAMPED, because the grids do not fit an arbitrary terminal: vaxis keeps a `Screen` and an
- // `InternalScreen`, pardes keeps its own `Surface` and `previous_cells`, so every cell is
- // paid for four times. Measured on the die - 40x12 initialises with room to spare, 80x24
- // exhausts the heap and `Pardes.init` returns OutOfMemory with 9,128 bytes left. The host's
- // terminal is normally larger than the board can render, so the editor takes a corner of it
- // instead of refusing to start.
- //
- // ATOMIC, because `Vaxis.resize` deinits both screens BEFORE allocating the replacements
- // (Vaxis.zig:194-206), so a failed resize leaves vaxis with freed screens and renders
- // nothing at all. That is exactly how this was found: the host bridge injects a size report
- // on attach, the 80x24 it reported could not be allocated, and an editor that had just drawn
- // its interface went silent. A failure now puts the previous geometry back.
- .winsize => |ws| {
- const want: vaxis.Winsize = .{
- .rows = @min(ws.rows, max_rows),
- .cols = @min(ws.cols, max_cols),
- .x_pixel = ws.x_pixel,
- .y_pixel = ws.y_pixel,
- };
- if (want.cols == cur_winsize.cols and want.rows == cur_winsize.rows) return;
- const previous = cur_winsize;
- // vaxis is only resized when it is the thing doing the rendering. Under `direct_emit` its
- // grids are a single cell and stay that way - see `vaxisSize` - so there is nothing here
- // to reallocate, which also means a resize can no longer fail for want of two grids.
- if (!direct_emit) {
- vx.resize(gpa(), &out, want) catch {
- vx.resize(gpa(), &out, previous) catch {};
- return;
- };
- }
- cur_winsize = want;
- c.update(.{ .resize = .{ .cols = want.cols, .rows = want.rows } });
- dirty = true;
- },
- // A TTY cannot report a pointer leaving its grid, so losing focus is the only reliable
- // pointer-leave signal there is.
- .focus_out => {
- c.update(.pointer_leave);
- dirty = true;
- },
- .focus_in, .mouse_leave => {},
- // Capability replies. vaxis's own Loop sets these fields directly (`Loop.zig:377-403`);
- // with no Loop, this is where they land. DA1 is the terminator: every terminal answers it
- // last, so it is the signal that the whole handshake is in and the detected features can be
- // switched on.
- .cap_kitty_keyboard => vx.caps.kitty_keyboard = true,
- .cap_kitty_graphics => vx.caps.kitty_graphics = true,
- .cap_rgb => vx.caps.rgb = true,
- .cap_unicode => {
- vx.caps.unicode = .unicode;
- vx.screen.width_method = .unicode;
- },
- .cap_sgr_pixels => vx.caps.sgr_pixels = true,
- .cap_color_scheme_updates => vx.caps.color_scheme_updates = true,
- .cap_multi_cursor => vx.caps.multi_cursor = true,
- .cap_da1 => {
- vx.enableDetectedFeatures(&out) catch {};
- out.flush() catch {};
- dirty = true;
- },
- .color_report, .color_scheme => {},
- .key_release => {},
- }
-}
-
-/// The effective codepoint the way vaxis's own `Key.matches` sees it: a single-character `text`
-/// wins, because the terminal has already resolved shift; otherwise the shifted codepoint.
-fn effCp(key: vaxis.Key) u21 {
- if (key.text) |t| {
- const view = std.unicode.Utf8View.init(t) catch return key.codepoint;
- var it = view.iterator();
- if (it.nextCodepoint()) |cp| {
- if (it.nextCodepoint() == null) return cp;
- }
- }
- return key.shifted_codepoint orelse key.codepoint;
-}
-
-/// vaxis functional-key codepoints -> core constants. The ASCII ones already coincide, so
-/// enter/tab/escape/backspace pass straight through.
-fn mapKey(cp: u21) u21 {
- return switch (cp) {
- vaxis.Key.up => pardes.Key.up,
- vaxis.Key.down => pardes.Key.down,
- vaxis.Key.left => pardes.Key.left,
- vaxis.Key.right => pardes.Key.right,
- vaxis.Key.home => pardes.Key.home,
- vaxis.Key.end => pardes.Key.end,
- vaxis.Key.page_up => pardes.Key.page_up,
- vaxis.Key.page_down => pardes.Key.page_down,
- vaxis.Key.delete => pardes.Key.delete,
- else => cp,
- };
-}
-export fn pardes_p4_tick(now_ms: u64) callconv(.c) void {
- const c = core orelse return;
- last_now_ms = now_ms;
- // The held Escape, released. Anything still waiting after this long is a key the human pressed,
- // not the head of a sequence: the next byte of a real sequence is 87 us behind on this line, and
- // even a slow terminal emulator answers a query in well under a millisecond. Ten is generous by
- // two orders of magnitude and imperceptible to the person pressing it - the same trade every
- // terminal editor makes for the same reason.
- if (esc_held_at) |at| {
- if (now_ms -% at >= esc_hold_ms) {
- drainInput(c, true);
- dirty = true;
- }
- }
- if (c.animationActive()) {
- c.update(.tick);
- dirty = true;
- }
-}
-
-/// How long a lone ESC waits for a second byte before it counts as the Escape key.
-const esc_hold_ms = 10;
-
-/// The last timestamp `pardes_p4_tick` was given, so `drainInput` can date a hold without needing a
-/// clock of its own - there is no clock on this side of the ABI.
-var last_now_ms: u64 = 0;
-
-/// When the buffer became a lone ESC, or null when it is not holding one.
-var esc_held_at: ?u64 = null;
-
-export fn pardes_p4_wants_frame() callconv(.c) bool {
- const c = core orelse return false;
- return dirty or c.animationActive();
-}
-
-export fn pardes_p4_render() callconv(.c) u32 {
- const c = core orelse return 0;
- c.pump(.{ .ctx = null, .vtable = &pardes_host }) catch |err| return errCode(err);
- dirty = false;
- return 0;
-}
-
-export fn pardes_p4_quit() callconv(.c) bool {
- const c = core orelse return true;
- return c.quit;
-}
-
-// ------------------------------------------------------------------------------------ the host
-
-const pardes_host: pardes.Host.VTable = .{ .push_present = present, .pull_gpio_toggle = gpioToggle };
-
-/// The `Gpio` word's one seam to the board. Nothing here knows what a pad is; it forwards, and
-/// answers false when the firmware brought none, which is what puts "gpio: NoPads" on the message
-/// row rather than a trap.
-fn gpioToggle(_: ?*anyopaque, pin: u16, was: *u8, now: *u8) bool {
- const f = host_gpio orelse return false;
- return f(out_ctx, pin, was, now);
-}
-
-/// The canonical surface -> the wire. Same shape as the tty shell's (`src/tty/tty.zig:1096`) minus
-/// the panel compositor and the kitty image path: neither has a reason to exist on a board with no
-/// pixels. Where the tty shell hands every cell to vaxis and lets it diff, this diffs against the
-/// Surface itself and can then emit the ANSI directly - see `direct_emit`.
-fn present(_: ?*anyopaque, surface: *const pardes.Surface) void {
- const t0 = cycles();
- const win = vx.window();
- const n = @as(usize, surface.cols) * @as(usize, surface.rows);
-
- // THE SHADOW GRID. Copying all 480 cells into vaxis every frame cost 6.75 ms on the die - 57%
- // of a keystroke, and it was paid whether or not anything changed: a second render with nothing
- // new measured the same as the first. vaxis already diffs its own grid against the terminal, but
- // it can only do that AFTER being told every cell, and being told is the expensive part
- // (`writeCell` builds a vaxis `Cell`, which carries an always-null image placement).
- //
- // So keep the previous Surface and tell vaxis only what moved. `Cell.visuallyEqual` is the
- // right comparison and already exists for the panel compositor's benefit: it ignores scratch
- // bytes past `len` and treats any two default cells as equal, so it cannot manufacture a write.
- //
- // STATIC, and that is not a micro-optimisation - it is a bug fix. The first version allocated
- // this from the editor's heap, and on a board whose 384 KiB is already nearly spoken for that
- // was enough to make `vx.resize` fail: a resize then hit its OOM path, restored the previous
- // geometry and returned, so the screen was never repainted. Measured as a resize emitting 80
- // bytes where it had emitted 1,392. The grid is bounded by `max_cols` x `max_rows` at comptime,
- // so it belongs in `.bss` where it cannot compete with anything.
- const full = !shadow_grid or prev_cols != surface.cols or prev_rows != surface.rows;
- emit_bytes = 0;
- if (full) {
- prev_cols = surface.cols;
- prev_rows = surface.rows;
- if (direct_emit) {
- // Reset first: a `2J` while a non-default background is active fills the screen with it.
- emitRaw("\x1b[0m\x1b[2J") catch return;
- emit_style = .{};
- emit_col = -1;
- } else win.clear();
- }
- const usable = shadow_grid and n <= prev_cells.len;
-
- var y: u16 = 0;
- while (y < surface.rows) : (y += 1) {
- const row0 = @as(usize, y) * @as(usize, surface.cols);
- const src = surface.cells[row0..][0..surface.cols];
-
- // A ROW AT A TIME FIRST. `Surface.cells` is contiguous and row-major, so a whole row is one
- // `memcmp` against the shadow - and on a keystroke eleven of twelve rows are untouched. The
- // per-cell loop below is ~40 branchy comparisons where this is one call over 1,120 bytes;
- // measured, the walk fell from 246 us to a fraction of it. Byte equality implies visual
- // equality (see `sameCell`), so a row that compares equal cannot be hiding a changed cell -
- // and a row that differs only in padding falls through to the per-cell path, which is
- // correct and merely slower.
- if (usable and !full) {
- const shadow = prev_cells[row0..][0..surface.cols];
- if (sameBytes(std.mem.sliceAsBytes(src), std.mem.sliceAsBytes(shadow))) continue;
- }
-
- var x: u16 = 0;
- while (x < surface.cols) : (x += 1) {
- const cell = &src[x];
- const idx = row0 + @as(usize, x);
- if (usable) {
- if (!full and sameCell(cell, &prev_cells[idx])) continue;
- prev_cells[idx] = cell.*;
- } else if (cell.default) continue;
-
- writeOne(win, x, y, cell, surface.cols) catch return;
- }
- }
- if (direct_emit) {
- // BOTH branches have to reach the packet boundary, and the second one is easy to forget:
- // measured, a frame that only hid the cursor was 6 bytes and cost 4014 us at 640 characters
- // against 3863 at 320, because 6 bytes never fills a packet and waited out the bridge's
- // timer. Hiding an already-hidden cursor is as idempotent as positioning it twice.
- if (surface.cursor) |cur| {
- cup(cur.y, cur.x) catch return;
- emitRaw("\x1b[?25h") catch return;
- emit_col = -1;
- while (emit_bytes < emit_min_frame) cup(cur.y, cur.x) catch return;
- } else {
- while (emit_bytes < emit_min_frame) emitRaw("\x1b[?25l") catch return;
- }
- } else if (surface.cursor) |cur| {
- win.showCursor(cur.x, cur.y);
- } else win.hideCursor();
- const t1 = cycles();
-
- // vaxis diffs against its own shadow grid, so this writes only what changed - which is what
- // makes an editor usable at 11.9 KB/s. With `direct_emit` that diff has already happened, one
- // stage earlier and against the Surface itself, so there is nothing left here to do.
- if (!direct_emit) vx.render(&out) catch return;
- const t2 = cycles();
- out.flush() catch return;
- const t3 = cycles();
-
- prof_copy_cy = t1 -% t0;
- prof_render_cy = t2 -% t1;
- prof_flush_cy = t3 -% t2;
-}
-
-/// One cell to the wire, either through vaxis or straight out.
-inline fn writeOne(win: vaxis.Window, x: u16, y: u16, cell: *const pardes.Cell, cols: u16) !void {
- if (!direct_emit) {
- // Changed TO default. `win.clear()` is what used to blank these, and it is not run on an
- // incremental frame, so say it explicitly.
- if (cell.default) return win.writeCell(x, y, .{ .char = .{ .grapheme = " " }, .style = .{} });
- return win.writeCell(x, y, .{
- .char = .{ .grapheme = cell.grapheme() },
- .style = vaxisStyle(cell.style),
- });
- }
-
- if (emit_row != y or emit_col != x) {
- try cup(y, x);
- emit_row = y;
- emit_col = @intCast(x);
- }
-
- const style: pardes.CellStyle = if (cell.default) .{} else cell.style;
- if (!std.meta.eql(emit_style, style)) {
- try emitStyle(style);
- emit_style = style;
- }
-
- try emitRaw(if (cell.default) " " else cell.grapheme());
-
- // Where the terminal's cursor now is. A single printable ASCII byte advanced it exactly one
- // column; anything else - a wide glyph, a cluster, the spacer cell pardes writes after a wide
- // one - is not worth predicting, so give up and let the next cell emit an absolute CUP. The last
- // column is given up on too, because whether the cursor rests on it or has wrapped past it
- // depends on the terminal's deferred-wrap behaviour, and the two disagree by a whole row.
- if (x + 1 < cols and cell.len == 1 and cell.text[0] >= 0x20 and cell.text[0] < 0x7f) {
- emit_col += 1;
- } else emit_col = -1;
-}
-
-/// A style as an absolute SGR, always opening with a reset.
-///
-/// Absolute rather than a delta from whatever is currently on, and that is what keeps it short
-/// enough to be worth having: no per-attribute off-codes, no state to keep beyond the last style
-/// emitted, and a frame that gets cut off cannot leave a later cell wearing an earlier one's colour.
-/// It costs a few bytes on a style change, against the ~9 of CUP a changed cell is paying anyway.
-fn emitStyle(s: pardes.CellStyle) !void {
- try emitRaw("\x1b[0");
- if (s.bold) try emitRaw(";1");
- if (s.dim) try emitRaw(";2");
- if (s.italic) try emitRaw(";3");
- if (s.blink) try emitRaw(";5");
- if (s.reverse) try emitRaw(";7");
- if (s.invisible) try emitRaw(";8");
- if (s.strikethrough) try emitRaw(";9");
- try emitRaw(switch (s.ul) {
- .off => "",
- .single => ";4",
- .double => ";4:2",
- .curly => ";4:3",
- .dotted => ";4:4",
- .dashed => ";4:5",
- });
- try emitColor(s.fg, 30);
- try emitColor(s.bg, 40);
- try emitRaw("m");
-}
-
-/// `base` is 30 for a foreground and 40 for a background, which is the only thing separating the two
-/// in every form SGR has for a colour: 30-37 against 40-47, 90-97 against 100-107, 38 against 48.
-fn emitColor(c: pardes.Color, comptime base: u16) !void {
- var b: [20]u8 = undefined;
- var i: usize = 0;
- switch (c) {
- // Already said by the reset this SGR opens with.
- .default => return,
- .index => |n| {
- b[i] = ';';
- i += 1;
- if (n < 8) {
- i += dec(b[i..], base + n);
- } else if (n < 16) {
- i += dec(b[i..], base + 60 + (n - 8));
- } else {
- i += dec(b[i..], base + 8);
- i += lit(b[i..], ";5;");
- i += dec(b[i..], n);
- }
- },
- .rgb => |v| {
- b[i] = ';';
- i += 1;
- i += dec(b[i..], base + 8);
- i += lit(b[i..], ";2;");
- for (v, 0..) |component, k| {
- if (k != 0) {
- b[i] = ';';
- i += 1;
- }
- i += dec(b[i..], component);
- }
- },
- }
- try emitRaw(b[0..i]);
-}
-
-/// Absolute cursor positioning, hand-rolled rather than through `out.print`.
-///
-/// Not for elegance: this is the single most frequent sequence the emitter produces, at least one per
-/// changed run, and `std.fmt` brings a whole format-string interpreter to write at most two digits.
-/// The grid is bounded by `max_cols` x `max_rows`, so nothing here can exceed three.
-fn cup(row: u16, col: u16) !void {
- var b: [12]u8 = undefined;
- var i: usize = lit(&b, "\x1b[");
- i += dec(b[i..], row + 1);
- b[i] = ';';
- i += 1;
- i += dec(b[i..], col + 1);
- b[i] = 'H';
- i += 1;
- try emitRaw(b[0..i]);
-}
-
-/// Decimal, least significant digit first into a scratch buffer and then reversed. Five digits is
-/// every `u16`, so there is no fallback to `std.fmt` and no value this cannot write.
-fn dec(buf: []u8, v: u16) usize {
- var digits: [5]u8 = undefined;
- var n: usize = 0;
- var rest = v;
- while (true) {
- digits[n] = '0' + @as(u8, @intCast(rest % 10));
- n += 1;
- rest /= 10;
- if (rest == 0) break;
- }
- for (0..n) |k| buf[k] = digits[n - 1 - k];
- return n;
-}
-
-inline fn lit(buf: []u8, comptime s: []const u8) usize {
- @memcpy(buf[0..s.len], s);
- return s.len;
-}
-
-/// Every direct-emit byte goes through here, because the count is what the padding below needs.
-inline fn emitRaw(bytes: []const u8) !void {
- emit_bytes += bytes.len;
- try out.writeAll(bytes);
-}
-
-/// A/B switch for the emitter above, on the same terms as `shadow_grid`: false routes every cell back
-/// through vaxis, which is the reference. vaxis's own diff measured 631 us of a 4.37 ms keystroke and
-/// all of it was redundant - `present` has already worked out which cells moved, so vaxis was being
-/// told the answer and then computing it again from scratch.
-const direct_emit = true;
-
-/// THE FRAME HAS A MINIMUM SIZE, and it is the USB bridge's, not the terminal's.
-///
-/// The board is wired to the host through a CH340, and 32 is not a guess: it is `wMaxPacketSize` of
-/// endpoint 0x82, the bulk IN, as the device itself reports it - a full-speed 0x0020. The bridge
-/// forwards a packet when the packet is FULL, so a frame shorter than that sits there until an
-/// internal timer gives up on more, which is worth about a millisecond - a quarter of the budget.
-///
-/// Measured, at the same board cost and with the screen byte-identical: a 21-byte frame round-trips
-/// in 4817 us and the same frame padded to 49 bytes in 3814 us. MORE BYTES, ARRIVING SOONER. It also
-/// explains why routing through vaxis looked competitive - its frames are 81 bytes, so they fill a
-/// packet by accident and never wait.
-///
-/// So pad to the packet boundary. The filler is repeated absolute cursor positioning: idempotent,
-/// already the sequence the emitter ends on, and it cannot alter a cell. This is the same bargain as
-/// an Ethernet runt frame - the medium has a minimum and the sender pays it - and it is a real
-/// trade, not free: the wasted bytes are wire time that delays a LATER frame, so it is only worth it
-/// while the frame is small, which is exactly when it applies.
-const emit_min_frame: usize = 32;
-
-/// Bytes emitted this frame, for `emit_min_frame`.
-var emit_bytes: usize = 0;
-
-/// What the terminal is currently wearing and where its cursor is, so that a run of changed cells in
-/// one row costs one CUP and one SGR rather than one of each per cell. `emit_col` is signed because
-/// -1 means "no longer known" - see `writeOne`.
-var emit_style: pardes.CellStyle = .{};
-var emit_row: u16 = 0;
-var emit_col: i32 = -1;
-
-/// A/B switch, kept because this optimisation is exactly the kind that can be right about latency
-/// and wrong about the screen. With it false, `present` behaves as it did before the shadow grid -
-/// clear and write every cell - which is the reference any measurement of it should be compared
-/// against, and the way to tell a rendering bug from a rendering difference.
-const shadow_grid = true;
-
-/// The previous Surface, cell for cell, sized for the largest grid this board can drive. In `.bss`
-/// rather than on the heap: see `present`. `prev_cols`/`prev_rows` being zero on the first frame is
-/// what makes that frame a full one.
-var prev_cells: [@as(usize, max_cols) * @as(usize, max_rows)]pardes.Cell = if (shadow_grid) @splat(.{}) else undefined;
-var prev_cols: u16 = 0;
-var prev_rows: u16 = 0;
-
-/// Cell equality for the shadow grid, as bytes.
-///
-/// `Cell.visuallyEqual` is the semantically exact answer and it is too slow to ask 480 times a
-/// frame: `std.meta.eql` on a `CellStyle` recurses through a colour union and eight booleans, and
-/// the walk measured 1.45 ms - about 270 cycles per comparison of a ~28-byte struct.
-///
-/// Byte equality IMPLIES visual equality, so this can never claim two different cells are the same.
-/// It can miss an equality - scratch bytes past `len`, or padding - and the only cost of that is one
-/// redundant `writeCell` that vaxis then diffs away. Defaults are still handled by meaning rather
-/// than by bytes, because an unpainted cell's text and style are whatever the last frame left there.
-inline fn sameCell(a: *const pardes.Cell, b: *const pardes.Cell) bool {
- if (a.default or b.default) return a.default and b.default;
- return sameBytes(std.mem.asBytes(a), std.mem.asBytes(b));
-}
-
-/// Exact byte equality, a word at a time when both spans are aligned for it.
-///
-/// This comparison is the firmware's largest read by a wide margin - two 13 KB streams every frame -
-/// and it measured 3.2 cycles per byte, about four times what word-wide loads should need, which is
-/// what a byte-at-a-time loop looks like. The answer is identical either way: this is still exact
-/// byte equality, so it keeps the property the whole diff rests on, that byte equality implies
-/// visual equality.
-///
-/// The alignment test is a RUNTIME one because `Cell` has alignment 1 - it is all `u8` fields - so
-/// whether a row begins on a word boundary is a property of whoever allocated the Surface and not of
-/// the type. A row is 40 cells of 26 bytes, which is divisible by four, so if the base is aligned
-/// every row is. When it is not, the byte loop is still here.
-inline fn sameBytes(a: []const u8, b: []const u8) bool {
- if (a.len != b.len) return false;
- if ((@intFromPtr(a.ptr) | @intFromPtr(b.ptr)) & 3 == 0) {
- const n = a.len / 4;
- const wa: [*]align(4) const u32 = @ptrCast(@alignCast(a.ptr));
- const wb: [*]align(4) const u32 = @ptrCast(@alignCast(b.ptr));
- for (wa[0..n], wb[0..n]) |x, y| {
- if (x != y) return false;
- }
- return std.mem.eql(u8, a[n * 4 ..], b[n * 4 ..]);
- }
- return std.mem.eql(u8, a, b);
-}
-
-// ------------------------------------------------------------------ where a frame's time goes
-//
-// A frame has three stages and they want different fixes, so the firmware is given all three rather
-// than one total. Measured on the die, a render costs ~11 ms whether or not anything changed, which
-// says the cost is the unconditional walk and not the edit - but "the walk" is two walks, the copy
-// into vaxis's grid and vaxis's own diff, and only one of them is ours to change.
-//
-// Two CSR reads per stage. `cycle` is the unprivileged counter, read high-low-high because two
-// 32-bit halves can straddle a wrap.
-var prof_copy_cy: u64 = 0;
-var prof_render_cy: u64 = 0;
-var prof_flush_cy: u64 = 0;
-
-inline fn cycles() u64 {
- if (builtin.cpu.arch != .riscv32) return 0;
- while (true) {
- const hi0 = asm volatile ("csrr %[o], cycleh"
- : [o] "=r" (-> u32),
- );
- const lo = asm volatile ("csrr %[o], cycle"
- : [o] "=r" (-> u32),
- );
- const hi1 = asm volatile ("csrr %[o], cycleh"
- : [o] "=r" (-> u32),
- );
- if (hi0 == hi1) return (@as(u64, hi0) << 32) | lo;
- }
-}
-
-/// The last frame's three stages, in cycles. Zero on any platform without the CSR.
-export fn pardes_p4_frame_prof(copy: *u64, render: *u64, flush: *u64) callconv(.c) void {
- copy.* = prof_copy_cy;
- render.* = prof_render_cy;
- flush.* = prof_flush_cy;
-}
-
-fn vaxisStyle(s: pardes.CellStyle) vaxis.Style {
- return .{
- .fg = vaxisColor(s.fg),
- .bg = vaxisColor(s.bg),
- .bold = s.bold,
- .dim = s.dim,
- .italic = s.italic,
- .blink = s.blink,
- .reverse = s.reverse,
- .invisible = s.invisible,
- .strikethrough = s.strikethrough,
- .ul_style = switch (s.ul) {
- .off => .off,
- .single => .single,
- .double => .double,
- .curly => .curly,
- .dotted => .dotted,
- .dashed => .dashed,
- },
- };
-}
-
-fn vaxisColor(c: pardes.Color) vaxis.Color {
- return switch (c) {
- .default => .default,
- .index => |i| .{ .index = i },
- .rgb => |rgb| .{ .rgb = rgb },
- };
-}
-
-/// Errors cross the ABI as small non-zero integers. `@intFromError` is not stable across builds, so
-/// it is not used: the firmware only reports the number, and a stable-looking value that silently
-/// changed meaning would be worse than an opaque one.
-fn errCode(err: anyerror) u32 {
- return switch (err) {
- error.OutOfMemory => 1,
- error.WriteFailed => 2,
- else => 255,
- };
-}