diff options
| author | Gabriel Schneider <[email protected]> | 2026-08-26 02:04:47 -0300 |
|---|---|---|
| committer | Gabriel Schneider <[email protected]> | 2026-08-26 02:04:47 -0300 |
| commit | fc263fc4b1ee36a3c4bfd7d730cccb57ebe1064a (patch) | |
| tree | ec4deebe06728c2a2d147823e18e545be1aee2b0 | |
| parent | 2d3148247e6555b7b33bd532ae538d7a4358e160 (diff) | |
| download | pardes-fc263fc4b1ee36a3c4bfd7d730cccb57ebe1064a.tar.gz pardes-fc263fc4b1ee36a3c4bfd7d730cccb57ebe1064a.zip | |
Raise the board's grid to 56x14, and make it a build option
The 40x12 ceiling was never about the screen. It was about memory, and the comment above
`max_cols` said so: "every cell is paid for four times over: vaxis keeps a Screen and an
InternalScreen, pardes keeps its own Surface and previous_cells". Two of those four are now
dead weight - with `direct_emit` the emitter diffs the Surface against its own shadow and
writes the escapes itself, so vaxis's two grids are allocated, never read, and were the
largest single claim on a 384 KiB heap. `init` sizes them to ONE CELL. vaxis still does the
work only it can do: the alternate screen, the capability queries, and parsing everything
that comes back.
That removes the memory ceiling entirely - the heap now reports 336 KB free at every
geometry tried, including ones that used to fail - and leaves latency as the only limit,
which is the honest one: every frame walks the whole grid.
## Measured on the die, 0.87 us per cell
geometry cells round trip
40x12 480 3,628 us the old default
56x14 784 3,930 us the new one
56x16 896 3,965 us
60x18 1,080 4,114 us
64x20 1,280 4,281 us
80x24 1,920 4,809 us
100x30 3,000 5,743 us
120x36 4,320 6,923 us
140x42 5,880 8,310 us the largest that runs
160x48 7,680 links, then traps
200x60 12,000 does not link
56x14 is 63% more area and 40% more width than 40x12 and still holds the 4 ms this port was
built to. 56x16 was tried first: 3,965 us on the bench instrument but 4,029 on the
phase-randomised one, which is over, and the two instruments differ by about 50 us
systematically - so the wider grid went and two rows stayed behind. Width is worth more than
height for reading code.
The two failures at the top are worth naming precisely because they are different failures.
200x60 does not link: `.bss will not fit in region l2mem, overflowed by 76036 bytes`, that
`.bss` being the shell's shadow copy of the grid, sized at comptime. 160x48 links and then
TRAPS at boot - the same region pressure arriving at runtime as a collision rather than as a
diagnostic. Neither is a heap problem any more, which is the interesting part: the heap has
336 KB spare while `.bss` runs out.
`-Dp4-cols` / `-Dp4-rows` because none of the above is a constant. 80x24 is one flag away for
anyone who would rather have the classic terminal than the millisecond.
Verified at the new geometry rather than assumed: the A/B against the reference path - vaxis
rendering, full repaint, `shadow_grid` and `direct_emit` both off - is identical in every
cell, characters and resolved style. That matters more here than usual because the emitter's
column arithmetic has a special case at the last column, and 40 was the only width it had
ever been asked about. snap 95/95, hxdiff 481/0, hxparity 561/0, unit-test, both A/B arms,
tty/p4/gui, and the board's own `p4-bench --check`.
| -rw-r--r-- | build.zig | 62 | ||||
| -rw-r--r-- | src/p4.zig | 51 |
2 files changed, 77 insertions, 36 deletions
@@ -70,30 +70,32 @@ pub fn build(b: *std.Build) void { .cpu_model = .{ .explicit = &std.Target.riscv.cpu.generic_rv32 }, .cpu_features_add = riscvFeatures(&.{ .m, .a, .f, .c, .zicsr, .zifencei }), }; - const requested_target = b.standardTargetOptions(.{ .default_target = switch (platform) { - .macos => if (builtin.os.tag.isDarwin()) .{ - .cpu_arch = builtin.target.cpu.arch, - .os_tag = .macos, - .os_version_min = .{ .semver = macos_min_version }, - } else .{}, - // The P4 firmware target, spelled out here so `-Dplatform=p4` alone is a - // working command line. The CPU FEATURES are part of that spelling and - // are not optional: the object this build emits is linked into an image - // whose other halves are compiled `generic_rv32+m+a+f+c+zicsr+zifencei`, - // and `f` decides the float ABI. Leaving the model implicit produced a - // soft-float object and `ld.lld: cannot link object files with different - // floating-point ABI` — at LINK time in the other repo, far from here. - // Espressif's GCC adds the vendor extensions xesploop/xespv2p1 on top; - // upstream LLVM has neither and ordinary code never emits them, so this - // matches the base ISA the firmware uses exactly. - .p4 => p4_target, - .tty, .gui, .web => .{ - .cpu_arch = .x86_64, - .os_tag = .linux, - .abi = .gnu, - .glibc_version = .{ .major = 2, .minor = 38, .patch = 0 }, + const requested_target = b.standardTargetOptions(.{ + .default_target = switch (platform) { + .macos => if (builtin.os.tag.isDarwin()) .{ + .cpu_arch = builtin.target.cpu.arch, + .os_tag = .macos, + .os_version_min = .{ .semver = macos_min_version }, + } else .{}, + // The P4 firmware target, spelled out here so `-Dplatform=p4` alone is a + // working command line. The CPU FEATURES are part of that spelling and + // are not optional: the object this build emits is linked into an image + // whose other halves are compiled `generic_rv32+m+a+f+c+zicsr+zifencei`, + // and `f` decides the float ABI. Leaving the model implicit produced a + // soft-float object and `ld.lld: cannot link object files with different + // floating-point ABI` — at LINK time in the other repo, far from here. + // Espressif's GCC adds the vendor extensions xesploop/xespv2p1 on top; + // upstream LLVM has neither and ordinary code never emits them, so this + // matches the base ISA the firmware uses exactly. + .p4 => p4_target, + .tty, .gui, .web => .{ + .cpu_arch = .x86_64, + .os_tag = .linux, + .abi = .gnu, + .glibc_version = .{ .major = 2, .minor = 38, .patch = 0 }, + }, }, - } }); + }); // `standardTargetOptions` honours `default_target` ONLY when `-Dtarget` is absent, so the // documented `-Dplatform=p4 -Dtarget=riscv32-freestanding` discarded the CPU features above and // silently produced a soft-float object. The features are not a preference here - `f` decides @@ -346,6 +348,20 @@ pub fn build(b: *std.Build) void { opts.addOption(bool, "enable_tracy", tracy != null); opts.addOption(bool, "zls_backend", zls_backend); opts.addOption(bool, "mupdf", enable_mupdf); + // THE BOARD'S GRID, a build option because the right size is a measurement rather than a + // constant, and because two independent things limit it. + // + // LATENCY binds first, and linearly: every frame walks the whole grid, at 0.87 us per cell + // measured on the die. 56x14 is 784 cells and a 3,930 us round trip - 63% more area and 40% more + // width than the 40x12 it replaces, and still inside the 4 ms this port was built to hold. 56x16 + // was tried first and measured 4,029, which is over; 80x24 works and costs 4,809. The full curve + // is in `05-zig-p4/experiments/report.typ`. + // + // MEMORY binds much later, and only since vaxis's two unused grids stopped being allocated: + // 140x42 runs, 160x48 links and then traps, and 200x60 does not link at all - `.bss will not fit + // in region l2mem, overflowed by 76036 bytes`, that being the shell's shadow copy of the grid. + opts.addOption(u16, "p4_cols", b.option(u16, "p4-cols", "board grid width in cells (p4 only)") orelse 56); + opts.addOption(u16, "p4_rows", b.option(u16, "p4-rows", "board grid height in cells (p4 only)") orelse 14); // Meaningful only for the SDL shell. Keeping the platform condition here // prevents Config/EffectCode from describing tty/macOS/web as "prebuilt". opts.addOption(bool, "gui_shader_sources_prebuilt", gui_shader_sources_prebuilt); @@ -207,19 +207,39 @@ var in_paste: bool = false; /// starves input. var dirty: bool = true; -/// The largest grid this board can render, and the reason it is not the host's terminal size. +/// The largest grid this board can render, and the reason it is not just the host's terminal size. /// -/// Every cell is paid for four times over: vaxis keeps a `Screen` and an `InternalScreen`, pardes -/// keeps its own `Surface` and `previous_cells`. Against a 384 KiB heap that puts a hard ceiling on -/// the geometry, and it was measured rather than guessed - 40x12 initialises with room to spare, -/// 80x24 exhausts the heap and `Pardes.init` returns OutOfMemory with 9,128 bytes left. +/// A cell is paid for TWICE now, not four times: pardes keeps its `Surface` and this shell keeps a +/// shadow copy of it to diff against. vaxis used to keep a `Screen` and an `InternalScreen` as well, +/// and with `direct_emit` neither is ever read - the emitter diffs against the Surface and writes +/// the wire itself - so `init` sizes vaxis to a single cell and those two grids cost nothing. /// -/// Raising these is what PSRAM would buy: this board has 32 MB fitted and untrained. -pub const max_cols: u16 = 40; -pub const max_rows: u16 = 12; +/// Set with `-Dp4-cols` / `-Dp4-rows`, because the ceiling is a measurement rather than a constant +/// and it moves for two independent reasons: the 384 KiB heap, and the round trip, which grows with +/// the cell count because every frame walks the whole grid. See the geometry table in +/// `05-zig-p4/experiments/report.typ` for both curves. +/// +/// Raising these further is what PSRAM would buy: this board has 32 MB fitted and untrained. +pub const max_cols: u16 = @import("pardes_config").p4_cols; +pub const max_rows: u16 = @import("pardes_config").p4_rows; var cur_winsize: vaxis.Winsize = .{ .rows = max_rows, .cols = max_cols, .x_pixel = 0, .y_pixel = 0 }; +/// How big vaxis's own grids need to be. +/// +/// ONE CELL under `direct_emit`, because neither of them is ever read: vaxis keeps a `Screen` and an +/// `InternalScreen`, and the emitter diffs the Surface against its own shadow and writes the escapes +/// itself. Those two grids were the largest single claim on a 384 KiB heap and the reason the board +/// was held to 40x12 - the comment above `max_cols` used to say a cell was paid for four times over, +/// and this is what took it down to two. vaxis is still doing the work only it can do here: entering +/// the alternate screen, asking the terminal what it is, and parsing everything that comes back. +fn vaxisSize() vaxis.Winsize { + return if (direct_emit) + .{ .rows = 1, .cols = 1, .x_pixel = 0, .y_pixel = 0 } + else + cur_winsize; +} + // -------------------------------------------------------------------------------------- exports /// Hand over the allocator and the output sink, state the initial window size, and bring the editor @@ -253,7 +273,7 @@ export fn pardes_p4_init( pardes.syntax.start(allocs.tree_sitter); vx = vaxis.init(std.Io.failing, a, &env_map, .{}) catch |err| return errCode(err); - vx.resize(a, &out, cur_winsize) catch |err| return errCode(err); + vx.resize(a, &out, vaxisSize()) catch |err| return errCode(err); // Ask the terminal what it is. Both halves are pure byte writers, which is the whole reason this // works over a serial line: the replies arrive as ordinary input and are parsed like any key. @@ -452,10 +472,15 @@ fn apply(c: *pardes.Pardes, ev: vaxis.Event) void { }; if (want.cols == cur_winsize.cols and want.rows == cur_winsize.rows) return; const previous = cur_winsize; - vx.resize(gpa(), &out, want) catch { - vx.resize(gpa(), &out, previous) catch {}; - return; - }; + // vaxis is only resized when it is the thing doing the rendering. Under `direct_emit` its + // grids are a single cell and stay that way - see `vaxisSize` - so there is nothing here + // to reallocate, which also means a resize can no longer fail for want of two grids. + if (!direct_emit) { + vx.resize(gpa(), &out, want) catch { + vx.resize(gpa(), &out, previous) catch {}; + return; + }; + } cur_winsize = want; c.update(.{ .resize = .{ .cols = want.cols, .rows = want.rows } }); dirty = true; |
