From 767a35d3ae3b87d1bfdd960f2187f0a51ad32333 Mon Sep 17 00:00:00 2001 From: Gabriel Schneider Date: Tue, 25 Aug 2026 20:20:24 -0300 Subject: Make a keystroke 2.6x cheaper by not asking Unicode about ASCII A keystroke on the ESP32-P4 cost 17.0 ms and the goal is 4. Profiling the core in that board's exact configuration - 40x12, tree-sitter disabled, via `zig build perf -Dtree-sitter=disabled -- --cols 40 --rows 12 --only small` - named the cost, and it was Unicode machinery answering questions about the letter `y`. Four changes, each a fast path guarded so that non-ASCII text takes exactly the road it took before. `modal.graphemeStart` was 21.5% of a keystroke, the single largest item. It iterates graphemes FROM THE START of the text with the full UAX #29 break state machine until it passes the offset, and the render path calls it once per visible row with a column offset - so the cost followed the cursor's distance along its line. That is the shape measured on the die, where inserting at column 320 of a fixed 320-character line cost 7.8 ms more than inserting at column 0 of the same line. In UAX #29 every ASCII scalar is its own cluster with ONE exception, GB3 (CR joined to LF); every other rule that could extend a cluster - Extend, ZWJ, SpacingMark, Prepend, Regional_Indicator - is spelled with non-ASCII scalars. So an ASCII byte whose predecessor is also ASCII, and not that CR-LF pair, IS a boundary. O(1), and sound rather than approximate. `Surface.print` then became the largest at 26.2%: per character it took a UTF-8 length, a decode, a FRESHLY CONSTRUCTED grapheme iterator, a slice validation and a width lookup, to conclude that `y` is one cell. Printable ASCII followed by ASCII takes none of that now. Same guard, same reason. `file_pane.graphemeDisplayWidth` was 6.9%, essentially all of it asking `gwidth` about ASCII. Bounded to 0x20..0x7e on purpose: DEL and the C0 controls are not one printable cell and `gwidth` stays the authority on them. `modal.lineSlice` searched for "\n" with the generic substring search where a memchr does; it is called once per visible row per frame. Measured at the P4's geometry and configuration, on the host: render 55 -> 12 us, key-down 483 -> 24 us, key-right 327 -> 13 us, edit-char 205 -> 46 us. On the die, the per-character cost of a keystroke fell from 54.3 to 6.9 us - 7.9x - and a keystroke at a 160-character line from 25.56 ms to 15.36 ms. ## The shadow grid, and why it is static `src/p4.zig`'s `present` copied all 480 cells into vaxis every frame, which measured 6.75 ms on the die - 57% of a keystroke - and was paid whether or not anything changed: a second render with nothing new cost the same as the first. vaxis diffs its own grid, but only after being told every cell, and being told is the expensive part. So `present` now keeps the previous Surface and tells vaxis only what moved. `Cell.visuallyEqual` is the right comparison and already existed. Copy: 6.75 -> 1.45 ms. The grid lives in `.bss`, sized by `max_cols` x `max_rows` at comptime, and that is not a micro-optimisation. The first version allocated it from the editor's heap; on a board whose 384 KiB is nearly spoken for, that is exactly the kind of change that works and then breaks something else three steps away. `shadow_grid` is a comptime A/B switch, kept deliberately. With it false, `present` behaves as it did before - clear and write every cell - which is the reference any measurement should be compared against, and the way to tell a rendering bug from a rendering difference. It earned its keep immediately: the two paths were run against the same 19-step workload on the die - inserts, deletes, motions that move the modified-marker, a line outgrowing the viewport, backspaces that shrink it - and the reconstructed screens are byte-identical. ## Verification `snap` 95/95 scripts, `hxdiff` 481 cases 0 mismatches, `hxparity` 561 cases 0 mismatches, `unit-test`, `image-harness`, `pdf-harness`, `mupdf-check`, and tty / p4 / gui all build. The rendering changes are exactly the sort that pass a latency benchmark while corrupting a screen, so the snapshot parity suite is the one that matters here and it is unchanged. `test/perf.zig` gains `--cols`/`--rows`/`--only`. The screen's shape is one of the things that table exists to hold constant, and 40x12 is not a scaled guess at the board - it is the board. `--only` exists because under `perf record` one 63 ms cell on the largest fixture swamps every sample from the case being asked about. ## Found, not fixed `vx.resize` fails on this board: a runtime geometry change hits its allocation failure path, restores the previous size and returns, so 80 bytes go out where 1,392 should. Verified independent of everything above - it reproduces with `shadow_grid` false. The board therefore has one geometry for the life of a session, which is why the staleness test above compares two firmwares rather than resizing one. --- test/perf.zig | 31 ++++++++++++++++++++++++++----- 1 file changed, 26 insertions(+), 5 deletions(-) (limited to 'test/perf.zig') diff --git a/test/perf.zig b/test/perf.zig index 1387a1ac..087eea80 100644 --- a/test/perf.zig +++ b/test/perf.zig @@ -47,8 +47,13 @@ fn nowNs() u64 { /// harness pins (that one matches helix's own harness; this one wants a screen /// somebody actually works in), and fixed, because half of these numbers scale /// with the number of cells on the screen. -const screen_cols: u16 = 120; -const screen_rows: u16 = 40; +/// The viewport every measurement runs in. Overridable, because the SHAPE of the +/// screen is one of the things this table exists to hold constant while something +/// else varies - and the ESP32-P4 firmware runs a 40x12 grid, where a render costs +/// 47x what it costs here for a tenth of the cells. Profiling that needs the same +/// geometry, not a scaled guess. +var screen_cols: u16 = 120; +var screen_rows: u16 = 40; /// The four shapes a file comes in. `lines` x `cols` is the generated body; /// the point of the pair is that "big" has two different meanings and they @@ -173,6 +178,7 @@ pub fn main(init: std.process.Init) !void { var json = false; var reps: usize = 25; var base_path: ?[]const u8 = null; + var only: ?[]const u8 = null; var i: usize = 1; while (i < args.len) : (i += 1) { const a = args[i]; @@ -184,7 +190,16 @@ pub fn main(init: std.process.Init) !void { } else if (std.mem.eql(u8, a, "--base") and i + 1 < args.len) { i += 1; base_path = args[i]; - } else fatal("usage: pardes-perf [--json] [--reps N] [--base old.json]", .{}); + } else if (std.mem.eql(u8, a, "--cols") and i + 1 < args.len) { + i += 1; + screen_cols = std.fmt.parseInt(u16, args[i], 10) catch screen_cols; + } else if (std.mem.eql(u8, a, "--rows") and i + 1 < args.len) { + i += 1; + screen_rows = std.fmt.parseInt(u16, args[i], 10) catch screen_rows; + } else if (std.mem.eql(u8, a, "--only") and i + 1 < args.len) { + i += 1; + only = args[i]; + } else fatal("usage: pardes-perf [--json] [--reps N] [--base old.json] [--cols N] [--rows N] [--only NAME]", .{}); } // fixtures live in a temp dir and are rewritten every run: they are inputs @@ -206,17 +221,23 @@ pub fn main(init: std.process.Init) !void { gpa.free(text); } - var cells: [std.enums.values(Op).len][fixtures.len]Cell = undefined; + // `--only` leaves the other cells zeroed rather than reshaping the table. Its purpose is + // profiling, not reporting: under `perf record` a single 63 ms cell on the largest fixture + // swamps the samples, and the question "what does ONE keystroke on a small document spend its + // time in" cannot be answered from a profile dominated by a different one. + var cells: [std.enums.values(Op).len][fixtures.len]Cell = @splat(@splat(.{})); for (std.enums.values(Op), 0..) |op, oi| { for (fixtures, 0..) |fx, fi| { + if (only) |name| if (!std.mem.eql(u8, name, fx.name)) continue; cells[oi][fi] = try measure(op, fx, paths[fi], reps); } } term_chunk = try buildTermText(4 * 1024); - var term_cells: [std.enums.values(TermOp).len][term_fixtures.len]Cell = undefined; + var term_cells: [std.enums.values(TermOp).len][term_fixtures.len]Cell = @splat(@splat(.{})); for (std.enums.values(TermOp), 0..) |op, oi| { for (term_fixtures, 0..) |fx, fi| { + if (only != null) continue; term_cells[oi][fi] = try measureTerm(op, fx, reps); } } -- cgit v1.3