From 4ab24352873ed7bf8db93ef6bfec36a34b0357e8 Mon Sep 17 00:00:00 2001 From: Gabriel Schneider Date: Tue, 25 Aug 2026 22:10:36 -0300 Subject: The frame diff was comparing byte at a time; compare words, and pay the bridge on every frame Two findings, both in the P4 shell's own `present`. ## std.mem.eql was the largest read in the firmware, one byte at a time The shadow-grid diff compares each row against the previous frame: two 13 KB streams, every frame, and by far the biggest memory access the firmware makes. It measured 3.2 cycles per byte, which is about four times what word-wide loads need - the shape of a byte-at-a-time loop, and `std.mem.eql` is what it was. `sameBytes` compares a `u32` at a time and falls back to the byte loop when the spans are not aligned for it. The alignment test has to be a RUNTIME one because `Cell` is all `u8` fields and therefore has alignment 1: whether a row begins on a word boundary is a property of whoever allocated the Surface, not of the type. A row is 40 cells of 26 bytes, divisible by four, so an aligned base makes every row aligned. The answer is bit-for-bit the same - this is still exact byte equality - so it keeps the property the whole diff rests on: byte equality implies visual equality, so the diff can never claim two different cells are the same. Measured on the die: the grid walk 223 -> 66 us, 0.95 cycles per byte. 157 us off every keystroke at every document length, and the single largest win since the clock raise. ## A frame that only hides the cursor still has to fill a USB packet The padding added for the bridge's 32-byte bulk-IN packet covered the branch that positions the cursor and not the branch that hides it. A frame that only hid the cursor was six bytes and waited out the bridge's timer. Hiding an already-hidden cursor is as idempotent as positioning it twice, so it pads the same way. The packet size is no longer inferred from an experiment either: 32 is `wMaxPacketSize` of endpoint 0x82 as the device reports it, and the sweep over pad targets confirms what it implies - 0 and 16 sit at 4.7-5.1 ms, while 32, 48 and 64 all sit at 3.6-3.8 ms. Crossing the boundary is worth about 950 us; going past it buys nothing. ## Result length 0 20 40 80 160 320 640 chars RTT 3602 3624 3638 3790 3868 4026 4192 us Fixed cost 3652 us against a 4 ms target, from 16.99 ms where this started. A phase-randomised instrument agrees over 80 trials: median 3687 us, minimum 3571, maximum 3912 - every trial under 4 ms. The two lengths still above 4 ms are the ones where the line has outgrown the viewport, so the cursor is off screen and the keystroke changes NOTHING: the frame is 36 bytes of cursor-hide and padding, zero cells changed, while pardes still rebuilds all 480 cells of the Surface for 858-985 us. That is the one architectural item left and it is not a micro -optimisation: nothing in this repository can avoid work pardes has already done. Verified: screen byte-identical to the vaxis reference on the 18-step workload, with canonical style decoding rather than escape history. snap 95/95, hxdiff 481/0, hxparity 561/0, unit-test, both A/B arms build, tty, p4 and gui all build. --- src/p4.zig | 50 +++++++++++++++++++++++++++++++++++++++----------- 1 file changed, 39 insertions(+), 11 deletions(-) diff --git a/src/p4.zig b/src/p4.zig index 5319ad90..bf9d1c7d 100644 --- a/src/p4.zig +++ b/src/p4.zig @@ -220,7 +220,6 @@ pub const max_rows: u16 = 12; var cur_winsize: vaxis.Winsize = .{ .rows = max_rows, .cols = max_cols, .x_pixel = 0, .y_pixel = 0 }; - // -------------------------------------------------------------------------------------- exports /// Hand over the allocator and the output sink, state the initial window size, and bring the editor @@ -560,7 +559,7 @@ fn present(_: ?*anyopaque, surface: *const pardes.Surface) void { // correct and merely slower. if (usable and !full) { const shadow = prev_cells[row0..][0..surface.cols]; - if (std.mem.eql(u8, std.mem.sliceAsBytes(src), std.mem.sliceAsBytes(shadow))) continue; + if (sameBytes(std.mem.sliceAsBytes(src), std.mem.sliceAsBytes(shadow))) continue; } var x: u16 = 0; @@ -576,15 +575,18 @@ fn present(_: ?*anyopaque, surface: *const pardes.Surface) void { } } if (direct_emit) { + // BOTH branches have to reach the packet boundary, and the second one is easy to forget: + // measured, a frame that only hid the cursor was 6 bytes and cost 4014 us at 640 characters + // against 3863 at 320, because 6 bytes never fills a packet and waited out the bridge's + // timer. Hiding an already-hidden cursor is as idempotent as positioning it twice. if (surface.cursor) |cur| { cup(cur.y, cur.x) catch return; emitRaw("\x1b[?25h") catch return; emit_col = -1; - // Up to the bridge's packet boundary, and no further. See `emit_min_frame`: this is the - // one place that knows how many bytes the frame came to, and repeating the sequence the - // frame already ended on is the only filler that cannot change what is on the screen. while (emit_bytes < emit_min_frame) cup(cur.y, cur.x) catch return; - } else emitRaw("\x1b[?25l") catch return; + } else { + while (emit_bytes < emit_min_frame) emitRaw("\x1b[?25l") catch return; + } } else if (surface.cursor) |cur| { win.showCursor(cur.x, cur.y); } else win.hideCursor(); @@ -757,10 +759,10 @@ const direct_emit = true; /// THE FRAME HAS A MINIMUM SIZE, and it is the USB bridge's, not the terminal's. /// -/// The board is wired to the host through a CH340, a full-speed device whose bulk IN endpoint takes -/// 32-byte packets. It forwards a packet when the packet is FULL, and a frame shorter than that sits -/// in the bridge until an internal timer gives up on more - which is worth about a millisecond, and -/// a millisecond is a quarter of the entire keystroke budget. +/// The board is wired to the host through a CH340, and 32 is not a guess: it is `wMaxPacketSize` of +/// endpoint 0x82, the bulk IN, as the device itself reports it - a full-speed 0x0020. The bridge +/// forwards a packet when the packet is FULL, so a frame shorter than that sits there until an +/// internal timer gives up on more, which is worth about a millisecond - a quarter of the budget. /// /// Measured, at the same board cost and with the screen byte-identical: a 21-byte frame round-trips /// in 4817 us and the same frame padded to 49 bytes in 3814 us. MORE BYTES, ARRIVING SOONER. It also @@ -809,7 +811,33 @@ var prev_rows: u16 = 0; /// than by bytes, because an unpainted cell's text and style are whatever the last frame left there. inline fn sameCell(a: *const pardes.Cell, b: *const pardes.Cell) bool { if (a.default or b.default) return a.default and b.default; - return std.mem.eql(u8, std.mem.asBytes(a), std.mem.asBytes(b)); + return sameBytes(std.mem.asBytes(a), std.mem.asBytes(b)); +} + +/// Exact byte equality, a word at a time when both spans are aligned for it. +/// +/// This comparison is the firmware's largest read by a wide margin - two 13 KB streams every frame - +/// and it measured 3.2 cycles per byte, about four times what word-wide loads should need, which is +/// what a byte-at-a-time loop looks like. The answer is identical either way: this is still exact +/// byte equality, so it keeps the property the whole diff rests on, that byte equality implies +/// visual equality. +/// +/// The alignment test is a RUNTIME one because `Cell` has alignment 1 - it is all `u8` fields - so +/// whether a row begins on a word boundary is a property of whoever allocated the Surface and not of +/// the type. A row is 40 cells of 26 bytes, which is divisible by four, so if the base is aligned +/// every row is. When it is not, the byte loop is still here. +inline fn sameBytes(a: []const u8, b: []const u8) bool { + if (a.len != b.len) return false; + if ((@intFromPtr(a.ptr) | @intFromPtr(b.ptr)) & 3 == 0) { + const n = a.len / 4; + const wa: [*]align(4) const u32 = @ptrCast(@alignCast(a.ptr)); + const wb: [*]align(4) const u32 = @ptrCast(@alignCast(b.ptr)); + for (wa[0..n], wb[0..n]) |x, y| { + if (x != y) return false; + } + return std.mem.eql(u8, a[n * 4 ..], b[n * 4 ..]); + } + return std.mem.eql(u8, a, b); } // ------------------------------------------------------------------ where a frame's time goes -- cgit v1.3