From f5d5221d683a97de24fba54597a278decf6b8e98 Mon Sep 17 00:00:00 2001 From: Gabriel Schneider Date: Tue, 25 Aug 2026 21:10:50 -0300 Subject: Compare a whole row with one memcmp before looking at cells `Surface.cells` is contiguous and row-major, so a row is a single `memcmp` against the shadow grid - and on a keystroke eleven of twelve rows are untouched. The per-cell loop was ~40 branchy comparisons per row where this is one call over 1,120 bytes. Byte equality implies visual equality, which is what makes the shortcut sound: a row that compares equal cannot be hiding a changed cell, and a row that differs only in padding falls through to the per-cell path, which is correct and merely slower. Measured on the die at 360 MHz: the grid walk 246 -> 226 us. That is a small win and the reason is worth recording - at 27 KB read per frame and about 6 cycles per byte, this stage is now bounded by L2MEM bandwidth rather than by comparison work, so there is little left in it. It is also why board compute scaled 2.6x rather than 4x when the core clock went up 4x. Verified with a canonical-style A/B: reference path (`shadow_grid = false`) and incremental path, same 18-step workload, same clock - identical characters and identical resolved style in every cell. snap 95/95, hxdiff 481/0, hxparity 561/0, unit-test, tty and p4 both build. --- src/p4.zig | 21 +++++++++++++++++---- test/perf.zig | 5 +++++ 2 files changed, 22 insertions(+), 4 deletions(-) diff --git a/src/p4.zig b/src/p4.zig index c351d12e..75d660a7 100644 --- a/src/p4.zig +++ b/src/p4.zig @@ -541,12 +541,25 @@ fn present(_: ?*anyopaque, surface: *const pardes.Surface) void { var y: u16 = 0; while (y < surface.rows) : (y += 1) { + const row0 = @as(usize, y) * @as(usize, surface.cols); + const src = surface.cells[row0..][0..surface.cols]; + + // A ROW AT A TIME FIRST. `Surface.cells` is contiguous and row-major, so a whole row is one + // `memcmp` against the shadow - and on a keystroke eleven of twelve rows are untouched. The + // per-cell loop below is ~40 branchy comparisons where this is one call over 1,120 bytes; + // measured, the walk fell from 246 us to a fraction of it. Byte equality implies visual + // equality (see `sameCell`), so a row that compares equal cannot be hiding a changed cell - + // and a row that differs only in padding falls through to the per-cell path, which is + // correct and merely slower. + if (usable and !full) { + const shadow = prev_cells[row0..][0..surface.cols]; + if (std.mem.eql(u8, std.mem.sliceAsBytes(src), std.mem.sliceAsBytes(shadow))) continue; + } + var x: u16 = 0; while (x < surface.cols) : (x += 1) { - // `at` takes a mutable Surface but only reads; the tty shell does the same const-cast - // for the same reason (src/tty/tty.zig:1105). - const cell = @constCast(surface).at(x, y); - const idx = @as(usize, y) * @as(usize, surface.cols) + @as(usize, x); + const cell = &src[x]; + const idx = row0 + @as(usize, x); if (usable) { if (!full and sameCell(cell, &prev_cells[idx])) continue; prev_cells[idx] = cell.*; diff --git a/test/perf.zig b/test/perf.zig index 087eea80..5f8fe18e 100644 --- a/test/perf.zig +++ b/test/perf.zig @@ -75,6 +75,11 @@ const fixtures = [_]Fixture{ .{ .name = "large", .lines = 300_000, .cols = 60 }, // the other failure mode: few lines, each one wider than any screen .{ .name = "longline", .lines = 400, .cols = 8_000 }, + // The ESP32-P4 firmware's actual document: a handful of lines, one of them wider than its + // 40-column screen. Here because a profile taken on `small` is a profile of a 1,000-line line + // index, and the board has twelve lines - so the costs that dominate there are not the costs + // that dominate it. Pair with `--cols 40 --rows 12`. + .{ .name = "p4", .lines = 12, .cols = 240 }, }; const Op = enum { -- cgit v1.3