diff options
| author | Gabriel Schneider <[email protected]> | 2026-08-25 20:30:52 -0300 |
|---|---|---|
| committer | Gabriel Schneider <[email protected]> | 2026-08-25 20:30:52 -0300 |
| commit | 649fd0e983d6a207932f1ec43f7ded00e26a69c0 (patch) | |
| tree | 7c072506e5a2ccbc8bf6c2afbe5414903b8fa1cc /src/p4.zig | |
| parent | 767a35d3ae3b87d1bfdd960f2187f0a51ad32333 (diff) | |
| download | pardes-649fd0e983d6a207932f1ec43f7ded00e26a69c0.tar.gz pardes-649fd0e983d6a207932f1ec43f7ded00e26a69c0.zip | |
Compare shadow-grid cells as bytes, not through std.meta.eql
`Cell.visuallyEqual` is the semantically exact answer and too slow to ask 480 times a
frame: `std.meta.eql` on a `CellStyle` recurses through a colour union and eight
booleans, and the walk measured 1.45 ms on the die - about 270 cycles to compare a
28-byte struct.
`sameCell` in src/p4.zig does it as bytes. That is safe in the direction that
matters: byte equality IMPLIES visual equality, so it can never claim two different
cells are the same. It can miss an equality - scratch bytes past `len`, or padding -
and the only cost of that is one redundant `writeCell` which vaxis then diffs away.
Defaults are still compared by meaning, because an unpainted cell's text and style are
whatever the previous frame left in them.
Measured: the grid walk 1.45 -> 0.98 ms, a keystroke 8.87 -> 8.37 ms fixed.
Verified the way a rendering change has to be. The A/B harness now hashes the SGR
state of every cell as well as its character, because the first version compared text
only and would have passed a colour regression in silence. Reference path
(`shadow_grid = false`, clear and write everything) and incremental path were each run
against the same 19-step workload on the die and the reconstructed screens are
identical in both text and per-row style hash.
snap 95/95, hxdiff 481 cases 0 mismatches, hxparity 561 cases 0 mismatches, unit-test,
and tty / p4 / gui all build.
Diffstat (limited to 'src/p4.zig')
| -rw-r--r-- | src/p4.zig | 23 |
1 files changed, 19 insertions, 4 deletions
@@ -548,7 +548,7 @@ fn present(_: ?*anyopaque, surface: *const pardes.Surface) void { const cell = @constCast(surface).at(x, y); const idx = @as(usize, y) * @as(usize, surface.cols) + @as(usize, x); if (usable) { - if (!full and cell.visuallyEqual(&prev_cells[idx])) continue; + if (!full and sameCell(cell, &prev_cells[idx])) continue; prev_cells[idx] = cell.*; } else if (cell.default) continue; @@ -581,19 +581,34 @@ fn present(_: ?*anyopaque, surface: *const pardes.Surface) void { prof_flush_cy = t3 -% t2; } -/// The previous Surface, cell for cell, sized for the largest grid this board can drive. In `.bss` -/// rather than on the heap: see `present`. `prev_cols`/`prev_rows` being zero on the first frame is -/// what makes that frame a full one. /// A/B switch, kept because this optimisation is exactly the kind that can be right about latency /// and wrong about the screen. With it false, `present` behaves as it did before the shadow grid - /// clear and write every cell - which is the reference any measurement of it should be compared /// against, and the way to tell a rendering bug from a rendering difference. const shadow_grid = true; +/// The previous Surface, cell for cell, sized for the largest grid this board can drive. In `.bss` +/// rather than on the heap: see `present`. `prev_cols`/`prev_rows` being zero on the first frame is +/// what makes that frame a full one. var prev_cells: [@as(usize, max_cols) * @as(usize, max_rows)]pardes.Cell = if (shadow_grid) @splat(.{}) else undefined; var prev_cols: u16 = 0; var prev_rows: u16 = 0; +/// Cell equality for the shadow grid, as bytes. +/// +/// `Cell.visuallyEqual` is the semantically exact answer and it is too slow to ask 480 times a +/// frame: `std.meta.eql` on a `CellStyle` recurses through a colour union and eight booleans, and +/// the walk measured 1.45 ms - about 270 cycles per comparison of a ~28-byte struct. +/// +/// Byte equality IMPLIES visual equality, so this can never claim two different cells are the same. +/// It can miss an equality - scratch bytes past `len`, or padding - and the only cost of that is one +/// redundant `writeCell` that vaxis then diffs away. Defaults are still handled by meaning rather +/// than by bytes, because an unpainted cell's text and style are whatever the last frame left there. +inline fn sameCell(a: *const pardes.Cell, b: *const pardes.Cell) bool { + if (a.default or b.default) return a.default and b.default; + return std.mem.eql(u8, std.mem.asBytes(a), std.mem.asBytes(b)); +} + // ------------------------------------------------------------------ where a frame's time goes // // A frame has three stages and they want different fixes, so the firmware is given all three rather |
