summaryrefslogtreecommitdiff
diff options
context:
space:
mode:
-rw-r--r--src/p4.zig23
1 files changed, 19 insertions, 4 deletions
diff --git a/src/p4.zig b/src/p4.zig
index 57ee88cf..c351d12e 100644
--- a/src/p4.zig
+++ b/src/p4.zig
@@ -548,7 +548,7 @@ fn present(_: ?*anyopaque, surface: *const pardes.Surface) void {
const cell = @constCast(surface).at(x, y);
const idx = @as(usize, y) * @as(usize, surface.cols) + @as(usize, x);
if (usable) {
- if (!full and cell.visuallyEqual(&prev_cells[idx])) continue;
+ if (!full and sameCell(cell, &prev_cells[idx])) continue;
prev_cells[idx] = cell.*;
} else if (cell.default) continue;
@@ -581,19 +581,34 @@ fn present(_: ?*anyopaque, surface: *const pardes.Surface) void {
prof_flush_cy = t3 -% t2;
}
-/// The previous Surface, cell for cell, sized for the largest grid this board can drive. In `.bss`
-/// rather than on the heap: see `present`. `prev_cols`/`prev_rows` being zero on the first frame is
-/// what makes that frame a full one.
/// A/B switch, kept because this optimisation is exactly the kind that can be right about latency
/// and wrong about the screen. With it false, `present` behaves as it did before the shadow grid -
/// clear and write every cell - which is the reference any measurement of it should be compared
/// against, and the way to tell a rendering bug from a rendering difference.
const shadow_grid = true;
+/// The previous Surface, cell for cell, sized for the largest grid this board can drive. In `.bss`
+/// rather than on the heap: see `present`. `prev_cols`/`prev_rows` being zero on the first frame is
+/// what makes that frame a full one.
var prev_cells: [@as(usize, max_cols) * @as(usize, max_rows)]pardes.Cell = if (shadow_grid) @splat(.{}) else undefined;
var prev_cols: u16 = 0;
var prev_rows: u16 = 0;
+/// Cell equality for the shadow grid, as bytes.
+///
+/// `Cell.visuallyEqual` is the semantically exact answer and it is too slow to ask 480 times a
+/// frame: `std.meta.eql` on a `CellStyle` recurses through a colour union and eight booleans, and
+/// the walk measured 1.45 ms - about 270 cycles per comparison of a ~28-byte struct.
+///
+/// Byte equality IMPLIES visual equality, so this can never claim two different cells are the same.
+/// It can miss an equality - scratch bytes past `len`, or padding - and the only cost of that is one
+/// redundant `writeCell` that vaxis then diffs away. Defaults are still handled by meaning rather
+/// than by bytes, because an unpainted cell's text and style are whatever the last frame left there.
+inline fn sameCell(a: *const pardes.Cell, b: *const pardes.Cell) bool {
+ if (a.default or b.default) return a.default and b.default;
+ return std.mem.eql(u8, std.mem.asBytes(a), std.mem.asBytes(b));
+}
+
// ------------------------------------------------------------------ where a frame's time goes
//
// A frame has three stages and they want different fixes, so the firmware is given all three rather