diff options
Diffstat (limited to 'src/p4.zig')
| -rw-r--r-- | src/p4.zig | 98 |
1 files changed, 96 insertions, 2 deletions
@@ -30,6 +30,7 @@ //! any other input. Firmware has no `TIOCGWINSZ`, so the host-side bridge synthesises the first one. const std = @import("std"); +const builtin = @import("builtin"); const pardes = @import("pardes.zig"); const vaxis = @import("vaxis"); @@ -510,8 +511,34 @@ const pardes_host: pardes.Host.VTable = .{ .push_present = present }; /// (`src/tty/tty.zig:1096`) minus the panel compositor and the kitty image path: neither has a /// reason to exist on a board with no pixels. fn present(_: ?*anyopaque, surface: *const pardes.Surface) void { + const t0 = cycles(); const win = vx.window(); - win.clear(); + const n = @as(usize, surface.cols) * @as(usize, surface.rows); + + // THE SHADOW GRID. Copying all 480 cells into vaxis every frame cost 6.75 ms on the die - 57% + // of a keystroke, and it was paid whether or not anything changed: a second render with nothing + // new measured the same as the first. vaxis already diffs its own grid against the terminal, but + // it can only do that AFTER being told every cell, and being told is the expensive part + // (`writeCell` builds a vaxis `Cell`, which carries an always-null image placement). + // + // So keep the previous Surface and tell vaxis only what moved. `Cell.visuallyEqual` is the + // right comparison and already exists for the panel compositor's benefit: it ignores scratch + // bytes past `len` and treats any two default cells as equal, so it cannot manufacture a write. + // + // STATIC, and that is not a micro-optimisation - it is a bug fix. The first version allocated + // this from the editor's heap, and on a board whose 384 KiB is already nearly spoken for that + // was enough to make `vx.resize` fail: a resize then hit its OOM path, restored the previous + // geometry and returned, so the screen was never repainted. Measured as a resize emitting 80 + // bytes where it had emitted 1,392. The grid is bounded by `max_cols` x `max_rows` at comptime, + // so it belongs in `.bss` where it cannot compete with anything. + const full = !shadow_grid or prev_cols != surface.cols or prev_rows != surface.rows; + if (full) { + prev_cols = surface.cols; + prev_rows = surface.rows; + win.clear(); + } + const usable = shadow_grid and n <= prev_cells.len; + var y: u16 = 0; while (y < surface.rows) : (y += 1) { var x: u16 = 0; @@ -519,7 +546,18 @@ fn present(_: ?*anyopaque, surface: *const pardes.Surface) void { // `at` takes a mutable Surface but only reads; the tty shell does the same const-cast // for the same reason (src/tty/tty.zig:1105). const cell = @constCast(surface).at(x, y); - if (cell.default) continue; + const idx = @as(usize, y) * @as(usize, surface.cols) + @as(usize, x); + if (usable) { + if (!full and cell.visuallyEqual(&prev_cells[idx])) continue; + prev_cells[idx] = cell.*; + } else if (cell.default) continue; + + if (cell.default) { + // Changed TO default. `win.clear()` is what used to blank these, and it is not run + // on an incremental frame, so say it explicitly. + win.writeCell(x, y, .{ .char = .{ .grapheme = " " }, .style = .{} }); + continue; + } win.writeCell(x, y, .{ .char = .{ .grapheme = cell.grapheme() }, .style = vaxisStyle(cell.style), @@ -529,11 +567,67 @@ fn present(_: ?*anyopaque, surface: *const pardes.Surface) void { if (surface.cursor) |cur| { win.showCursor(cur.x, cur.y); } else win.hideCursor(); + const t1 = cycles(); // vaxis diffs against its own shadow grid, so this writes only what changed - which is what // makes an editor usable at 11.9 KB/s. vx.render(&out) catch return; + const t2 = cycles(); out.flush() catch return; + const t3 = cycles(); + + prof_copy_cy = t1 -% t0; + prof_render_cy = t2 -% t1; + prof_flush_cy = t3 -% t2; +} + +/// The previous Surface, cell for cell, sized for the largest grid this board can drive. In `.bss` +/// rather than on the heap: see `present`. `prev_cols`/`prev_rows` being zero on the first frame is +/// what makes that frame a full one. +/// A/B switch, kept because this optimisation is exactly the kind that can be right about latency +/// and wrong about the screen. With it false, `present` behaves as it did before the shadow grid - +/// clear and write every cell - which is the reference any measurement of it should be compared +/// against, and the way to tell a rendering bug from a rendering difference. +const shadow_grid = true; + +var prev_cells: [@as(usize, max_cols) * @as(usize, max_rows)]pardes.Cell = if (shadow_grid) @splat(.{}) else undefined; +var prev_cols: u16 = 0; +var prev_rows: u16 = 0; + +// ------------------------------------------------------------------ where a frame's time goes +// +// A frame has three stages and they want different fixes, so the firmware is given all three rather +// than one total. Measured on the die, a render costs ~11 ms whether or not anything changed, which +// says the cost is the unconditional walk and not the edit - but "the walk" is two walks, the copy +// into vaxis's grid and vaxis's own diff, and only one of them is ours to change. +// +// Two CSR reads per stage. `cycle` is the unprivileged counter, read high-low-high because two +// 32-bit halves can straddle a wrap. +var prof_copy_cy: u64 = 0; +var prof_render_cy: u64 = 0; +var prof_flush_cy: u64 = 0; + +inline fn cycles() u64 { + if (builtin.cpu.arch != .riscv32) return 0; + while (true) { + const hi0 = asm volatile ("csrr %[o], cycleh" + : [o] "=r" (-> u32), + ); + const lo = asm volatile ("csrr %[o], cycle" + : [o] "=r" (-> u32), + ); + const hi1 = asm volatile ("csrr %[o], cycleh" + : [o] "=r" (-> u32), + ); + if (hi0 == hi1) return (@as(u64, hi0) << 32) | lo; + } +} + +/// The last frame's three stages, in cycles. Zero on any platform without the CSR. +export fn pardes_p4_frame_prof(copy: *u64, render: *u64, flush: *u64) callconv(.c) void { + copy.* = prof_copy_cy; + render.* = prof_render_cy; + flush.* = prof_flush_cy; } fn vaxisStyle(s: pardes.CellStyle) vaxis.Style { |
