diff options
Diffstat (limited to 'src')
| -rw-r--r-- | src/file_pane.zig | 6 | ||||
| -rw-r--r-- | src/modal.zig | 26 | ||||
| -rw-r--r-- | src/p4.zig | 98 | ||||
| -rw-r--r-- | src/pardes.zig | 24 |
4 files changed, 149 insertions, 5 deletions
diff --git a/src/file_pane.zig b/src/file_pane.zig index 8fc0fa0d..1c6d5e2f 100644 --- a/src/file_pane.zig +++ b/src/file_pane.zig @@ -92,6 +92,12 @@ pub fn dumpPane( pub fn graphemeDisplayWidth(grapheme: []const u8) usize { if (std.mem.eql(u8, grapheme, "\t")) return config.tab_width; + // A one-byte printable ASCII grapheme is one cell, and saying so here rather than asking + // `gwidth` costs a comparison instead of a Unicode table walk. `gwidth` was 6.9% of a profiled + // keystroke at the P4's geometry, essentially all of it answering this question about `y`. + // Bounded to 0x20..0x7e on purpose: DEL and the C0 controls are not one printable cell, and + // `gwidth` is still the authority on them. + if (grapheme.len == 1 and grapheme[0] >= 0x20 and grapheme[0] < 0x7f) return 1; return @max(1, @as(usize, vaxis.gwidth.gwidth(grapheme, .unicode))); } diff --git a/src/modal.zig b/src/modal.zig index f8093ed8..953aa900 100644 --- a/src/modal.zig +++ b/src/modal.zig @@ -499,7 +499,9 @@ pub fn lineStartOffset(content: []const u8, row: usize) usize { pub fn lineSlice(content: []const u8, row: usize) []const u8 { const start = lineStartOffset(content, row); if (start >= content.len) return ""; - const nl = std.mem.indexOfPos(u8, content, start, "\n") orelse content.len; + // indexOfScalarPos, not indexOfPos with a one-byte needle: the latter runs the generic + // substring search where a memchr will do, and this is called once per visible row per frame. + const nl = std.mem.indexOfScalarPos(u8, content, start, '\n') orelse content.len; return content[start..nl]; } @@ -849,11 +851,29 @@ pub fn deleteSpan(alloc: std.mem.Allocator, content: []const u8, a: Cursor, b: C pub const HxRange = struct { anchor: usize, head: usize }; -/// The grapheme containing `off`, or text.len at EOF. This is also the repair -/// path for stale/external byte columns that happen to point into UTF-8. +/// The first byte of the grapheme cluster containing `off`. +/// +/// The general answer needs UAX #29, which is why the slow path below iterates from the start of +/// `text` with the full break state machine - and that made this the single hottest function in a +/// keystroke: 21.5% of a profiled edit at the ESP32-P4's 40x12 geometry, because the render path +/// calls it once per visible row with a column offset, so the cost follows the cursor's distance +/// along its line. That is exactly the shape measured on the die, where inserting at column 320 of +/// a fixed line cost 7.8 ms more than inserting at column 0 of the same line. +/// +/// The fast path is sound rather than approximate. In UAX #29 every ASCII scalar is its own +/// grapheme cluster with ONE exception, GB3: CR is joined to a following LF. Every other rule that +/// could extend a cluster across `off` - Extend, ZWJ, SpacingMark, Prepend, Regional_Indicator - +/// is spelled with non-ASCII scalars. So if the byte at `off` and the byte before it are both +/// ASCII and are not that CR-LF pair, `off` already IS a cluster boundary and there is nothing to +/// search for. Text that is not all ASCII still takes the slow path, byte for byte as before. pub fn graphemeStart(text: []const u8, off: usize) usize { const bounded = @min(off, text.len); if (bounded == text.len) return text.len; + if (text[bounded] < 0x80) { + if (bounded == 0) return 0; + const prev = text[bounded - 1]; + if (prev < 0x80 and !(prev == '\r' and text[bounded] == '\n')) return bounded; + } var it = uucode.grapheme.utf8Iterator(text); while (it.nextGrapheme()) |g| { if (bounded < g.end) return g.start; @@ -30,6 +30,7 @@ //! any other input. Firmware has no `TIOCGWINSZ`, so the host-side bridge synthesises the first one. const std = @import("std"); +const builtin = @import("builtin"); const pardes = @import("pardes.zig"); const vaxis = @import("vaxis"); @@ -510,8 +511,34 @@ const pardes_host: pardes.Host.VTable = .{ .push_present = present }; /// (`src/tty/tty.zig:1096`) minus the panel compositor and the kitty image path: neither has a /// reason to exist on a board with no pixels. fn present(_: ?*anyopaque, surface: *const pardes.Surface) void { + const t0 = cycles(); const win = vx.window(); - win.clear(); + const n = @as(usize, surface.cols) * @as(usize, surface.rows); + + // THE SHADOW GRID. Copying all 480 cells into vaxis every frame cost 6.75 ms on the die - 57% + // of a keystroke, and it was paid whether or not anything changed: a second render with nothing + // new measured the same as the first. vaxis already diffs its own grid against the terminal, but + // it can only do that AFTER being told every cell, and being told is the expensive part + // (`writeCell` builds a vaxis `Cell`, which carries an always-null image placement). + // + // So keep the previous Surface and tell vaxis only what moved. `Cell.visuallyEqual` is the + // right comparison and already exists for the panel compositor's benefit: it ignores scratch + // bytes past `len` and treats any two default cells as equal, so it cannot manufacture a write. + // + // STATIC, and that is not a micro-optimisation - it is a bug fix. The first version allocated + // this from the editor's heap, and on a board whose 384 KiB is already nearly spoken for that + // was enough to make `vx.resize` fail: a resize then hit its OOM path, restored the previous + // geometry and returned, so the screen was never repainted. Measured as a resize emitting 80 + // bytes where it had emitted 1,392. The grid is bounded by `max_cols` x `max_rows` at comptime, + // so it belongs in `.bss` where it cannot compete with anything. + const full = !shadow_grid or prev_cols != surface.cols or prev_rows != surface.rows; + if (full) { + prev_cols = surface.cols; + prev_rows = surface.rows; + win.clear(); + } + const usable = shadow_grid and n <= prev_cells.len; + var y: u16 = 0; while (y < surface.rows) : (y += 1) { var x: u16 = 0; @@ -519,7 +546,18 @@ fn present(_: ?*anyopaque, surface: *const pardes.Surface) void { // `at` takes a mutable Surface but only reads; the tty shell does the same const-cast // for the same reason (src/tty/tty.zig:1105). const cell = @constCast(surface).at(x, y); - if (cell.default) continue; + const idx = @as(usize, y) * @as(usize, surface.cols) + @as(usize, x); + if (usable) { + if (!full and cell.visuallyEqual(&prev_cells[idx])) continue; + prev_cells[idx] = cell.*; + } else if (cell.default) continue; + + if (cell.default) { + // Changed TO default. `win.clear()` is what used to blank these, and it is not run + // on an incremental frame, so say it explicitly. + win.writeCell(x, y, .{ .char = .{ .grapheme = " " }, .style = .{} }); + continue; + } win.writeCell(x, y, .{ .char = .{ .grapheme = cell.grapheme() }, .style = vaxisStyle(cell.style), @@ -529,11 +567,67 @@ fn present(_: ?*anyopaque, surface: *const pardes.Surface) void { if (surface.cursor) |cur| { win.showCursor(cur.x, cur.y); } else win.hideCursor(); + const t1 = cycles(); // vaxis diffs against its own shadow grid, so this writes only what changed - which is what // makes an editor usable at 11.9 KB/s. vx.render(&out) catch return; + const t2 = cycles(); out.flush() catch return; + const t3 = cycles(); + + prof_copy_cy = t1 -% t0; + prof_render_cy = t2 -% t1; + prof_flush_cy = t3 -% t2; +} + +/// The previous Surface, cell for cell, sized for the largest grid this board can drive. In `.bss` +/// rather than on the heap: see `present`. `prev_cols`/`prev_rows` being zero on the first frame is +/// what makes that frame a full one. +/// A/B switch, kept because this optimisation is exactly the kind that can be right about latency +/// and wrong about the screen. With it false, `present` behaves as it did before the shadow grid - +/// clear and write every cell - which is the reference any measurement of it should be compared +/// against, and the way to tell a rendering bug from a rendering difference. +const shadow_grid = true; + +var prev_cells: [@as(usize, max_cols) * @as(usize, max_rows)]pardes.Cell = if (shadow_grid) @splat(.{}) else undefined; +var prev_cols: u16 = 0; +var prev_rows: u16 = 0; + +// ------------------------------------------------------------------ where a frame's time goes +// +// A frame has three stages and they want different fixes, so the firmware is given all three rather +// than one total. Measured on the die, a render costs ~11 ms whether or not anything changed, which +// says the cost is the unconditional walk and not the edit - but "the walk" is two walks, the copy +// into vaxis's grid and vaxis's own diff, and only one of them is ours to change. +// +// Two CSR reads per stage. `cycle` is the unprivileged counter, read high-low-high because two +// 32-bit halves can straddle a wrap. +var prof_copy_cy: u64 = 0; +var prof_render_cy: u64 = 0; +var prof_flush_cy: u64 = 0; + +inline fn cycles() u64 { + if (builtin.cpu.arch != .riscv32) return 0; + while (true) { + const hi0 = asm volatile ("csrr %[o], cycleh" + : [o] "=r" (-> u32), + ); + const lo = asm volatile ("csrr %[o], cycle" + : [o] "=r" (-> u32), + ); + const hi1 = asm volatile ("csrr %[o], cycleh" + : [o] "=r" (-> u32), + ); + if (hi0 == hi1) return (@as(u64, hi0) << 32) | lo; + } +} + +/// The last frame's three stages, in cycles. Zero on any platform without the CSR. +export fn pardes_p4_frame_prof(copy: *u64, render: *u64, flush: *u64) callconv(.c) void { + copy.* = prof_copy_cy; + render.* = prof_render_cy; + flush.* = prof_flush_cy; } fn vaxisStyle(s: pardes.CellStyle) vaxis.Style { diff --git a/src/pardes.zig b/src/pardes.zig index 91811cc9..dcc99aea 100644 --- a/src/pardes.zig +++ b/src/pardes.zig @@ -2848,6 +2848,30 @@ pub const Surface = struct { var i: usize = 0; while (i < text.len) { if (col >= end) break; + // ASCII FAST PATH. Printable ASCII is one byte, one cell, one column, and the general + // path below reaches that answer through a UTF-8 length, a decode, a freshly + // constructed grapheme iterator, a slice validation and a width lookup - per character. + // That made this function 26% of a keystroke when profiled in the ESP32-P4's + // configuration (40x12, no tree-sitter), which is the largest single item there. + // + // The guard on the NEXT byte is what makes it correct rather than merely fast: an ASCII + // base joins a following combining mark, ZWJ or spacing mark into ONE cluster, and every + // scalar that can do that is non-ASCII. So an ASCII byte followed by another ASCII byte + // (or by nothing) is a complete grapheme cluster on its own. Same condition + // `modal.nextGrapheme` uses, for the same reason. + // + // `\t`, `\r` and the C0 controls are excluded by the range test and keep their existing + // handling below; DEL is excluded too. + { + const b = text[i]; + if (b >= 0x20 and b < 0x7f and (i + 1 == text.len or text[i + 1] < 0x80)) { + s.set(col, y, text[i .. i + 1], style); + i += 1; + col += 1; + continue; + } + } + if (col >= end) break; // n == 0: not a start byte at all. A short tail or a bad // continuation decodes to null the same way — one U+FFFD, one byte. const n = std.unicode.utf8ByteSequenceLength(text[i]) catch 0; |
