summaryrefslogtreecommitdiff
path: root/src
diff options
context:
space:
mode:
Diffstat (limited to 'src')
-rw-r--r--src/file_pane.zig6
-rw-r--r--src/modal.zig26
-rw-r--r--src/p4.zig98
-rw-r--r--src/pardes.zig24
4 files changed, 149 insertions, 5 deletions
diff --git a/src/file_pane.zig b/src/file_pane.zig
index 8fc0fa0d..1c6d5e2f 100644
--- a/src/file_pane.zig
+++ b/src/file_pane.zig
@@ -92,6 +92,12 @@ pub fn dumpPane(
pub fn graphemeDisplayWidth(grapheme: []const u8) usize {
if (std.mem.eql(u8, grapheme, "\t")) return config.tab_width;
+ // A one-byte printable ASCII grapheme is one cell, and saying so here rather than asking
+ // `gwidth` costs a comparison instead of a Unicode table walk. `gwidth` was 6.9% of a profiled
+ // keystroke at the P4's geometry, essentially all of it answering this question about `y`.
+ // Bounded to 0x20..0x7e on purpose: DEL and the C0 controls are not one printable cell, and
+ // `gwidth` is still the authority on them.
+ if (grapheme.len == 1 and grapheme[0] >= 0x20 and grapheme[0] < 0x7f) return 1;
return @max(1, @as(usize, vaxis.gwidth.gwidth(grapheme, .unicode)));
}
diff --git a/src/modal.zig b/src/modal.zig
index f8093ed8..953aa900 100644
--- a/src/modal.zig
+++ b/src/modal.zig
@@ -499,7 +499,9 @@ pub fn lineStartOffset(content: []const u8, row: usize) usize {
pub fn lineSlice(content: []const u8, row: usize) []const u8 {
const start = lineStartOffset(content, row);
if (start >= content.len) return "";
- const nl = std.mem.indexOfPos(u8, content, start, "\n") orelse content.len;
+ // indexOfScalarPos, not indexOfPos with a one-byte needle: the latter runs the generic
+ // substring search where a memchr will do, and this is called once per visible row per frame.
+ const nl = std.mem.indexOfScalarPos(u8, content, start, '\n') orelse content.len;
return content[start..nl];
}
@@ -849,11 +851,29 @@ pub fn deleteSpan(alloc: std.mem.Allocator, content: []const u8, a: Cursor, b: C
pub const HxRange = struct { anchor: usize, head: usize };
-/// The grapheme containing `off`, or text.len at EOF. This is also the repair
-/// path for stale/external byte columns that happen to point into UTF-8.
+/// The first byte of the grapheme cluster containing `off`.
+///
+/// The general answer needs UAX #29, which is why the slow path below iterates from the start of
+/// `text` with the full break state machine - and that made this the single hottest function in a
+/// keystroke: 21.5% of a profiled edit at the ESP32-P4's 40x12 geometry, because the render path
+/// calls it once per visible row with a column offset, so the cost follows the cursor's distance
+/// along its line. That is exactly the shape measured on the die, where inserting at column 320 of
+/// a fixed line cost 7.8 ms more than inserting at column 0 of the same line.
+///
+/// The fast path is sound rather than approximate. In UAX #29 every ASCII scalar is its own
+/// grapheme cluster with ONE exception, GB3: CR is joined to a following LF. Every other rule that
+/// could extend a cluster across `off` - Extend, ZWJ, SpacingMark, Prepend, Regional_Indicator -
+/// is spelled with non-ASCII scalars. So if the byte at `off` and the byte before it are both
+/// ASCII and are not that CR-LF pair, `off` already IS a cluster boundary and there is nothing to
+/// search for. Text that is not all ASCII still takes the slow path, byte for byte as before.
pub fn graphemeStart(text: []const u8, off: usize) usize {
const bounded = @min(off, text.len);
if (bounded == text.len) return text.len;
+ if (text[bounded] < 0x80) {
+ if (bounded == 0) return 0;
+ const prev = text[bounded - 1];
+ if (prev < 0x80 and !(prev == '\r' and text[bounded] == '\n')) return bounded;
+ }
var it = uucode.grapheme.utf8Iterator(text);
while (it.nextGrapheme()) |g| {
if (bounded < g.end) return g.start;
diff --git a/src/p4.zig b/src/p4.zig
index 7356dc73..57ee88cf 100644
--- a/src/p4.zig
+++ b/src/p4.zig
@@ -30,6 +30,7 @@
//! any other input. Firmware has no `TIOCGWINSZ`, so the host-side bridge synthesises the first one.
const std = @import("std");
+const builtin = @import("builtin");
const pardes = @import("pardes.zig");
const vaxis = @import("vaxis");
@@ -510,8 +511,34 @@ const pardes_host: pardes.Host.VTable = .{ .push_present = present };
/// (`src/tty/tty.zig:1096`) minus the panel compositor and the kitty image path: neither has a
/// reason to exist on a board with no pixels.
fn present(_: ?*anyopaque, surface: *const pardes.Surface) void {
+ const t0 = cycles();
const win = vx.window();
- win.clear();
+ const n = @as(usize, surface.cols) * @as(usize, surface.rows);
+
+ // THE SHADOW GRID. Copying all 480 cells into vaxis every frame cost 6.75 ms on the die - 57%
+ // of a keystroke, and it was paid whether or not anything changed: a second render with nothing
+ // new measured the same as the first. vaxis already diffs its own grid against the terminal, but
+ // it can only do that AFTER being told every cell, and being told is the expensive part
+ // (`writeCell` builds a vaxis `Cell`, which carries an always-null image placement).
+ //
+ // So keep the previous Surface and tell vaxis only what moved. `Cell.visuallyEqual` is the
+ // right comparison and already exists for the panel compositor's benefit: it ignores scratch
+ // bytes past `len` and treats any two default cells as equal, so it cannot manufacture a write.
+ //
+ // STATIC, and that is not a micro-optimisation - it is a bug fix. The first version allocated
+ // this from the editor's heap, and on a board whose 384 KiB is already nearly spoken for that
+ // was enough to make `vx.resize` fail: a resize then hit its OOM path, restored the previous
+ // geometry and returned, so the screen was never repainted. Measured as a resize emitting 80
+ // bytes where it had emitted 1,392. The grid is bounded by `max_cols` x `max_rows` at comptime,
+ // so it belongs in `.bss` where it cannot compete with anything.
+ const full = !shadow_grid or prev_cols != surface.cols or prev_rows != surface.rows;
+ if (full) {
+ prev_cols = surface.cols;
+ prev_rows = surface.rows;
+ win.clear();
+ }
+ const usable = shadow_grid and n <= prev_cells.len;
+
var y: u16 = 0;
while (y < surface.rows) : (y += 1) {
var x: u16 = 0;
@@ -519,7 +546,18 @@ fn present(_: ?*anyopaque, surface: *const pardes.Surface) void {
// `at` takes a mutable Surface but only reads; the tty shell does the same const-cast
// for the same reason (src/tty/tty.zig:1105).
const cell = @constCast(surface).at(x, y);
- if (cell.default) continue;
+ const idx = @as(usize, y) * @as(usize, surface.cols) + @as(usize, x);
+ if (usable) {
+ if (!full and cell.visuallyEqual(&prev_cells[idx])) continue;
+ prev_cells[idx] = cell.*;
+ } else if (cell.default) continue;
+
+ if (cell.default) {
+ // Changed TO default. `win.clear()` is what used to blank these, and it is not run
+ // on an incremental frame, so say it explicitly.
+ win.writeCell(x, y, .{ .char = .{ .grapheme = " " }, .style = .{} });
+ continue;
+ }
win.writeCell(x, y, .{
.char = .{ .grapheme = cell.grapheme() },
.style = vaxisStyle(cell.style),
@@ -529,11 +567,67 @@ fn present(_: ?*anyopaque, surface: *const pardes.Surface) void {
if (surface.cursor) |cur| {
win.showCursor(cur.x, cur.y);
} else win.hideCursor();
+ const t1 = cycles();
// vaxis diffs against its own shadow grid, so this writes only what changed - which is what
// makes an editor usable at 11.9 KB/s.
vx.render(&out) catch return;
+ const t2 = cycles();
out.flush() catch return;
+ const t3 = cycles();
+
+ prof_copy_cy = t1 -% t0;
+ prof_render_cy = t2 -% t1;
+ prof_flush_cy = t3 -% t2;
+}
+
+/// The previous Surface, cell for cell, sized for the largest grid this board can drive. In `.bss`
+/// rather than on the heap: see `present`. `prev_cols`/`prev_rows` being zero on the first frame is
+/// what makes that frame a full one.
+/// A/B switch, kept because this optimisation is exactly the kind that can be right about latency
+/// and wrong about the screen. With it false, `present` behaves as it did before the shadow grid -
+/// clear and write every cell - which is the reference any measurement of it should be compared
+/// against, and the way to tell a rendering bug from a rendering difference.
+const shadow_grid = true;
+
+var prev_cells: [@as(usize, max_cols) * @as(usize, max_rows)]pardes.Cell = if (shadow_grid) @splat(.{}) else undefined;
+var prev_cols: u16 = 0;
+var prev_rows: u16 = 0;
+
+// ------------------------------------------------------------------ where a frame's time goes
+//
+// A frame has three stages and they want different fixes, so the firmware is given all three rather
+// than one total. Measured on the die, a render costs ~11 ms whether or not anything changed, which
+// says the cost is the unconditional walk and not the edit - but "the walk" is two walks, the copy
+// into vaxis's grid and vaxis's own diff, and only one of them is ours to change.
+//
+// Two CSR reads per stage. `cycle` is the unprivileged counter, read high-low-high because two
+// 32-bit halves can straddle a wrap.
+var prof_copy_cy: u64 = 0;
+var prof_render_cy: u64 = 0;
+var prof_flush_cy: u64 = 0;
+
+inline fn cycles() u64 {
+ if (builtin.cpu.arch != .riscv32) return 0;
+ while (true) {
+ const hi0 = asm volatile ("csrr %[o], cycleh"
+ : [o] "=r" (-> u32),
+ );
+ const lo = asm volatile ("csrr %[o], cycle"
+ : [o] "=r" (-> u32),
+ );
+ const hi1 = asm volatile ("csrr %[o], cycleh"
+ : [o] "=r" (-> u32),
+ );
+ if (hi0 == hi1) return (@as(u64, hi0) << 32) | lo;
+ }
+}
+
+/// The last frame's three stages, in cycles. Zero on any platform without the CSR.
+export fn pardes_p4_frame_prof(copy: *u64, render: *u64, flush: *u64) callconv(.c) void {
+ copy.* = prof_copy_cy;
+ render.* = prof_render_cy;
+ flush.* = prof_flush_cy;
}
fn vaxisStyle(s: pardes.CellStyle) vaxis.Style {
diff --git a/src/pardes.zig b/src/pardes.zig
index 91811cc9..dcc99aea 100644
--- a/src/pardes.zig
+++ b/src/pardes.zig
@@ -2848,6 +2848,30 @@ pub const Surface = struct {
var i: usize = 0;
while (i < text.len) {
if (col >= end) break;
+ // ASCII FAST PATH. Printable ASCII is one byte, one cell, one column, and the general
+ // path below reaches that answer through a UTF-8 length, a decode, a freshly
+ // constructed grapheme iterator, a slice validation and a width lookup - per character.
+ // That made this function 26% of a keystroke when profiled in the ESP32-P4's
+ // configuration (40x12, no tree-sitter), which is the largest single item there.
+ //
+ // The guard on the NEXT byte is what makes it correct rather than merely fast: an ASCII
+ // base joins a following combining mark, ZWJ or spacing mark into ONE cluster, and every
+ // scalar that can do that is non-ASCII. So an ASCII byte followed by another ASCII byte
+ // (or by nothing) is a complete grapheme cluster on its own. Same condition
+ // `modal.nextGrapheme` uses, for the same reason.
+ //
+ // `\t`, `\r` and the C0 controls are excluded by the range test and keep their existing
+ // handling below; DEL is excluded too.
+ {
+ const b = text[i];
+ if (b >= 0x20 and b < 0x7f and (i + 1 == text.len or text[i + 1] < 0x80)) {
+ s.set(col, y, text[i .. i + 1], style);
+ i += 1;
+ col += 1;
+ continue;
+ }
+ }
+ if (col >= end) break;
// n == 0: not a start byte at all. A short tail or a bad
// continuation decodes to null the same way — one U+FFFD, one byte.
const n = std.unicode.utf8ByteSequenceLength(text[i]) catch 0;