summaryrefslogtreecommitdiff
path: root/test/perf.zig
diff options
context:
space:
mode:
authorGabriel Schneider <[email protected]>2026-07-31 07:32:15 -0300
committerGabriel Schneider <[email protected]>2026-08-01 15:02:08 -0300
commit560ae0a63f7b5d354bad3ba13dd8ab26b5f0a870 (patch)
treedb2762cc2bc7bdeaae187e15cfc68a33ef2d1c48 /test/perf.zig
parent5bf8d6dd077517270377e5d8551108ecf252374f (diff)
downloadpardes-560ae0a63f7b5d354bad3ba13dd8ab26b5f0a870.tar.gz
pardes-560ae0a63f7b5d354bad3ba13dd8ab26b5f0a870.zip
motion, scrolling and redraw are flat in file size now
zig build perf drives the core directly — event, effects, one frame, no pty — over four generated fixtures: 1k lines, 50k, 300k, and 400 lines of 8000 columns, because a file that is long and a file that is wide fail differently. Every sample seeks somewhere else in the file first, since measuring at line 3 of a 300k-line file hides exactly the bug. perf record said half the run was scanning for newlines from byte 0. So File carries a line index, built on demand and invalidated in exactly ONE place — setContent, the funnel every content swap already goes through. That killed the scrollbar's per-frame line count (12.6% of the whole run by itself), scrollBy, ensureCursorVisible, lastNavRow, the syntax window bounds and two O(scroll) walks. normalKey computed max_line as a const at the top: two full passes over the buffer on every keystroke of every kind, for three g/G branches. It is lazy now. The modal primitives each walked the text twice for the same line. And the visible window was re-parsed on every scrolled row — a third of a megabyte per keypress on the wide fixture. The highlighted range is remembered, a scroll inside it is free, and only a re-parse that FOLLOWS a scroll takes slack: doing it unconditionally made typing 2.1x slower, since every character paid for a band it could never amortise. One j on a 19 MB file: 37.8ms -> 266us. Render: 4.1ms -> 77us. Open costs 1.25x more for the one extra pass, which buys 54x on every frame after, and 8 bytes per line of memory. Left standing, measured and named: edit-char is 14ms on 19MB because content is immutable and every keystroke copies the buffer. A third of that is the index rebuild, which could be a shift if setContent knew the edit offset; the rest wants a rope. bodyText's double copy and Surface.print's per-cell decode never rose above 2% of the profile afterwards, so they were left alone. No golden moved.
Diffstat (limited to 'test/perf.zig')
-rw-r--r--test/perf.zig466
1 files changed, 466 insertions, 0 deletions
diff --git a/test/perf.zig b/test/perf.zig
new file mode 100644
index 00000000..368e6ce1
--- /dev/null
+++ b/test/perf.zig
@@ -0,0 +1,466 @@
+//! The editing scoreboard: how long one user gesture costs against file size.
+//!
+//! zig build perf -- the table
+//! zig build perf -- --json -- the same, machine-readable
+//! zig build perf -- --base old.json -- table + a before/after delta column
+//! zig build perf -- --reps 60 -- more samples (default 25)
+//!
+//! It drives the sans-IO core directly — feed an Event, drain the effects,
+//! call render — exactly like test/hxdiff.zig, so there is no pty, no shell
+//! and no scheduler between the clock and the code under test. A run is
+//! self-contained: the fixtures are generated into /tmp/pardes-perf here and
+//! rewritten every run, so the numbers do not move when this repo's own
+//! sources do and there is no stale input to chase.
+//!
+//! WHAT A CELL MEANS. Every input row is the WHOLE gesture: update(event),
+//! drain the effect queue, render one frame into a fresh arena — what a user
+//! waits for between pressing a key and seeing the screen. `render` on its own
+//! is the redraw with nothing changed (a resize, a refresh), which is the floor
+//! every other row sits on.
+//!
+//! WHERE THE CURSOR IS matters more than anything else here: several of the
+//! core's line lookups are scans from byte 0, so a measurement taken at line 3
+//! of a 200k-line file reports a cost the user never pays. Each sample seeks to
+//! a different position spread across the whole file BEFORE the clock starts,
+//! so the median is the cost at a typical place in the document, not at its
+//! top.
+const std = @import("std");
+const libc = std.c;
+const pardes = @import("pardes");
+
+// ghostty-vt narrates every sequence it does not implement, and the fixture
+// text goes nowhere near a terminal here — cut the libraries to errors so the
+// table is the only thing on the screen.
+pub const std_options: std.Options = .{ .log_level = .err };
+
+const gpa = std.heap.page_allocator;
+
+/// std.time.Timer is gone in 0.16; clock_gettime is what dump.zig and
+/// test/lspbench.zig already use.
+fn nowNs() u64 {
+ var ts: std.c.timespec = undefined;
+ _ = std.c.clock_gettime(.MONOTONIC, &ts);
+ return @as(u64, @intCast(ts.sec)) *| 1_000_000_000 +| @as(u64, @intCast(ts.nsec));
+}
+
+/// The viewport every fixture is measured in. Bigger than the 80x24 the helix
+/// harness pins (that one matches helix's own harness; this one wants a screen
+/// somebody actually works in), and fixed, because half of these numbers scale
+/// with the number of cells on the screen.
+const screen_cols: u16 = 120;
+const screen_rows: u16 = 40;
+
+/// The four shapes a file comes in. `lines` x `cols` is the generated body;
+/// the point of the pair is that "big" has two different meanings and they
+/// break different code — a 200k-line file is long in the LINE index, an
+/// 8000-column file is long inside ONE line, and a scan that is fine for one
+/// is quadratic for the other.
+const Fixture = struct {
+ name: []const u8,
+ lines: usize,
+ cols: usize,
+};
+
+const fixtures = [_]Fixture{
+ // the control: a normal source file. Anything that shows up here is a
+ // constant cost, not a scaling one.
+ .{ .name = "small", .lines = 1_000, .cols = 60 },
+ .{ .name = "medium", .lines = 50_000, .cols = 60 },
+ // "a few MB, hundreds of thousands of lines"
+ .{ .name = "large", .lines = 300_000, .cols = 60 },
+ // the other failure mode: few lines, each one wider than any screen
+ .{ .name = "longline", .lines = 400, .cols = 8_000 },
+};
+
+const Op = enum {
+ /// Look at the path: disk read, pane creation, layout, first frame.
+ open,
+ /// one redraw, nothing changed
+ render,
+ /// `j` — one row of cursor movement
+ key_down,
+ /// `l` — one column. On `longline` the cursor sits deep in the line, so
+ /// this is the horizontal-scroll path; elsewhere it is a plain step.
+ key_right,
+ /// PageDown
+ page_down,
+ /// one mouse wheel tick
+ wheel,
+ /// one printable typed in insert mode
+ edit_char,
+
+ fn label(o: Op) []const u8 {
+ return switch (o) {
+ .open => "open",
+ .render => "render",
+ .key_down => "key-down",
+ .key_right => "key-right",
+ .page_down => "page-down",
+ .wheel => "wheel",
+ .edit_char => "edit-char",
+ };
+ }
+};
+
+const Cell = struct {
+ min_us: u64 = 0,
+ med_us: u64 = 0,
+ p90_us: u64 = 0,
+ max_us: u64 = 0,
+};
+
+pub fn main(init: std.process.Init) !void {
+ const args = try init.minimal.args.toSlice(init.arena.allocator());
+ var json = false;
+ var reps: usize = 25;
+ var base_path: ?[]const u8 = null;
+ var i: usize = 1;
+ while (i < args.len) : (i += 1) {
+ const a = args[i];
+ if (std.mem.eql(u8, a, "--json")) {
+ json = true;
+ } else if (std.mem.eql(u8, a, "--reps") and i + 1 < args.len) {
+ i += 1;
+ reps = std.fmt.parseInt(usize, args[i], 10) catch reps;
+ } else if (std.mem.eql(u8, a, "--base") and i + 1 < args.len) {
+ i += 1;
+ base_path = args[i];
+ } else fatal("usage: pardes-perf [--json] [--reps N] [--base old.json]", .{});
+ }
+
+ // fixtures live in a temp dir and are rewritten every run: they are inputs
+ // to a measurement, and a stale one silently changes what the table means.
+ // .zig extensions on purpose — that is what puts tree-sitter in the frame.
+ const dir = "/tmp/pardes-perf";
+ _ = libc.mkdir(dir, 0o755); // EEXIST is fine
+ var paths: [fixtures.len][]const u8 = undefined;
+ var bytes: [fixtures.len]usize = undefined;
+ for (fixtures, 0..) |fx, fi| {
+ const path = try std.fmt.allocPrint(gpa, "{s}/{s}.zig", .{ dir, fx.name });
+ const text = try generate(fx);
+ try writeFile(path, text);
+ paths[fi] = path;
+ // what got WRITTEN, not lines*cols: a line whose code is already wider
+ // than `cols` is left alone rather than truncated, so the nominal
+ // product would understate the file the editor actually opens
+ bytes[fi] = text.len;
+ gpa.free(text);
+ }
+
+ var cells: [std.enums.values(Op).len][fixtures.len]Cell = undefined;
+ for (std.enums.values(Op), 0..) |op, oi| {
+ for (fixtures, 0..) |fx, fi| {
+ cells[oi][fi] = try measure(op, fx, paths[fi], reps);
+ }
+ }
+
+ if (json) return reportJson(init.io, &cells, &bytes, reps);
+ reportText(&cells, &bytes, reps, base_path);
+}
+
+// ---- the measured gestures ----
+
+/// One (op, fixture) cell: `reps` timed samples plus three warmup rounds,
+/// reported as min / median / p90 / max. The spread between median and p90 is
+/// the run's noise, and reportText prints the worst one so a reader knows how
+/// big a difference has to be before it is real.
+fn measure(op: Op, fx: Fixture, path: []const u8, reps: usize) !Cell {
+ const warmup = 3;
+ var samples: std.ArrayList(u64) = .empty;
+ defer samples.deinit(gpa);
+
+ if (op == .open) {
+ // fresh core per sample: opening is a one-shot, and the second Look at
+ // the same path only refocuses the pane already holding it.
+ for (0..warmup + reps) |n| {
+ const core = try boot();
+ defer core.deinit();
+ const t0 = nowNs();
+ core.lookAt(0, path);
+ pump(core);
+ _ = try frame(core);
+ const dt = nowNs() -| t0;
+ if (n >= warmup) try samples.append(gpa, dt / 1000);
+ }
+ return summarize(samples.items);
+ }
+
+ const core = try boot();
+ defer core.deinit();
+ core.lookAt(0, path);
+ pump(core);
+ const id = filePane(core) orelse fatal("Look {s} opened no file pane", .{path});
+ core.active = id;
+ const pane = core.panes[id].?;
+ _ = try frame(core);
+ // insert mode is entered ONCE: `i` is a mode change, not a keystroke of
+ // typing, and measuring it inside every edit sample would report the wrong
+ // thing entirely.
+ if (op == .edit_char) {
+ core.update(.{ .key = .{ .cp = 'i', .text = "i" } });
+ pump(core);
+ }
+
+ for (0..warmup + reps) |n| {
+ seek(pane, fx, n);
+ _ = try frame(core); // settle the view at the new spot, untimed
+ const t0 = nowNs();
+ switch (op) {
+ .open => unreachable,
+ .render => {},
+ .key_down => core.update(.{ .key = .{ .cp = 'j', .text = "j" } }),
+ .key_right => core.update(.{ .key = .{ .cp = 'l', .text = "l" } }),
+ .page_down => core.update(.{ .key = .{ .cp = pardes.Key.page_down } }),
+ .wheel => core.update(.{ .mouse = .{ .button = .wheel_down, .kind = .press, .col = 4, .row = 8 } }),
+ .edit_char => core.update(.{ .key = .{ .cp = 'x', .text = "x" } }),
+ }
+ pump(core);
+ _ = try frame(core);
+ const dt = nowNs() -| t0;
+ if (n >= warmup) try samples.append(gpa, dt / 1000);
+ }
+ return summarize(samples.items);
+}
+
+/// Park the cursor and the view at sample `n`'s position, walked across the
+/// file by a prime stride so consecutive samples land nowhere near each other
+/// and a whole run covers the document rather than one neighbourhood of it.
+/// Rows sit in the middle 80% (a position at the very top or bottom measures
+/// the clamp, not the work), and on a wide fixture the column is deep inside
+/// the line so `key-right` exercises the hscroll cut rather than column 1.
+/// The sequence depends only on `n`, so two builds see the same positions in
+/// the same order and their cells are comparable one for one.
+fn seek(pane: *pardes.Pane, fx: Fixture, n: usize) void {
+ const f = &pane.file.?;
+ const span = @max(1, fx.lines * 8 / 10);
+ const row = fx.lines / 10 + (n *% 7919) % span;
+ f.scroll = @min(row, fx.lines -| 1);
+ f.syntax_dirty = true;
+ pane.cur_row = @intCast(f.scroll);
+ pane.cur_col = if (fx.cols > 200) @intCast(fx.cols * 3 / 4) else 0;
+ pane.cur_pinned = true;
+ pane.hscroll = @max(0, pane.cur_col - 40);
+ pane.sticky_col = -1;
+}
+
+fn boot() !*pardes.Pardes {
+ const core = try pardes.Pardes.init(gpa, .{ .tty_only = true });
+ core.update(.{ .resize = .{ .cols = screen_cols, .rows = screen_rows } });
+ pump(core);
+ return core;
+}
+
+/// one frame into a throwaway arena — the shell's per-frame arena, which is
+/// what keeps the retained-render contract honest (vaxis stores slices into
+/// whatever it was handed, so the frame's text must outlive vx.render)
+fn frame(core: *pardes.Pardes) !*pardes.Surface {
+ var arena: std.heap.ArenaAllocator = .init(gpa);
+ defer arena.deinit();
+ return core.render(arena.allocator());
+}
+
+fn filePane(core: *pardes.Pardes) ?usize {
+ for (core.panes, 0..) |slot, i| {
+ const pane = slot orelse continue;
+ if (pane.file != null) return i;
+ }
+ return null;
+}
+
+/// Drain queued effects, all ignored (spawn, write, watch, ...): there is no
+/// shell here, and nothing measured depends on one answering.
+fn pump(core: *pardes.Pardes) void {
+ while (core.nextEffect()) |_| {}
+}
+
+fn summarize(samples: []u64) Cell {
+ if (samples.len == 0) return .{};
+ std.mem.sort(u64, samples, {}, std.sort.asc(u64));
+ return .{
+ .min_us = samples[0],
+ .med_us = samples[samples.len / 2],
+ .p90_us = samples[(samples.len * 9) / 10 -| 1],
+ .max_us = samples[samples.len - 1],
+ };
+}
+
+// ---- the fixture generator ----
+
+/// Plausible Zig, so the tree-sitter pass has real nodes to walk rather than
+/// one giant error node: a repeating four-line shape padded to `cols`. Long
+/// lines get their width from a comment tail — the alternative (an enormous
+/// string literal) makes the whole file one token and flatters every scan that
+/// looks for a newline.
+fn generate(fx: Fixture) ![]u8 {
+ var out: std.ArrayList(u8) = .empty;
+ try out.ensureTotalCapacity(gpa, fx.lines * (fx.cols + 1) + 64);
+ var n: usize = 0;
+ while (n < fx.lines) : (n += 1) {
+ const start = out.items.len;
+ switch (n % 4) {
+ 0 => try out.print(gpa, "const value_{d}: u32 = {d}; // ", .{ n, n *% 2654435761 }),
+ 1 => try out.print(gpa, "pub fn helper_{d}(a: u32, b: u32) u32 {{ return a +% b *% {d}; }} // ", .{ n, n }),
+ 2 => try out.print(gpa, " const text_{d} = \"lorem ipsum dolor sit amet {d}\"; // ", .{ n, n }),
+ else => try out.print(gpa, "// comment line {d} — ", .{n}),
+ }
+ while (out.items.len - start < fx.cols) try out.append(gpa, 'x');
+ try out.append(gpa, '\n');
+ }
+ return out.toOwnedSlice(gpa);
+}
+
+fn writeFile(path: []const u8, text: []const u8) !void {
+ var pathbuf: [4096]u8 = undefined;
+ const path_z = try std.fmt.bufPrintSentinel(&pathbuf, "{s}", .{path}, 0);
+ const fd = libc.open(path_z, .{ .ACCMODE = .WRONLY, .CREAT = true, .TRUNC = true }, @as(libc.mode_t, 0o644));
+ if (fd < 0) return error.OpenFailed;
+ defer _ = libc.close(fd);
+ var off: usize = 0;
+ while (off < text.len) {
+ const n = libc.write(fd, text.ptr + off, text.len - off);
+ if (n < 0) {
+ if (libc.errno(n) == .INTR) continue;
+ return error.WriteFailed;
+ }
+ off += @intCast(n);
+ }
+}
+
+// ---- reporting ----
+
+fn reportText(cells: *const [std.enums.values(Op).len][fixtures.len]Cell, bytes: *const [fixtures.len]usize, reps: usize, base_path: ?[]const u8) void {
+ const o = std.debug.print;
+ const base = if (base_path) |bp| readBase(bp) else null;
+
+ o("pardes perf — {d}x{d} viewport, {d} samples/cell, median us\n\n", .{ screen_cols, screen_rows, reps });
+ o("{s:<12}", .{"fixture"});
+ for (fixtures) |fx| o(" {s:>12}", .{fx.name});
+ o("\n{s:<12}", .{"lines"});
+ for (fixtures) |fx| o(" {d:>12}", .{fx.lines});
+ o("\n{s:<12}", .{"cols"});
+ for (fixtures) |fx| o(" {d:>12}", .{fx.cols});
+ o("\n{s:<12}", .{"size"});
+ for (bytes) |b| o(" {d:>10} KB", .{b / 1024});
+ o("\n\n", .{});
+
+ // one row per gesture. `open` is a whole Look; every other row is
+ // update+effects+one frame, which is the latency a user actually sees.
+ o("{s:<12}", .{"op"});
+ for (fixtures) |fx| {
+ if (base != null) o(" {s:>19}", .{fx.name}) else o(" {s:>12}", .{fx.name});
+ }
+ // 12 for the op name + one column per fixture, wider when a baseline adds
+ // its ratio to each cell
+ o("\n{s}\n", .{if (base != null) "-" ** (12 + fixtures.len * 20) else "-" ** (12 + fixtures.len * 13)});
+ var worst_jitter: f64 = 0;
+ for (std.enums.values(Op), 0..) |op, oi| {
+ o("{s:<12}", .{op.label()});
+ for (0..fixtures.len) |fi| {
+ const c = cells[oi][fi];
+ if (c.med_us > 0) {
+ const j = (@as(f64, @floatFromInt(c.p90_us)) - @as(f64, @floatFromInt(c.med_us))) / @as(f64, @floatFromInt(c.med_us));
+ if (j > worst_jitter) worst_jitter = j;
+ }
+ if (base) |b| {
+ const prev = b.med[oi][fi];
+ if (prev == 0) {
+ o(" {d:>12}{s:>7}", .{ c.med_us, "-" });
+ } else {
+ const ratio = @as(f64, @floatFromInt(c.med_us)) / @as(f64, @floatFromInt(prev));
+ o(" {d:>12} {d:>6.2}x", .{ c.med_us, ratio });
+ }
+ } else o(" {d:>12}", .{c.med_us});
+ }
+ o("\n", .{});
+ }
+ // The spread is not only scheduler noise: samples are taken at DIFFERENT
+ // places in the file on purpose (see seek), so a cost that still depends on
+ // where the cursor is shows up here as well. Both are reasons not to
+ // believe a small difference, and the sample positions are identical from
+ // run to run, so two builds are still comparable cell for cell.
+ o("\nnoise: worst cell p90 is {d:.0}% over its median (scheduler + the spread\n", .{worst_jitter * 100});
+ o("of sample positions through the file). Treat a difference smaller than\n", .{});
+ o("that as nothing, and re-run before believing a small win.\n", .{});
+ if (base_path) |bp| o("baseline: {s} (x column = now / then; under 1.00 is faster)\n", .{bp});
+}
+
+/// Real stdout, not std.debug.print's stderr: this is the form `--base` reads
+/// back, and `zig build perf -- --json > runs/old.json` writing an empty file
+/// would make the next comparison silently print no ratios at all.
+fn reportJson(io: std.Io, cells: *const [std.enums.values(Op).len][fixtures.len]Cell, bytes: *const [fixtures.len]usize, reps: usize) void {
+ var out: std.ArrayList(u8) = .empty;
+ defer out.deinit(gpa);
+ out.print(gpa, "{{\"cols\":{d},\"rows\":{d},\"reps\":{d},\"fixtures\":[", .{ screen_cols, screen_rows, reps }) catch return;
+ for (fixtures, 0..) |fx, fi| {
+ out.print(gpa, "{s}{{\"name\":\"{s}\",\"lines\":{d},\"cols\":{d},\"bytes\":{d}}}", .{
+ if (fi > 0) "," else "", fx.name, fx.lines, fx.cols, bytes[fi],
+ }) catch return;
+ }
+ out.appendSlice(gpa, "],\"cells\":[") catch return;
+ var first = true;
+ for (std.enums.values(Op), 0..) |op, oi| {
+ for (fixtures, 0..) |fx, fi| {
+ const c = cells[oi][fi];
+ out.print(gpa, "{s}{{\"op\":\"{s}\",\"fixture\":\"{s}\",\"min_us\":{d},\"med_us\":{d},\"p90_us\":{d},\"max_us\":{d}}}", .{
+ if (first) "" else ",", op.label(), fx.name, c.min_us, c.med_us, c.p90_us, c.max_us,
+ }) catch return;
+ first = false;
+ }
+ }
+ out.appendSlice(gpa, "]}\n") catch return;
+ std.Io.File.stdout().writeStreamingAll(io, out.items) catch {};
+}
+
+const Base = struct { med: [std.enums.values(Op).len][fixtures.len]u64 };
+
+/// A previous --json run, reduced to the medians this table compares against.
+/// Cells the old run did not have stay 0 and print as "-": the op list may
+/// have grown since, and a missing number is not a regression. An unreadable
+/// file is fatal rather than silently ratio-less — a comparison you asked for
+/// and did not get is worse than no comparison.
+fn readBase(path: []const u8) ?Base {
+ const src = readFileAlloc(path) catch fatal("--base: cannot read {s}", .{path});
+ var b: Base = .{ .med = @splat(@splat(0)) };
+ const parsed = std.json.parseFromSlice(struct {
+ cells: []const struct {
+ op: []const u8,
+ fixture: []const u8,
+ med_us: u64,
+ },
+ }, gpa, src, .{ .ignore_unknown_fields = true }) catch return null;
+ defer parsed.deinit();
+ for (parsed.value.cells) |c| {
+ for (std.enums.values(Op), 0..) |op, oi| {
+ if (!std.mem.eql(u8, op.label(), c.op)) continue;
+ for (fixtures, 0..) |fx, fi| {
+ if (std.mem.eql(u8, fx.name, c.fixture)) b.med[oi][fi] = c.med_us;
+ }
+ }
+ }
+ return b;
+}
+
+fn readFileAlloc(path: []const u8) ![]u8 {
+ var pathbuf: [4096]u8 = undefined;
+ const path_z = try std.fmt.bufPrintSentinel(&pathbuf, "{s}", .{path}, 0);
+ const fd = libc.open(path_z, .{ .ACCMODE = .RDONLY });
+ if (fd < 0) return error.OpenFailed;
+ defer _ = libc.close(fd);
+ var buf: std.ArrayList(u8) = .empty;
+ var chunk: [16384]u8 = undefined;
+ while (true) {
+ const n = libc.read(fd, &chunk, chunk.len);
+ if (n < 0) {
+ if (libc.errno(n) == .INTR) continue;
+ return error.ReadFailed;
+ }
+ if (n == 0) break;
+ try buf.appendSlice(gpa, chunk[0..@intCast(n)]);
+ }
+ return buf.items;
+}
+
+fn fatal(comptime fmt: []const u8, args: anytype) noreturn {
+ std.debug.print("pardes-perf: " ++ fmt ++ "\n", args);
+ std.process.exit(1);
+}