diff options
| author | Gabriel Schneider <[email protected]> | 2026-07-31 07:32:15 -0300 |
|---|---|---|
| committer | Gabriel Schneider <[email protected]> | 2026-08-01 15:02:08 -0300 |
| commit | 560ae0a63f7b5d354bad3ba13dd8ab26b5f0a870 (patch) | |
| tree | db2762cc2bc7bdeaae187e15cfc68a33ef2d1c48 /test/perf.zig | |
| parent | 5bf8d6dd077517270377e5d8551108ecf252374f (diff) | |
| download | pardes-560ae0a63f7b5d354bad3ba13dd8ab26b5f0a870.tar.gz pardes-560ae0a63f7b5d354bad3ba13dd8ab26b5f0a870.zip | |
motion, scrolling and redraw are flat in file size now
zig build perf drives the core directly — event, effects, one frame, no pty —
over four generated fixtures: 1k lines, 50k, 300k, and 400 lines of 8000
columns, because a file that is long and a file that is wide fail differently.
Every sample seeks somewhere else in the file first, since measuring at line 3
of a 300k-line file hides exactly the bug.
perf record said half the run was scanning for newlines from byte 0. So File
carries a line index, built on demand and invalidated in exactly ONE place —
setContent, the funnel every content swap already goes through. That killed the
scrollbar's per-frame line count (12.6% of the whole run by itself), scrollBy,
ensureCursorVisible, lastNavRow, the syntax window bounds and two O(scroll)
walks. normalKey computed max_line as a const at the top: two full passes over
the buffer on every keystroke of every kind, for three g/G branches. It is lazy
now. The modal primitives each walked the text twice for the same line.
And the visible window was re-parsed on every scrolled row — a third of a
megabyte per keypress on the wide fixture. The highlighted range is remembered,
a scroll inside it is free, and only a re-parse that FOLLOWS a scroll takes
slack: doing it unconditionally made typing 2.1x slower, since every character
paid for a band it could never amortise.
One j on a 19 MB file: 37.8ms -> 266us. Render: 4.1ms -> 77us. Open costs 1.25x
more for the one extra pass, which buys 54x on every frame after, and 8 bytes
per line of memory.
Left standing, measured and named: edit-char is 14ms on 19MB because content is
immutable and every keystroke copies the buffer. A third of that is the index
rebuild, which could be a shift if setContent knew the edit offset; the rest
wants a rope. bodyText's double copy and Surface.print's per-cell decode never
rose above 2% of the profile afterwards, so they were left alone.
No golden moved.
Diffstat (limited to 'test/perf.zig')
| -rw-r--r-- | test/perf.zig | 466 |
1 files changed, 466 insertions, 0 deletions
diff --git a/test/perf.zig b/test/perf.zig new file mode 100644 index 00000000..368e6ce1 --- /dev/null +++ b/test/perf.zig @@ -0,0 +1,466 @@ +//! The editing scoreboard: how long one user gesture costs against file size. +//! +//! zig build perf -- the table +//! zig build perf -- --json -- the same, machine-readable +//! zig build perf -- --base old.json -- table + a before/after delta column +//! zig build perf -- --reps 60 -- more samples (default 25) +//! +//! It drives the sans-IO core directly — feed an Event, drain the effects, +//! call render — exactly like test/hxdiff.zig, so there is no pty, no shell +//! and no scheduler between the clock and the code under test. A run is +//! self-contained: the fixtures are generated into /tmp/pardes-perf here and +//! rewritten every run, so the numbers do not move when this repo's own +//! sources do and there is no stale input to chase. +//! +//! WHAT A CELL MEANS. Every input row is the WHOLE gesture: update(event), +//! drain the effect queue, render one frame into a fresh arena — what a user +//! waits for between pressing a key and seeing the screen. `render` on its own +//! is the redraw with nothing changed (a resize, a refresh), which is the floor +//! every other row sits on. +//! +//! WHERE THE CURSOR IS matters more than anything else here: several of the +//! core's line lookups are scans from byte 0, so a measurement taken at line 3 +//! of a 200k-line file reports a cost the user never pays. Each sample seeks to +//! a different position spread across the whole file BEFORE the clock starts, +//! so the median is the cost at a typical place in the document, not at its +//! top. +const std = @import("std"); +const libc = std.c; +const pardes = @import("pardes"); + +// ghostty-vt narrates every sequence it does not implement, and the fixture +// text goes nowhere near a terminal here — cut the libraries to errors so the +// table is the only thing on the screen. +pub const std_options: std.Options = .{ .log_level = .err }; + +const gpa = std.heap.page_allocator; + +/// std.time.Timer is gone in 0.16; clock_gettime is what dump.zig and +/// test/lspbench.zig already use. +fn nowNs() u64 { + var ts: std.c.timespec = undefined; + _ = std.c.clock_gettime(.MONOTONIC, &ts); + return @as(u64, @intCast(ts.sec)) *| 1_000_000_000 +| @as(u64, @intCast(ts.nsec)); +} + +/// The viewport every fixture is measured in. Bigger than the 80x24 the helix +/// harness pins (that one matches helix's own harness; this one wants a screen +/// somebody actually works in), and fixed, because half of these numbers scale +/// with the number of cells on the screen. +const screen_cols: u16 = 120; +const screen_rows: u16 = 40; + +/// The four shapes a file comes in. `lines` x `cols` is the generated body; +/// the point of the pair is that "big" has two different meanings and they +/// break different code — a 200k-line file is long in the LINE index, an +/// 8000-column file is long inside ONE line, and a scan that is fine for one +/// is quadratic for the other. +const Fixture = struct { + name: []const u8, + lines: usize, + cols: usize, +}; + +const fixtures = [_]Fixture{ + // the control: a normal source file. Anything that shows up here is a + // constant cost, not a scaling one. + .{ .name = "small", .lines = 1_000, .cols = 60 }, + .{ .name = "medium", .lines = 50_000, .cols = 60 }, + // "a few MB, hundreds of thousands of lines" + .{ .name = "large", .lines = 300_000, .cols = 60 }, + // the other failure mode: few lines, each one wider than any screen + .{ .name = "longline", .lines = 400, .cols = 8_000 }, +}; + +const Op = enum { + /// Look at the path: disk read, pane creation, layout, first frame. + open, + /// one redraw, nothing changed + render, + /// `j` — one row of cursor movement + key_down, + /// `l` — one column. On `longline` the cursor sits deep in the line, so + /// this is the horizontal-scroll path; elsewhere it is a plain step. + key_right, + /// PageDown + page_down, + /// one mouse wheel tick + wheel, + /// one printable typed in insert mode + edit_char, + + fn label(o: Op) []const u8 { + return switch (o) { + .open => "open", + .render => "render", + .key_down => "key-down", + .key_right => "key-right", + .page_down => "page-down", + .wheel => "wheel", + .edit_char => "edit-char", + }; + } +}; + +const Cell = struct { + min_us: u64 = 0, + med_us: u64 = 0, + p90_us: u64 = 0, + max_us: u64 = 0, +}; + +pub fn main(init: std.process.Init) !void { + const args = try init.minimal.args.toSlice(init.arena.allocator()); + var json = false; + var reps: usize = 25; + var base_path: ?[]const u8 = null; + var i: usize = 1; + while (i < args.len) : (i += 1) { + const a = args[i]; + if (std.mem.eql(u8, a, "--json")) { + json = true; + } else if (std.mem.eql(u8, a, "--reps") and i + 1 < args.len) { + i += 1; + reps = std.fmt.parseInt(usize, args[i], 10) catch reps; + } else if (std.mem.eql(u8, a, "--base") and i + 1 < args.len) { + i += 1; + base_path = args[i]; + } else fatal("usage: pardes-perf [--json] [--reps N] [--base old.json]", .{}); + } + + // fixtures live in a temp dir and are rewritten every run: they are inputs + // to a measurement, and a stale one silently changes what the table means. + // .zig extensions on purpose — that is what puts tree-sitter in the frame. + const dir = "/tmp/pardes-perf"; + _ = libc.mkdir(dir, 0o755); // EEXIST is fine + var paths: [fixtures.len][]const u8 = undefined; + var bytes: [fixtures.len]usize = undefined; + for (fixtures, 0..) |fx, fi| { + const path = try std.fmt.allocPrint(gpa, "{s}/{s}.zig", .{ dir, fx.name }); + const text = try generate(fx); + try writeFile(path, text); + paths[fi] = path; + // what got WRITTEN, not lines*cols: a line whose code is already wider + // than `cols` is left alone rather than truncated, so the nominal + // product would understate the file the editor actually opens + bytes[fi] = text.len; + gpa.free(text); + } + + var cells: [std.enums.values(Op).len][fixtures.len]Cell = undefined; + for (std.enums.values(Op), 0..) |op, oi| { + for (fixtures, 0..) |fx, fi| { + cells[oi][fi] = try measure(op, fx, paths[fi], reps); + } + } + + if (json) return reportJson(init.io, &cells, &bytes, reps); + reportText(&cells, &bytes, reps, base_path); +} + +// ---- the measured gestures ---- + +/// One (op, fixture) cell: `reps` timed samples plus three warmup rounds, +/// reported as min / median / p90 / max. The spread between median and p90 is +/// the run's noise, and reportText prints the worst one so a reader knows how +/// big a difference has to be before it is real. +fn measure(op: Op, fx: Fixture, path: []const u8, reps: usize) !Cell { + const warmup = 3; + var samples: std.ArrayList(u64) = .empty; + defer samples.deinit(gpa); + + if (op == .open) { + // fresh core per sample: opening is a one-shot, and the second Look at + // the same path only refocuses the pane already holding it. + for (0..warmup + reps) |n| { + const core = try boot(); + defer core.deinit(); + const t0 = nowNs(); + core.lookAt(0, path); + pump(core); + _ = try frame(core); + const dt = nowNs() -| t0; + if (n >= warmup) try samples.append(gpa, dt / 1000); + } + return summarize(samples.items); + } + + const core = try boot(); + defer core.deinit(); + core.lookAt(0, path); + pump(core); + const id = filePane(core) orelse fatal("Look {s} opened no file pane", .{path}); + core.active = id; + const pane = core.panes[id].?; + _ = try frame(core); + // insert mode is entered ONCE: `i` is a mode change, not a keystroke of + // typing, and measuring it inside every edit sample would report the wrong + // thing entirely. + if (op == .edit_char) { + core.update(.{ .key = .{ .cp = 'i', .text = "i" } }); + pump(core); + } + + for (0..warmup + reps) |n| { + seek(pane, fx, n); + _ = try frame(core); // settle the view at the new spot, untimed + const t0 = nowNs(); + switch (op) { + .open => unreachable, + .render => {}, + .key_down => core.update(.{ .key = .{ .cp = 'j', .text = "j" } }), + .key_right => core.update(.{ .key = .{ .cp = 'l', .text = "l" } }), + .page_down => core.update(.{ .key = .{ .cp = pardes.Key.page_down } }), + .wheel => core.update(.{ .mouse = .{ .button = .wheel_down, .kind = .press, .col = 4, .row = 8 } }), + .edit_char => core.update(.{ .key = .{ .cp = 'x', .text = "x" } }), + } + pump(core); + _ = try frame(core); + const dt = nowNs() -| t0; + if (n >= warmup) try samples.append(gpa, dt / 1000); + } + return summarize(samples.items); +} + +/// Park the cursor and the view at sample `n`'s position, walked across the +/// file by a prime stride so consecutive samples land nowhere near each other +/// and a whole run covers the document rather than one neighbourhood of it. +/// Rows sit in the middle 80% (a position at the very top or bottom measures +/// the clamp, not the work), and on a wide fixture the column is deep inside +/// the line so `key-right` exercises the hscroll cut rather than column 1. +/// The sequence depends only on `n`, so two builds see the same positions in +/// the same order and their cells are comparable one for one. +fn seek(pane: *pardes.Pane, fx: Fixture, n: usize) void { + const f = &pane.file.?; + const span = @max(1, fx.lines * 8 / 10); + const row = fx.lines / 10 + (n *% 7919) % span; + f.scroll = @min(row, fx.lines -| 1); + f.syntax_dirty = true; + pane.cur_row = @intCast(f.scroll); + pane.cur_col = if (fx.cols > 200) @intCast(fx.cols * 3 / 4) else 0; + pane.cur_pinned = true; + pane.hscroll = @max(0, pane.cur_col - 40); + pane.sticky_col = -1; +} + +fn boot() !*pardes.Pardes { + const core = try pardes.Pardes.init(gpa, .{ .tty_only = true }); + core.update(.{ .resize = .{ .cols = screen_cols, .rows = screen_rows } }); + pump(core); + return core; +} + +/// one frame into a throwaway arena — the shell's per-frame arena, which is +/// what keeps the retained-render contract honest (vaxis stores slices into +/// whatever it was handed, so the frame's text must outlive vx.render) +fn frame(core: *pardes.Pardes) !*pardes.Surface { + var arena: std.heap.ArenaAllocator = .init(gpa); + defer arena.deinit(); + return core.render(arena.allocator()); +} + +fn filePane(core: *pardes.Pardes) ?usize { + for (core.panes, 0..) |slot, i| { + const pane = slot orelse continue; + if (pane.file != null) return i; + } + return null; +} + +/// Drain queued effects, all ignored (spawn, write, watch, ...): there is no +/// shell here, and nothing measured depends on one answering. +fn pump(core: *pardes.Pardes) void { + while (core.nextEffect()) |_| {} +} + +fn summarize(samples: []u64) Cell { + if (samples.len == 0) return .{}; + std.mem.sort(u64, samples, {}, std.sort.asc(u64)); + return .{ + .min_us = samples[0], + .med_us = samples[samples.len / 2], + .p90_us = samples[(samples.len * 9) / 10 -| 1], + .max_us = samples[samples.len - 1], + }; +} + +// ---- the fixture generator ---- + +/// Plausible Zig, so the tree-sitter pass has real nodes to walk rather than +/// one giant error node: a repeating four-line shape padded to `cols`. Long +/// lines get their width from a comment tail — the alternative (an enormous +/// string literal) makes the whole file one token and flatters every scan that +/// looks for a newline. +fn generate(fx: Fixture) ![]u8 { + var out: std.ArrayList(u8) = .empty; + try out.ensureTotalCapacity(gpa, fx.lines * (fx.cols + 1) + 64); + var n: usize = 0; + while (n < fx.lines) : (n += 1) { + const start = out.items.len; + switch (n % 4) { + 0 => try out.print(gpa, "const value_{d}: u32 = {d}; // ", .{ n, n *% 2654435761 }), + 1 => try out.print(gpa, "pub fn helper_{d}(a: u32, b: u32) u32 {{ return a +% b *% {d}; }} // ", .{ n, n }), + 2 => try out.print(gpa, " const text_{d} = \"lorem ipsum dolor sit amet {d}\"; // ", .{ n, n }), + else => try out.print(gpa, "// comment line {d} — ", .{n}), + } + while (out.items.len - start < fx.cols) try out.append(gpa, 'x'); + try out.append(gpa, '\n'); + } + return out.toOwnedSlice(gpa); +} + +fn writeFile(path: []const u8, text: []const u8) !void { + var pathbuf: [4096]u8 = undefined; + const path_z = try std.fmt.bufPrintSentinel(&pathbuf, "{s}", .{path}, 0); + const fd = libc.open(path_z, .{ .ACCMODE = .WRONLY, .CREAT = true, .TRUNC = true }, @as(libc.mode_t, 0o644)); + if (fd < 0) return error.OpenFailed; + defer _ = libc.close(fd); + var off: usize = 0; + while (off < text.len) { + const n = libc.write(fd, text.ptr + off, text.len - off); + if (n < 0) { + if (libc.errno(n) == .INTR) continue; + return error.WriteFailed; + } + off += @intCast(n); + } +} + +// ---- reporting ---- + +fn reportText(cells: *const [std.enums.values(Op).len][fixtures.len]Cell, bytes: *const [fixtures.len]usize, reps: usize, base_path: ?[]const u8) void { + const o = std.debug.print; + const base = if (base_path) |bp| readBase(bp) else null; + + o("pardes perf — {d}x{d} viewport, {d} samples/cell, median us\n\n", .{ screen_cols, screen_rows, reps }); + o("{s:<12}", .{"fixture"}); + for (fixtures) |fx| o(" {s:>12}", .{fx.name}); + o("\n{s:<12}", .{"lines"}); + for (fixtures) |fx| o(" {d:>12}", .{fx.lines}); + o("\n{s:<12}", .{"cols"}); + for (fixtures) |fx| o(" {d:>12}", .{fx.cols}); + o("\n{s:<12}", .{"size"}); + for (bytes) |b| o(" {d:>10} KB", .{b / 1024}); + o("\n\n", .{}); + + // one row per gesture. `open` is a whole Look; every other row is + // update+effects+one frame, which is the latency a user actually sees. + o("{s:<12}", .{"op"}); + for (fixtures) |fx| { + if (base != null) o(" {s:>19}", .{fx.name}) else o(" {s:>12}", .{fx.name}); + } + // 12 for the op name + one column per fixture, wider when a baseline adds + // its ratio to each cell + o("\n{s}\n", .{if (base != null) "-" ** (12 + fixtures.len * 20) else "-" ** (12 + fixtures.len * 13)}); + var worst_jitter: f64 = 0; + for (std.enums.values(Op), 0..) |op, oi| { + o("{s:<12}", .{op.label()}); + for (0..fixtures.len) |fi| { + const c = cells[oi][fi]; + if (c.med_us > 0) { + const j = (@as(f64, @floatFromInt(c.p90_us)) - @as(f64, @floatFromInt(c.med_us))) / @as(f64, @floatFromInt(c.med_us)); + if (j > worst_jitter) worst_jitter = j; + } + if (base) |b| { + const prev = b.med[oi][fi]; + if (prev == 0) { + o(" {d:>12}{s:>7}", .{ c.med_us, "-" }); + } else { + const ratio = @as(f64, @floatFromInt(c.med_us)) / @as(f64, @floatFromInt(prev)); + o(" {d:>12} {d:>6.2}x", .{ c.med_us, ratio }); + } + } else o(" {d:>12}", .{c.med_us}); + } + o("\n", .{}); + } + // The spread is not only scheduler noise: samples are taken at DIFFERENT + // places in the file on purpose (see seek), so a cost that still depends on + // where the cursor is shows up here as well. Both are reasons not to + // believe a small difference, and the sample positions are identical from + // run to run, so two builds are still comparable cell for cell. + o("\nnoise: worst cell p90 is {d:.0}% over its median (scheduler + the spread\n", .{worst_jitter * 100}); + o("of sample positions through the file). Treat a difference smaller than\n", .{}); + o("that as nothing, and re-run before believing a small win.\n", .{}); + if (base_path) |bp| o("baseline: {s} (x column = now / then; under 1.00 is faster)\n", .{bp}); +} + +/// Real stdout, not std.debug.print's stderr: this is the form `--base` reads +/// back, and `zig build perf -- --json > runs/old.json` writing an empty file +/// would make the next comparison silently print no ratios at all. +fn reportJson(io: std.Io, cells: *const [std.enums.values(Op).len][fixtures.len]Cell, bytes: *const [fixtures.len]usize, reps: usize) void { + var out: std.ArrayList(u8) = .empty; + defer out.deinit(gpa); + out.print(gpa, "{{\"cols\":{d},\"rows\":{d},\"reps\":{d},\"fixtures\":[", .{ screen_cols, screen_rows, reps }) catch return; + for (fixtures, 0..) |fx, fi| { + out.print(gpa, "{s}{{\"name\":\"{s}\",\"lines\":{d},\"cols\":{d},\"bytes\":{d}}}", .{ + if (fi > 0) "," else "", fx.name, fx.lines, fx.cols, bytes[fi], + }) catch return; + } + out.appendSlice(gpa, "],\"cells\":[") catch return; + var first = true; + for (std.enums.values(Op), 0..) |op, oi| { + for (fixtures, 0..) |fx, fi| { + const c = cells[oi][fi]; + out.print(gpa, "{s}{{\"op\":\"{s}\",\"fixture\":\"{s}\",\"min_us\":{d},\"med_us\":{d},\"p90_us\":{d},\"max_us\":{d}}}", .{ + if (first) "" else ",", op.label(), fx.name, c.min_us, c.med_us, c.p90_us, c.max_us, + }) catch return; + first = false; + } + } + out.appendSlice(gpa, "]}\n") catch return; + std.Io.File.stdout().writeStreamingAll(io, out.items) catch {}; +} + +const Base = struct { med: [std.enums.values(Op).len][fixtures.len]u64 }; + +/// A previous --json run, reduced to the medians this table compares against. +/// Cells the old run did not have stay 0 and print as "-": the op list may +/// have grown since, and a missing number is not a regression. An unreadable +/// file is fatal rather than silently ratio-less — a comparison you asked for +/// and did not get is worse than no comparison. +fn readBase(path: []const u8) ?Base { + const src = readFileAlloc(path) catch fatal("--base: cannot read {s}", .{path}); + var b: Base = .{ .med = @splat(@splat(0)) }; + const parsed = std.json.parseFromSlice(struct { + cells: []const struct { + op: []const u8, + fixture: []const u8, + med_us: u64, + }, + }, gpa, src, .{ .ignore_unknown_fields = true }) catch return null; + defer parsed.deinit(); + for (parsed.value.cells) |c| { + for (std.enums.values(Op), 0..) |op, oi| { + if (!std.mem.eql(u8, op.label(), c.op)) continue; + for (fixtures, 0..) |fx, fi| { + if (std.mem.eql(u8, fx.name, c.fixture)) b.med[oi][fi] = c.med_us; + } + } + } + return b; +} + +fn readFileAlloc(path: []const u8) ![]u8 { + var pathbuf: [4096]u8 = undefined; + const path_z = try std.fmt.bufPrintSentinel(&pathbuf, "{s}", .{path}, 0); + const fd = libc.open(path_z, .{ .ACCMODE = .RDONLY }); + if (fd < 0) return error.OpenFailed; + defer _ = libc.close(fd); + var buf: std.ArrayList(u8) = .empty; + var chunk: [16384]u8 = undefined; + while (true) { + const n = libc.read(fd, &chunk, chunk.len); + if (n < 0) { + if (libc.errno(n) == .INTR) continue; + return error.ReadFailed; + } + if (n == 0) break; + try buf.appendSlice(gpa, chunk[0..@intCast(n)]); + } + return buf.items; +} + +fn fatal(comptime fmt: []const u8, args: anytype) noreturn { + std.debug.print("pardes-perf: " ++ fmt ++ "\n", args); + std.process.exit(1); +} |
