From be2a9957708cbf0c478ca861c4a1f0f227bbfe10 Mon Sep 17 00:00:00 2001 From: Gabriel Schneider Date: Fri, 14 Aug 2026 22:38:50 -0300 Subject: pdf: continuous scroll bench harness and per-frame render path --- test/pdf_scroll_bench.zig | 156 +++++++++++++++++++++++++++++++++++++--------- 1 file changed, 125 insertions(+), 31 deletions(-) (limited to 'test') diff --git a/test/pdf_scroll_bench.zig b/test/pdf_scroll_bench.zig index 7bd644e4..31aab270 100644 --- a/test/pdf_scroll_bench.zig +++ b/test/pdf_scroll_bench.zig @@ -1,9 +1,17 @@ -//! ReleaseFast scoreboard for FAST scrolling of the continuous PDF strip. +//! Scoreboard for FAST scrolling of the continuous PDF strip. //! -//! zig build pdf-scroll-bench -Doptimize=ReleaseFast +//! zig build pdf-scroll-bench -Doptimize=ReleaseFast the numbers //! zig build pdf-scroll-bench -Doptimize=ReleaseFast -- --json //! zig build pdf-scroll-bench -Doptimize=ReleaseFast -- --warmup 1 --reps 15 //! zig build pdf-scroll-bench -Doptimize=ReleaseFast -- --path other.pdf +//! zig build pdf-scroll-bench -Doptimize=Debug -- --reps 1 the diagnosis +//! zig build pdf-scroll-bench -Dtracy=~/05-genizah/tracy the zones +//! +//! Every optimize mode runs. A profile or a crash wants safety checks, line +//! numbers and un-inlined frames; only the SCOREBOARD wants ReleaseFast, so the +//! mode is stamped in both output formats and shouted about on stderr rather +//! than being refused — a benchmark you cannot run under a debugger is a +//! benchmark whose regressions you cannot explain. //! //! What a fling IS, and why one-tick-per-frame benchmarks miss it: a shell //! pump drains every input the OS queued since the last present and then calls @@ -75,6 +83,9 @@ const Config = struct { json: bool = false, warmup: usize = 2, reps: usize = 25, + /// Override `max_frames` for one run: a 5000-page manual is 2000 frames of + /// travel per round, which is a fine scoreboard and a poor edit-run loop. + frames: ?usize = null, }; /// The bit-exact description of one scrolling run. `visual` is the invariant; @@ -147,8 +158,11 @@ const Plan = struct { }; pub fn main(init: std.process.Init) !void { - if (!bench_config.release_fast_core) - fatal("the production core must be ReleaseFast; add -Doptimize=ReleaseFast", .{}); + if (!bench_config.release_fast_core) std.debug.print( + "WARNING: the core is {s}, not ReleaseFast — these timings are for" ++ + " reading a profile, NOT for the scoreboard\n", + .{bench_config.core_optimize}, + ); const args = try init.minimal.args.toSlice(init.arena.allocator()); const config = parseArgs(args); @@ -157,16 +171,17 @@ pub fn main(init: std.process.Init) !void { // viewport rather than of the machine. const travel = probeTravel(config.path); + const cap = config.frames orelse max_frames; const results = [_]Result{ - try measure(config, planFor(.fling_down, travel)), - try measure(config, planFor(.fling_reverse, travel)), - try measure(config, planFor(.slow_scroll, travel)), + try measure(config, planFor(.fling_down, travel, cap)), + try measure(config, planFor(.fling_reverse, travel, cap)), + try measure(config, planFor(.slow_scroll, travel, cap)), }; if (config.json) - reportJson(init.io, config, &results) + reportJson(init.io, config, travel, &results) else - reportText(config, &results); + reportText(config, travel, &results); } fn parseArgs(args: []const []const u8) Config { @@ -184,8 +199,12 @@ fn parseArgs(args: []const []const u8) Config { } else if (std.mem.eql(u8, args[i], "--path") and i + 1 < args.len) { i += 1; config.path = args[i]; + } else if (std.mem.eql(u8, args[i], "--frames") and i + 1 < args.len) { + i += 1; + config.frames = parseCount("--frames", args[i]); } else fatal( - "usage: pardes-pdf-scroll-bench [--json] [--warmup N] [--reps N] [--path FILE]", + "usage: pardes-pdf-scroll-bench [--json] [--warmup N] [--reps N]" ++ + " [--path FILE] [--frames N]", .{}, ); } @@ -205,6 +224,11 @@ const Travel = struct { max_scroll: f64, /// Display pixels one fling frame's notch batch covers. per_frame: f64, + /// What the document turned out to be, so a row is readable next to a row + /// from another file: a 7-page paper and a 5000-page manual are not the + /// same benchmark and the report must not pretend otherwise. + pages: usize, + strip_px: u64, }; fn probeTravel(path: []const u8) Travel { @@ -227,20 +251,40 @@ fn probeTravel(path: []const u8) Travel { // lands on the document end instead of adding a no-op frame to the row. const viewport_h: u64 = @as(u64, bench_rows -| pardes.BOX_H) * bench_cell_pixels.h; const notch: f64 = @floatFromInt(bench_cell_pixels.h); + if (!std.math.isFinite(notch) or notch <= 0) + fatal("a notch of {d} display pixels cannot size a fling", .{notch}); + const strip = @as(f64, @floatFromInt(pv.document_height -| viewport_h)); + if (!std.math.isFinite(strip)) + fatal("a {d}-page strip does not fit a float; refusing to invent a plan", .{pv.page_count}); return .{ - .max_scroll = @max(notch, @as(f64, @floatFromInt(pv.document_height -| viewport_h))), + .max_scroll = @max(notch, strip), .per_frame = notch * @as(f64, @floatFromInt(fling_ticks_per_frame)), + .pages = pv.page_count, + .strip_px = pv.document_height, }; } -fn planFor(scenario: Scenario, travel: Travel) Plan { - const down = @min( - max_frames, - @as(usize, @intFromFloat(@ceil(travel.max_scroll / travel.per_frame))), - ); +/// Frames are derived from the document, then capped — `max_frames` is a +/// property of the BENCHMARK (a row must not become a multi-minute run on a +/// 5000-page manual), and it caps the whole plan, not merely its downward half. +/// The cap is applied before any arithmetic on the count, so a document the +/// float division cannot describe still produces a runnable plan instead of an +/// overflow. +fn planFor(scenario: Scenario, travel: Travel, cap: usize) Plan { + const wanted = travel.max_scroll / travel.per_frame; + const whole: usize = if (std.math.isFinite(wanted) and wanted >= 1) + @intFromFloat(@min(@as(f64, @floatFromInt(cap)), @ceil(wanted))) + else + 1; + const down = @min(cap, whole); return switch (scenario) { .fling_down => .{ .scenario = scenario, .frames = down, .down_frames = down }, - .fling_reverse => .{ .scenario = scenario, .frames = down * 2, .down_frames = down }, + // half down, half back: the reverse row keeps the plan's frame budget + .fling_reverse => .{ + .scenario = scenario, + .frames = @max(2, down), + .down_frames = @max(1, down / 2), + }, .slow_scroll => .{ .scenario = scenario, .frames = slow_frames, .down_frames = slow_frames }, }; } @@ -314,6 +358,9 @@ fn measure(config: Config, plan: Plan) !Result { const elapsed = nowNs() -| started; total += elapsed; recorder.observe(frame, surface); + // The same frame boundary src/tty/tty.zig marks, so a Tracy capture + // of this benchmark reads like a capture of the real shell. + pardes.frameMark(); if (round >= config.warmup) { pool[pool_len] = @max(1, elapsed); pool_len += 1; @@ -369,6 +416,14 @@ const Recorder = struct { }; } + /// The `visual` half hashes WHAT THE READER SEES and nothing else: the + /// destination rectangle, and the pixels inside the source rectangle the + /// backend samples — walked row by row out of the texture, so a page + /// delivered as a narrow band and the same page delivered whole hash the + /// same. Texture dimensions, buffer lengths and revisions are transport, + /// and they live in the other checksum on purpose: fast scrolling is + /// ALLOWED to change how pixels are carried and is never allowed to change + /// which pixels arrive. fn observe(self: *Recorder, frame: usize, surface: *const pardes.Surface) void { self.frames += 1; self.visual = mix(self.visual, surface.nimages); @@ -385,31 +440,59 @@ const Recorder = struct { self.visual = mix(self.visual, place.y); self.visual = mix(self.visual, place.w); self.visual = mix(self.visual, place.h); - self.visual = mix(self.visual, place.iw); - self.visual = mix(self.visual, place.ih); - self.visual = mix(self.visual, place.rgba.len); self.visual = mix(self.visual, place.native.page); self.visual = mix(self.visual, @intFromEnum(place.native.fit)); self.visual = mix(self.visual, place.native.pan_x); self.visual = mix(self.visual, place.native.pan_y); self.visual = mix(self.visual, @as(u32, @bitCast(place.native.pixel_offset_y))); if (place.native.geometry) |geometry| { - inline for (.{ geometry.src, geometry.dst }) |rect| { + inline for (.{geometry.dst}) |rect| { self.visual = mix(self.visual, rect.x); self.visual = mix(self.visual, rect.y); self.visual = mix(self.visual, rect.w); self.visual = mix(self.visual, rect.h); } - } else self.visual = mix(self.visual, std.math.maxInt(u64)); - self.visual = if (full) - std.hash.Wyhash.hash(self.visual, place.rgba) - else - stridedHash(self.visual, place.rgba); + self.visual = self.hashSource(self.visual, place, geometry.src, full); + } else { + self.visual = mix(self.visual, std.math.maxInt(u64)); + self.visual = if (full) + std.hash.Wyhash.hash(self.visual, place.rgba) + else + stridedHash(self.visual, place.rgba); + } self.transport = mix(self.transport, place.native.page); + self.transport = mix(self.transport, place.iw); + self.transport = mix(self.transport, place.ih); + self.transport = mix(self.transport, place.rgba.len); self.transport = mix(self.transport, place.native.revision -% self.base_revision); } } + /// Hash the source rectangle's pixels out of a texture of `place.iw` pixels + /// per row. `full` walks every row; otherwise the same stride the whole- + /// buffer hash used, so the cost stays a fraction of a frame. + fn hashSource( + self: *Recorder, + seed: u64, + place: pardes.ImagePlace, + src: anytype, + full: bool, + ) u64 { + _ = self; + const stride = place.iw * 4; + if (stride == 0 or place.rgba.len < stride) return mix(seed, std.math.maxInt(u64)); + const rows = @min(src.h, @as(u32, @intCast(place.rgba.len / stride)) -| src.y); + var hash = mix(seed, rows); + var row: u32 = 0; + while (row < rows) : (row += if (full) 1 else 7) { + const start = (@as(usize, src.y) + row) * stride + @as(usize, src.x) * 4; + const len = @min(@as(usize, src.w) * 4, place.rgba.len -| start); + if (len == 0) break; + hash = std.hash.Wyhash.hash(hash, place.rgba[start..][0..len]); + } + return hash; + } + fn finish(self: *Recorder, pv: anytype) Identity { var visual = mix(self.visual, @as(u64, @bitCast(pv.document_scroll_y))); visual = mix(visual, pv.page); @@ -481,8 +564,15 @@ fn nowNs() u64 { @as(u64, @intCast(ts.nsec)); } -fn reportText(config: Config, results: []const Result) void { - std.debug.print("pardes PDF fast-scroll benchmark (ReleaseFast)\n", .{}); +fn reportText(config: Config, travel: Travel, results: []const Result) void { + std.debug.print("pardes PDF fast-scroll benchmark ({s})\n", .{bench_config.core_optimize}); + std.debug.print( + "{s}: {d} pages, {d}px strip at {d}x{d} cells of {d}x{d}px\n", + .{ + config.path, travel.pages, travel.strip_px, bench_cols, + bench_rows, bench_cell_pixels.w, bench_cell_pixels.h, + }, + ); std.debug.print( "warmup: {d}, sampled reps: {d}; every number is ONE FRAME (notch batch + render + drain)\n\n", .{ config.warmup, config.reps }, @@ -521,12 +611,16 @@ fn reportText(config: Config, results: []const Result) void { const json_report_max_bytes = 16 * 1024; -fn reportJson(io: std.Io, config: Config, results: []const Result) void { +fn reportJson(io: std.Io, config: Config, travel: Travel, results: []const Result) void { var storage: [json_report_max_bytes]u8 = undefined; var out: std.Io.Writer = .fixed(&storage); out.print( - "{{\"benchmark\":\"pardes-pdf-scroll\",\"build\":\"ReleaseFast\",\"warmup\":{d},\"reps\":{d},\"results\":[", - .{ config.warmup, config.reps }, + "{{\"benchmark\":\"pardes-pdf-scroll\",\"build\":\"{s}\",\"path\":\"{s}\"," ++ + "\"pages\":{d},\"strip_px\":{d},\"warmup\":{d},\"reps\":{d},\"results\":[", + .{ + bench_config.core_optimize, config.path, travel.pages, + travel.strip_px, config.warmup, config.reps, + }, ) catch return; for (results, 0..) |result, i| out.print( "{s}{{\"scenario\":\"{s}\",\"frames\":{d},\"notches_per_frame\":{d}," ++ -- cgit v1.3