diff options
Diffstat (limited to 'test/pdf_scroll_bench.zig')
| -rw-r--r-- | test/pdf_scroll_bench.zig | 567 |
1 files changed, 567 insertions, 0 deletions
diff --git a/test/pdf_scroll_bench.zig b/test/pdf_scroll_bench.zig new file mode 100644 index 00000000..7bd644e4 --- /dev/null +++ b/test/pdf_scroll_bench.zig @@ -0,0 +1,567 @@ +//! ReleaseFast scoreboard for FAST scrolling of the continuous PDF strip. +//! +//! zig build pdf-scroll-bench -Doptimize=ReleaseFast +//! zig build pdf-scroll-bench -Doptimize=ReleaseFast -- --json +//! zig build pdf-scroll-bench -Doptimize=ReleaseFast -- --warmup 1 --reps 15 +//! zig build pdf-scroll-bench -Doptimize=ReleaseFast -- --path other.pdf +//! +//! What a fling IS, and why one-tick-per-frame benchmarks miss it: a shell +//! pump drains every input the OS queued since the last present and then calls +//! render() once. A trackpad fling therefore arrives as a BATCH of wheel +//! notches inside one frame, and the frame budget is spent on the batch plus +//! the single render — not on one notch. Every timed region below is exactly +//! that: `ticks_per_frame` real `Mouse.wheel_down`/`wheel_up` presses through +//! `Pardes.update`, one `render()` into a reset frame arena, and the effect +//! drain, which is the loop src/tty/tty.zig and src/gui/gui.zig both run. +//! +//! Rows are per-FRAME latency, not per-notch: a dropped frame is what a reader +//! feels, and the batch is indivisible from the reader's point of view. +//! +//! Two checksums, because they answer two different questions: +//! +//! `visual` is the no-regression proof. Frame count, emitted image count, +//! every placement rectangle, the presented RGBA (strided every +//! frame, in full every `full_rgba_stride`-th frame), and the +//! final scroll offset and page. Any optimization that changes +//! ONE presented pixel or ONE placement changes this. It must be +//! bit-identical across builds. +//! `transport` is the texture-upload trace: the (page, raster generation) +//! pair per emitted image, with generations rebased on the +//! counter's value at run start. Caching work legitimately +//! changes this while `visual` stays fixed — reusing a retained +//! raster keeps its old generation, which is precisely the +//! retransmit the revision mechanism exists to suppress. +//! +//! Setup, teardown, checksumming, and reporting are outside every timer. +const std = @import("std"); +const libc = std.c; +const pardes = @import("pardes"); +const pdf = @import("mupdf"); +const bench_config = @import("pdf_scroll_bench_config"); + +pub const std_options: std.Options = .{ .log_level = .err }; + +const gpa = std.heap.smp_allocator; + +// A reading-sized window. Wide enough that fit-width pages are taller than the +// viewport (so a fling genuinely crosses page boundaries) and small enough +// that the Kitty raster policy's 1200px cap is the binding quality term, which +// is what the tty shell actually ships. +const bench_cols: u16 = 120; +const bench_rows: u16 = 40; +const bench_cell_pixels: pardes.CellPixels = .{ .w = 8, .h = 16 }; + +/// config.wheel_rows is one cell per notch, so a notch is `cell_pixels.h` +/// display pixels. 192 notches is ~3072px — between two and three fit-width +/// pages of travel inside a single frame, i.e. a hard flick whose momentum the +/// OS delivers as one burst. This is the number that makes the benchmark about +/// flinging rather than about nudging. +const fling_ticks_per_frame: usize = 192; +const slow_ticks_per_frame: usize = 1; +/// The control row is a fixed wall of frames rather than a whole-document +/// traversal: one notch per frame over a real document is tens of thousands of +/// frames, and 240 already crosses several page boundaries at reading speed. +const slow_frames: usize = 240; +/// Guard against a pathological document turning one row into a multi-minute +/// run. Deterministic either way: the frame count is derived from layout. +const max_frames: usize = 2048; +/// Full-RGBA hashing is O(document) per frame, so it samples the run instead of +/// every frame; the strided hash below runs on every frame and every image. +const full_rgba_stride: usize = 8; +const checksum_seed: u64 = 0x7061_7264_6573_5343; + +const Config = struct { + path: []const u8 = "docs/design.pdf", + json: bool = false, + warmup: usize = 2, + reps: usize = 25, +}; + +/// The bit-exact description of one scrolling run. `visual` is the invariant; +/// see the file header for why `transport` is reported beside it rather than +/// folded into it. +const Identity = struct { + frames: u64, + images: u64, + visual: u64, + transport: u64, + + fn eql(a: Identity, b: Identity) bool { + return a.frames == b.frames and a.images == b.images and + a.visual == b.visual and a.transport == b.transport; + } +}; + +const Result = struct { + name: []const u8, + ticks_per_frame: usize, + frames: usize, + min_ns: u64, + median_ns: u64, + p90_ns: u64, + p99_ns: u64, + max_ns: u64, + total_ns: u64, + rasterizations: u64, + rasters_retained: u64, + identity: Identity, +}; + +const Scenario = enum { + /// Sustained high-velocity travel from page zero to the last page. + fling_down, + /// Down to the end and straight back up. Worst case for any cache that + /// only retains what the current frame shows: the way back finds nothing. + fling_reverse, + /// One notch per frame. The control: it must not regress. + slow_scroll, + + fn label(scenario: Scenario) []const u8 { + return switch (scenario) { + .fling_down => "fling_down", + .fling_reverse => "fling_reverse", + .slow_scroll => "slow_scroll", + }; + } + + fn ticks(scenario: Scenario) usize { + return switch (scenario) { + .fling_down, .fling_reverse => fling_ticks_per_frame, + .slow_scroll => slow_ticks_per_frame, + }; + } +}; + +const Plan = struct { + scenario: Scenario, + frames: usize, + /// Frames of downward travel before `fling_reverse` turns around. + down_frames: usize, + + fn button(plan: Plan, frame: usize) pardes.Mouse.Button { + return switch (plan.scenario) { + .fling_reverse => if (frame < plan.down_frames) .wheel_down else .wheel_up, + .fling_down, .slow_scroll => .wheel_down, + }; + } +}; + +pub fn main(init: std.process.Init) !void { + if (!bench_config.release_fast_core) + fatal("the production core must be ReleaseFast; add -Doptimize=ReleaseFast", .{}); + const args = try init.minimal.args.toSlice(init.arena.allocator()); + const config = parseArgs(args); + + // One untimed probe boot supplies the strip height every plan is sized + // from, so the frame counts below are a property of the document and the + // viewport rather than of the machine. + const travel = probeTravel(config.path); + + const results = [_]Result{ + try measure(config, planFor(.fling_down, travel)), + try measure(config, planFor(.fling_reverse, travel)), + try measure(config, planFor(.slow_scroll, travel)), + }; + + if (config.json) + reportJson(init.io, config, &results) + else + reportText(config, &results); +} + +fn parseArgs(args: []const []const u8) Config { + var config: Config = .{}; + var i: usize = 1; + while (i < args.len) : (i += 1) { + if (std.mem.eql(u8, args[i], "--json")) { + config.json = true; + } else if (std.mem.eql(u8, args[i], "--warmup") and i + 1 < args.len) { + i += 1; + config.warmup = parseCount("--warmup", args[i]); + } else if (std.mem.eql(u8, args[i], "--reps") and i + 1 < args.len) { + i += 1; + config.reps = parseCount("--reps", args[i]); + } else if (std.mem.eql(u8, args[i], "--path") and i + 1 < args.len) { + i += 1; + config.path = args[i]; + } else fatal( + "usage: pardes-pdf-scroll-bench [--json] [--warmup N] [--reps N] [--path FILE]", + .{}, + ); + } + return config; +} + +fn parseCount(flag: []const u8, text: []const u8) usize { + const value = std.fmt.parseUnsigned(usize, text, 10) catch + fatal("{s} expects a positive integer", .{flag}); + if (value == 0 or value > 10_000) + fatal("{s} must be in 1..10000", .{flag}); + return value; +} + +const Travel = struct { + /// Scrollable display pixels: the whole strip minus one screenful. + max_scroll: f64, + /// Display pixels one fling frame's notch batch covers. + per_frame: f64, +}; + +fn probeTravel(path: []const u8) Travel { + const core = bootCore(path) catch |err| + fatal("cannot boot a PDF pane on {s}: {t}", .{ path, err }); + defer core.deinit(); + var frame_arena = std.heap.ArenaAllocator.init(gpa); + defer frame_arena.deinit(); + _ = core.render(frame_arena.allocator()) catch |err| + fatal("first render failed: {t}", .{err}); + drainEffects(core); + const pane = core.panes[0] orelse fatal("PDF pane disappeared", .{}); + if (pane.pdf == null) fatal("{s} did not open as a PDF", .{path}); + const pv = &pane.pdf.?; + if (pv.page_count < 4) + fatal("{s} has {d} pages; a fling benchmark needs at least 4", .{ path, pv.page_count }); + if (!pv.layout_valid or pv.document_height == 0) + fatal("the pixel strip never laid out; is native_images/cell_pixels wired?", .{}); + // Exactly the clamp scrollPdfDocument enforces, so the last fling frame + // lands on the document end instead of adding a no-op frame to the row. + const viewport_h: u64 = @as(u64, bench_rows -| pardes.BOX_H) * bench_cell_pixels.h; + const notch: f64 = @floatFromInt(bench_cell_pixels.h); + return .{ + .max_scroll = @max(notch, @as(f64, @floatFromInt(pv.document_height -| viewport_h))), + .per_frame = notch * @as(f64, @floatFromInt(fling_ticks_per_frame)), + }; +} + +fn planFor(scenario: Scenario, travel: Travel) Plan { + const down = @min( + max_frames, + @as(usize, @intFromFloat(@ceil(travel.max_scroll / travel.per_frame))), + ); + return switch (scenario) { + .fling_down => .{ .scenario = scenario, .frames = down, .down_frames = down }, + .fling_reverse => .{ .scenario = scenario, .frames = down * 2, .down_frames = down }, + .slow_scroll => .{ .scenario = scenario, .frames = slow_frames, .down_frames = slow_frames }, + }; +} + +/// A pane on a real multi-page document with the native pixel-strip path live +/// and a genuine pixel viewport delivered the way a shell delivers one. +fn bootCore(path: []const u8) !*pardes.Pardes { + const core = try pardes.Pardes.init(gpa, .{ + .file = path, + .cols = bench_cols, + .rows = bench_rows, + }); + errdefer core.deinit(); + core.native_images = true; + core.update(.{ .resize = .{ + .cols = bench_cols, + .rows = bench_rows, + .cell_pixels = bench_cell_pixels, + } }); + drainEffects(core); + return core; +} + +fn measure(config: Config, plan: Plan) !Result { + const name = plan.scenario.label(); + const ticks = plan.scenario.ticks(); + const totals = try gpa.alloc(u64, config.reps); + defer gpa.free(totals); + const pool = try gpa.alloc(u64, config.reps * plan.frames); + defer gpa.free(pool); + var pool_len: usize = 0; + const recorder = try gpa.create(Recorder); + defer gpa.destroy(recorder); + var expected: ?Identity = null; + var rasterizations: u64 = 0; + var rasters_retained: u64 = 0; + + for (0..config.warmup + config.reps) |round| { + const core = try bootCore(config.path); + defer core.deinit(); + var frame_arena = std.heap.ArenaAllocator.init(gpa); + defer frame_arena.deinit(); + // Untimed: the steady state a reader flings FROM already has page zero + // resident and the strip laid out. + _ = try core.render(frame_arena.allocator()); + drainEffects(core); + const pane = core.panes[0] orelse fatal("PDF pane disappeared", .{}); + const pv = &pane.pdf.?; + const rect = core.rects[0]; + // Anywhere inside the pane body resolves to this pane's hover, which + // is what routes the notch; the exact cell is irrelevant to a wheel. + const wheel: pardes.Mouse = .{ + .button = .wheel_down, + .kind = .press, + .col = rect.x + rect.w / 2, + .row = rect.y + rect.h / 2, + }; + const base_revision = pv.next_raster_revision; + recorder.reset(base_revision); + + var total: u64 = 0; + for (0..plan.frames) |frame| { + var event = wheel; + event.button = plan.button(frame); + const started = nowNs(); + for (0..ticks) |_| core.update(.{ .mouse = event }); + _ = frame_arena.reset(.retain_capacity); + const surface = core.render(frame_arena.allocator()) catch |err| + fatal("render failed in {s} frame {d}: {t}", .{ name, frame, err }); + drainEffects(core); + const elapsed = nowNs() -| started; + total += elapsed; + recorder.observe(frame, surface); + if (round >= config.warmup) { + pool[pool_len] = @max(1, elapsed); + pool_len += 1; + } + } + + const identity = recorder.finish(pv); + try verifyIdentity(name, &expected, identity, round); + if (round >= config.warmup) { + totals[round - config.warmup] = total; + rasterizations = pv.next_raster_revision -% base_revision; + rasters_retained = pv.rasters_len; + } + } + + std.debug.assert(pool_len == pool.len); + std.mem.sort(u64, pool, {}, std.sort.asc(u64)); + std.mem.sort(u64, totals, {}, std.sort.asc(u64)); + return .{ + .name = name, + .ticks_per_frame = ticks, + .frames = plan.frames, + .min_ns = pool[0], + .median_ns = pool[pool.len / 2], + .p90_ns = pool[(pool.len * 9) / 10 -| 1], + .p99_ns = pool[(pool.len * 99) / 100 -| 1], + .max_ns = pool[pool.len - 1], + .total_ns = totals[totals.len / 2], + .rasterizations = rasterizations, + .rasters_retained = rasters_retained, + .identity = expected.?, + }; +} + +/// Accumulates the two checksums across one run. Raster generations are +/// rebased on the counter's value at run start so the trace describes the +/// ORDER textures were produced in rather than absolute counter values, which +/// depend on how much the untimed boot happened to rasterize. +const Recorder = struct { + frames: u64, + images: u64, + visual: u64, + transport: u64, + base_revision: u32, + + fn reset(self: *Recorder, base_revision: u32) void { + self.* = .{ + .frames = 0, + .images = 0, + .visual = checksum_seed, + .transport = checksum_seed, + .base_revision = base_revision, + }; + } + + fn observe(self: *Recorder, frame: usize, surface: *const pardes.Surface) void { + self.frames += 1; + self.visual = mix(self.visual, surface.nimages); + self.transport = mix(self.transport, surface.nimages); + const full = frame % full_rgba_stride == 0; + for (surface.images[0..surface.nimages]) |maybe| { + const place = maybe orelse { + self.visual = mix(self.visual, std.math.maxInt(u64)); + continue; + }; + self.images += 1; + self.visual = mix(self.visual, place.pane); + self.visual = mix(self.visual, place.x); + self.visual = mix(self.visual, place.y); + self.visual = mix(self.visual, place.w); + self.visual = mix(self.visual, place.h); + self.visual = mix(self.visual, place.iw); + self.visual = mix(self.visual, place.ih); + self.visual = mix(self.visual, place.rgba.len); + self.visual = mix(self.visual, place.native.page); + self.visual = mix(self.visual, @intFromEnum(place.native.fit)); + self.visual = mix(self.visual, place.native.pan_x); + self.visual = mix(self.visual, place.native.pan_y); + self.visual = mix(self.visual, @as(u32, @bitCast(place.native.pixel_offset_y))); + if (place.native.geometry) |geometry| { + inline for (.{ geometry.src, geometry.dst }) |rect| { + self.visual = mix(self.visual, rect.x); + self.visual = mix(self.visual, rect.y); + self.visual = mix(self.visual, rect.w); + self.visual = mix(self.visual, rect.h); + } + } else self.visual = mix(self.visual, std.math.maxInt(u64)); + self.visual = if (full) + std.hash.Wyhash.hash(self.visual, place.rgba) + else + stridedHash(self.visual, place.rgba); + self.transport = mix(self.transport, place.native.page); + self.transport = mix(self.transport, place.native.revision -% self.base_revision); + } + } + + fn finish(self: *Recorder, pv: anytype) Identity { + var visual = mix(self.visual, @as(u64, @bitCast(pv.document_scroll_y))); + visual = mix(visual, pv.page); + visual = mix(visual, pv.page_count); + visual = mix(visual, pv.document_height); + return .{ + .frames = self.frames, + .images = self.images, + .visual = visual, + .transport = self.transport, + }; + } +}; + +/// 256 fixed-stride RGBA words. Cheap enough for every image of every frame +/// while still failing on any change that touches a band of the page, and the +/// full hash above closes the gap on sampled frames. +fn stridedHash(seed: u64, rgba: []const u8) u64 { + if (rgba.len < 4) return mix(seed, rgba.len); + const stride = @max(@as(usize, 4), (rgba.len / 256) & ~@as(usize, 3)); + var hash = seed; + var at: usize = 0; + while (at + 4 <= rgba.len) : (at += stride) + hash = std.hash.Wyhash.hash(hash, rgba[at..][0..4]); + return hash; +} + +fn mix(seed: u64, value: anytype) u64 { + const Value = @TypeOf(value); + const Stable = switch (@typeInfo(Value)) { + .comptime_int => u64, + else => Value, + }; + var stable: Stable = value; + return std.hash.Wyhash.hash(seed, std.mem.asBytes(&stable)); +} + +fn drainEffects(core: *pardes.Pardes) void { + while (core.nextEffect()) |_| {} +} + +fn verifyIdentity( + name: []const u8, + expected: *?Identity, + got: Identity, + round: usize, +) !void { + if (expected.*) |want| { + if (!want.eql(got)) { + std.debug.print( + "pdf-scroll-bench: unstable identity in {s} round {d}:\n" ++ + " expected frames={d} images={d} visual={x:0>16} transport={x:0>16}\n" ++ + " got frames={d} images={d} visual={x:0>16} transport={x:0>16}\n", + .{ + name, round, want.frames, want.images, + want.visual, want.transport, got.frames, got.images, + got.visual, got.transport, + }, + ); + return error.UnstableIdentity; + } + } else expected.* = got; +} + +fn nowNs() u64 { + var ts: std.c.timespec = undefined; + _ = std.c.clock_gettime(.MONOTONIC, &ts); + return @as(u64, @intCast(ts.sec)) *| 1_000_000_000 +| + @as(u64, @intCast(ts.nsec)); +} + +fn reportText(config: Config, results: []const Result) void { + std.debug.print("pardes PDF fast-scroll benchmark (ReleaseFast)\n", .{}); + std.debug.print( + "warmup: {d}, sampled reps: {d}; every number is ONE FRAME (notch batch + render + drain)\n\n", + .{ config.warmup, config.reps }, + ); + std.debug.print("{s:<16} {s:>6} {s:>7} {s:>10} {s:>10} {s:>10} {s:>10} {s:>10} {s:>11} {s:>7} {s:>5} {s:<16} {s}\n", .{ + "scenario", "frames", "notches", "min", + "median", "p90", "p99", "max", + "total_us", "rasters", "kept", "visual", + "transport", + }); + std.debug.print("{s}\n", .{"-" ** 160}); + for (results) |result| std.debug.print( + "{s:<16} {d:>6} {d:>7} {d:>10} {d:>10} {d:>10} {d:>10} {d:>10} {d:>11} {d:>7} {d:>5} {x:0>16} {x:0>16}\n", + .{ + result.name, + result.frames, + result.ticks_per_frame, + result.min_ns, + result.median_ns, + result.p90_ns, + result.p99_ns, + result.max_ns, + result.total_ns / 1000, + result.rasterizations, + result.rasters_retained, + result.identity.visual, + result.identity.transport, + }, + ); + std.debug.print( + "\n`rasters` counts MuPDF page rasterizations over one run; `kept` is the retained raster count at the end.\n" ++ + "`visual` pins presented pixels and placement and must never change; `transport` traces texture uploads.\n", + .{}, + ); +} + +const json_report_max_bytes = 16 * 1024; + +fn reportJson(io: std.Io, config: Config, results: []const Result) void { + var storage: [json_report_max_bytes]u8 = undefined; + var out: std.Io.Writer = .fixed(&storage); + out.print( + "{{\"benchmark\":\"pardes-pdf-scroll\",\"build\":\"ReleaseFast\",\"warmup\":{d},\"reps\":{d},\"results\":[", + .{ config.warmup, config.reps }, + ) catch return; + for (results, 0..) |result, i| out.print( + "{s}{{\"scenario\":\"{s}\",\"frames\":{d},\"notches_per_frame\":{d}," ++ + "\"min_ns\":{d},\"median_ns\":{d},\"p90_ns\":{d},\"p99_ns\":{d},\"max_ns\":{d},\"total_ns\":{d}," ++ + "\"rasterizations\":{d},\"rasters_retained\":{d},\"images\":{d}," ++ + "\"visual\":\"{x:0>16}\",\"transport\":\"{x:0>16}\"}}", + .{ + if (i == 0) "" else ",", + result.name, + result.frames, + result.ticks_per_frame, + result.min_ns, + result.median_ns, + result.p90_ns, + result.p99_ns, + result.max_ns, + result.total_ns, + result.rasterizations, + result.rasters_retained, + result.identity.images, + result.identity.visual, + result.identity.transport, + }, + ) catch return; + out.writeAll("]}\n") catch return; + std.Io.File.stdout().writeStreamingAll(io, out.buffered()) catch {}; +} + +fn fatal(comptime format: []const u8, args: anytype) noreturn { + std.debug.print("pdf-scroll-bench: " ++ format ++ "\n", args); + std.process.exit(1); +} + +comptime { + // The bench links the real MuPDF wrapper the core uses; keep the import + // load-bearing so a build that silently dropped it fails here. + _ = pdf.Document; +} |
