diff options
Diffstat (limited to 'test/fs_bench.zig')
| -rw-r--r-- | test/fs_bench.zig | 263 |
1 files changed, 263 insertions, 0 deletions
diff --git a/test/fs_bench.zig b/test/fs_bench.zig new file mode 100644 index 00000000..7315676a --- /dev/null +++ b/test/fs_bench.zig @@ -0,0 +1,263 @@ +//! THE FILESYSTEM SCOREBOARD: what one acme-fs request costs the core, and +//! what serving one costs an editor that nobody is scripting. +//! +//! zig build fs-bench -- the table +//! zig build fs-bench -- --json -- the same, machine-readable +//! zig build fs-bench -- --reps 200000 -- more samples per row +//! +//! There is no FUSE here, and no thread: `acmefs.handle` IS the transaction +//! (src/acmefs.zig), so driving it directly is measuring the whole of what the +//! core does per request. That is the point of the split — a transport adds a +//! `read(2)`, a `write(2)` and a wake, and those are the kernel's numbers, not +//! ours (the mounted end-to-end figures are measured with real clients; see +//! examples/README.md). +//! +//! THREE QUESTIONS THIS ANSWERS. +//! +//! 1. Is a request cheap enough to serve thousands per frame? Each row is +//! one `handle` call, median of `--reps`. +//! 2. Does the steady state ALLOCATE? Every row is measured through a +//! counting allocator and the table prints the allocation count. A +//! non-zero number in a read row is a bug: reads answer with a range of +//! the pane's live text (`Payload.region`) or with the staging buffer, +//! and the staging buffer is cleared, never freed. +//! 3. What does an editor with NO script attached pay? The last two rows are +//! the same keystroke with zero listeners and with one. Zero listeners +//! must be indistinguishable from an editor with no filesystem compiled +//! in at all: one branch in `file_pane.setContent`. +const std = @import("std"); +const pardes = @import("pardes"); +const acmefs = pardes.acmefs; + +pub const std_options: std.Options = .{ .log_level = .err }; + +const backing = std.heap.page_allocator; + +/// std.time.Timer is gone in 0.16; clock_gettime is what test/perf.zig uses. +fn nowNs() u64 { + var ts: std.c.timespec = undefined; + _ = std.c.clock_gettime(.MONOTONIC, &ts); + return @as(u64, @intCast(ts.sec)) *| 1_000_000_000 +| @as(u64, @intCast(ts.nsec)); +} + +const Row = struct { + name: []const u8, + ns: u64, + allocs: usize, + note: []const u8 = "", +}; + +var rows: std.ArrayList(Row) = .empty; + +fn record(name: []const u8, total_ns: u64, reps: usize, counting: *std.testing.FailingAllocator, note: []const u8) void { + rows.append(backing, .{ + .name = name, + .ns = total_ns / @max(1, reps), + .allocs = counting.allocations, + .note = note, + }) catch {}; +} + +/// A session with something to measure against: one big file pane, one shell, +/// and a scripted pane whose event queue has records waiting. +const Session = struct { + core: *pardes.Pardes, + counting: *std.testing.FailingAllocator, + file_id: usize, + file_serial: u32, + + fn init(counting: *std.testing.FailingAllocator, body_bytes: usize) !Session { + const gpa = counting.allocator(); + const core = try pardes.Pardes.init(gpa, .{ .tty_only = true, .cols = 120, .rows = 40 }); + while (core.nextEffect()) |_| {} + + // A body big enough that a copy would show up in the numbers. + const line = "the quick brown fox jumps over the lazy dog\n"; + var content: std.ArrayList(u8) = .empty; + while (content.items.len < body_bytes) try content.appendSlice(backing, line); + const pane = try core.hxOpenFileContent(content.items); + content.deinit(backing); + const id = core.active; + while (core.nextEffect()) |_| {} + return .{ + .core = core, + .counting = counting, + .file_id = id, + .file_serial = pane.serial, + }; + } + + fn deinit(s: *Session) void { + s.core.deinit(); + } + + fn node(s: *const Session, file: acmefs.PaneFile) u64 { + return acmefs.Node.of(s.file_serial, file); + } +}; + +/// One row: run `req` `reps` times and report the mean cost plus how many +/// allocations the whole run made. +/// +/// The per-update scratch arena is reset each rep because that is what the +/// real path does — `Pardes.update` resets it at the end of every event, and +/// `handle` called bare would otherwise let one arena grow across a hundred +/// thousand requests and count its CHUNKS as allocations. Resetting here +/// measures the request, not the harness. +fn bench(s: *Session, name: []const u8, reps: usize, req: acmefs.Req, note: []const u8) void { + // warm the staging buffer and any lazy index the first call builds + for (0..64) |_| { + _ = acmefs.handle(s.core, req); + _ = s.core.scratch.reset(.retain_capacity); + } + s.counting.allocations = 0; + const start = nowNs(); + for (0..reps) |_| { + const reply = acmefs.handle(s.core, req); + std.mem.doNotOptimizeAway(reply.status); + _ = s.core.scratch.reset(.retain_capacity); + } + record(name, nowNs() - start, reps, s.counting, note); +} + +pub fn main(init: std.process.Init) !void { + const args = try init.minimal.args.toSlice(init.arena.allocator()); + var reps: usize = 100_000; + var json = false; + var i: usize = 1; + while (i < args.len) : (i += 1) { + if (std.mem.eql(u8, args[i], "--json")) { + json = true; + } else if (std.mem.eql(u8, args[i], "--reps") and i + 1 < args.len) { + i += 1; + reps = try std.fmt.parseInt(usize, args[i], 10); + } + } + + // std.testing.FailingAllocator with the default options never induces a + // failure and counts every allocation, which is the whole of what this + // benchmark wanted from a wrapper. + var counting: std.testing.FailingAllocator = .init(backing, .{}); + var s = try Session.init(&counting, 1 << 20); // a 1 MiB body + defer s.deinit(); + + // ---- the three shapes of request ------------------------------------- + bench(&s, "getattr body", reps, .{ + .tag = 1, + .op = .getattr, + .node = s.node(.body), + }, "stat of a 1 MiB body"); + bench(&s, "lookup ctl", reps, .{ + .tag = 2, + .op = .lookup, + .node = acmefs.Node.of(s.file_serial, .dir), + .data = "ctl", + }, "name -> node"); + bench(&s, "read body 4K", reps, .{ + .tag = 3, + .op = .read, + .node = s.node(.body), + .off = 4096, + .size = 4096, + }, "must be zero-copy"); + bench(&s, "read body 1M", reps / 10, .{ + .tag = 4, + .op = .read, + .node = s.node(.body), + .off = 0, + .size = 1 << 20, + }, "same cost as 4K if truly zero-copy"); + bench(&s, "read ctl", reps, .{ + .tag = 5, + .op = .read, + .node = s.node(.ctl), + .size = 256, + }, "formatted into the staging buffer"); + bench(&s, "read index", reps, .{ + .tag = 6, + .op = .read, + .node = @intFromEnum(acmefs.TopFile.index), + .size = 4096, + }, "one line per pane"); + bench(&s, "readdir root", reps, .{ + .tag = 7, + .op = .readdir, + .node = @intFromEnum(acmefs.TopFile.root), + .size = 4096, + }, "staged dirents"); + bench(&s, "read event (empty)", reps, .{ + .tag = 8, + .op = .read, + .node = s.node(.event), + .size = 256, + }, "Status.again — the blocking primitive"); + // Appending GROWS the fixture, and every append is a whole-body swap plus + // an undo snapshot (that is the core's edit model, not this filesystem's), + // so this row is quadratic in its own rep count against a 1 MiB body. + // Bounded on purpose: the question is what one write costs, and 200 of + // them answer it without spending a quarter of an hour proving that + // appending ten megabytes a kilobyte at a time is slow. + bench(&s, "write body 1K", @min(reps, 200), .{ + .tag = 9, + .op = .write, + .node = s.node(.body), + .data = "x" ** 1024, + .size = 1024, + }, "append: whole-body swap + undo snapshot"); + + // ---- what an unscripted editor pays ---------------------------------- + // The same keystroke, twice: with nobody listening and with one listener. + // The first number is the honest answer to "what does this feature cost a + // session that never uses it", and the pair is what recording costs. + // + // A FRESH SESSION PER ROW, and this matters: typing inserts at the cursor, + // so the line under it grows by one character per rep, and the core's + // per-keystroke cost is dominated by walking that line's grapheme widths + // (measured: `file_pane.graphemeDisplayWidth` is 65% of this benchmark's + // cycles). Reusing one session made the second row type into a body the + // first had already lengthened, and reported a 2.8x "overhead" that was + // entirely the fixture. Small bodies for the same reason: on the megabyte + // fixture a keystroke costs ~40 ms whatever this filesystem does. + const key_reps = @min(reps, 2000); + for ([_]bool{ false, true }) |scripted| { + var keys = try Session.init(&counting, 32 * 1024); + defer keys.deinit(); + const pane = keys.core.panes[keys.file_id].?; + pane.mode = .insert; + if (scripted) { + keys.core.fs.panes[keys.file_id].readers = 1; + keys.core.fs.listeners = 1; + } + counting.allocations = 0; + const start = nowNs(); + for (0..key_reps) |_| { + keys.core.update(.{ .key = .{ .cp = 'x', .text = "x" } }); + while (keys.core.nextEffect()) |_| {} + } + record( + if (scripted) "keystroke, 1 listener" else "keystroke, no listener", + nowNs() - start, + key_reps, + &counting, + if (scripted) "diff + record" else "one branch", + ); + } + + var out: std.Io.Writer.Allocating = .init(backing); + defer out.deinit(); + const w = &out.writer; + if (json) { + try w.writeAll("{\"rows\":["); + for (rows.items, 0..) |r, n| { + if (n > 0) try w.writeAll(","); + try w.print("{{\"name\":\"{s}\",\"ns\":{d},\"allocs\":{d}}}", .{ r.name, r.ns, r.allocs }); + } + try w.writeAll("]}\n"); + } else { + try w.print("{s:<26} {s:>9} {s:>8} {s}\n", .{ "request", "ns/op", "allocs", "note" }); + try w.print("{s:<26} {s:>9} {s:>8} {s}\n", .{ "-" ** 26, "-" ** 9, "-" ** 8, "-" ** 20 }); + for (rows.items) |r| + try w.print("{s:<26} {d:>9} {d:>8} {s}\n", .{ r.name, r.ns, r.allocs, r.note }); + } + try std.Io.File.stdout().writeStreamingAll(init.io, out.written()); +} |
