summaryrefslogtreecommitdiff
path: root/test/fs_bench.zig
blob: 7315676ac444eacb109cdf8a733262a1ad31cb77 (plain) (blame)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
//! THE FILESYSTEM SCOREBOARD: what one acme-fs request costs the core, and
//! what serving one costs an editor that nobody is scripting.
//!
//!   zig build fs-bench                  -- the table
//!   zig build fs-bench -- --json        -- the same, machine-readable
//!   zig build fs-bench -- --reps 200000 -- more samples per row
//!
//! There is no FUSE here, and no thread: `acmefs.handle` IS the transaction
//! (src/acmefs.zig), so driving it directly is measuring the whole of what the
//! core does per request. That is the point of the split — a transport adds a
//! `read(2)`, a `write(2)` and a wake, and those are the kernel's numbers, not
//! ours (the mounted end-to-end figures are measured with real clients; see
//! examples/README.md).
//!
//! THREE QUESTIONS THIS ANSWERS.
//!
//!   1. Is a request cheap enough to serve thousands per frame? Each row is
//!      one `handle` call, median of `--reps`.
//!   2. Does the steady state ALLOCATE? Every row is measured through a
//!      counting allocator and the table prints the allocation count. A
//!      non-zero number in a read row is a bug: reads answer with a range of
//!      the pane's live text (`Payload.region`) or with the staging buffer,
//!      and the staging buffer is cleared, never freed.
//!   3. What does an editor with NO script attached pay? The last two rows are
//!      the same keystroke with zero listeners and with one. Zero listeners
//!      must be indistinguishable from an editor with no filesystem compiled
//!      in at all: one branch in `file_pane.setContent`.
const std = @import("std");
const pardes = @import("pardes");
const acmefs = pardes.acmefs;

pub const std_options: std.Options = .{ .log_level = .err };

const backing = std.heap.page_allocator;

/// std.time.Timer is gone in 0.16; clock_gettime is what test/perf.zig uses.
fn nowNs() u64 {
    var ts: std.c.timespec = undefined;
    _ = std.c.clock_gettime(.MONOTONIC, &ts);
    return @as(u64, @intCast(ts.sec)) *| 1_000_000_000 +| @as(u64, @intCast(ts.nsec));
}

const Row = struct {
    name: []const u8,
    ns: u64,
    allocs: usize,
    note: []const u8 = "",
};

var rows: std.ArrayList(Row) = .empty;

fn record(name: []const u8, total_ns: u64, reps: usize, counting: *std.testing.FailingAllocator, note: []const u8) void {
    rows.append(backing, .{
        .name = name,
        .ns = total_ns / @max(1, reps),
        .allocs = counting.allocations,
        .note = note,
    }) catch {};
}

/// A session with something to measure against: one big file pane, one shell,
/// and a scripted pane whose event queue has records waiting.
const Session = struct {
    core: *pardes.Pardes,
    counting: *std.testing.FailingAllocator,
    file_id: usize,
    file_serial: u32,

    fn init(counting: *std.testing.FailingAllocator, body_bytes: usize) !Session {
        const gpa = counting.allocator();
        const core = try pardes.Pardes.init(gpa, .{ .tty_only = true, .cols = 120, .rows = 40 });
        while (core.nextEffect()) |_| {}

        // A body big enough that a copy would show up in the numbers.
        const line = "the quick brown fox jumps over the lazy dog\n";
        var content: std.ArrayList(u8) = .empty;
        while (content.items.len < body_bytes) try content.appendSlice(backing, line);
        const pane = try core.hxOpenFileContent(content.items);
        content.deinit(backing);
        const id = core.active;
        while (core.nextEffect()) |_| {}
        return .{
            .core = core,
            .counting = counting,
            .file_id = id,
            .file_serial = pane.serial,
        };
    }

    fn deinit(s: *Session) void {
        s.core.deinit();
    }

    fn node(s: *const Session, file: acmefs.PaneFile) u64 {
        return acmefs.Node.of(s.file_serial, file);
    }
};

/// One row: run `req` `reps` times and report the mean cost plus how many
/// allocations the whole run made.
///
/// The per-update scratch arena is reset each rep because that is what the
/// real path does — `Pardes.update` resets it at the end of every event, and
/// `handle` called bare would otherwise let one arena grow across a hundred
/// thousand requests and count its CHUNKS as allocations. Resetting here
/// measures the request, not the harness.
fn bench(s: *Session, name: []const u8, reps: usize, req: acmefs.Req, note: []const u8) void {
    // warm the staging buffer and any lazy index the first call builds
    for (0..64) |_| {
        _ = acmefs.handle(s.core, req);
        _ = s.core.scratch.reset(.retain_capacity);
    }
    s.counting.allocations = 0;
    const start = nowNs();
    for (0..reps) |_| {
        const reply = acmefs.handle(s.core, req);
        std.mem.doNotOptimizeAway(reply.status);
        _ = s.core.scratch.reset(.retain_capacity);
    }
    record(name, nowNs() - start, reps, s.counting, note);
}

pub fn main(init: std.process.Init) !void {
    const args = try init.minimal.args.toSlice(init.arena.allocator());
    var reps: usize = 100_000;
    var json = false;
    var i: usize = 1;
    while (i < args.len) : (i += 1) {
        if (std.mem.eql(u8, args[i], "--json")) {
            json = true;
        } else if (std.mem.eql(u8, args[i], "--reps") and i + 1 < args.len) {
            i += 1;
            reps = try std.fmt.parseInt(usize, args[i], 10);
        }
    }

    // std.testing.FailingAllocator with the default options never induces a
    // failure and counts every allocation, which is the whole of what this
    // benchmark wanted from a wrapper.
    var counting: std.testing.FailingAllocator = .init(backing, .{});
    var s = try Session.init(&counting, 1 << 20); // a 1 MiB body
    defer s.deinit();

    // ---- the three shapes of request -------------------------------------
    bench(&s, "getattr body", reps, .{
        .tag = 1,
        .op = .getattr,
        .node = s.node(.body),
    }, "stat of a 1 MiB body");
    bench(&s, "lookup ctl", reps, .{
        .tag = 2,
        .op = .lookup,
        .node = acmefs.Node.of(s.file_serial, .dir),
        .data = "ctl",
    }, "name -> node");
    bench(&s, "read body 4K", reps, .{
        .tag = 3,
        .op = .read,
        .node = s.node(.body),
        .off = 4096,
        .size = 4096,
    }, "must be zero-copy");
    bench(&s, "read body 1M", reps / 10, .{
        .tag = 4,
        .op = .read,
        .node = s.node(.body),
        .off = 0,
        .size = 1 << 20,
    }, "same cost as 4K if truly zero-copy");
    bench(&s, "read ctl", reps, .{
        .tag = 5,
        .op = .read,
        .node = s.node(.ctl),
        .size = 256,
    }, "formatted into the staging buffer");
    bench(&s, "read index", reps, .{
        .tag = 6,
        .op = .read,
        .node = @intFromEnum(acmefs.TopFile.index),
        .size = 4096,
    }, "one line per pane");
    bench(&s, "readdir root", reps, .{
        .tag = 7,
        .op = .readdir,
        .node = @intFromEnum(acmefs.TopFile.root),
        .size = 4096,
    }, "staged dirents");
    bench(&s, "read event (empty)", reps, .{
        .tag = 8,
        .op = .read,
        .node = s.node(.event),
        .size = 256,
    }, "Status.again — the blocking primitive");
    // Appending GROWS the fixture, and every append is a whole-body swap plus
    // an undo snapshot (that is the core's edit model, not this filesystem's),
    // so this row is quadratic in its own rep count against a 1 MiB body.
    // Bounded on purpose: the question is what one write costs, and 200 of
    // them answer it without spending a quarter of an hour proving that
    // appending ten megabytes a kilobyte at a time is slow.
    bench(&s, "write body 1K", @min(reps, 200), .{
        .tag = 9,
        .op = .write,
        .node = s.node(.body),
        .data = "x" ** 1024,
        .size = 1024,
    }, "append: whole-body swap + undo snapshot");

    // ---- what an unscripted editor pays ----------------------------------
    // The same keystroke, twice: with nobody listening and with one listener.
    // The first number is the honest answer to "what does this feature cost a
    // session that never uses it", and the pair is what recording costs.
    //
    // A FRESH SESSION PER ROW, and this matters: typing inserts at the cursor,
    // so the line under it grows by one character per rep, and the core's
    // per-keystroke cost is dominated by walking that line's grapheme widths
    // (measured: `file_pane.graphemeDisplayWidth` is 65% of this benchmark's
    // cycles). Reusing one session made the second row type into a body the
    // first had already lengthened, and reported a 2.8x "overhead" that was
    // entirely the fixture. Small bodies for the same reason: on the megabyte
    // fixture a keystroke costs ~40 ms whatever this filesystem does.
    const key_reps = @min(reps, 2000);
    for ([_]bool{ false, true }) |scripted| {
        var keys = try Session.init(&counting, 32 * 1024);
        defer keys.deinit();
        const pane = keys.core.panes[keys.file_id].?;
        pane.mode = .insert;
        if (scripted) {
            keys.core.fs.panes[keys.file_id].readers = 1;
            keys.core.fs.listeners = 1;
        }
        counting.allocations = 0;
        const start = nowNs();
        for (0..key_reps) |_| {
            keys.core.update(.{ .key = .{ .cp = 'x', .text = "x" } });
            while (keys.core.nextEffect()) |_| {}
        }
        record(
            if (scripted) "keystroke, 1 listener" else "keystroke, no listener",
            nowNs() - start,
            key_reps,
            &counting,
            if (scripted) "diff + record" else "one branch",
        );
    }

    var out: std.Io.Writer.Allocating = .init(backing);
    defer out.deinit();
    const w = &out.writer;
    if (json) {
        try w.writeAll("{\"rows\":[");
        for (rows.items, 0..) |r, n| {
            if (n > 0) try w.writeAll(",");
            try w.print("{{\"name\":\"{s}\",\"ns\":{d},\"allocs\":{d}}}", .{ r.name, r.ns, r.allocs });
        }
        try w.writeAll("]}\n");
    } else {
        try w.print("{s:<26} {s:>9} {s:>8}  {s}\n", .{ "request", "ns/op", "allocs", "note" });
        try w.print("{s:<26} {s:>9} {s:>8}  {s}\n", .{ "-" ** 26, "-" ** 9, "-" ** 8, "-" ** 20 });
        for (rows.items) |r|
            try w.print("{s:<26} {d:>9} {d:>8}  {s}\n", .{ r.name, r.ns, r.allocs, r.note });
    }
    try std.Io.File.stdout().writeStreamingAll(init.io, out.written());
}