summaryrefslogtreecommitdiff
path: root/src/fs9_client.zig
blob: ad46dee4e29f473b40a4f642c48a34a20e3b5cc9 (plain) (blame)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
//! `9p <dial> <path>`: one pardes reading a file out of another pardes's tree.
//!
//! `src/fs9_service.zig`'s MIRROR, and the other half of `9P-2`: that file is a
//! listener with connections and hands each one a `ninep.Server`, this one
//! dials a single socket and drives a `ninep.Client` over it. They share the
//! socket NAMING and nothing else — `socketPath` is imported verbatim, so a
//! bare `9p work /1/body` resolves to exactly the path a `pardes --fs9 work`
//! bound, which is the whole point of having one spelling of it.
//!
//! WHAT IS HERE, and it is the same four things any non-blocking byte stream
//! needs: connect, read into `push`, `output` out through `send` and back
//! through `wrote`, and close. Everything above that is `src/9p.zig`, which is
//! freestanding and knows about neither sockets nor panes.
//!
//! IT BLOCKS, BRIEFLY AND BOUNDED, and that is a decision rather than an
//! oversight. `src/look.zig`'s `readFile` already blocks the frame on a disk
//! read — opening a file pane is a person waiting for a file — and a remote
//! read over a unix socket on the same machine is the same wait with a context
//! switch in it. What makes it safe to say that is the BUDGET: `budget_ms` is
//! the deadline for the whole transaction, `poll` is what waits, and the
//! descriptor is non-blocking, so a peer that stops answering costs one
//! `budget_ms` pause and a message on the message row rather than a wedged
//! editor. The alternative — a request queued into the frame loop, a state
//! machine per outstanding fetch, a pane that fills in later — is an async
//! runtime, and `docs/9p.typ` §12.5 is explicit that this design does not get
//! one.
//!
//! WHY THE HOST AND NOT THE CORE. A socket is `std.c`, and `src/9p.zig` must
//! keep compiling for `wasm32-freestanding` and the board's
//! `riscv32-freestanding`; the same `ninep.Client` runs over
//! `src/esp32p4/uart.zig` with no line of this file involved. So the split is
//! the one the server half already made: protocol in the freestanding file,
//! descriptor here.
//!
//! Linux and darwin, like every other unix socket in the tree. Anywhere else
//! `supported` is false and the word is not registered at all.
const std = @import("std");
const libc = std.c;
const nested = @import("nested.zig");
const ninep = @import("9p.zig");
const fs9_service = @import("fs9_service.zig");
const pardes = @import("pardes.zig");
const Pardes = pardes.Pardes;
const output_pane = @import("output_pane.zig");

/// Unix sockets, which is all this needs — `fs9_service`'s own predicate, so a
/// build that can serve 9P can dial it and one that cannot has neither.
pub const supported = fs9_service.supported;

/// `sun_path`, from the kernel's struct. See `nested.sun_path_len`.
const sun_path_len = nested.sun_path_len;

/// The deadline for the WHOLE transaction: connect, handshake, attach, walk,
/// open, every read, clunk.
///
/// TWO SECONDS, and the number is about the human rather than about the wire.
/// On a local socket the whole exchange is six round trips and some memcpys —
/// microseconds — so any wait long enough to notice means the far end is not
/// answering, and the useful thing to do about that is say so. Two seconds is
/// long enough that a busy editor on the other side finishing its frame is
/// never mistaken for a dead one, and short enough that a mistyped socket name
/// on a path that happens to exist does not feel like a hang.
pub const budget_ms: i64 = 2000;

/// The most bytes one `9p` will carry into a pane.
///
/// A MEGABYTE, which is a quarter of `look.zig`'s cap for a virtual file
/// (`read_stream_max_bytes`) and for a sharper reason: what is on the other end
/// is a synthetic tree of live editor state, where the biggest file is one
/// pane's `body`. A megabyte of it is a large source file; ten megabytes is
/// somebody pointing this word at a `/dev/zero` equivalent, and a read loop
/// with no cap would spend the whole budget filling the heap.
pub const max_bytes: u64 = 1 << 20;

/// The deepest path this word will walk.
///
/// `MAXWELEM` is sixteen elements per `Twalk` and a deeper path is legal — the
/// client splits it into chunks and `transact` does — so this is not a protocol
/// bound. It is a bound on the ARGUMENT: acme's tree is two deep (`/1/body`),
/// two chunks is thirty-two, and a path with more elements than that is a typo
/// or a loop rather than a file. Refused rather than truncated, because a
/// truncated path names a different file.
pub const max_depth: usize = 2 * ninep.max_welem;

/// What the far end reports in `Rstat`'s three name fields (`uid`, `gid`,
/// `muid`) for everything we touch, because `ninep.Server` records the
/// attach's `uname` and quotes it back.
///
/// A CONSTANT AND NOT `$USER`: the socket's permissions are the identity here
/// (0600, in a 0700 per-user directory — docs/9p.typ §10), so this string is
/// not a credential and cannot become one. What it is for is the operator
/// reading `ls -l` on the far side, and "pardes" tells them which program
/// walked their tree, which a login name they already share with it does not.
const uname = "pardes";

/// Everything this word can refuse, and each one is a different thing to do
/// about it.
pub const Error = error{
    MissingDial,
    MissingPath,
    /// More than `max_depth` elements.
    PathTooDeep,
    /// The dial names no address we can form: an empty name, a name with a
    /// separator or a NUL in it, a path past `sun_path`, or no runtime
    /// directory to resolve a bare name against.
    BadDial,
    /// `socket(2)` or `connect(2)` said no: nothing is listening on that
    /// socket, or its permissions are not ours. The overwhelmingly common
    /// case, and it means "that pardes is not running with `--fs9`".
    Dial,
    /// The peer closed mid-transaction.
    Hangup,
    /// `budget_ms` elapsed. See there.
    Timeout,
    /// The stream stopped being 9P: `ninep.Client` went dead, or a reply
    /// arrived whose shape does not answer the request it was tagged for.
    /// Nothing can be resynchronised from here.
    Botch,
    /// The far end answered `Rerror`. Its own string goes on the message row —
    /// see `fetch` — so this value only says "reported already".
    Remote,
    /// The path names a directory. A directory READ is a run of `stat`
    /// records rather than text, so opening it in a pane would show a person
    /// the wire format; `9p` names files.
    IsDirectory,
    /// The walk stopped short: some element of the path is not there. Distinct
    /// from `Remote` because a partial walk is a SUCCESSFUL `Rwalk` with fewer
    /// qids and carries no message to report.
    NotFound,
    /// Past `max_bytes`.
    FileTooLarge,
    /// The pane that asked went away while this was in flight.
    MissingPane,
};

/// The far end's own words, copied out of the client's input buffer before the
/// connection is torn down and the buffer with it. `ninep.errmax` is the buffer
/// a Plan 9 client has for an error string, so it is the right size for one.
const RemoteError = struct {
    buf: [ninep.errmax]u8 = undefined,
    len: usize = 0,

    /// Returns the error so that every call site is `return remote.set(e)`.
    fn set(r: *RemoteError, msg: []const u8) error{Remote} {
        r.len = @min(msg.len, r.buf.len);
        @memcpy(r.buf[0..r.len], msg[0..r.len]);
        return error.Remote;
    }

    fn text(r: *const RemoteError) []const u8 {
        return r.buf[0..r.len];
    }
};

/// `9p <dial> <path>` — walk to a remote file, read it, and open the bytes in a
/// pane.
///
/// The pane is an ORDINARY OUTPUT BUFFER, which is what `src/board_memory.zig`
/// puts a hexdump in and what `Grep` puts its rows in: a file pane with an
/// `output` origin, so every motion, chord, search and Look works on it for
/// free. Deliberately NOT a real file pane, even though the bytes came from
/// `look.readFile`'s own kind of read: `file_pane.open` arms a file WATCH on
/// the path it was given, and there is no local path here for `inotify` to
/// watch — the bytes live in another process's memory. An output buffer is the
/// existing answer to "text with no file behind it".
///
/// Identified by the WHOLE argument, so `9p work /1/body` and `9p work /2/body`
/// are two panes and running either again refills its own.
pub fn fetch(p: *Pardes, id: usize, argument: []const u8) !void {
    const a = std.mem.trim(u8, argument, " \t\r\n");
    const args = try parse(a);
    var names: [max_depth][]const u8 = undefined;
    const n = try elements(args.path, &names);

    var sock_buf: [sun_path_len]u8 = undefined;
    const sock = resolve(&sock_buf, args.dial) orelse return Error.BadDial;

    var remote: RemoteError = .{};
    const content = fetchBytes(p.gpa, sock, names[0..n], &remote) catch |err| {
        if (err != Error.Remote) return err;
        // The far end's own wording, which is the whole error ABI in base
        // 9P2000 (docs/registry.typ `9P-4`) — reported verbatim rather than
        // mapped to one of ours, because it is the only thing that says which
        // of the eight operations the other side objected to and why.
        var buf: [ninep.errmax + 8]u8 = undefined;
        p.setMessage(id, std.fmt.bufPrint(&buf, "9p: {s}", .{remote.text()}) catch "9p: refused");
        return;
    };
    const pane = p.panes[id] orelse {
        p.gpa.free(content);
        return Error.MissingPane;
    };
    const dir = if (pane.file) |f| (std.fs.path.dirname(f.path) orelse "/") else pane.cwdSlice();
    try output_pane.fillResults(p, id, dir, .{ .cmd = .@"9p" }, a, content, null);
}

const Args = struct { dial: []const u8, path: []const u8 };

/// `<dial> <path>`, split at the FIRST run of whitespace and not tokenized.
///
/// The path keeps its spaces, because a pane's name in acme's tree can have
/// them and a path is the last argument: `9p work /1/tag` and
/// `9p work /a name/body` both have exactly one reading. The dial cannot have
/// them, and does not need to — it is a socket name or a socket path.
fn parse(a: []const u8) Error!Args {
    if (a.len == 0) return Error.MissingDial;
    const cut = std.mem.indexOfAny(u8, a, " \t") orelse return Error.MissingPath;
    const path = std.mem.trim(u8, a[cut..], " \t\r\n");
    if (path.len == 0) return Error.MissingPath;
    return .{ .dial = a[0..cut], .path = path };
}

/// A path into `Twalk` elements. Separators are collapsed and a trailing one is
/// dropped, so `/1/body`, `1/body` and `//1/body/` are one file — the
/// normalisation every shell already does, done here because 9P has no
/// pathnames at all and a client that forwarded an empty element would be
/// asking for a file called "".
///
/// ZERO ELEMENTS is the root, which is legal and is a directory; `transact`
/// refuses it there, where every other directory is refused too.
fn elements(path: []const u8, out: *[max_depth][]const u8) Error!usize {
    var n: usize = 0;
    var it = std.mem.tokenizeScalar(u8, path, '/');
    while (it.next()) |name| {
        if (n == out.len) return Error.PathTooDeep;
        out[n] = name;
        n += 1;
    }
    return n;
}

/// The dial, as an address.
///
/// TWO SPELLINGS, told apart by a separator, and the distinction is the one a
/// person already makes: a NAME is what `pardes --fs9 work` was started with,
/// and it resolves through `fs9_service.socketPath` — the same function that
/// bound it, so the two can never drift. A PATH is taken as given, which is
/// what you need for a socket somewhere else entirely: a bind-mounted
/// container, a different user's runtime directory, an `ssh -L` forward.
fn resolve(buf: *[sun_path_len]u8, dial: []const u8) ?[:0]const u8 {
    if (dial.len == 0) return null;
    if (std.mem.indexOfScalar(u8, dial, '/') != null) {
        if (std.mem.indexOfScalar(u8, dial, 0) != null) return null;
        return std.fmt.bufPrintSentinel(buf, "{s}", .{dial}, 0) catch null;
    }
    var dir_buf: [sun_path_len:0]u8 = undefined;
    const dir = nested.socketDir(&dir_buf) orelse return null;
    return fs9_service.socketPath(buf, dir, dial);
}

/// One dialled connection: the descriptor, the deadline, the client and its
/// three buffers.
///
/// HEAP-ALLOCATED by `fetchBytes`, for `fs9_service.Listener`'s reason and one
/// more: `cl.in` and `cl.out` are slices INTO this struct, so it must never be
/// moved once `cl` is initialised, and at three msizes it is 24 KiB, which does
/// not belong on the frame's stack.
///
/// The msize is `fs9_service.msize`, the one number the serving side is already
/// sized from. A client on the same machine reading the same tree has no reason
/// to pick a different one, and picking the same one means the handshake never
/// clamps.
const Session = struct {
    fd: c_int,
    /// `nowMs()` past which every wait gives up.
    deadline: i64,
    cl: ninep.Client = undefined,
    in: [fs9_service.msize]u8 = undefined,
    out: [fs9_service.msize]u8 = undefined,
    /// A frame-local staging buffer rather than a read straight into the
    /// client's tail: advancing `in_len` is `push`'s business, and reaching
    /// past it to do it here would make this file a second author of
    /// `9p.zig`'s invariants for the sake of one memcpy per 8 KiB. Exactly
    /// `fs9_service.fill`'s reasoning, from the other side.
    stage: [fs9_service.msize]u8 = undefined,

    /// Wait for `events` on the descriptor, or give up. THE ONLY PLACE THIS
    /// FILE BLOCKS, and the only place the budget is spent.
    fn wait(s: *Session, events: i16) Error!void {
        while (true) {
            const left = s.deadline - nowMs();
            if (left <= 0) return Error.Timeout;
            var fds = [1]libc.pollfd{.{ .fd = s.fd, .events = events, .revents = 0 }};
            const ready = libc.poll(&fds, 1, @intCast(@min(left, budget_ms)));
            if (ready < 0) {
                if (libc.errno(ready) == .INTR) continue;
                return Error.Hangup;
            }
            if (ready == 0) return Error.Timeout;
            // What we asked for wins over HUP: a peer that wrote a reply and
            // then closed reports both at once, and those bytes are ours.
            if (fds[0].revents & events != 0) return;
            return Error.Hangup;
        }
    }

    /// Push everything the client owes the wire, and nothing else. Split out of
    /// `settle` so that `dropNoWait` can send a message it will never collect a
    /// reply for. Bounded by the same deadline `wait` enforces.
    fn flush(s: *Session) Error!void {
        while (s.cl.output().len != 0) {
            try s.wait(poll_out);
            const bytes = s.cl.output();
            const sent = libc.send(s.fd, bytes.ptr, bytes.len, nosignal);
            if (sent < 0) switch (libc.errno(sent)) {
                .INTR, .AGAIN => continue,
                else => return Error.Hangup,
            };
            // No progress and no error: looping on it is a spin, and a
            // spin in here is the editor at 100% of a core.
            if (sent == 0) return Error.Hangup;
            s.cl.wrote(@intCast(sent));
        }
    }

    /// Drive the client until the one outstanding request answers: flush what
    /// we owe, collect if a reply is already buffered, otherwise wait and read.
    ///
    /// LOCK-STEP, deliberately, and it is worth saying why given that
    /// `ninep.Client` allows sixteen requests in flight. A `9p` word is one
    /// person waiting for one file, and its round trips are strictly ordered
    /// anyway — you cannot read a fid you have not opened, or open one you have
    /// not walked to. The one place pipelining would pay is the read loop, and
    /// on a local socket at an 8 KiB msize a megabyte is 128 round trips of a
    /// few microseconds each; buying that back would cost this file a request
    /// window, an out-of-order reassembly buffer and a reason for both. The
    /// CLIENT is where the sixteen tags live, so the board's runtime and any
    /// future caller get them without this file having spent them.
    fn settle(s: *Session) Error!ninep.Client.Done {
        while (true) {
            try s.flush();
            if (s.cl.take()) |done| return done;
            if (s.cl.dead) return Error.Botch;
            try s.wait(poll_in);
            const room = s.cl.in.len - s.cl.in_len;
            // Cannot happen: one reply is at most one msize and the buffer is
            // exactly that, so a full buffer with nothing to take would mean
            // the far end sent a frame it told us it would not.
            if (room == 0) return Error.Botch;
            const got = libc.read(s.fd, &s.stage, @min(room, s.stage.len));
            if (got == 0) return Error.Hangup;
            if (got < 0) switch (libc.errno(got)) {
                .INTR, .AGAIN => continue,
                else => return Error.Hangup,
            };
            const n = s.cl.push(s.stage[0..@intCast(got)]);
            // The read was clamped to the room, so this cannot be short; it is
            // asserted rather than ignored because silently dropping wire
            // bytes desynchronises the stream, which is the one failure 9P
            // cannot resynchronise from.
            std.debug.assert(n == @as(usize, @intCast(got)));
        }
    }

    /// One request, one reply, and the two answers that are not the one asked
    /// for folded into errors here so that `transact` reads as a script.
    fn ask(s: *Session, req: ninep.Client.Request, remote: *RemoteError) Error!ninep.Client.Result {
        _ = s.cl.submit(req) catch return Error.Botch;
        const done = try s.settle();
        if (done.result == .fail) return remote.set(done.result.fail);
        // The client already refuses a reply whose shape does not match the
        // request's op (it kills the connection), so this can only be an
        // `Rerror` we have just handled. Checked anyway: a `switch` here would
        // be a second copy of that table.
        if (std.mem.eql(u8, @tagName(done.result), @tagName(std.meta.activeTag(req)))) return done.result;
        return Error.Botch;
    }

    /// Clunk a fid and WAIT for the answer, because the caller is about to
    /// reuse the number. `transact`'s walk alternates between fids 1 and 2, and
    /// in-order processing is the only thing that makes that safe.
    ///
    /// Best effort otherwise: a refused clunk still frees the fid on both sides
    /// (`clunk(5)`), so there is nothing here worth failing a fetch over.
    fn drop(s: *Session, fid: u32) void {
        _ = s.cl.submit(.{ .clunk = .{ .fid = fid } }) catch return;
        _ = s.settle() catch {};
    }

    /// Clunk a fid and do NOT wait. For the last one, where the descriptor is
    /// closed on the next line and closing it frees every fid the connection
    /// held — so the `Rclunk` is not merely unwanted, it is unobservable.
    ///
    /// Waiting for it cost the whole budget against a peer that answers
    /// everything else and ignores clunks: measured at 2.005 s to deliver a
    /// file that was already in hand, and 2.003 s to report an error decided
    /// 1.4 ms in. The bytes still go out — a well-behaved peer gets its clunk
    /// and frees the fid immediately rather than at hangup — but nothing here
    /// reads the reply.
    fn dropNoWait(s: *Session, fid: u32) void {
        _ = s.cl.submit(.{ .clunk = .{ .fid = fid } }) catch return;
        s.flush() catch {};
    }
};

/// The whole transaction, and the only function here that knows 9P's order of
/// operations: version, attach, walk, open, read to the end, clunk.
fn transact(
    s: *Session,
    names: []const []const u8,
    out: *std.Io.Writer.Allocating,
    remote: *RemoteError,
) !void {
    // The handshake. A server that answers "unknown" has no dialect in common
    // with us and leaves `msize` at zero, which is a connection nothing can be
    // submitted on — reported as a botch, because there is no fallback ladder
    // here to climb down.
    _ = try s.ask(.{ .version = .{} }, remote);
    if (s.cl.msize == 0) return Error.Botch;

    // Fid 0 is the root for the life of the connection; 1 and 2 alternate as
    // the walk descends, so a chunked walk never needs a third.
    const root: u32 = 0;
    var here = (try s.ask(.{ .attach = .{ .fid = root, .uname = uname } }, remote)).attach;
    var cur: u32 = root;
    var next: u32 = 1;

    var i: usize = 0;
    while (i < names.len) {
        // `MAXWELEM` elements at a time, which is what makes a path deeper than
        // sixteen work at all: every implementation refuses a seventeenth
        // element, so a deep path is several walks with an intermediate fid.
        const n = @min(ninep.max_welem, names.len - i);
        const w = (try s.ask(.{ .walk = .{
            .fid = cur,
            .newfid = next,
            .names = names[i..][0..n],
        } }, remote)).walk;
        // A PARTIAL WALK IS A SUCCESS with fewer qids, and only a failure on
        // the first element is an `Rerror`. So this comparison is the whole of
        // "did the path exist", and skipping it is how a client ends up
        // reading the wrong file.
        if (w.nwqid != n) return Error.NotFound;
        here = w.wqid[n - 1];
        if (cur != root) s.drop(cur);
        cur = next;
        next = if (next == 1) 2 else 1;
        i += n;
    }
    defer s.dropNoWait(cur);

    // The qid the walk landed on already says what this is, so the refusal
    // costs no round trip — and it catches the bare `/` too, which is zero
    // elements and the root.
    if (here.type & ninep.qtdir != 0) return Error.IsDirectory;

    _ = try s.ask(.{ .open = .{ .fid = cur, .mode = ninep.oread } }, remote);

    // Read to the end. 9P has no EOF flag: a reply SHORTER than the count is
    // ordinary and means nothing, and a reply of ZERO bytes is the end of the
    // file (`read(5)`). The offset advances by what came back and never by what
    // was asked, which is the same rule a POSIX read loop follows.
    var off: u64 = 0;
    while (true) {
        if (off >= max_bytes) return Error.FileTooLarge;
        const want: u32 = @intCast(@min(@as(u64, s.cl.maxRead()), max_bytes - off));
        const data = (try s.ask(.{ .read = .{ .fid = cur, .offset = off, .count = want } }, remote)).read;
        if (data.len == 0) return;
        try out.writer.writeAll(data);
        off += data.len;
    }
}

/// Dial, transact, and hand back the bytes — gpa-owned, the way
/// `look.readFile`'s are, so the pane adopts them with no second copy.
fn fetchBytes(
    gpa: std.mem.Allocator,
    sock: [:0]const u8,
    names: []const []const u8,
    remote: *RemoteError,
) ![]u8 {
    if (comptime !supported) return Error.Dial;
    // The deadline starts BEFORE the dial, because the dial is part of the
    // transaction and used not to be bounded by anything at all. See `connect`.
    const deadline = nowMs() +| budget_ms;
    const fd = try connect(sock, deadline);
    const s = gpa.create(Session) catch return error.OutOfMemory;
    defer {
        _ = libc.close(fd);
        gpa.destroy(s);
    }
    s.* = .{ .fd = fd, .deadline = deadline };
    s.cl = .init(.{ .in = &s.in, .out = &s.out });

    var out: std.Io.Writer.Allocating = .init(gpa);
    errdefer out.deinit();
    try transact(s, names, &out, remote);
    return out.toOwnedSlice();
}

/// Connect to a unix socket, non-blocking from the first moment there is
/// anything to wait for — which is the connect itself.
///
/// This used to leave the connect BLOCKING, on the argument that a unix socket
/// either completes at once or refuses at once. That is true only while the
/// listener's accept queue has room. When it is full, Linux's
/// `unix_stream_connect` waits in `unix_wait_for_peer` for `sk_sndtimeo`, which
/// defaults to MAX_SCHEDULE_TIMEOUT — forever. The core is single-threaded, so
/// that is the whole editor: no frame, no keystroke, no filesystem request
/// served. Measured at 177 seconds against a peer that had called `listen` and
/// never `accept`, and it ended only because the peer was killed. Nothing in
/// pardes would have ended it, and the trigger needs no hostility — a peer that
/// is itself wedged does it, and a bare path names sockets pardes does not own.
///
/// So the descriptor is non-blocking before the connect and the wait is spent
/// against the caller's deadline. Both refusals have to be handled and they are
/// different: on AF_UNIX a full backlog is EAGAIN, NOT the EINPROGRESS a TCP
/// connect would give, so EAGAIN retries until the deadline and EINPROGRESS
/// waits for POLLOUT and then asks SO_ERROR what actually happened.
fn connect(sock: [:0]const u8, deadline: i64) Error!c_int {
    if (sock.len + 1 > sun_path_len) return Error.BadDial;
    var addr: libc.sockaddr.un = .{ .path = @splat(0) };
    @memcpy(addr.path[0 .. sock.len + 1], sock[0 .. sock.len + 1]);
    const fd = libc.socket(libc.AF.UNIX, libc.SOCK.STREAM, 0);
    if (fd < 0) return Error.Dial;
    nested.setCloexec(fd);
    setNonblock(fd);
    errdefer _ = libc.close(fd);
    while (true) {
        if (libc.connect(fd, @ptrCast(&addr), @sizeOf(@TypeOf(addr))) == 0) break;
        switch (libc._errno().*) {
            // The backlog is full. Nobody is obliged to drain it, so this is a
            // poll on the clock rather than on the descriptor: there is no
            // event to wait for, only room that may or may not appear.
            @intFromEnum(libc.E.AGAIN), @intFromEnum(libc.E.INTR) => {
                if (nowMs() >= deadline) return Error.Dial;
                nap(2);
            },
            // Someone is listening and the connect is under way. This one IS a
            // descriptor event, so wait for it and then ask what it was.
            @intFromEnum(libc.E.INPROGRESS), @intFromEnum(libc.E.ALREADY) => {
                const left = deadline - nowMs();
                if (left <= 0) return Error.Dial;
                var pfd: [1]libc.pollfd = .{.{ .fd = fd, .events = poll_out, .revents = 0 }};
                if (libc.poll(&pfd, 1, @intCast(@min(left, 1000))) <= 0) continue;
                var err: c_int = 0;
                var len: libc.socklen_t = @sizeOf(c_int);
                if (libc.getsockopt(fd, libc.SOL.SOCKET, libc.SO.ERROR, @ptrCast(&err), &len) != 0)
                    return Error.Dial;
                if (err == 0) break;
                return Error.Dial;
            },
            // Already connected by a previous round of this loop.
            @intFromEnum(libc.E.ISCONN) => break,
            else => return Error.Dial,
        }
    }
    if (comptime nested.darwin) {
        // linux says MSG_NOSIGNAL per write, darwin once per socket. A peer
        // that dies mid-transaction must not take the editor down with it.
        const on: c_int = 1;
        _ = libc.setsockopt(fd, libc.SOL.SOCKET, libc.SO.NOSIGPIPE, &on, @sizeOf(c_int));
    }
    return fd;
}

/// The descriptor is non-blocking and `poll` does the waiting, because that is
/// the only shape in which the budget above is enforceable: a blocking `read`
/// has no deadline to give it. `fs9_service`'s own `setNonblock` is not reused
/// for its stated reason — importing a daemon into a path the tty and GUI
/// shells take would make a frontend transport a dependency of a builtin.
fn setNonblock(fd: c_int) void {
    const flags = libc.fcntl(fd, libc.F.GETFL, @as(c_int, 0));
    if (flags < 0) return;
    var o: libc.O = @bitCast(@as(u32, @bitCast(flags)));
    o.NONBLOCK = true;
    _ = libc.fcntl(fd, libc.F.SETFL, @as(c_int, @bitCast(@as(u32, @bitCast(o)))));
}

/// Milliseconds on the MONOTONIC clock, which is the only clock a deadline may
/// be measured against: the wall clock can be stepped, and an NTP correction
/// landing mid-fetch would turn a two-second budget into a hang or into an
/// instant timeout depending on which way it went.
///
/// A clock that will not answer is reported as THE END OF TIME, so the budget
/// expires on the first wait rather than never — the saturating `+|` at the one
/// call site that adds to it is what makes that safe.
fn nowMs() i64 {
    var ts: libc.timespec = undefined;
    if (libc.clock_gettime(.MONOTONIC, &ts) != 0) return std.math.maxInt(i64);
    return @as(i64, ts.sec) * std.time.ms_per_s + @divTrunc(ts.nsec, std.time.ns_per_ms);
}

const poll_in: i16 = @intCast(libc.POLL.IN);
const poll_out: i16 = @intCast(libc.POLL.OUT);

/// Sleep a couple of milliseconds while a full accept backlog drains. There is
/// no descriptor to wait on for that — the room either appears or the deadline
/// arrives — so this is the one place here that waits on the clock. `poll` with
/// no descriptors is the portable spelling and needs no `nanosleep` import.
fn nap(ms: c_int) void {
    _ = libc.poll(&[0]libc.pollfd{}, 0, ms);
}

/// A dead peer must never kill the editor. linux says it per write, darwin once
/// per socket (see `connect`).
const nosignal: u32 = if (nested.darwin) 0 else libc.MSG.NOSIGNAL;

const testing = std.testing;

test "the argument is a dial and then the rest of the line" {
    const a = try parse("work /1/body");
    try testing.expectEqualStrings("work", a.dial);
    try testing.expectEqualStrings("/1/body", a.path);

    // A pane's name in acme's tree may contain spaces, and the path is the last
    // argument, so it keeps them.
    const spaced = try parse("work /a name/body");
    try testing.expectEqualStrings("work", spaced.dial);
    try testing.expectEqualStrings("/a name/body", spaced.path);

    // A socket path as the dial, which is the other spelling.
    const p = try parse("/run/user/1000/pardes-9p-work.sock  /index");
    try testing.expectEqualStrings("/run/user/1000/pardes-9p-work.sock", p.dial);
    try testing.expectEqualStrings("/index", p.path);

    try testing.expectError(Error.MissingDial, parse(""));
    try testing.expectError(Error.MissingPath, parse("work"));
    try testing.expectError(Error.MissingPath, parse("work   "));
}

test "a path becomes walk elements, normalised the way a shell would" {
    var out: [max_depth][]const u8 = undefined;
    try testing.expectEqual(@as(usize, 2), try elements("/1/body", &out));
    try testing.expectEqualStrings("1", out[0]);
    try testing.expectEqualStrings("body", out[1]);

    // Leading, trailing and doubled separators are one file, not four.
    try testing.expectEqual(@as(usize, 2), try elements("1/body", &out));
    try testing.expectEqual(@as(usize, 2), try elements("//1//body//", &out));
    try testing.expectEqual(@as(usize, 1), try elements("/index", &out));

    // Zero elements is the root, which is legal here and refused as a
    // directory where every other directory is.
    try testing.expectEqual(@as(usize, 0), try elements("/", &out));

    // Deeper than two full walks is a typo, and truncating it would name a
    // different file.
    var deep: [8 * max_depth]u8 = @splat('/');
    for (0..max_depth + 1) |i| deep[i * 2 + 1] = 'a';
    try testing.expectError(Error.PathTooDeep, elements(deep[0 .. (max_depth + 1) * 2], &out));
}

test "a bare dial resolves to the socket --fs9 binds, and a path is taken as given" {
    if (comptime !supported) return error.SkipZigTest;
    var buf: [sun_path_len]u8 = undefined;

    // The one spelling both halves share: this must be the same name
    // `fs9_service.socketPath` produces, or the friendly form dials nothing.
    const named = resolve(&buf, "work").?;
    try testing.expect(std.mem.endsWith(u8, named, "/pardes-9p-work.sock"));
    var expect: [sun_path_len]u8 = undefined;
    var dir_buf: [sun_path_len:0]u8 = undefined;
    const dir = nested.socketDir(&dir_buf).?;
    try testing.expectEqualStrings(fs9_service.socketPath(&expect, dir, "work").?, named);

    // A separator makes it a path, verbatim.
    const path = resolve(&buf, "/tmp/somewhere.sock").?;
    try testing.expectEqualStrings("/tmp/somewhere.sock", path);

    // And the refusals: nothing to dial, and a NUL that would truncate the
    // address into something else entirely.
    try testing.expect(resolve(&buf, "") == null);
    try testing.expect(resolve(&buf, "/tmp/a\x00b") == null);
}

test "a dial with nothing listening is one error and not a wait" {
    if (comptime !supported) return error.SkipZigTest;
    // The overwhelmingly common failure — "that pardes is not running with
    // --fs9" — and it must be immediate: `connect` on a unix socket with no
    // listener is refused by the kernel with no timeout in it, which is why
    // `budget_ms` is never spent here.
    var names: [max_depth][]const u8 = undefined;
    const n = try elements("/1/body", &names);
    var remote: RemoteError = .{};
    const before = nowMs();
    try testing.expectError(
        Error.Dial,
        fetchBytes(testing.allocator, "/tmp/pardes-9p-no-such-socket.sock", names[0..n], &remote),
    );
    try testing.expect(nowMs() - before < budget_ms);
}

test "one fetch costs three msize buffers and nothing that grows" {
    // The number this word adds to a session WHILE IT RUNS, and nothing after:
    // the `Session` is freed before `fetch` returns and only the content
    // survives, adopted by the pane. Heap rather than stack for the reason
    // `Session` states.
    try testing.expectEqual(@as(usize, fs9_service.msize), @as(usize, (Session{ .fd = -1, .deadline = 0 }).in.len));
    try testing.expect(@sizeOf(Session) <= 3 * fs9_service.msize + 256);
    // The client's own state is a rounding error beside its buffers, which is
    // the whole point of borrowing payloads out of `in` instead of copying
    // them per tag.
    try testing.expect(@sizeOf(ninep.Client) <= 256);
}