summaryrefslogtreecommitdiff
path: root/src/fs9_service.zig
blob: eebfe7bf98935cb69f4e1c3226dcde406714ab4d (plain) (blame)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
//! `pardes --fs9`: what a native HOST has to decide to serve acme's control
//! filesystem over 9P2000 on a unix socket.
//!
//! `src/fs_service.zig`'s sibling, and deliberately a separate file: that one
//! is three decisions about a MOUNT (where to mount, when to drain, what a
//! pane shell is told), and this one is a listener with connections, buffers
//! and a socket path. They share the seam and nothing else — `Transport`,
//! `drain` and `Drained` all live there and are used verbatim here, which is
//! the whole point of `9P-2`: a second answer to the same three functions.
//!
//! WHAT IS HERE: a bound listening socket, a small fixed table of connections,
//! and the four things a non-blocking byte stream needs — accept, read into
//! `push`, `output` out through `write` and back through `wrote`, and hangup.
//! Everything above that is `src/9p.zig`, which is freestanding and knows
//! about neither sockets nor `acmefs.zig`; the instantiation
//! `ninep.Server(pardes.acmefs)` happens here and nowhere else.
//!
//! WHAT IS NOT HERE: the drain. One connection is one `Transport`, and the
//! host calls `fs_service.drain` per connection per frame at the same point in
//! the frame it drains the mount — see `drainAll`, and see
//! `detached/server.zig`'s `Source.ninep` for why a filesystem request must
//! not be served from inside a poll dispatch.
//!
//! Linux and darwin, like every other unix socket in the tree. On anything
//! else `open` returns null and the flag is quietly off.
const std = @import("std");
const libc = std.c;
const nested = @import("nested.zig");
const ninep = @import("9p.zig");
const pardes = @import("pardes.zig");
const fs_service = @import("fs_service.zig");

/// Diagnostics land where `fs_service`'s do and for its reason: stderr IS the
/// screen in the tty shell, so this is the `PARDES_LOG=1` copy.
const log = std.log.scoped(.fs9);

/// Unix sockets, which is all this needs. `nested.supported` also demands a
/// way to name an arbitrary pid's executable, which no part of this asks.
pub const supported = builtin_unix;
const builtin_unix = @import("builtin").os.tag == .linux or nested.darwin;

/// `sun_path`, from the kernel's struct. See `nested.sun_path_len`.
const sun_path_len = nested.sun_path_len;

/// A THIRD prefix in the one per-user directory, beside nested.zig's
/// `pardes-<pid>.sock` and detached/server.zig's `pardes-detached-<name>.sock`,
/// for the reason `nested.zig:39-44` gives: one directory vetted by different
/// predicates is exactly the divergence that naming prevents. That file's
/// sweeper unlinks any name whose digits name a dead pid, and this socket must
/// never look like one; the detached transport's `vetted` accepts only its own
/// prefix, so a 9P socket cannot be dialled by a frontend expecting `wire.zig`
/// either.
const prefix = "pardes-9p-";

/// The msize this host serves, and the ONE number both buffers are sized from.
///
/// 8 KiB, which is `9P-17`'s clamp applied to the thing a client actually
/// reads: the largest single answer this tree produces is one pane's `body`,
/// and a client reading a megabyte of it does so in msize-sized `Tread`s
/// whatever this number is. So the only thing a larger msize buys is fewer
/// round trips on a LOCAL socket, and the only thing it costs is resident
/// memory in a daemon nobody is talking to. 8 KiB is two `Tread`s per screen
/// of text and comfortably above `ninep.min_msize` (4096), which is the floor
/// below which plan9port's `9p` and Linux's `v9fs` start refusing mounts with
/// `EINVAL` and no message.
pub const msize: u32 = 8192;

/// Connections one session serves at once.
///
/// FOUR, and it is not a guess about load: a 9P client here is a SCRIPT, and
/// the thing a script does is walk, read and clunk. What holds a connection
/// open for minutes is a blocked reader on `event` or `cons` — one per script
/// that is watching the editor — and beyond a handful of those the honest
/// answer is that somebody is using the wrong tool. The number is small on
/// purpose because a connection costs its buffers whether it is busy or idle:
/// MEASURED at 34,072 B each (a `Server` of 9,488 B plus `msize` in and twice
/// `msize` out) for 136,408 B of table, which is the whole of what `--fs9`
/// adds to a daemon's resident memory. A refused connect is also a diagnostic
/// a script author sees immediately, where a silently queued one is not. The
/// detached transport's `max_clients` is 32 because a frontend is a human's
/// window; this is not that.
pub const max_conns = 4;

/// The 9P server, over the filesystem ABI `acmefs.zig` defines. This
/// instantiation is the only coupling between the freestanding protocol file
/// and the core, and it is a type parameter rather than an import for the
/// reason `9p.zig`'s `Server` doc comment gives.
const Srv = ninep.Server(pardes.acmefs);

/// One connection: a socket, a server, and the server's two buffers.
///
/// THE BUFFERS ARE FIELDS HERE, which is what `Srv`'s "no allocator" means
/// from the caller's side: `srv.in` and `srv.out` are slices INTO this struct,
/// so a `Conn` must never be moved or copied once `srv` is initialised. That
/// is why `Listener` is heap-allocated by `open` and held by pointer, and why
/// nothing below takes a `Conn` by value.
///
/// `out` is twice `msize` because `Srv` requires it: one reply being written
/// out and one being built, which is what lets a reply be encoded the moment
/// the core answers with no "can I write yet" question anywhere in `9p.zig`.
const Conn = struct {
    /// Non-negative exactly while the peer is connected. It goes to -1 the
    /// moment the connection ends, which is BEFORE this slot is free: see
    /// `draining`.
    fd: c_int = -1,
    /// The peer has gone and the server still owes the core `release` calls
    /// for the fids it held. A dropped `event` fid without one leaves the
    /// pane's reader count high forever (`acmefs.zig:1053-1061`), so the slot
    /// stays occupied, with no descriptor, until `next()` runs dry. See
    /// `Srv.hangup` and `drainAll`.
    draining: bool = false,
    /// When this peer connected, on the monotonic clock, or 0 when the clock
    /// is unavailable. Read by `expire`: a connection that has not sent
    /// `Tversion` within `greet_deadline_ms` is holding a slot by silence,
    /// which with only four of them is a cheaper denial than the frontend
    /// socket's thirty-two. `Server.msize == 0` is the "has not versioned yet"
    /// flag, and version(5) requires `Tversion` before any other message, so
    /// there is no legitimate client this can catch.
    accepted_ms: i64 = 0,
    /// Undefined until `accept` initialises it in place, which it may only do
    /// through a pointer to this exact storage.
    srv: Srv = undefined,
    in: [msize]u8 = undefined,
    out: [2 * msize]u8 = undefined,

    /// This connection's answer to `fs_service.Transport`. Thunked exactly
    /// like `fuse.Fs.transport()`, and for its reason: a `*Conn` is not an
    /// `*anyopaque` and a vtable cannot hold the typed function.
    fn transport(c: *Conn) fs_service.Transport {
        return .{ .ctx = c, .vtable = &transport_vtable };
    }

    const transport_vtable: fs_service.Transport.VTable = .{
        .retry = transportRetry,
        .next = transportNext,
        .reply = transportReply,
    };

    fn transportRetry(ctx: *anyopaque) ?pardes.acmefs.Req {
        const c: *Conn = @ptrCast(@alignCast(ctx));
        return c.srv.retry();
    }

    fn transportNext(ctx: *anyopaque) ?pardes.acmefs.Req {
        const c: *Conn = @ptrCast(@alignCast(ctx));
        return c.srv.next();
    }

    fn transportReply(ctx: *anyopaque, r: *const pardes.acmefs.Reply, bytes: []const u8) void {
        const c: *Conn = @ptrCast(@alignCast(ctx));
        c.srv.reply(r, bytes);
    }
};

/// How long the 9P listener stays out of the poll set after an `accept` that
/// failed for a reason that persists — EMFILE and ENFILE above all. The same
/// number and the same argument as the frontend listener's own pause: the
/// connection is still in the backlog, `poll` is level triggered, and coming
/// straight back spins the core until some unrelated descriptor is freed.
const accept_pause_ms: i64 = 100;

/// The listening socket and its connections. Heap-allocated because a `Conn`
/// holds slices into itself (see there) and because at three buffers per
/// connection this is ≈100 KiB, which does not belong in a host's struct.
pub const Listener = struct {
    fd: c_int = -1,
    /// Do not accept before this moment on the monotonic clock. Set when
    /// `accept(2)` fails for a reason that leaves the connection in the backlog
    /// — EMFILE and ENFILE — because a level-triggered poll then reports the
    /// listener ready forever and coming straight back spins the core. Zero
    /// means accepting normally.
    paused_ms: i64 = 0,
    /// The bound path, kept so teardown unlinks exactly what was created —
    /// guarded on the fd, like nested.zig's and detached/server.zig's
    /// `unlisten`.
    path_buf: [sun_path_len]u8 = undefined,
    path_len: usize = 0,
    conns: [max_conns]Conn = @splat(.{}),

    /// The socket path, so a host can tell the operator where to dial.
    pub fn path(l: *const Listener) []const u8 {
        return l.path_buf[0..l.path_len];
    }

    /// Accept whatever is waiting, bounded.
    ///
    /// BOUNDED for detached/server.zig's `accept` reason: `poll` is level
    /// triggered, so an unaccepted backlog reports ready forever and a peer
    /// dialling in a loop would otherwise hold the core in here. And always
    /// accepting, even with a full table, for the same reason — the surplus is
    /// accepted and closed rather than left to spin the poll.
    pub fn accept(l: *Listener) void {
        if (comptime !supported) return;
        for (0..max_conns + 1) |_| {
            const fd = libc.accept(l.fd, null, null);
            if (fd < 0) switch (libc.errno(fd)) {
                // The ordinary exit: nothing more is queued.
                .AGAIN => return,
                // Retry: a signal, or a peer that gave up between the poll and
                // the accept. Neither says anything about our capacity.
                .INTR, .CONNABORTED => continue,
                // Out of descriptors. The connection STAYS in the backlog, so a
                // level-triggered poll reports the listener ready again at once
                // and coming straight back spins the core until something
                // unrelated frees an fd — measured at 99.8% of one, sustained.
                // The frontend listener one file over solves it the same way.
                else => {
                    l.paused_ms = nowMs() +| accept_pause_ms;
                    log.warn("--fs9: accept failed; pausing the listener for {d} ms", .{accept_pause_ms});
                    return;
                },
            };
            nested.setCloexec(fd);
            setNonblock(fd);
            if (comptime nested.darwin) {
                // linux says MSG_NOSIGNAL per write, darwin once per socket. A
                // script that dies mid-reply must not take the editor down.
                const on: c_int = 1;
                _ = libc.setsockopt(fd, libc.SOL.SOCKET, libc.SO.NOSIGPIPE, &on, @sizeOf(c_int));
            }
            const c = for (&l.conns) |*cand| {
                if (cand.fd < 0 and !cand.draining) break cand;
            } else {
                // No slot. Closing is the whole refusal: a 9P client that
                // reads EOF instead of an `Rversion` reports a dial failure,
                // which is the honest thing for it to say.
                log.debug("--fs9: refusing a connection, all {d} slots busy", .{max_conns});
                _ = libc.close(fd);
                continue;
            };
            c.fd = fd;
            c.draining = false;
            c.accepted_ms = nowMs();
            c.srv = .init(.{
                .in = &c.in,
                .out = &c.out,
                .root = @intFromEnum(pardes.acmefs.TopFile.root),
            });
        }
    }

    /// How long a connection may hold a slot without saying `Tversion`. The
    /// same five seconds and the same argument as the frontend socket's
    /// `greet_deadline_ms` (`detached/server.zig`): a slot held by silence is
    /// the same denial as a full queue, arrived at from the other end. Cheaper
    /// here, because there are four slots rather than thirty-two and no
    /// handshake to fake.
    pub const greet_deadline_ms: i64 = 5000;

    /// Take back any slot whose peer connected and then said nothing. Called
    /// once per frame beside the drain; the host folds `nextDue` into its poll
    /// timeout so the deadline is kept on an otherwise idle session rather than
    /// whenever some other descriptor happens to wake it.
    pub fn expire(l: *Listener) void {
        if (comptime !supported) return;
        const now = nowMs();
        if (now == 0) return; // no clock; see `nowMs`
        for (&l.conns, 0..) |*c, i| {
            if (c.fd < 0 or c.srv.msize != 0) continue;
            if (now - c.accepted_ms < greet_deadline_ms) continue;
            log.debug("--fs9: slot {d} never sent Tversion; taking it back", .{i});
            l.drop(@intCast(i));
        }
    }

    /// Is the listener worth polling this round? False while it is paused after
    /// a persistent `accept` failure — leaving it in the set is exactly the
    /// spin the pause exists to stop.
    pub fn accepting(l: *const Listener) bool {
        if (comptime !supported) return false;
        if (l.fd < 0) return false;
        if (l.paused_ms == 0) return true;
        const now = nowMs();
        return now == 0 or now >= l.paused_ms;
    }

    /// Milliseconds until the earliest greet deadline, or null when nothing is
    /// waiting on the clock. Floored at zero so a deadline already past polls
    /// once without blocking instead of blocking on a negative timeout.
    pub fn nextDue(l: *const Listener) ?i32 {
        if (comptime !supported) return null;
        const now = nowMs();
        if (now == 0) return null;
        var due: ?i64 = null;
        // The pause is a clock deadline like the greet ones: without it here,
        // an idle session would sleep through the moment the listener is
        // allowed back and only notice on the next unrelated wake.
        if (l.paused_ms > now) due = l.paused_ms;
        for (&l.conns) |*c| {
            if (c.fd < 0 or c.srv.msize != 0) continue;
            const at = c.accepted_ms + greet_deadline_ms;
            due = if (due) |d| @min(d, at) else at;
        }
        const at = due orelse return null;
        return @intCast(@max(0, at - now));
    }

    /// The monotonic clock in milliseconds, or 0 when there is none — which
    /// every caller reads as "no deadlines this round" rather than as a time.
    fn nowMs() i64 {
        var ts: libc.timespec = undefined;
        if (libc.clock_gettime(.MONOTONIC, &ts) != 0) return 0;
        return @as(i64, ts.sec) * std.time.ms_per_s + @divTrunc(ts.nsec, std.time.ns_per_ms);
    }

    /// Read one chunk off connection `i` and hand it to the server.
    ///
    /// ONE read per connection per round, which is detached/server.zig's
    /// `receive` rule: a script in a `while true` loop gets one turn and then
    /// the loop moves on to the other connections and to the frame.
    ///
    /// Sized to what the server can TAKE rather than to the socket, because
    /// `push` returns short on back-pressure and bytes read past that point
    /// would have nowhere to go. Zero room is not an error and not a hangup:
    /// the buffer holds a message the core has not finished with, and the next
    /// drain frees it.
    pub fn fill(l: *Listener, i: u8) void {
        if (comptime !supported) return;
        const c = &l.conns[i];
        // FIRST, and before the room guard below, which is the trap: once
        // `startFrame` gives up on the framing, `in_len` is stuck at `in.len`
        // for good, so `room == 0` returns without reading, `poll` is level
        // triggered, the descriptor reports ready again immediately, and the
        // loop never sleeps. Measured at 99.7% of a core, sustained, reachable
        // by any process with the uid in one `write(2)`.
        if (c.srv.dead) return l.drop(i);
        const room = c.srv.in.len - c.srv.in_len;
        if (room == 0) return;
        // A frame-local staging buffer rather than a read straight into the
        // server's tail: advancing `in_len` is `push`'s business, and reaching
        // past it to do it here would make this file a second author of
        // `9p.zig`'s invariants for the sake of one memcpy per 8 KiB.
        var buf: [msize]u8 = undefined;
        const got = libc.read(c.fd, &buf, @min(room, buf.len));
        if (got == 0) return l.drop(i); // clean EOF: the script left
        if (got < 0) return switch (libc.errno(got)) {
            .INTR, .AGAIN => {},
            else => l.drop(i),
        };
        const n = c.srv.push(buf[0..@intCast(got)]);
        // The stream stopped being 9P. `Server.startFrame` sets `dead` when the
        // framing is unrecoverable — a `size[4]` of zero, or one larger than the
        // input buffer — and `push` then takes NOTHING, for good, because there
        // is nowhere to resynchronise to in a protocol whose only frame marker
        // is the length you were just lied to about.
        //
        // This has to be checked before the assert below, and the assert is why:
        // it used to fire, and firing meant `unreachable` on the daemon's own
        // thread — every pane, every attached frontend and the FUSE mount gone,
        // reached by any client that sends one bad length and then one more
        // byte. The socket is 0600 in a 0700 directory, but the whole point of
        // `--fs9` is that other programs dial it, so a buggy one is enough.
        if (c.srv.dead) return l.drop(i);
        // NOW it cannot happen: the read was clamped to the room and the only
        // other refusal is the one handled above. Asserted rather than ignored
        // because silently dropping wire bytes desynchronises the stream, which
        // is the one failure 9P cannot resynchronise from.
        std.debug.assert(n == @as(usize, @intCast(got)));
    }

    /// Push what the kernel will take of what this connection owes, and leave
    /// the rest for a POLLOUT. detached/server.zig's `flush` on a 9P byte
    /// FIFO instead of an `ArrayList`, and the rule it exists for is the same:
    /// a peer that will not read must never block the editor.
    pub fn flush(l: *Listener, i: u8) void {
        if (comptime !supported) return;
        const c = &l.conns[i];
        if (c.fd < 0) return;
        while (true) {
            const bytes = c.srv.output();
            if (bytes.len == 0) return;
            const n = libc.send(c.fd, bytes.ptr, bytes.len, nosignal);
            if (n < 0) switch (libc.errno(n)) {
                .INTR => continue,
                .AGAIN => return,
                else => return l.drop(i),
            };
            // No progress and no error. Looping on it is a spin, and a spin in
            // here is the whole session at 100% of a core with no syscall for
            // a signal to interrupt — host_io.zig's `writeFd` rule.
            if (n == 0) return;
            c.srv.wrote(@intCast(n));
        }
    }

    /// Whether this connection wants POLLOUT: only while it owes bytes, which
    /// is the same rule and the same reason as a client's and a pty's — asking
    /// for it unconditionally makes every idle socket a ready descriptor and
    /// turns the poll into a spin.
    pub fn owes(l: *const Listener, i: u8) bool {
        return l.conns[i].srv.output().len != 0;
    }

    /// True when this slot has a live descriptor to poll.
    pub fn live(l: *const Listener, i: u8) bool {
        return l.conns[i].fd >= 0;
    }

    /// This connection is over: out of the poll set, out of the process — but
    /// NOT out of the table, because the server still owes the core a
    /// `release` per open fid. See `Conn.draining`.
    pub fn drop(l: *Listener, i: u8) void {
        const c = &l.conns[i];
        if (c.fd >= 0) {
            _ = libc.close(c.fd);
            c.fd = -1;
        }
        if (c.draining) return;
        c.srv.hangup();
        c.draining = true;
    }

    /// This connection's `Transport`, or null when the slot has no work: the
    /// peer never arrived, or it left and its fids are already released.
    ///
    /// THE HOST DRAINS, not this file, and that is not a style choice. A reply
    /// reaches a transport through the host's `push_fs_reply`, so the host has
    /// to know WHICH transport the request being served came from — the
    /// "routing origin" docs/9p.typ's layering table (`O2 two listeners`)
    /// names as the thing a second listener costs. Handing the transport out
    /// here and taking the `Drained` back in `settle` is that origin made
    /// explicit: the host sets it, calls `fs_service.drain`, clears it. A
    /// `drainAll` that hid the loop in this file could not, and every 9P reply
    /// went to the FUSE mount instead — measured, as a `Tattach` that never
    /// came back.
    pub fn transport(l: *Listener, i: u8) ?fs_service.Transport {
        if (comptime !supported) return null;
        const c = &l.conns[i];
        if (c.fd < 0 and !c.draining) return null;
        return c.transport();
    }

    /// What one connection's drain came to: write the replies, and reclaim the
    /// slot when a hung-up peer's last fid is released. Returns whether this
    /// connection still owes work, which the caller or's into the flag that
    /// keeps the loop from sleeping.
    ///
    /// The flush is HERE rather than left to a POLLOUT, so a reply the core
    /// produced this frame is on the wire this frame; what the kernel would not
    /// take waits for POLLOUT as usual.
    pub fn settle(l: *Listener, i: u8, d: fs_service.Drained) bool {
        if (comptime !supported) return false;
        const c = &l.conns[i];
        if (c.draining) {
            // A hung-up connection has no descriptor and therefore no event of
            // its own: its orphaned fids are pumped out over successive frames,
            // and the loop must stay hot until the pump runs dry. It does run
            // dry — the debt is one release per fid.
            if (d.count == 0) {
                c.draining = false;
                return false;
            }
            return true;
        }
        l.flush(i);
        return d.pending;
    }

    /// Every connection down, the socket closed, and the path unlinked.
    ///
    /// The orphaned fids are NOT pumped here: `deinit` runs when the session is
    /// being torn down, and the core it would report the releases to is going
    /// with it. `Fs.deinit` makes the same choice about the mount.
    pub fn deinit(l: *Listener, gpa: std.mem.Allocator) void {
        for (0..max_conns) |i| l.drop(@intCast(i));
        if (l.fd >= 0) {
            _ = libc.close(l.fd);
            l.fd = -1;
            var z: [sun_path_len:0]u8 = undefined;
            @memcpy(z[0..l.path_len], l.path_buf[0..l.path_len]);
            z[l.path_len] = 0;
            _ = libc.unlink(z[0..l.path_len :0]);
        }
        gpa.destroy(l);
    }
};

/// `<dir>/pardes-9p-<name>.sock`. A name is one path component and nothing
/// clever, for detached/server.zig's `socketPath` reason: a `/` would put the
/// socket somewhere else entirely and a NUL would truncate the address. The
/// buffer is `sun_path`-sized, so a name that does not fit is no address at
/// all rather than a truncated one pointing somewhere else.
pub fn socketPath(buf: *[sun_path_len]u8, dir: []const u8, name: []const u8) ?[:0]const u8 {
    if (name.len == 0) return null;
    if (std.mem.indexOfAny(u8, name, "/\x00") != null) return null;
    return std.fmt.bufPrintSentinel(buf, "{s}/" ++ prefix ++ "{s}.sock", .{ dir, name }, 0) catch null;
}

/// Bind, listen, and hand back a listener — or null, which is the same answer
/// for "this platform has no unix sockets", "there is no runtime directory"
/// and "the bind failed". That is `fs_service.start`'s posture and
/// nested.zig's: a transport that will not come up must cost the operator
/// their scripting, never their session.
///
/// `named` is `Options.fs9`: EMPTY means a bare `--fs9`, so `fallback` names
/// it (the session name in a daemon, this pid anywhere else), and anything
/// else is the name the user gave, which wins — scripts need a path they can
/// predict.
pub fn open(gpa: std.mem.Allocator, named: []const u8, fallback: []const u8) ?*Listener {
    if (comptime !supported) return null;
    var dir_buf: [sun_path_len:0]u8 = undefined;
    const dir = nested.socketDir(&dir_buf) orelse {
        log.warn("--fs9: no runtime directory for the socket", .{});
        return null;
    };
    if (!nested.ensureSocketDir(dir)) return null;
    const l = gpa.create(Listener) catch return null;
    l.* = .{};
    // No sweep of the directory, unlike detached/server.zig's `listen`. That
    // sweeper connects to every socket of its OWN prefix to retire dead ones;
    // this prefix has no handshake to probe with, so the only stale file worth
    // removing is the one this bind collides with, immediately below.
    const p = socketPath(&l.path_buf, dir, if (named.len != 0) named else fallback) orelse {
        gpa.destroy(l);
        return null;
    };
    var addr: libc.sockaddr.un = .{ .path = @splat(0) };
    @memcpy(addr.path[0 .. p.len + 1], p[0 .. p.len + 1]);
    const fd = libc.socket(libc.AF.UNIX, libc.SOCK.STREAM, 0);
    if (fd < 0) {
        gpa.destroy(l);
        return null;
    }
    nested.setCloexec(fd);
    // `bind` IS the exclusive create, so it and nothing else decides who owns
    // a name — detached/server.zig's rule, and its reason: unlinking
    // unconditionally is how a second daemon takes a live one's socket away.
    // The one case that is not a collision is a session killed rather than
    // quit, whose file outlived it, and `alive` is the only thing allowed to
    // say so.
    if (libc.bind(fd, @ptrCast(&addr), @sizeOf(@TypeOf(addr))) != 0) {
        if (alive(p)) {
            log.warn("--fs9: something is already listening on {s}", .{p});
            _ = libc.close(fd);
            gpa.destroy(l);
            return null;
        }
        _ = libc.unlink(p);
        if (libc.bind(fd, @ptrCast(&addr), @sizeOf(@TypeOf(addr))) != 0) {
            _ = libc.close(fd);
            gpa.destroy(l);
            return null;
        }
    }
    // Owner-only, and BEFORE listen(2), which is the first moment anyone could
    // connect. The directory is already 0700; this is the second wall, and
    // this socket can write into every pane of a live editor.
    _ = libc.chmod(p, 0o600);
    if (libc.listen(fd, max_conns) != 0) {
        _ = libc.close(fd);
        gpa.destroy(l);
        return null;
    }
    setNonblock(fd);
    l.fd = fd;
    l.path_len = p.len;
    log.info("--fs9: serving 9P2000 on {s}", .{p});
    return l;
}

/// Whether a socket file at `path` has a listener behind it. Only ever asked
/// about a bind that failed, and it answers on the CONNECT: a refusal means
/// the file outlived its process and may be unlinked, and anything else —
/// including a success — means somebody is there.
fn alive(path: [:0]const u8) bool {
    var addr: libc.sockaddr.un = .{ .path = @splat(0) };
    if (path.len + 1 > sun_path_len) return true; // cannot ask; assume occupied
    @memcpy(addr.path[0 .. path.len + 1], path[0 .. path.len + 1]);
    const fd = libc.socket(libc.AF.UNIX, libc.SOCK.STREAM, 0);
    if (fd < 0) return true;
    defer _ = libc.close(fd);
    return libc.connect(fd, @ptrCast(&addr), @sizeOf(@TypeOf(addr))) == 0;
}

/// Every descriptor here is non-blocking, for detached/server.zig's reason:
/// the core must never park on a peer. Its `setNonblock` is not reused because
/// importing the daemon into a module the tty and GUI shells may also serve
/// from would make a frontend transport a dependency of a filesystem.
fn setNonblock(fd: c_int) void {
    const flags = libc.fcntl(fd, libc.F.GETFL, @as(c_int, 0));
    if (flags < 0) return;
    var o: libc.O = @bitCast(@as(u32, @bitCast(flags)));
    o.NONBLOCK = true;
    _ = libc.fcntl(fd, libc.F.SETFL, @as(c_int, @bitCast(@as(u32, @bitCast(o)))));
}

/// A dead script must never kill the editor. linux says it per write, darwin
/// once per socket (see `accept`).
const nosignal: u32 = if (nested.darwin) 0 else libc.MSG.NOSIGNAL;

const testing = std.testing;

test "the socket name is a third prefix in the shared directory" {
    // Asserted rather than described: nested.zig's sweeper unlinks any name
    // whose digits name a dead pid, and the detached transport's `vetted`
    // accepts only its own prefix. A 9P socket must be invisible to both.
    var buf: [sun_path_len]u8 = undefined;
    const p = socketPath(&buf, "/run/user/1000", "t9srv").?;
    try testing.expectEqualStrings("/run/user/1000/pardes-9p-t9srv.sock", p);
    try testing.expect(!std.mem.startsWith(u8, std.fs.path.basename(p), "pardes-detached-"));
}

test "a name that is not one path component is no address at all" {
    var buf: [sun_path_len]u8 = undefined;
    try testing.expect(socketPath(&buf, "/run", "") == null);
    try testing.expect(socketPath(&buf, "/run", "a/b") == null);
    try testing.expect(socketPath(&buf, "/run", "a\x00b") == null);
}

test "one connection's buffers are sized from the one msize constant" {
    // The `Srv` asserts both of these at `init`, where a violation is a panic
    // in a live daemon; here it is a build failure instead.
    try testing.expect(msize >= ninep.min_msize);
    const c: Conn = .{};
    try testing.expectEqual(@as(usize, msize), c.in.len);
    try testing.expectEqual(@as(usize, 2 * msize), c.out.len);
}