summaryrefslogtreecommitdiff
path: root/src/fs9_service.zig
diff options
context:
space:
mode:
Diffstat (limited to 'src/fs9_service.zig')
-rw-r--r--src/fs9_service.zig132
1 files changed, 127 insertions, 5 deletions
diff --git a/src/fs9_service.zig b/src/fs9_service.zig
index f02e179e..eebfe7bf 100644
--- a/src/fs9_service.zig
+++ b/src/fs9_service.zig
@@ -109,6 +109,14 @@ const Conn = struct {
/// stays occupied, with no descriptor, until `next()` runs dry. See
/// `Srv.hangup` and `drainAll`.
draining: bool = false,
+ /// When this peer connected, on the monotonic clock, or 0 when the clock
+ /// is unavailable. Read by `expire`: a connection that has not sent
+ /// `Tversion` within `greet_deadline_ms` is holding a slot by silence,
+ /// which with only four of them is a cheaper denial than the frontend
+ /// socket's thirty-two. `Server.msize == 0` is the "has not versioned yet"
+ /// flag, and version(5) requires `Tversion` before any other message, so
+ /// there is no legitimate client this can catch.
+ accepted_ms: i64 = 0,
/// Undefined until `accept` initialises it in place, which it may only do
/// through a pointer to this exact storage.
srv: Srv = undefined,
@@ -144,11 +152,24 @@ const Conn = struct {
}
};
+/// How long the 9P listener stays out of the poll set after an `accept` that
+/// failed for a reason that persists — EMFILE and ENFILE above all. The same
+/// number and the same argument as the frontend listener's own pause: the
+/// connection is still in the backlog, `poll` is level triggered, and coming
+/// straight back spins the core until some unrelated descriptor is freed.
+const accept_pause_ms: i64 = 100;
+
/// The listening socket and its connections. Heap-allocated because a `Conn`
/// holds slices into itself (see there) and because at three buffers per
/// connection this is ≈100 KiB, which does not belong in a host's struct.
pub const Listener = struct {
fd: c_int = -1,
+ /// Do not accept before this moment on the monotonic clock. Set when
+ /// `accept(2)` fails for a reason that leaves the connection in the backlog
+ /// — EMFILE and ENFILE — because a level-triggered poll then reports the
+ /// listener ready forever and coming straight back spins the core. Zero
+ /// means accepting normally.
+ paused_ms: i64 = 0,
/// The bound path, kept so teardown unlinks exactly what was created —
/// guarded on the fd, like nested.zig's and detached/server.zig's
/// `unlisten`.
@@ -172,7 +193,23 @@ pub const Listener = struct {
if (comptime !supported) return;
for (0..max_conns + 1) |_| {
const fd = libc.accept(l.fd, null, null);
- if (fd < 0) return; // EAGAIN is this loop's ordinary exit
+ if (fd < 0) switch (libc.errno(fd)) {
+ // The ordinary exit: nothing more is queued.
+ .AGAIN => return,
+ // Retry: a signal, or a peer that gave up between the poll and
+ // the accept. Neither says anything about our capacity.
+ .INTR, .CONNABORTED => continue,
+ // Out of descriptors. The connection STAYS in the backlog, so a
+ // level-triggered poll reports the listener ready again at once
+ // and coming straight back spins the core until something
+ // unrelated frees an fd — measured at 99.8% of one, sustained.
+ // The frontend listener one file over solves it the same way.
+ else => {
+ l.paused_ms = nowMs() +| accept_pause_ms;
+ log.warn("--fs9: accept failed; pausing the listener for {d} ms", .{accept_pause_ms});
+ return;
+ },
+ };
nested.setCloexec(fd);
setNonblock(fd);
if (comptime nested.darwin) {
@@ -193,6 +230,7 @@ pub const Listener = struct {
};
c.fd = fd;
c.draining = false;
+ c.accepted_ms = nowMs();
c.srv = .init(.{
.in = &c.in,
.out = &c.out,
@@ -201,6 +239,70 @@ pub const Listener = struct {
}
}
+ /// How long a connection may hold a slot without saying `Tversion`. The
+ /// same five seconds and the same argument as the frontend socket's
+ /// `greet_deadline_ms` (`detached/server.zig`): a slot held by silence is
+ /// the same denial as a full queue, arrived at from the other end. Cheaper
+ /// here, because there are four slots rather than thirty-two and no
+ /// handshake to fake.
+ pub const greet_deadline_ms: i64 = 5000;
+
+ /// Take back any slot whose peer connected and then said nothing. Called
+ /// once per frame beside the drain; the host folds `nextDue` into its poll
+ /// timeout so the deadline is kept on an otherwise idle session rather than
+ /// whenever some other descriptor happens to wake it.
+ pub fn expire(l: *Listener) void {
+ if (comptime !supported) return;
+ const now = nowMs();
+ if (now == 0) return; // no clock; see `nowMs`
+ for (&l.conns, 0..) |*c, i| {
+ if (c.fd < 0 or c.srv.msize != 0) continue;
+ if (now - c.accepted_ms < greet_deadline_ms) continue;
+ log.debug("--fs9: slot {d} never sent Tversion; taking it back", .{i});
+ l.drop(@intCast(i));
+ }
+ }
+
+ /// Is the listener worth polling this round? False while it is paused after
+ /// a persistent `accept` failure — leaving it in the set is exactly the
+ /// spin the pause exists to stop.
+ pub fn accepting(l: *const Listener) bool {
+ if (comptime !supported) return false;
+ if (l.fd < 0) return false;
+ if (l.paused_ms == 0) return true;
+ const now = nowMs();
+ return now == 0 or now >= l.paused_ms;
+ }
+
+ /// Milliseconds until the earliest greet deadline, or null when nothing is
+ /// waiting on the clock. Floored at zero so a deadline already past polls
+ /// once without blocking instead of blocking on a negative timeout.
+ pub fn nextDue(l: *const Listener) ?i32 {
+ if (comptime !supported) return null;
+ const now = nowMs();
+ if (now == 0) return null;
+ var due: ?i64 = null;
+ // The pause is a clock deadline like the greet ones: without it here,
+ // an idle session would sleep through the moment the listener is
+ // allowed back and only notice on the next unrelated wake.
+ if (l.paused_ms > now) due = l.paused_ms;
+ for (&l.conns) |*c| {
+ if (c.fd < 0 or c.srv.msize != 0) continue;
+ const at = c.accepted_ms + greet_deadline_ms;
+ due = if (due) |d| @min(d, at) else at;
+ }
+ const at = due orelse return null;
+ return @intCast(@max(0, at - now));
+ }
+
+ /// The monotonic clock in milliseconds, or 0 when there is none — which
+ /// every caller reads as "no deadlines this round" rather than as a time.
+ fn nowMs() i64 {
+ var ts: libc.timespec = undefined;
+ if (libc.clock_gettime(.MONOTONIC, &ts) != 0) return 0;
+ return @as(i64, ts.sec) * std.time.ms_per_s + @divTrunc(ts.nsec, std.time.ns_per_ms);
+ }
+
/// Read one chunk off connection `i` and hand it to the server.
///
/// ONE read per connection per round, which is detached/server.zig's
@@ -215,6 +317,13 @@ pub const Listener = struct {
pub fn fill(l: *Listener, i: u8) void {
if (comptime !supported) return;
const c = &l.conns[i];
+ // FIRST, and before the room guard below, which is the trap: once
+ // `startFrame` gives up on the framing, `in_len` is stuck at `in.len`
+ // for good, so `room == 0` returns without reading, `poll` is level
+ // triggered, the descriptor reports ready again immediately, and the
+ // loop never sleeps. Measured at 99.7% of a core, sustained, reachable
+ // by any process with the uid in one `write(2)`.
+ if (c.srv.dead) return l.drop(i);
const room = c.srv.in.len - c.srv.in_len;
if (room == 0) return;
// A frame-local staging buffer rather than a read straight into the
@@ -229,10 +338,23 @@ pub const Listener = struct {
else => l.drop(i),
};
const n = c.srv.push(buf[0..@intCast(got)]);
- // Cannot happen — the read was clamped to the room — and it is
- // asserted rather than ignored because silently dropping wire bytes
- // desynchronises the stream, which is the one failure 9P cannot
- // resynchronise from.
+ // The stream stopped being 9P. `Server.startFrame` sets `dead` when the
+ // framing is unrecoverable — a `size[4]` of zero, or one larger than the
+ // input buffer — and `push` then takes NOTHING, for good, because there
+ // is nowhere to resynchronise to in a protocol whose only frame marker
+ // is the length you were just lied to about.
+ //
+ // This has to be checked before the assert below, and the assert is why:
+ // it used to fire, and firing meant `unreachable` on the daemon's own
+ // thread — every pane, every attached frontend and the FUSE mount gone,
+ // reached by any client that sends one bad length and then one more
+ // byte. The socket is 0600 in a 0700 directory, but the whole point of
+ // `--fs9` is that other programs dial it, so a buggy one is enough.
+ if (c.srv.dead) return l.drop(i);
+ // NOW it cannot happen: the read was clamped to the room and the only
+ // other refusal is the one handled above. Asserted rather than ignored
+ // because silently dropping wire bytes desynchronises the stream, which
+ // is the one failure 9P cannot resynchronise from.
std.debug.assert(n == @as(usize, @intCast(got)));
}