From b7fc01550c7bde290cf14276d94193b5b4031dc8 Mon Sep 17 00:00:00 2001 From: Gabriel Schneider Date: Tue, 22 Sep 2026 11:18:05 -0300 Subject: 9ns --mntgen: a server that never answers stalls only its own name MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Opening a fish (self-wrapped in `9ns --mntgen`) and running an agent in it would sometimes freeze the whole session: no input reached it and nothing under /mnt/9p answered, until the shell was killed from outside. The cause was one posted server that accepted a connection and then never spoke 9P — pardes, answering its 9P from the same loop that was walking its own mount, was the one on this machine, but any wedged or half-dead server does it. Three things conspired, and each is fixed on its own: * The dispatcher dialed. A LOOKUP of an undialed name ran connect, Tversion, Tattach and Tstat on the one thread that reads /dev/fuse, so while that server kept quiet no request for any name was read, and no FUSE_INTERRUPT either. Now the dispatcher makes a Mount without touching the network and queues the walk to the mount's worker, which dials while serving it. The dial is the request in flight, so an interrupt of the walk abandons it at once (`Session.abort_on_cancel`: nothing to flush before a session exists) and the walk answers EINTR; a failed dial leaves the mount undialed for the next walk to retry; a full listen backlog (the server stopped accepting) is retried for 5s and then EIO. Every later LOOKUP of the name goes through the same queue and is answered from the remembered root attr, so the dispatcher never holds a session at all. * Once the dispatcher had read a request the process behind it was unkillable (FUSE waits out a request userspace has taken), and an INTERRUPT for a request still sitting in a mount's queue was dropped. The dispatcher now takes a queued request out and answers EINTR itself, and forwards only in-flight ones to the worker; queue and in-flight unique are read under the mount's mutex, where the worker moves a request from one to the other. A Tflush the server never answers is given 3s (`Session.flush_grace_ms`) and then the session is declared wedged: the request answers EINTR, the mount dies, the next walk makes a new one. * The kernel serialized the directory. Without FUSE_PARALLEL_DIROPS in the INIT reply every LOOKUP and READDIR in a directory takes its inode lock, so one parked walk held up every other name under /mnt/9p however free the dispatcher was (`cat` sat in fuse_lock_inode). The flag is now negotiated when the kernel offers it. What remains is the kernel's own serialization of lookups of one *name*: a second walker into the parked name waits for the first walk to end, and only then proceeds (and can be interrupted in its turn). An adversarial review of the above found three more things, fixed here: the single-connection bridge's one-slot stash stopped polling the FUSE fd while a second request was parked, so an INTERRUPT could not arrive (and parallel dirops make a second request routine) — the stash is now a queue of copies and the fd is always watched; a dead or wedged mount kept its socket open until exit, where a late-answering single-threaded server could block on it — the session is closed when the mount dies; and teardown after DESTROY or ENODEV (the child still alive, so stop_fd says nothing) could join a worker parked on a mute server forever — the sockets are shut down before the join. The flush grace is a deadline now, not a timer restarted on every wakeup. A black-box run against the binary (hostile servers: mute, garbage, close-after-accept, full backlog, 100 mute names, interrupt storms, 300 deaths of one server) found that a dead mount kept its socket, its interrupt pipe and a megabyte of buffers until exit — three descriptors per death — so `retire` now frees all of it and keeps only the slot; descriptors, threads and RSS stay flat across 400 deaths. The 4096-slot cap per process remains and is documented. Reproduced with a socket that accepts and never writes, posted beside 9agents in a scratch registry: before, `cat /mnt/9p/agents/pid` parked behind `stat /mnt/9p/hang` and SIGINT did nothing; after, it answers at once, the parked walker dies of its signal within milliseconds, and a server that answers the handshake but ignores reads and Tflush releases its reader after the grace. mntgen.sh and adv_bridge_interrupt.sh now check exactly that; nine.zig gains unit tests for the grace and the abort. Also in this change: the uncommitted ESTALE-on-death and FUSE_NOTIFY_INVAL_ENTRY work from the working copy, which the dead-mount path here builds on. Co-Authored-By: Claude Fable 5.1 --- 9ns/src/fuse.zig | 45 +++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 45 insertions(+) (limited to '9ns/src/fuse.zig') diff --git a/9ns/src/fuse.zig b/9ns/src/fuse.zig index 216e616..e45abc2 100644 --- a/9ns/src/fuse.zig +++ b/9ns/src/fuse.zig @@ -37,6 +37,12 @@ pub const FUSE_BIG_WRITES: u32 = 1 << 5; /// Required for `--no-direct-io` correctness: 9P sizes change under us, and /// without this the kernel trusts a stale cached size and truncates reads. pub const FUSE_AUTO_INVAL_DATA: u32 = 1 << 12; +/// Without this the kernel takes the directory's inode lock around every +/// LOOKUP and READDIR in it, so one walk parked on a server that never +/// answers holds up every other name in the same directory — the whole +/// registry root, for a mntgen mount. With it, walks into different names +/// proceed side by side and only the parked one waits. +pub const FUSE_PARALLEL_DIROPS: u32 = 1 << 18; pub const FUSE_MAX_PAGES: u32 = 1 << 22; pub const FATTR_MODE: u32 = 1 << 0; @@ -389,6 +395,44 @@ pub fn replyError(fd: i32, unique: u64, err: linux.E) Error!void { return writeAll(fd, &iov, 1, @sizeOf(OutHeader)); } +/// Notification codes, sent to the kernel unsolicited: they ride an +/// `OutHeader` with `unique = 0` and the code (positive) in `error`. +pub const notify_inval_entry: i32 = 3; + +/// The body of a `FUSE_NOTIFY_INVAL_ENTRY`, followed by the name and a NUL. +pub const NotifyInvalEntryOut = extern struct { + parent: u64, + namelen: u32, + padding: u32 = 0, +}; + +/// Drops the kernel's cached dentry for `name` under `parent`, so the next +/// access of that path comes back as a fresh LOOKUP instead of reusing a +/// node id the server no longer knows. +/// +/// `unique = 0` marks the message as a notification rather than a reply, so +/// it is safe to interleave with replies on the same fd, from any thread. +/// +/// Best-effort by nature: the kernel answers ENOENT when it had nothing +/// cached under that name (`writeAll` already treats that as success) and +/// EINVAL when it does not support the notification, which surfaces here as +/// `error.Io`. Callers ignore both — a notification that does not land +/// leaves them exactly where they were without it. +pub fn notifyInvalEntry(fd: i32, parent: u64, name: []const u8) Error!void { + if (name.len == 0 or name.len > std.math.maxInt(u32) - 1) return error.Protocol; + const out = NotifyInvalEntryOut{ .parent = parent, .namelen = @intCast(name.len) }; + const total = @sizeOf(OutHeader) + @sizeOf(NotifyInvalEntryOut) + name.len + 1; + const header = OutHeader{ .len = @intCast(total), .@"error" = notify_inval_entry, .unique = 0 }; + const nul = [_]u8{0}; + var iov = [_]std.posix.iovec_const{ + .{ .base = @ptrCast(&header), .len = @sizeOf(OutHeader) }, + .{ .base = @ptrCast(&out), .len = @sizeOf(NotifyInvalEntryOut) }, + .{ .base = name.ptr, .len = name.len }, + .{ .base = &nul, .len = 1 }, + }; + return writeAll(fd, &iov, iov.len, total); +} + fn writeAll(fd: i32, iov: [*]const std.posix.iovec_const, count: usize, total: usize) Error!void { while (true) { const rc = linux.writev(fd, iov, count); @@ -462,6 +506,7 @@ pub fn initReply(in: *const InitIn, max_write: u32) InitOut { out.flags |= FUSE_MAX_PAGES; out.max_pages = 256; } + if (in.flags & FUSE_PARALLEL_DIROPS != 0) out.flags |= FUSE_PARALLEL_DIROPS; return out; } -- cgit v1.3