summaryrefslogtreecommitdiff
path: root/9ns/src
diff options
context:
space:
mode:
authorGabriel Schneider <[email protected]>2026-09-19 23:28:22 -0300
committerGabriel Schneider <[email protected]>2026-09-19 23:28:22 -0300
commitba996acfcad1698adbf4a1834fe50e73b1c6cab9 (patch)
tree282ba00ce5b10d7416aecb9f2f0f0a439340a57d /9ns/src
parentb05abcba3ea09ea106ad28364c6e40a3ec31b890 (diff)
downloadcloud9-ba996acfcad1698adbf4a1834fe50e73b1c6cab9.tar.gz
cloud9-ba996acfcad1698adbf4a1834fe50e73b1c6cab9.zip
Rename programs: 9player -> 9ns, introspect -> 9proc, app -> web (9web)
Directories, binaries, build options (-D9ns, -D9proc), step names, module name (9proc), thread and fs names, env var NINEPLAYER_MOUNT -> NINE_MOUNT, docs and test scripts. Browser assets move to web/static. Co-Authored-By: Claude Fable 5.1 <[email protected]>
Diffstat (limited to '9ns/src')
-rw-r--r--9ns/src/bridge.zig971
-rw-r--r--9ns/src/fuse.zig653
-rw-r--r--9ns/src/main.zig444
-rw-r--r--9ns/src/nine.zig756
-rw-r--r--9ns/src/ns.zig1082
5 files changed, 3906 insertions, 0 deletions
diff --git a/9ns/src/bridge.zig b/9ns/src/bridge.zig
new file mode 100644
index 0000000..3a072ec
--- /dev/null
+++ b/9ns/src/bridge.zig
@@ -0,0 +1,971 @@
+//! FUSE ↔ 9P2000 translation: the request loop that turns kernel FUSE requests
+//! into synchronous 9P calls on a `nine.Session` and sends the replies back.
+//!
+//! Everything here is single-threaded and one request at a time. State is three
+//! tables: inodes (nodeid → fid/qid, deduplicated by qid.path), open handles
+//! (fh → fid plus a cached directory listing), and the reverse qid map.
+const std = @import("std");
+const cloud9 = @import("cloud9");
+const fuse = @import("fuse.zig");
+const nine = @import("nine.zig");
+const linux = std.os.linux;
+
+pub const Options = struct {
+ /// Reported owner of every file.
+ uid: u32,
+ gid: u32,
+ /// attr/entry cache validity (0 = none).
+ attr_timeout_ns: u64 = 1_000_000_000,
+ /// FOPEN_DIRECT_IO on every regular file.
+ direct_io: bool = true,
+ /// Trace every request, reply and 9P call to stderr.
+ debug: bool = false,
+};
+
+/// Largest single READ/WRITE payload we accept from the kernel.
+pub const max_write: u32 = 1 << 20;
+/// Upper bound on the raw bytes of one directory listing (about a million entries);
+/// past it the listing fails with EIO instead of eating memory.
+pub const max_dir_bytes: u64 = 64 << 20;
+/// The kernel refuses dirents longer than this (FUSE_NAME_MAX) with EIO.
+pub const max_name_len: usize = 1024;
+/// Request buffer: `max_write` plus room for the header and the largest in-struct.
+pub const request_buf_len: usize = max_write + 4096;
+
+const Inode = struct {
+ fid: u32,
+ qid: cloud9.Qid,
+ nlookup: u64,
+ /// nodeid of the directory this inode was looked up in (root: itself). Used for "..".
+ parent: u64,
+};
+
+pub const Entry = struct { name: []u8, ino: u64, dtype: u32 };
+
+pub const DirList = struct {
+ entries: std.ArrayList(Entry) = .empty,
+
+ pub fn deinit(d: *DirList, gpa: std.mem.Allocator) void {
+ for (d.entries.items) |e| gpa.free(e.name);
+ d.entries.deinit(gpa);
+ }
+};
+
+const Handle = struct {
+ fid: u32,
+ nodeid: u64,
+ dir: ?DirList,
+};
+
+/// Errors a request handler may surface. Policy failures are ordinary errors
+/// that the dispatcher maps to an errno; `FuseIo` means the kernel side is broken.
+const HandlerError = nine.Session.Error || error{
+ BadRequest,
+ NoEntry,
+ BadHandle,
+ Exdev,
+ Perm,
+ NotSup,
+ /// A directory listing the server sent could not be parsed (EIO, not fatal).
+ BadDir,
+ FuseIo,
+};
+
+/// Runs until the FUSE fd reports ENODEV, a DESTROY arrives, or `stop_fd`
+/// becomes readable (also while a 9P reply is outstanding). Returns
+/// `error.Closed` if the 9P server went away.
+pub fn serve(gpa: std.mem.Allocator, fuse_fd: i32, session: *nine.Session, root_fid: u32, stop_fd: i32, opts: Options) !void {
+ var effective = opts;
+ // With page caching on, a nonzero attr cache lets the kernel trust a stale
+ // (often zero) size and truncate reads: 9P sizes are authoritative and change
+ // under us. direct_io ignores the cached size, so the cache is safe only there.
+ if (!effective.direct_io) effective.attr_timeout_ns = 0;
+ var b: Bridge = .{
+ .gpa = gpa,
+ .fuse_fd = fuse_fd,
+ .nine = session,
+ .opts = effective,
+ };
+ defer b.deinit();
+
+ b.req_buf = try gpa.alignedAlloc(u8, .@"8", request_buf_len);
+ b.data_buf = try gpa.alloc(u8, max_write);
+
+ // Abandon any pending 9P reply once the child is gone (stop_fd readable),
+ // including the initial root stat below: a silent server must not pin us.
+ session.stop_fd = stop_fd;
+ defer session.stop_fd = -1;
+
+ // Node 1 is the root; its qid comes from a stat so lookups resolving back to
+ // it (e.g. via a walk) dedupe onto node 1.
+ var root_qid: cloud9.Qid = .{ .type = cloud9.qtdir, .version = 0, .path = 0 };
+ if (b.stat(root_fid)) |st| {
+ root_qid = st.qid;
+ b.root_path = st.qid.path;
+ try b.by_qid.put(gpa, st.qid.path, fuse.root_id);
+ } else |e| switch (e) {
+ error.Nine => {},
+ error.Stopped => return,
+ else => return error.Closed,
+ }
+ try b.inodes.put(gpa, fuse.root_id, .{ .fid = root_fid, .qid = root_qid, .nlookup = 1, .parent = fuse.root_id });
+
+ var pfds = [_]linux.pollfd{
+ .{ .fd = fuse_fd, .events = linux.POLL.IN, .revents = 0 },
+ .{ .fd = stop_fd, .events = linux.POLL.IN, .revents = 0 },
+ };
+ while (true) {
+ pfds[0].revents = 0;
+ pfds[1].revents = 0;
+ const rc = linux.poll(&pfds, pfds.len, -1);
+ switch (linux.errno(rc)) {
+ .SUCCESS => {},
+ .INTR, .AGAIN => continue,
+ else => return error.Io,
+ }
+ if (pfds[1].revents != 0) {
+ b.trace("stop_fd readable; leaving serve loop", .{});
+ return;
+ }
+ if (pfds[0].revents == 0) continue;
+ const req = (fuse.readRequest(fuse_fd, b.req_buf) catch |e| switch (e) {
+ error.Protocol => return error.FuseProtocol,
+ else => return error.FuseIo,
+ }) orelse {
+ b.trace("fuse fd reports ENODEV; unmounted", .{});
+ return;
+ };
+ if (!try b.dispatch(req)) return;
+ }
+}
+
+const Bridge = struct {
+ gpa: std.mem.Allocator,
+ fuse_fd: i32,
+ nine: *nine.Session,
+ opts: Options,
+ req_buf: []align(8) u8 = &.{},
+ data_buf: []u8 = &.{},
+ inodes: std.AutoHashMapUnmanaged(u64, Inode) = .empty,
+ by_qid: std.AutoHashMapUnmanaged(u64, u64) = .empty,
+ handles: std.AutoHashMapUnmanaged(u64, Handle) = .empty,
+ next_node: u64 = 2,
+ next_fh: u64 = 1,
+ /// qid.path of the root, reported as ino 1 wherever it shows up.
+ root_path: u64 = 0,
+ /// errno of the most recent Rerror that a handler did not swallow. Kept here
+ /// because `Session.rpc` clears its ename on every call, and error paths
+ /// clunk (an rpc) before the dispatcher maps the failure to an errno.
+ last_err: linux.E = .IO,
+
+ fn deinit(b: *Bridge) void {
+ var it = b.handles.valueIterator();
+ while (it.next()) |h| if (h.dir) |*d| d.deinit(b.gpa);
+ b.handles.deinit(b.gpa);
+ b.inodes.deinit(b.gpa);
+ b.by_qid.deinit(b.gpa);
+ if (b.req_buf.len != 0) b.gpa.free(b.req_buf);
+ if (b.data_buf.len != 0) b.gpa.free(b.data_buf);
+ }
+
+ fn trace(b: *const Bridge, comptime fmt: []const u8, args: anytype) void {
+ if (b.opts.debug) std.debug.print("9ns: " ++ fmt ++ "\n", args);
+ }
+
+ // -- dispatch --------------------------------------------------------------------
+
+ /// Handles one request. Returns false when the loop should stop (DESTROY).
+ /// Fatal errors (dead 9P session, broken FUSE fd) propagate.
+ fn dispatch(b: *Bridge, req: fuse.Request) !bool {
+ const h = req.header;
+ const op = h.op();
+ b.trace("<- {s} unique={d} nodeid={d} len={d} (fids={d} inodes={d} handles={d})", .{ opName(op), h.unique, h.nodeid, h.len, b.nine.fidsInUse(), b.inodes.count(), b.handles.count() });
+ const wants_reply = switch (op) {
+ .forget, .batch_forget, .interrupt => false,
+ else => true,
+ };
+ if (op == .destroy) {
+ b.reply(h.unique, &.{}) catch {};
+ return false;
+ }
+ b.handle(req) catch |e| {
+ const code: linux.E = switch (e) {
+ error.Nine => b.last_err,
+ error.BadRequest => .INVAL,
+ error.NoEntry => .NOENT,
+ error.BadHandle => .BADF,
+ error.Exdev => .XDEV,
+ error.Perm => .PERM,
+ error.NotSup => .NOSYS,
+ error.OutOfMemory => .NOMEM,
+ error.TooLarge => .NAMETOOLONG,
+ error.BadDir => .IO,
+ error.Closed, error.Protocol, error.Io, error.Stopped => .IO,
+ error.FuseIo => return error.FuseIo,
+ };
+ if (wants_reply) try b.replyError(h.unique, code);
+ switch (e) {
+ error.Closed, error.Protocol, error.Io => return error.Closed,
+ error.Stopped => return false, // the child is gone; the mount is being torn down
+ else => {},
+ }
+ };
+ return true;
+ }
+
+ fn handle(b: *Bridge, req: fuse.Request) HandlerError!void {
+ const u = req.header.unique;
+ switch (req.header.op()) {
+ .init => {
+ const in = try body(fuse.InitIn, req);
+ const out = fuse.initReply(in, max_write);
+ try b.reply(u, &.{std.mem.asBytes(&out)});
+ },
+ .lookup => {
+ const name = try nameAfter(void, req);
+ const entry = try b.lookupEntry(req.header.nodeid, name);
+ try b.reply(u, &.{std.mem.asBytes(&entry)});
+ },
+ .forget => {
+ const in = try body(fuse.ForgetIn, req);
+ try b.forget(req.header.nodeid, in.nlookup);
+ },
+ .batch_forget => {
+ const in = try body(fuse.BatchForgetIn, req);
+ const rest = req.body[@sizeOf(fuse.BatchForgetIn)..];
+ const count: usize = in.count;
+ if (rest.len < count * @sizeOf(fuse.ForgetOne)) return error.BadRequest;
+ for (0..count) |i| {
+ const one = std.mem.bytesToValue(fuse.ForgetOne, rest[i * @sizeOf(fuse.ForgetOne) ..][0..@sizeOf(fuse.ForgetOne)]);
+ try b.forget(one.nodeid, one.nlookup);
+ }
+ },
+ .getattr => {
+ const ino = b.inodes.get(req.header.nodeid) orelse return error.NoEntry;
+ const st = try b.stat(ino.fid);
+ const out = b.attrOut(st, b.inoOf(req.header.nodeid, ino.qid));
+ try b.reply(u, &.{std.mem.asBytes(&out)});
+ },
+ .setattr => try b.setattr(req),
+ .open => try b.openFile(req, false),
+ .opendir => try b.openFile(req, true),
+ .read => {
+ const in = try body(fuse.ReadIn, req);
+ const h = b.handles.get(in.fh) orelse return error.BadHandle;
+ const want: usize = @min(in.size, max_write);
+ const n = try b.read(h.fid, in.offset, b.data_buf[0..want]);
+ try b.reply(u, &.{b.data_buf[0..n]});
+ },
+ .write => {
+ const in = try body(fuse.WriteIn, req);
+ const h = b.handles.get(in.fh) orelse return error.BadHandle;
+ const rest = req.body[@sizeOf(fuse.WriteIn)..];
+ if (rest.len < in.size) return error.BadRequest;
+ const n = try b.write(h.fid, in.offset, rest[0..in.size]);
+ const out = fuse.WriteOut{ .size = @intCast(n) };
+ try b.reply(u, &.{std.mem.asBytes(&out)});
+ },
+ .readdir => try b.readdir(req),
+ .release, .releasedir => {
+ const in = try body(fuse.ReleaseIn, req);
+ const kv = b.handles.fetchRemove(in.fh) orelse return error.BadHandle;
+ var h = kv.value;
+ if (h.dir) |*d| d.deinit(b.gpa);
+ try b.clunk(h.fid);
+ try b.reply(u, &.{});
+ },
+ .flush, .fsync, .fsyncdir => try b.reply(u, &.{}),
+ .create => try b.create(req),
+ .mkdir => {
+ const in = try body(fuse.MkdirIn, req);
+ const name = try nameAfter(fuse.MkdirIn, req);
+ const parent = b.inodes.get(req.header.nodeid) orelse return error.NoEntry;
+ const fid = try b.clone(parent.fid);
+ _ = b.create9(fid, name, cloud9.dmdir | (in.mode & 0o777), cloud9.oread) catch |e| {
+ b.clunkQuiet(fid);
+ return e;
+ };
+ try b.clunk(fid);
+ const entry = try b.lookupEntry(req.header.nodeid, name);
+ try b.reply(u, &.{std.mem.asBytes(&entry)});
+ },
+ .unlink, .rmdir => {
+ const name = try nameAfter(void, req);
+ const parent = b.inodes.get(req.header.nodeid) orelse return error.NoEntry;
+ const tmp = try b.walkName(parent.fid, name);
+ try b.remove(tmp);
+ try b.reply(u, &.{});
+ },
+ .rename => {
+ const in = try body(fuse.RenameIn, req);
+ const old = try nameAfter(fuse.RenameIn, req);
+ const new = try secondName(req, old, @sizeOf(fuse.RenameIn));
+ try b.rename(req.header.nodeid, in.newdir, old, new, 0);
+ try b.reply(u, &.{});
+ },
+ .rename2 => {
+ const in = try body(fuse.Rename2In, req);
+ const old = try nameAfter(fuse.Rename2In, req);
+ const new = try secondName(req, old, @sizeOf(fuse.Rename2In));
+ try b.rename(req.header.nodeid, in.newdir, old, new, in.flags);
+ try b.reply(u, &.{});
+ },
+ .statfs => {
+ const out = fuse.StatfsOut{ .st = .{ .bsize = 4096, .namelen = 255, .frsize = 4096 } };
+ try b.reply(u, &.{std.mem.asBytes(&out)});
+ },
+ .interrupt => {},
+ .destroy => unreachable, // handled in dispatch
+ .access => return error.NotSup,
+ else => return error.NotSup,
+ }
+ }
+
+ // -- handlers ----------------------------------------------------------------------
+
+ /// walk(parent → new fid, [name]) + stat, deduplicated by qid.path. Bumps nlookup.
+ fn lookupEntry(b: *Bridge, parent_id: u64, name: []const u8) HandlerError!fuse.EntryOut {
+ const parent = b.inodes.get(parent_id) orelse return error.NoEntry;
+ const newfid = try b.walkName(parent.fid, name);
+ const st = b.stat(newfid) catch |e| {
+ b.clunkQuiet(newfid);
+ return e;
+ };
+ const qid = st.qid;
+ var nodeid: u64 = undefined;
+ if (b.by_qid.get(qid.path)) |existing| {
+ // A directory and a file sharing a qid.path (a server bug) must not
+ // share a node: the kernel would mark the inode bad, and for the
+ // root that is fatal for the whole mount.
+ const merge = if (b.inodes.getPtr(existing)) |ino| (ino.qid.type & cloud9.qtdir) == (qid.type & cloud9.qtdir) else false;
+ if (merge) {
+ const ino = b.inodes.getPtr(existing).?;
+ ino.nlookup += 1;
+ ino.qid = qid;
+ nodeid = existing;
+ if (existing == fuse.root_id) {
+ b.clunkQuiet(newfid);
+ } else {
+ // Keep the fresh fid (it is bound to the current file at this
+ // name) and retire the older one.
+ const stale = ino.fid;
+ ino.fid = newfid;
+ b.clunkQuiet(stale);
+ }
+ } else {
+ // Stale reverse entry, or a type clash: bind a fresh node to it.
+ nodeid = try b.newInode(newfid, qid, parent_id);
+ }
+ } else {
+ nodeid = try b.newInode(newfid, qid, parent_id);
+ }
+ var out = fuse.EntryOut{
+ .nodeid = nodeid,
+ .generation = 0,
+ .attr = b.attrFrom(st, b.inoOf(nodeid, qid)),
+ };
+ out.entry_valid = b.opts.attr_timeout_ns / 1_000_000_000;
+ out.entry_valid_nsec = @intCast(b.opts.attr_timeout_ns % 1_000_000_000);
+ out.attr_valid = out.entry_valid;
+ out.attr_valid_nsec = out.entry_valid_nsec;
+ return out;
+ }
+
+ fn newInode(b: *Bridge, fid: u32, qid: cloud9.Qid, parent: u64) HandlerError!u64 {
+ const nodeid = b.next_node;
+ b.inodes.put(b.gpa, nodeid, .{ .fid = fid, .qid = qid, .nlookup = 1, .parent = parent }) catch |e| {
+ b.clunkQuiet(fid);
+ return e;
+ };
+ b.by_qid.put(b.gpa, qid.path, nodeid) catch |e| {
+ _ = b.inodes.remove(nodeid);
+ b.clunkQuiet(fid);
+ return e;
+ };
+ b.next_node += 1;
+ return nodeid;
+ }
+
+ fn forget(b: *Bridge, nodeid: u64, n: u64) HandlerError!void {
+ if (nodeid == fuse.root_id) return;
+ const ino = b.inodes.getPtr(nodeid) orelse return;
+ if (ino.nlookup > n) {
+ ino.nlookup -= n;
+ return;
+ }
+ const fid = ino.fid;
+ const path = ino.qid.path;
+ _ = b.inodes.remove(nodeid);
+ if (b.by_qid.get(path)) |mapped| {
+ if (mapped == nodeid) _ = b.by_qid.remove(path);
+ }
+ b.clunk(fid) catch |e| switch (e) {
+ error.Nine => {},
+ else => return e,
+ };
+ }
+
+ fn setattr(b: *Bridge, req: fuse.Request) HandlerError!void {
+ const in = try body(fuse.SetattrIn, req);
+ const ino = b.inodes.get(req.header.nodeid) orelse return error.NoEntry;
+ const old = try b.stat(ino.fid);
+ const old_mode = old.mode;
+
+ var st = nine.dontcare;
+ var changed = false;
+ if (in.valid & fuse.FATTR_UID != 0 and in.uid != b.opts.uid) return error.Perm;
+ if (in.valid & fuse.FATTR_GID != 0 and in.gid != b.opts.gid) return error.Perm;
+ if (in.valid & fuse.FATTR_SIZE != 0) {
+ st.length = in.size;
+ changed = true;
+ }
+ if (in.valid & fuse.FATTR_MODE != 0) {
+ st.mode = (old_mode & ~@as(u32, 0o777)) | (in.mode & 0o777);
+ changed = true;
+ }
+ if (in.valid & fuse.FATTR_MTIME_NOW != 0) {
+ st.mtime = nowSeconds();
+ changed = true;
+ } else if (in.valid & fuse.FATTR_MTIME != 0) {
+ st.mtime = @truncate(in.mtime);
+ changed = true;
+ }
+ if (changed) try b.wstat(ino.fid, st);
+ const fresh = try b.stat(ino.fid);
+ const out = b.attrOut(fresh, b.inoOf(req.header.nodeid, ino.qid));
+ try b.reply(req.header.unique, &.{std.mem.asBytes(&out)});
+ }
+
+ fn openFile(b: *Bridge, req: fuse.Request, is_dir: bool) HandlerError!void {
+ const in = try body(fuse.OpenIn, req);
+ const ino = b.inodes.get(req.header.nodeid) orelse return error.NoEntry;
+ const mode: u8 = if (is_dir) cloud9.oread else openMode(in.flags);
+ const fid = try b.clone(ino.fid);
+ _ = b.open9(fid, mode) catch |e| {
+ b.clunkQuiet(fid);
+ return e;
+ };
+ const fh = try b.newHandle(fid, req.header.nodeid);
+ const out = fuse.OpenOut{
+ .fh = fh,
+ .open_flags = if (!is_dir and b.opts.direct_io) fuse.FOPEN_DIRECT_IO else 0,
+ };
+ try b.reply(req.header.unique, &.{std.mem.asBytes(&out)});
+ }
+
+ fn newHandle(b: *Bridge, fid: u32, nodeid: u64) HandlerError!u64 {
+ const fh = b.next_fh;
+ b.handles.put(b.gpa, fh, .{ .fid = fid, .nodeid = nodeid, .dir = null }) catch |e| {
+ b.clunkQuiet(fid);
+ return e;
+ };
+ b.next_fh += 1;
+ return fh;
+ }
+
+ fn create(b: *Bridge, req: fuse.Request) HandlerError!void {
+ const in = try body(fuse.CreateIn, req);
+ const name = try nameAfter(fuse.CreateIn, req);
+ const parent = b.inodes.get(req.header.nodeid) orelse return error.NoEntry;
+ // The created fid becomes the open file.
+ const fid = try b.clone(parent.fid);
+ _ = b.create9(fid, name, in.mode & 0o777, openMode(in.flags)) catch |e| {
+ b.clunkQuiet(fid);
+ return e;
+ };
+ const entry = b.lookupEntry(req.header.nodeid, name) catch |e| {
+ b.clunkQuiet(fid);
+ return e;
+ };
+ const fh = try b.newHandle(fid, entry.nodeid);
+ const oo = fuse.OpenOut{
+ .fh = fh,
+ .open_flags = if (b.opts.direct_io) fuse.FOPEN_DIRECT_IO else 0,
+ };
+ try b.reply(req.header.unique, &.{ std.mem.asBytes(&entry), std.mem.asBytes(&oo) });
+ }
+
+ fn rename(b: *Bridge, parent_id: u64, newdir: u64, old: []const u8, new: []const u8, flags: u32) HandlerError!void {
+ if (newdir != parent_id) return error.Exdev;
+ const rf: linux.RENAME = @bitCast(flags);
+ if (rf.EXCHANGE or rf.WHITEOUT) return error.BadRequest;
+ const parent = b.inodes.get(parent_id) orelse return error.NoEntry;
+ const tmp = try b.walkName(parent.fid, old);
+ defer b.clunkQuiet(tmp);
+ var st = nine.dontcare;
+ st.name = new;
+ b.wstat(tmp, st) catch |e| {
+ // 9P2000 rename never replaces an existing name; POSIX rename does.
+ if (e != error.Nine or rf.NOREPLACE or b.nine.errno() != .EXIST) return e;
+ try b.renameOver(parent.fid, tmp, new);
+ };
+ }
+
+ /// Replace `new` with the file behind `src`. An (empty) directory target is
+ /// removed first: it holds no data and the VFS already ruled out mismatched
+ /// types. A file target is parked under a temporary name so that a failing
+ /// second rename can put it back instead of having destroyed it.
+ fn renameOver(b: *Bridge, parent_fid: u32, src: u32, new: []const u8) HandlerError!void {
+ const victim = try b.walkName(parent_fid, new);
+ const vst = b.stat(victim) catch |e| {
+ b.clunkQuiet(victim);
+ return e;
+ };
+ var st = nine.dontcare;
+ st.name = new;
+ if (vst.mode & cloud9.dmdir != 0) {
+ b.trace(" rename target is a directory; removing it and retrying", .{});
+ try b.remove(victim);
+ return b.wstat(src, st);
+ }
+ var park_buf: [48]u8 = undefined;
+ const park = std.fmt.bufPrint(&park_buf, ".9ns-rename-{x}", .{randomU64()}) catch unreachable;
+ b.trace(" rename target exists; parking it as {s} and retrying", .{park});
+ var pst = nine.dontcare;
+ pst.name = park;
+ b.wstat(victim, pst) catch |e| {
+ b.clunkQuiet(victim);
+ return e;
+ };
+ b.wstat(src, st) catch |e| {
+ b.trace(" rename still failed; restoring the target", .{});
+ const saved = b.last_err;
+ b.wstat(victim, st) catch {};
+ b.last_err = saved;
+ b.clunkQuiet(victim);
+ return e;
+ };
+ b.remove(victim) catch b.trace(" could not remove the parked target {s}", .{park});
+ }
+
+ fn readdir(b: *Bridge, req: fuse.Request) HandlerError!void {
+ const in = try body(fuse.ReadIn, req);
+ const h = b.handles.getPtr(in.fh) orelse return error.BadHandle;
+ if (in.offset == 0 or h.dir == null) {
+ if (h.dir) |*d| d.deinit(b.gpa);
+ h.dir = null;
+ h.dir = try b.loadDir(h.fid, h.nodeid);
+ }
+ const dir = &h.dir.?;
+ const size: usize = @min(in.size, max_write);
+ const used = packDirents(dir.entries.items, in.offset, b.data_buf[0..size]);
+ try b.reply(req.header.unique, &.{b.data_buf[0..used]});
+ }
+
+ /// Reads the whole directory and builds its listing, "." and ".." first.
+ fn loadDir(b: *Bridge, fid: u32, nodeid: u64) HandlerError!DirList {
+ var list: DirList = .{};
+ errdefer list.deinit(b.gpa);
+ const self_ino = b.inoOfNode(nodeid);
+ const parent_ino = if (b.inodes.get(nodeid)) |ino| b.inoOfNode(ino.parent) else self_ino;
+ try list.entries.append(b.gpa, .{ .name = try b.gpa.dupe(u8, "."), .ino = self_ino, .dtype = fuse.DT_DIR });
+ try list.entries.append(b.gpa, .{ .name = try b.gpa.dupe(u8, ".."), .ino = parent_ino, .dtype = fuse.DT_DIR });
+
+ var offset: u64 = 0;
+ while (true) {
+ // A server that ignores the offset would otherwise feed us forever.
+ if (offset >= max_dir_bytes) return error.BadDir;
+ const n = try b.read(fid, offset, b.data_buf);
+ if (n == 0) break;
+ try parseDirRecords(b.gpa, b.data_buf[0..n], &list);
+ offset += n;
+ }
+ // Entries carrying the root's own qid.path get the root's ino (1), as GETATTR would report it.
+ for (list.entries.items[2..]) |*e| if (e.ino == b.root_path) {
+ e.ino = fuse.root_id;
+ };
+ return list;
+ }
+
+ // -- 9P wrappers (tracing) -----------------------------------------------------------
+
+ fn stat(b: *Bridge, fid: u32) nine.Session.Error!cloud9.Stat {
+ const st = b.nine.stat(fid) catch |e| return b.nineErr("stat", fid, e);
+ b.trace(" 9p stat fid={d} -> name={s} mode={o} len={d} qid={x}", .{ fid, st.name, st.mode, st.length, st.qid.path });
+ return st;
+ }
+
+ fn walkName(b: *Bridge, fid: u32, name: []const u8) nine.Session.Error!u32 {
+ const newfid = b.nine.allocFid();
+ _ = b.nine.walk(fid, newfid, &.{name}) catch |e| {
+ b.nine.freeFid(newfid);
+ return b.nineErr("walk", fid, e);
+ };
+ b.trace(" 9p walk fid={d} newfid={d} name={s} -> ok", .{ fid, newfid, name });
+ return newfid;
+ }
+
+ fn clone(b: *Bridge, fid: u32) nine.Session.Error!u32 {
+ const newfid = b.nine.clone(fid) catch |e| return b.nineErr("clone", fid, e);
+ b.trace(" 9p walk fid={d} newfid={d} (clone) -> ok", .{ fid, newfid });
+ return newfid;
+ }
+
+ fn open9(b: *Bridge, fid: u32, mode: u8) nine.Session.Error!nine.Session.Open {
+ const o = b.nine.open(fid, mode) catch |e| return b.nineErr("open", fid, e);
+ b.trace(" 9p open fid={d} mode={d} -> iounit={d}", .{ fid, mode, o.iounit });
+ return o;
+ }
+
+ fn create9(b: *Bridge, fid: u32, name: []const u8, perm: u32, mode: u8) nine.Session.Error!nine.Session.Open {
+ const o = b.nine.create(fid, name, perm, mode) catch |e| return b.nineErr("create", fid, e);
+ b.trace(" 9p create fid={d} name={s} perm={o} mode={d} -> iounit={d}", .{ fid, name, perm, mode, o.iounit });
+ return o;
+ }
+
+ fn read(b: *Bridge, fid: u32, offset: u64, buf: []u8) nine.Session.Error!usize {
+ const n = b.nine.read(fid, offset, buf) catch |e| return b.nineErr("read", fid, e);
+ b.trace(" 9p read fid={d} offset={d} count={d} -> {d}", .{ fid, offset, buf.len, n });
+ return n;
+ }
+
+ fn write(b: *Bridge, fid: u32, offset: u64, data: []const u8) nine.Session.Error!usize {
+ const n = b.nine.write(fid, offset, data) catch |e| return b.nineErr("write", fid, e);
+ b.trace(" 9p write fid={d} offset={d} count={d} -> {d}", .{ fid, offset, data.len, n });
+ return n;
+ }
+
+ fn wstat(b: *Bridge, fid: u32, st: cloud9.Stat) nine.Session.Error!void {
+ b.nine.wstat(fid, st) catch |e| return b.nineErr("wstat", fid, e);
+ b.trace(" 9p wstat fid={d} name={s} mode={x} len={x} mtime={x} -> ok", .{ fid, st.name, st.mode, st.length, st.mtime });
+ }
+
+ fn clunk(b: *Bridge, fid: u32) nine.Session.Error!void {
+ b.nine.clunk(fid) catch |e| return b.nineErr("clunk", fid, e);
+ b.trace(" 9p clunk fid={d} -> ok", .{fid});
+ }
+
+ /// Best-effort clunk during error unwinding; a dead session surfaces on the
+ /// next call. Does not disturb the errno of the failure being unwound.
+ fn clunkQuiet(b: *Bridge, fid: u32) void {
+ const saved = b.last_err;
+ defer b.last_err = saved;
+ b.clunk(fid) catch {};
+ }
+
+ fn remove(b: *Bridge, fid: u32) nine.Session.Error!void {
+ b.nine.remove(fid) catch |e| return b.nineErr("remove", fid, e);
+ b.trace(" 9p remove fid={d} -> ok", .{fid});
+ }
+
+ fn nineErr(b: *Bridge, what: []const u8, fid: u32, e: nine.Session.Error) nine.Session.Error {
+ if (e == error.Nine) {
+ b.last_err = b.nine.errno();
+ b.trace(" 9p {s} fid={d} -> Rerror \"{s}\" ({s})", .{ what, fid, b.nine.ename[0..b.nine.ename_len], @tagName(b.nine.errno()) });
+ } else {
+ b.trace(" 9p {s} fid={d} -> {s}", .{ what, fid, @errorName(e) });
+ }
+ return e;
+ }
+
+ // -- FUSE wrappers (tracing) --------------------------------------------------------
+
+ fn reply(b: *Bridge, unique: u64, payloads: []const []const u8) error{FuseIo}!void {
+ var total: usize = 0;
+ for (payloads) |p| total += p.len;
+ b.trace("-> unique={d} ok ({d} bytes)", .{ unique, total });
+ fuse.reply(b.fuse_fd, unique, payloads) catch return error.FuseIo;
+ }
+
+ fn replyError(b: *Bridge, unique: u64, code: linux.E) error{FuseIo}!void {
+ b.trace("-> unique={d} error E{s}", .{ unique, @tagName(code) });
+ fuse.replyError(b.fuse_fd, unique, code) catch return error.FuseIo;
+ }
+
+ // -- attrs ---------------------------------------------------------------------------
+
+ fn inoOf(b: *const Bridge, nodeid: u64, qid: cloud9.Qid) u64 {
+ return if (nodeid == fuse.root_id or qid.path == b.root_path) fuse.root_id else qid.path;
+ }
+
+ fn inoOfNode(b: *const Bridge, nodeid: u64) u64 {
+ if (nodeid == fuse.root_id) return fuse.root_id;
+ const ino = b.inodes.get(nodeid) orelse return nodeid;
+ return b.inoOf(nodeid, ino.qid);
+ }
+
+ fn attrFrom(b: *const Bridge, st: cloud9.Stat, ino: u64) fuse.Attr {
+ return attrFromStat(st, ino, b.opts.uid, b.opts.gid);
+ }
+
+ fn attrOut(b: *const Bridge, st: cloud9.Stat, ino: u64) fuse.AttrOut {
+ return .{
+ .attr_valid = b.opts.attr_timeout_ns / 1_000_000_000,
+ .attr_valid_nsec = @intCast(b.opts.attr_timeout_ns % 1_000_000_000),
+ .attr = b.attrFrom(st, ino),
+ };
+ }
+};
+
+// -- pure helpers (unit-tested) ------------------------------------------------------------
+
+/// Attr from a 9P Stat: DMDIR → S_IFDIR else S_IFREG, low 9 permission bits kept.
+pub fn attrFromStat(st: cloud9.Stat, ino: u64, uid: u32, gid: u32) fuse.Attr {
+ const ftype: u32 = if (st.mode & cloud9.dmdir != 0) fuse.S_IFDIR else fuse.S_IFREG;
+ return .{
+ .ino = ino,
+ // The kernel marks an inode bad when size > LLONG_MAX; clamp hostile lengths.
+ .size = @min(st.length, std.math.maxInt(i64)),
+ // Saturating: a hostile length of 2^64-1 must not overflow.
+ .blocks = st.length / 512 + @intFromBool(st.length % 512 != 0),
+ .atime = st.atime,
+ .mtime = st.mtime,
+ .ctime = st.mtime,
+ .mode = ftype | (st.mode & 0o777),
+ .nlink = 1,
+ .uid = uid,
+ .gid = gid,
+ .blksize = 4096,
+ };
+}
+
+/// Kernel open(2) flags → 9P open mode. O_APPEND has no 9P equivalent and is ignored.
+pub fn openMode(flags: u32) u8 {
+ const o: linux.O = @bitCast(flags);
+ var mode: u8 = switch (o.ACCMODE) {
+ .RDONLY => cloud9.oread,
+ .WRONLY => cloud9.owrite,
+ .RDWR => cloud9.ordwr,
+ };
+ if (o.TRUNC) mode |= cloud9.otrunc;
+ return mode;
+}
+
+/// Parses consecutive 9P directory records (2-byte size + Stat) and appends entries.
+pub fn parseDirRecords(gpa: std.mem.Allocator, bytes: []const u8, list: *DirList) error{ OutOfMemory, BadDir }!void {
+ var pos: usize = 0;
+ while (pos < bytes.len) {
+ if (bytes.len - pos < 2) return error.BadDir;
+ const size: usize = std.mem.readInt(u16, bytes[pos..][0..2], .little);
+ if (bytes.len - pos < 2 + size) return error.BadDir;
+ const st = cloud9.Stat.decode(bytes[pos..][0 .. 2 + size]) catch return error.BadDir;
+ pos += 2 + size;
+ // The kernel rejects a whole READDIR reply (EIO) over one bad name, and
+ // "." and ".." are synthesised by loadDir: drop such records instead.
+ if (!validDirentName(st.name)) continue;
+ const name = try gpa.dupe(u8, st.name);
+ errdefer gpa.free(name);
+ try list.entries.append(gpa, .{
+ .name = name,
+ .ino = st.qid.path,
+ .dtype = if (st.mode & cloud9.dmdir != 0) fuse.DT_DIR else fuse.DT_REG,
+ });
+ }
+}
+
+/// A name the kernel will accept in a dirent and that does not duplicate the synthetic "." / "..".
+pub fn validDirentName(name: []const u8) bool {
+ if (name.len == 0 or name.len > max_name_len) return false;
+ if (std.mem.indexOfAny(u8, name, "/\x00") != null) return false;
+ if (std.mem.eql(u8, name, ".") or std.mem.eql(u8, name, "..")) return false;
+ return true;
+}
+
+/// Packs dirents from `entries[offset..]` into `buf`; each record's `off` is its index + 1.
+/// Returns the number of bytes used.
+pub fn packDirents(entries: []const Entry, offset: u64, buf: []u8) usize {
+ var used: usize = 0;
+ var i: usize = @intCast(@min(offset, entries.len));
+ while (i < entries.len) : (i += 1) {
+ const e = entries[i];
+ if (!fuse.addDirent(buf, &used, e.ino, @as(u64, i) + 1, e.dtype, e.name)) break;
+ }
+ return used;
+}
+
+fn randomU64() u64 {
+ var bytes: [8]u8 = undefined;
+ if (linux.errno(linux.getrandom(&bytes, bytes.len, 0)) == .SUCCESS) return std.mem.readInt(u64, &bytes, .little);
+ var ts: linux.timespec = undefined;
+ _ = linux.clock_gettime(.MONOTONIC, &ts);
+ return @as(u64, @bitCast(ts.nsec)) ^ (@as(u64, @bitCast(ts.sec)) << 32);
+}
+
+fn nowSeconds() u32 {
+ var ts: linux.timespec = undefined;
+ if (linux.errno(linux.clock_gettime(.REALTIME, &ts)) != .SUCCESS) return 0;
+ return @intCast(@as(u64, @intCast(ts.sec)) & 0xFFFF_FFFF);
+}
+
+fn opName(op: fuse.Opcode) []const u8 {
+ return switch (op) {
+ _ => "unknown",
+ else => @tagName(op),
+ };
+}
+
+// Thin adapters so fuse.zig's parse errors become HandlerError.BadRequest.
+fn body(comptime T: type, req: fuse.Request) error{BadRequest}!*const T {
+ return fuse.body(T, req) catch error.BadRequest;
+}
+
+fn nameAfter(comptime T: type, req: fuse.Request) error{BadRequest}![]const u8 {
+ return fuse.nameAfter(T, req) catch error.BadRequest;
+}
+
+fn secondName(req: fuse.Request, first: []const u8, offset: usize) error{BadRequest}![]const u8 {
+ return fuse.secondName(req, first, offset) catch error.BadRequest;
+}
+
+// -- tests ------------------------------------------------------------------------------
+
+const testing = std.testing;
+
+test {
+ // Force semantic analysis of `serve` and the whole dispatch path, which no
+ // unit test can exercise without a FUSE mount.
+ testing.refAllDecls(@This());
+}
+
+fn testStat(name: []const u8, mode: u32, length: u64, path: u64) cloud9.Stat {
+ return .{
+ .type = 0,
+ .dev = 0,
+ .qid = .{ .type = if (mode & cloud9.dmdir != 0) cloud9.qtdir else 0, .version = 0, .path = path },
+ .mode = mode,
+ .atime = 100,
+ .mtime = 200,
+ .length = length,
+ .name = name,
+ .uid = "u",
+ .gid = "g",
+ .muid = "u",
+ };
+}
+
+test "attr mapping: DMDIR → S_IFDIR|perm, length → size/blocks" {
+ const d = attrFromStat(testStat("d", cloud9.dmdir | 0o755, 0, 9), 9, 1000, 1001);
+ try testing.expectEqual(fuse.S_IFDIR | 0o755, d.mode);
+ try testing.expectEqual(@as(u64, 9), d.ino);
+ try testing.expectEqual(@as(u64, 0), d.size);
+ try testing.expectEqual(@as(u64, 0), d.blocks);
+ try testing.expectEqual(@as(u32, 1000), d.uid);
+ try testing.expectEqual(@as(u32, 1001), d.gid);
+ try testing.expectEqual(@as(u32, 1), d.nlink);
+
+ const f = attrFromStat(testStat("f", 0o640 | cloud9.dmappend, 1025, 4), 4, 0, 0);
+ try testing.expectEqual(fuse.S_IFREG | 0o640, f.mode); // dmappend bit not leaked
+ try testing.expectEqual(@as(u64, 1025), f.size);
+ try testing.expectEqual(@as(u64, 3), f.blocks);
+ try testing.expectEqual(@as(u32, 4096), f.blksize);
+ try testing.expectEqual(@as(u64, 100), f.atime);
+ try testing.expectEqual(@as(u64, 200), f.mtime);
+ try testing.expectEqual(@as(u64, 200), f.ctime);
+
+ try testing.expectEqual(@as(u64, 1), attrFromStat(testStat("f", 0o600, 512, 4), 4, 0, 0).blocks);
+ try testing.expectEqual(@as(u64, 2), attrFromStat(testStat("f", 0o600, 513, 4), 4, 0, 0).blocks);
+}
+
+test "open flag → 9P mode mapping" {
+ const rdonly: u32 = @bitCast(linux.O{ .ACCMODE = .RDONLY });
+ const wronly: u32 = @bitCast(linux.O{ .ACCMODE = .WRONLY });
+ const rdwr: u32 = @bitCast(linux.O{ .ACCMODE = .RDWR });
+ const trunc: u32 = @bitCast(linux.O{ .TRUNC = true });
+ const append: u32 = @bitCast(linux.O{ .APPEND = true });
+ const creat: u32 = @bitCast(linux.O{ .CREAT = true });
+ try testing.expectEqual(cloud9.oread, openMode(rdonly));
+ try testing.expectEqual(cloud9.owrite, openMode(wronly));
+ try testing.expectEqual(cloud9.ordwr, openMode(rdwr));
+ try testing.expectEqual(cloud9.owrite | cloud9.otrunc, openMode(wronly | trunc));
+ try testing.expectEqual(cloud9.ordwr | cloud9.otrunc, openMode(rdwr | trunc | creat));
+ try testing.expectEqual(cloud9.owrite, openMode(wronly | append)); // O_APPEND ignored
+}
+
+test "dirlist parsing from two hand-encoded Stat records" {
+ var buf: [512]u8 = undefined;
+ const a = try cloud9.Stat.encode(testStat("alpha", 0o644, 10, 0x11), &buf);
+ const bb = try cloud9.Stat.encode(testStat("beta", cloud9.dmdir | 0o755, 0, 0x22), buf[a.len..]);
+ const bytes = buf[0 .. a.len + bb.len];
+ // Sanity: the record is prefixed by its own 2-byte size.
+ try testing.expectEqual(a.len - 2, std.mem.readInt(u16, bytes[0..2], .little));
+
+ var list: DirList = .{};
+ defer list.deinit(testing.allocator);
+ try parseDirRecords(testing.allocator, bytes, &list);
+ try testing.expectEqual(@as(usize, 2), list.entries.items.len);
+ try testing.expectEqualStrings("alpha", list.entries.items[0].name);
+ try testing.expectEqual(@as(u64, 0x11), list.entries.items[0].ino);
+ try testing.expectEqual(fuse.DT_REG, list.entries.items[0].dtype);
+ try testing.expectEqualStrings("beta", list.entries.items[1].name);
+ try testing.expectEqual(@as(u64, 0x22), list.entries.items[1].ino);
+ try testing.expectEqual(fuse.DT_DIR, list.entries.items[1].dtype);
+
+ // Truncated input is a protocol error and leaves earlier entries intact.
+ try testing.expectError(error.BadDir, parseDirRecords(testing.allocator, bytes[0 .. bytes.len - 1], &list));
+ try testing.expectEqual(@as(usize, 3), list.entries.items.len);
+}
+
+test "readdir packing and offset resumption" {
+ const names = [_][]const u8{ ".", "..", "one", "two", "three" };
+ var entries: [names.len]Entry = undefined;
+ for (&entries, names, 0..) |*e, n, i| e.* = .{ .name = @constCast(n), .ino = 100 + i, .dtype = if (i < 2) fuse.DT_DIR else fuse.DT_REG };
+
+ // Everything fits: five records, off = index + 1.
+ var big: [1024]u8 = undefined;
+ const used = packDirents(&entries, 0, &big);
+ var pos: usize = 0;
+ var idx: usize = 0;
+ while (pos < used) : (idx += 1) {
+ const d = std.mem.bytesToValue(fuse.Dirent, big[pos..][0..@sizeOf(fuse.Dirent)]);
+ try testing.expectEqual(@as(u64, 100 + idx), d.ino);
+ try testing.expectEqual(@as(u64, idx + 1), d.off);
+ try testing.expectEqualStrings(names[idx], big[pos + @sizeOf(fuse.Dirent) ..][0..d.namelen]);
+ pos += (@sizeOf(fuse.Dirent) + d.namelen + 7) & ~@as(usize, 7);
+ }
+ try testing.expectEqual(names.len, idx);
+
+ // A buffer that fits exactly two records ("." = 32, ".." = 32) stops there…
+ var small: [64]u8 = undefined;
+ const first_used = packDirents(&entries, 0, &small);
+ try testing.expectEqual(@as(usize, 64), first_used);
+ const last = std.mem.bytesToValue(fuse.Dirent, small[32..][0..@sizeOf(fuse.Dirent)]);
+ try testing.expectEqual(@as(u64, 2), last.off);
+ // …and resuming at the last `off` yields "one" next.
+ const second_used = packDirents(&entries, last.off, &small);
+ const next = std.mem.bytesToValue(fuse.Dirent, small[0..@sizeOf(fuse.Dirent)]);
+ try testing.expectEqualStrings("one", small[@sizeOf(fuse.Dirent)..][0..next.namelen]);
+ try testing.expectEqual(@as(u64, 3), next.off);
+ try testing.expect(second_used > 0);
+
+ // Past the end: nothing (EOF for the kernel).
+ try testing.expectEqual(@as(usize, 0), packDirents(&entries, names.len, &big));
+ try testing.expectEqual(@as(usize, 0), packDirents(&entries, 1000, &big));
+}
+
+test "attr mapping saturates hostile lengths instead of overflowing" {
+ const a = attrFromStat(testStat("f", 0o600, std.math.maxInt(u64), 4), 4, 0, 0);
+ try testing.expectEqual(@as(u64, std.math.maxInt(i64)), a.size);
+ try testing.expectEqual(@as(u64, std.math.maxInt(u64) / 512 + 1), a.blocks);
+ const b = attrFromStat(testStat("f", 0o600, 1024, 4), 4, 0, 0);
+ try testing.expectEqual(@as(u64, 2), b.blocks);
+ try testing.expectEqual(@as(u64, 1024), b.size);
+}
+
+test "dirent names the kernel would reject are dropped from listings" {
+ try testing.expect(validDirentName("a"));
+ try testing.expect(validDirentName("x" ** 1024));
+ try testing.expect(!validDirentName(""));
+ try testing.expect(!validDirentName("a/b"));
+ try testing.expect(!validDirentName("a\x00b"));
+ try testing.expect(!validDirentName("."));
+ try testing.expect(!validDirentName(".."));
+ try testing.expect(!validDirentName("x" ** 1025));
+
+ var buf: [4096]u8 = undefined;
+ var n: usize = 0;
+ for ([_][]const u8{ ".", "..", "", "a/b", "keep", "x" ** 1025, "also" }) |name| {
+ n += (try cloud9.Stat.encode(testStat(name, 0o644, 1, 0x30), buf[n..])).len;
+ }
+ var list: DirList = .{};
+ defer list.deinit(testing.allocator);
+ try parseDirRecords(testing.allocator, buf[0..n], &list);
+ try testing.expectEqual(@as(usize, 2), list.entries.items.len);
+ try testing.expectEqualStrings("keep", list.entries.items[0].name);
+ try testing.expectEqualStrings("also", list.entries.items[1].name);
+}
+
+test "DirList frees its names" {
+ var list: DirList = .{};
+ try list.entries.append(testing.allocator, .{ .name = try testing.allocator.dupe(u8, "x"), .ino = 1, .dtype = fuse.DT_REG });
+ list.deinit(testing.allocator);
+}
diff --git a/9ns/src/fuse.zig b/9ns/src/fuse.zig
new file mode 100644
index 0000000..682b3d5
--- /dev/null
+++ b/9ns/src/fuse.zig
@@ -0,0 +1,653 @@
+//! Kernel FUSE protocol subset (no libfuse, no libc, no policy).
+//!
+//! Extern structs mirror `/usr/include/linux/fuse.h` (kernel header 7.45);
+//! every layout is checked against the header's size at comptime. Only the
+//! opcodes and structs 9ns needs are here. The I/O helpers are blocking
+//! and allocation-free: the caller owns a single request buffer.
+//!
+//! Wire rules worth remembering:
+//! * The kernel delivers exactly one request per `read(2)` on `/dev/fuse`,
+//! and a reply must be exactly one `write(2)`/`writev(2)`.
+//! * Request bodies start right after the 40-byte `InHeader`; since every
+//! in-struct is 8-byte aligned in the header, `body()` requires the caller's
+//! buffer to be 8-byte aligned (`std.heap` page allocations and
+//! `align(8)` arrays both qualify).
+//! * A write that fails with `ENOENT` means the request was interrupted and
+//! the kernel already forgot it: the reply is silently dropped.
+//! * `ENODEV` on read means the filesystem was unmounted.
+
+const std = @import("std");
+const linux = std.os.linux;
+
+pub const kernel_version: u32 = 7;
+/// The minor we answer; the kernel adapts to the lower of the two.
+pub const kernel_minor: u32 = 31;
+pub const root_id: u64 = 1;
+
+pub const FOPEN_DIRECT_IO: u32 = 1 << 0;
+pub const FOPEN_KEEP_CACHE: u32 = 1 << 1;
+pub const FOPEN_NONSEEKABLE: u32 = 1 << 2;
+
+pub const FUSE_ASYNC_READ: u32 = 1 << 0;
+/// The kernel passes O_TRUNC in OPEN instead of a separate SETATTR(size=0); 9P has OTRUNC for exactly this.
+pub const FUSE_ATOMIC_O_TRUNC: u32 = 1 << 3;
+/// Without this the kernel's cached-write path (`--no-direct-io`) sends one 4 KiB WRITE per page.
+pub const FUSE_BIG_WRITES: u32 = 1 << 5;
+/// Re-fetch a cached inode's size/mtime and drop stale pages when they change.
+/// Required for `--no-direct-io` correctness: 9P sizes change under us, and
+/// without this the kernel trusts a stale cached size and truncates reads.
+pub const FUSE_AUTO_INVAL_DATA: u32 = 1 << 12;
+pub const FUSE_MAX_PAGES: u32 = 1 << 22;
+
+pub const FATTR_MODE: u32 = 1 << 0;
+pub const FATTR_UID: u32 = 1 << 1;
+pub const FATTR_GID: u32 = 1 << 2;
+pub const FATTR_SIZE: u32 = 1 << 3;
+pub const FATTR_ATIME: u32 = 1 << 4;
+pub const FATTR_MTIME: u32 = 1 << 5;
+pub const FATTR_FH: u32 = 1 << 6;
+pub const FATTR_ATIME_NOW: u32 = 1 << 7;
+pub const FATTR_MTIME_NOW: u32 = 1 << 8;
+pub const FATTR_LOCKOWNER: u32 = 1 << 9;
+pub const FATTR_CTIME: u32 = 1 << 10;
+
+/// File type bits for `Attr.mode` and `Dirent.type` (used by the bridge).
+pub const S_IFDIR: u32 = linux.S.IFDIR;
+pub const S_IFREG: u32 = linux.S.IFREG;
+pub const DT_DIR: u32 = linux.DT.DIR;
+pub const DT_REG: u32 = linux.DT.REG;
+
+pub const Opcode = enum(u32) {
+ lookup = 1,
+ forget = 2,
+ getattr = 3,
+ setattr = 4,
+ readlink = 5,
+ symlink = 6,
+ mknod = 8,
+ mkdir = 9,
+ unlink = 10,
+ rmdir = 11,
+ rename = 12,
+ link = 13,
+ open = 14,
+ read = 15,
+ write = 16,
+ statfs = 17,
+ release = 18,
+ fsync = 20,
+ setxattr = 21,
+ getxattr = 22,
+ listxattr = 23,
+ removexattr = 24,
+ flush = 25,
+ init = 26,
+ opendir = 27,
+ readdir = 28,
+ releasedir = 29,
+ fsyncdir = 30,
+ getlk = 31,
+ setlk = 32,
+ setlkw = 33,
+ access = 34,
+ create = 35,
+ interrupt = 36,
+ bmap = 37,
+ destroy = 38,
+ ioctl = 39,
+ poll = 40,
+ notify_reply = 41,
+ batch_forget = 42,
+ fallocate = 43,
+ readdirplus = 44,
+ rename2 = 45,
+ lseek = 46,
+ copy_file_range = 47,
+ setupmapping = 48,
+ removemapping = 49,
+ syncfs = 50,
+ tmpfile = 51,
+ statx = 52,
+ _,
+};
+
+// ---------------------------------------------------------------------------
+// Structs (field order and widths follow linux/fuse.h exactly)
+// ---------------------------------------------------------------------------
+
+pub const InHeader = extern struct {
+ len: u32,
+ opcode: u32,
+ unique: u64,
+ nodeid: u64,
+ uid: u32,
+ gid: u32,
+ pid: u32,
+ total_extlen: u16,
+ padding: u16,
+
+ pub fn op(h: InHeader) Opcode {
+ return @enumFromInt(h.opcode);
+ }
+};
+
+pub const OutHeader = extern struct {
+ len: u32,
+ @"error": i32,
+ unique: u64,
+};
+
+pub const Attr = extern struct {
+ ino: u64 = 0,
+ size: u64 = 0,
+ blocks: u64 = 0,
+ atime: u64 = 0,
+ mtime: u64 = 0,
+ ctime: u64 = 0,
+ atimensec: u32 = 0,
+ mtimensec: u32 = 0,
+ ctimensec: u32 = 0,
+ mode: u32 = 0,
+ nlink: u32 = 0,
+ uid: u32 = 0,
+ gid: u32 = 0,
+ rdev: u32 = 0,
+ blksize: u32 = 0,
+ flags: u32 = 0,
+};
+
+pub const EntryOut = extern struct {
+ nodeid: u64 = 0,
+ generation: u64 = 0,
+ entry_valid: u64 = 0,
+ attr_valid: u64 = 0,
+ entry_valid_nsec: u32 = 0,
+ attr_valid_nsec: u32 = 0,
+ attr: Attr = .{},
+};
+
+pub const AttrOut = extern struct {
+ attr_valid: u64 = 0,
+ attr_valid_nsec: u32 = 0,
+ dummy: u32 = 0,
+ attr: Attr = .{},
+};
+
+pub const GetattrIn = extern struct { getattr_flags: u32, dummy: u32, fh: u64 };
+
+pub const SetattrIn = extern struct {
+ valid: u32,
+ padding: u32,
+ fh: u64,
+ size: u64,
+ lock_owner: u64,
+ atime: u64,
+ mtime: u64,
+ ctime: u64,
+ atimensec: u32,
+ mtimensec: u32,
+ ctimensec: u32,
+ mode: u32,
+ unused4: u32,
+ uid: u32,
+ gid: u32,
+ unused5: u32,
+};
+
+pub const OpenIn = extern struct { flags: u32, open_flags: u32 };
+pub const OpenOut = extern struct { fh: u64 = 0, open_flags: u32 = 0, backing_id: i32 = 0 };
+pub const ReleaseIn = extern struct { fh: u64, flags: u32, release_flags: u32, lock_owner: u64 };
+pub const FlushIn = extern struct { fh: u64, unused: u32, padding: u32, lock_owner: u64 };
+
+pub const ReadIn = extern struct {
+ fh: u64,
+ offset: u64,
+ size: u32,
+ read_flags: u32,
+ lock_owner: u64,
+ flags: u32,
+ padding: u32,
+};
+
+pub const WriteIn = extern struct {
+ fh: u64,
+ offset: u64,
+ size: u32,
+ write_flags: u32,
+ lock_owner: u64,
+ flags: u32,
+ padding: u32,
+};
+
+pub const WriteOut = extern struct { size: u32, padding: u32 = 0 };
+pub const CreateIn = extern struct { flags: u32, mode: u32, umask: u32, open_flags: u32 };
+pub const MkdirIn = extern struct { mode: u32, umask: u32 };
+pub const RenameIn = extern struct { newdir: u64 };
+pub const Rename2In = extern struct { newdir: u64, flags: u32, padding: u32 };
+pub const ForgetIn = extern struct { nlookup: u64 };
+pub const BatchForgetIn = extern struct { count: u32, dummy: u32 };
+pub const ForgetOne = extern struct { nodeid: u64, nlookup: u64 };
+pub const FsyncIn = extern struct { fh: u64, fsync_flags: u32, padding: u32 };
+pub const AccessIn = extern struct { mask: u32, padding: u32 };
+pub const InterruptIn = extern struct { unique: u64 };
+pub const LseekIn = extern struct { fh: u64, offset: u64, whence: u32, padding: u32 };
+
+pub const Kstatfs = extern struct {
+ blocks: u64 = 0,
+ bfree: u64 = 0,
+ bavail: u64 = 0,
+ files: u64 = 0,
+ ffree: u64 = 0,
+ bsize: u32 = 0,
+ namelen: u32 = 0,
+ frsize: u32 = 0,
+ padding: u32 = 0,
+ spare: [6]u32 = [_]u32{0} ** 6,
+};
+
+pub const StatfsOut = extern struct { st: Kstatfs = .{} };
+
+pub const InitIn = extern struct {
+ major: u32,
+ minor: u32,
+ max_readahead: u32,
+ flags: u32,
+ flags2: u32,
+ unused: [11]u32,
+};
+
+/// 64 bytes; the kernel accepts this size whenever the answered minor >= 23.
+pub const InitOut = extern struct {
+ major: u32 = kernel_version,
+ minor: u32 = kernel_minor,
+ max_readahead: u32 = 0,
+ flags: u32 = 0,
+ max_background: u16 = 0,
+ congestion_threshold: u16 = 0,
+ max_write: u32 = 0,
+ time_gran: u32 = 0,
+ max_pages: u16 = 0,
+ map_alignment: u16 = 0,
+ flags2: u32 = 0,
+ max_stack_depth: u32 = 0,
+ request_timeout: u16 = 0,
+ unused: [11]u16 = [_]u16{0} ** 11,
+};
+
+/// Fixed 24-byte head of `fuse_dirent`; the name follows, padded to 8 bytes.
+pub const Dirent = extern struct { ino: u64, off: u64, namelen: u32, type: u32 };
+
+comptime {
+ std.debug.assert(@sizeOf(InHeader) == 40);
+ std.debug.assert(@sizeOf(OutHeader) == 16);
+ std.debug.assert(@sizeOf(Attr) == 88);
+ std.debug.assert(@sizeOf(EntryOut) == 128);
+ std.debug.assert(@sizeOf(AttrOut) == 104);
+ std.debug.assert(@sizeOf(GetattrIn) == 16);
+ std.debug.assert(@sizeOf(SetattrIn) == 88);
+ std.debug.assert(@sizeOf(OpenIn) == 8);
+ std.debug.assert(@sizeOf(OpenOut) == 16);
+ std.debug.assert(@sizeOf(ReleaseIn) == 24);
+ std.debug.assert(@sizeOf(FlushIn) == 24);
+ std.debug.assert(@sizeOf(ReadIn) == 40);
+ std.debug.assert(@sizeOf(WriteIn) == 40);
+ std.debug.assert(@sizeOf(WriteOut) == 8);
+ std.debug.assert(@sizeOf(CreateIn) == 16);
+ std.debug.assert(@sizeOf(MkdirIn) == 8);
+ std.debug.assert(@sizeOf(RenameIn) == 8);
+ std.debug.assert(@sizeOf(Rename2In) == 16);
+ std.debug.assert(@sizeOf(ForgetIn) == 8);
+ std.debug.assert(@sizeOf(BatchForgetIn) == 8);
+ std.debug.assert(@sizeOf(ForgetOne) == 16);
+ std.debug.assert(@sizeOf(FsyncIn) == 16);
+ std.debug.assert(@sizeOf(AccessIn) == 8);
+ std.debug.assert(@sizeOf(InterruptIn) == 8);
+ std.debug.assert(@sizeOf(Kstatfs) == 80);
+ std.debug.assert(@sizeOf(StatfsOut) == 80);
+ std.debug.assert(@sizeOf(InitIn) == 64);
+ std.debug.assert(@sizeOf(InitOut) == 64);
+ std.debug.assert(@sizeOf(Dirent) == 24);
+ std.debug.assert(@sizeOf(LseekIn) == 24);
+}
+
+// ---------------------------------------------------------------------------
+// Request / reply helpers
+// ---------------------------------------------------------------------------
+
+pub const Error = error{ Protocol, Io, TooManyPayloads };
+
+pub const Request = struct {
+ header: InHeader,
+ /// Bytes after the header; a slice into the caller's buffer.
+ body: []const u8,
+};
+
+/// Reads one kernel request with a single `read(2)`. Returns null on ENODEV
+/// (unmounted). Retries EINTR/EAGAIN/ENOENT. `buf` should be at least
+/// `max_write + 4096` bytes and 8-byte aligned so `body()` can view it.
+pub fn readRequest(fd: i32, buf: []u8) Error!?Request {
+ while (true) {
+ const rc = linux.read(fd, buf.ptr, buf.len);
+ switch (linux.errno(rc)) {
+ .SUCCESS => {
+ const n: usize = rc;
+ if (n < @sizeOf(InHeader)) return error.Protocol;
+ const header = std.mem.bytesToValue(InHeader, buf[0..@sizeOf(InHeader)]);
+ if (header.len != n) return error.Protocol;
+ return .{ .header = header, .body = buf[@sizeOf(InHeader)..n] };
+ },
+ .INTR, .AGAIN, .NOENT => continue,
+ .NODEV => return null,
+ else => return error.Io,
+ }
+ }
+}
+
+/// Maximum number of payload slices a single `reply` can carry.
+pub const max_payloads = 7;
+
+/// Success reply: `OutHeader` followed by the concatenated `payloads`, sent in
+/// one `writev(2)`. An ENOENT from the kernel means the request was
+/// interrupted; the reply is dropped and this returns normally.
+pub fn reply(fd: i32, unique: u64, payloads: []const []const u8) Error!void {
+ if (payloads.len > max_payloads) return error.TooManyPayloads;
+ var total: usize = @sizeOf(OutHeader);
+ for (payloads) |p| total += p.len;
+ if (total > std.math.maxInt(u32)) return error.Protocol;
+ const header = OutHeader{ .len = @intCast(total), .@"error" = 0, .unique = unique };
+ var iov: [max_payloads + 1]std.posix.iovec_const = undefined;
+ iov[0] = .{ .base = @ptrCast(&header), .len = @sizeOf(OutHeader) };
+ for (payloads, 1..) |p, i| iov[i] = .{ .base = p.ptr, .len = p.len };
+ return writeAll(fd, &iov, payloads.len + 1, total);
+}
+
+/// Error reply: an `OutHeader` carrying `-errno` and no payload.
+pub fn replyError(fd: i32, unique: u64, err: linux.E) Error!void {
+ const code: i32 = @intCast(@intFromEnum(err));
+ const header = OutHeader{ .len = @sizeOf(OutHeader), .@"error" = -code, .unique = unique };
+ var iov = [_]std.posix.iovec_const{.{ .base = @ptrCast(&header), .len = @sizeOf(OutHeader) }};
+ return writeAll(fd, &iov, 1, @sizeOf(OutHeader));
+}
+
+fn writeAll(fd: i32, iov: [*]const std.posix.iovec_const, count: usize, total: usize) Error!void {
+ while (true) {
+ const rc = linux.writev(fd, iov, count);
+ switch (linux.errno(rc)) {
+ .SUCCESS => return if (rc == total) {} else error.Protocol,
+ .INTR => continue,
+ .NOENT => return, // request was interrupted; reply dropped
+ else => return error.Io,
+ }
+ }
+}
+
+/// Appends a `fuse_dirent` (head + name, padded to a multiple of 8) at
+/// `buf[used.*..]`. Returns false and leaves `buf`/`used` unchanged if the
+/// record does not fit.
+pub fn addDirent(buf: []u8, used: *usize, ino: u64, off: u64, dtype: u32, name: []const u8) bool {
+ const raw = @sizeOf(Dirent) + name.len;
+ const rec = (raw + 7) & ~@as(usize, 7);
+ if (used.* > buf.len or buf.len - used.* < rec) return false;
+ const dst = buf[used.*..][0..rec];
+ const head = Dirent{ .ino = ino, .off = off, .namelen = @intCast(name.len), .type = dtype };
+ @memcpy(dst[0..@sizeOf(Dirent)], std.mem.asBytes(&head));
+ @memcpy(dst[@sizeOf(Dirent)..raw], name);
+ @memset(dst[raw..rec], 0);
+ used.* += rec;
+ return true;
+}
+
+/// Views the first `@sizeOf(T)` bytes of `req.body` as `T` (copy-free).
+/// Fails with `error.Protocol` if the body is too short or misaligned.
+pub fn body(comptime T: type, req: Request) Error!*const T {
+ if (req.body.len < @sizeOf(T)) return error.Protocol;
+ if (@intFromPtr(req.body.ptr) % @alignOf(T) != 0) return error.Protocol;
+ return @ptrCast(@alignCast(req.body.ptr));
+}
+
+/// The NUL-terminated string at `req.body[offset..]`, without the NUL.
+pub fn nameAt(req: Request, offset: usize) Error![]const u8 {
+ if (offset > req.body.len) return error.Protocol;
+ const rest = req.body[offset..];
+ const end = std.mem.indexOfScalar(u8, rest, 0) orelse return error.Protocol;
+ return rest[0..end];
+}
+
+/// The NUL-terminated string following a `T` body (or at offset 0 when
+/// `T == void`), e.g. LOOKUP's name (`void`) or MKDIR's name (`MkdirIn`).
+pub fn nameAfter(comptime T: type, req: Request) Error![]const u8 {
+ const offset = if (T == void) 0 else @sizeOf(T);
+ return nameAt(req, offset);
+}
+
+/// The string that follows `first` (obtained via `nameAt(req, offset)`),
+/// for "old\0new\0" pairs such as RENAME's.
+pub fn secondName(req: Request, first: []const u8, offset: usize) Error![]const u8 {
+ return nameAt(req, offset + first.len + 1);
+}
+
+/// Builds the INIT reply per docs/DESIGN.md.
+pub fn initReply(in: *const InitIn, max_write: u32) InitOut {
+ var out = InitOut{
+ .major = kernel_version,
+ .minor = @min(kernel_minor, in.minor),
+ .max_readahead = in.max_readahead,
+ .flags = FUSE_ASYNC_READ | FUSE_ATOMIC_O_TRUNC | FUSE_BIG_WRITES | FUSE_AUTO_INVAL_DATA,
+ .max_background = 16,
+ .congestion_threshold = 12,
+ .max_write = max_write,
+ .time_gran = 1,
+ };
+ if (in.flags & FUSE_MAX_PAGES != 0) {
+ out.flags |= FUSE_MAX_PAGES;
+ out.max_pages = 256;
+ }
+ return out;
+}
+
+// ---------------------------------------------------------------------------
+// Tests
+// ---------------------------------------------------------------------------
+
+const testing = std.testing;
+
+test "struct sizes match linux/fuse.h" {
+ // The comptime block above is the real check; this makes it run under
+ // `zig test` even if the module is otherwise unreferenced.
+ try testing.expectEqual(@as(usize, 40), @sizeOf(InHeader));
+ try testing.expectEqual(@as(usize, 64), @sizeOf(InitOut));
+ try testing.expectEqual(@as(usize, 24), @sizeOf(Dirent));
+ try testing.expectEqual(@as(u32, 26), @intFromEnum(Opcode.init));
+ try testing.expectEqual(Opcode.statx, @as(Opcode, @enumFromInt(52)));
+}
+
+test "addDirent pads records to 8 bytes and refuses when full" {
+ var buf: [1024]u8 = undefined;
+ var used: usize = 0;
+ const name = "abcdefghijklmnopq"; // 17 chars
+ var expect_total: usize = 0;
+ var n: usize = 1;
+ while (n <= 17) : (n += 1) {
+ const before = used;
+ try testing.expect(addDirent(&buf, &used, n, n, DT_REG, name[0..n]));
+ const rec = used - before;
+ try testing.expectEqual(@as(usize, 0), rec % 8);
+ try testing.expectEqual((24 + n + 7) & ~@as(usize, 7), rec);
+ // check head fields and NUL padding
+ const head = std.mem.bytesToValue(Dirent, buf[before..][0..24]);
+ try testing.expectEqual(n, head.ino);
+ try testing.expectEqual(@as(u32, @intCast(n)), head.namelen);
+ try testing.expectEqualStrings(name[0..n], buf[before + 24 ..][0..n]);
+ for (buf[before + 24 + n .. used]) |b| try testing.expectEqual(@as(u8, 0), b);
+ expect_total += rec;
+ }
+ try testing.expectEqual(expect_total, used);
+
+ // A record that does not fit leaves everything untouched.
+ var small: [40]u8 = undefined;
+ var used2: usize = 0;
+ try testing.expect(addDirent(&small, &used2, 1, 1, DT_DIR, "0123456789abcdef")); // 24+16 = 40
+ try testing.expectEqual(@as(usize, 40), used2);
+ try testing.expect(!addDirent(&small, &used2, 2, 2, DT_DIR, "x"));
+ try testing.expectEqual(@as(usize, 40), used2);
+ var tight: [31]u8 = undefined;
+ var used3: usize = 0;
+ try testing.expect(!addDirent(&tight, &used3, 1, 1, DT_REG, "a")); // needs 32
+ try testing.expectEqual(@as(usize, 0), used3);
+}
+
+test "body/nameAfter/secondName on hand-built requests" {
+ var buf: [128]u8 align(8) = undefined;
+ // LOOKUP(parent=1, "hello")
+ const name = "hello";
+ const hdr = InHeader{
+ .len = @intCast(@sizeOf(InHeader) + name.len + 1),
+ .opcode = @intFromEnum(Opcode.lookup),
+ .unique = 7,
+ .nodeid = root_id,
+ .uid = 1000,
+ .gid = 1000,
+ .pid = 42,
+ .total_extlen = 0,
+ .padding = 0,
+ };
+ @memcpy(buf[0..40], std.mem.asBytes(&hdr));
+ @memcpy(buf[40..45], name);
+ buf[45] = 0;
+ const req = Request{ .header = hdr, .body = buf[40..hdr.len] };
+ try testing.expectEqual(Opcode.lookup, req.header.op());
+ try testing.expectEqualStrings("hello", try nameAfter(void, req));
+ try testing.expectError(error.Protocol, body(MkdirIn, Request{ .header = hdr, .body = buf[40..44] }));
+
+ // MKDIR(mode=0o755) + "dir"
+ const mk = MkdirIn{ .mode = 0o755, .umask = 0o22 };
+ @memcpy(buf[40..48], std.mem.asBytes(&mk));
+ @memcpy(buf[48..51], "dir");
+ buf[51] = 0;
+ const mreq = Request{ .header = hdr, .body = buf[40..52] };
+ const got = try body(MkdirIn, mreq);
+ try testing.expectEqual(@as(u32, 0o755), got.mode);
+ try testing.expectEqualStrings("dir", try nameAfter(MkdirIn, mreq));
+
+ // RENAME(newdir) + "old\0new\0"
+ const rn = RenameIn{ .newdir = 9 };
+ @memcpy(buf[40..48], std.mem.asBytes(&rn));
+ @memcpy(buf[48..56], "old\x00new\x00");
+ const rreq = Request{ .header = hdr, .body = buf[40..56] };
+ try testing.expectEqual(@as(u64, 9), (try body(RenameIn, rreq)).newdir);
+ const old = try nameAfter(RenameIn, rreq);
+ try testing.expectEqualStrings("old", old);
+ try testing.expectEqualStrings("new", try secondName(rreq, old, @sizeOf(RenameIn)));
+ try testing.expectError(error.Protocol, secondName(rreq, "new", @sizeOf(RenameIn) + 4));
+
+ // Missing NUL and misalignment are protocol errors.
+ try testing.expectError(error.Protocol, nameAt(Request{ .header = hdr, .body = buf[48..51] }, 0));
+ try testing.expectError(error.Protocol, body(MkdirIn, Request{ .header = hdr, .body = buf[41..57] }));
+}
+
+test "initReply fields" {
+ var in = InitIn{ .major = 7, .minor = 45, .max_readahead = 131072, .flags = 0, .flags2 = 0, .unused = [_]u32{0} ** 11 };
+ const a = initReply(&in, 1 << 20);
+ try testing.expectEqual(@as(u32, 7), a.major);
+ try testing.expectEqual(@as(u32, 31), a.minor);
+ try testing.expectEqual(@as(u32, 131072), a.max_readahead);
+ try testing.expectEqual(FUSE_ASYNC_READ | FUSE_ATOMIC_O_TRUNC | FUSE_BIG_WRITES | FUSE_AUTO_INVAL_DATA, a.flags);
+ try testing.expectEqual(@as(u16, 0), a.max_pages);
+ try testing.expectEqual(@as(u16, 16), a.max_background);
+ try testing.expectEqual(@as(u16, 12), a.congestion_threshold);
+ try testing.expectEqual(@as(u32, 1 << 20), a.max_write);
+ try testing.expectEqual(@as(u32, 1), a.time_gran);
+
+ in.flags = FUSE_MAX_PAGES | FUSE_ASYNC_READ;
+ in.minor = 27;
+ const b = initReply(&in, 4096);
+ try testing.expectEqual(@as(u32, 27), b.minor);
+ try testing.expectEqual(FUSE_ASYNC_READ | FUSE_ATOMIC_O_TRUNC | FUSE_BIG_WRITES | FUSE_AUTO_INVAL_DATA | FUSE_MAX_PAGES, b.flags);
+ try testing.expectEqual(@as(u16, 256), b.max_pages);
+ try testing.expectEqual(@as(u32, 4096), b.max_write);
+}
+
+fn makePipe() ![2]i32 {
+ var fds: [2]i32 = undefined;
+ if (linux.errno(linux.pipe2(&fds, .{ .CLOEXEC = true })) != .SUCCESS) return error.Io;
+ return fds;
+}
+
+fn readExact(fd: i32, out: []u8) !void {
+ var got: usize = 0;
+ while (got < out.len) {
+ const rc = linux.read(fd, out[got..].ptr, out.len - got);
+ if (linux.errno(rc) != .SUCCESS or rc == 0) return error.Io;
+ got += rc;
+ }
+}
+
+test "reply writes header + payloads through a pipe" {
+ const fds = try makePipe();
+ defer _ = linux.close(fds[0]);
+ defer _ = linux.close(fds[1]);
+
+ const oo = OpenOut{ .fh = 0x1234, .open_flags = FOPEN_DIRECT_IO };
+ try reply(fds[1], 99, &.{ std.mem.asBytes(&oo), "tail" });
+
+ var out: [16 + 16 + 4]u8 = undefined;
+ try readExact(fds[0], &out);
+ const h = std.mem.bytesToValue(OutHeader, out[0..16]);
+ try testing.expectEqual(@as(u32, 36), h.len);
+ try testing.expectEqual(@as(i32, 0), h.@"error");
+ try testing.expectEqual(@as(u64, 99), h.unique);
+ try testing.expectEqualSlices(u8, std.mem.asBytes(&oo), out[16..32]);
+ try testing.expectEqualStrings("tail", out[32..36]);
+
+ // Empty payload list: header only.
+ try reply(fds[1], 5, &.{});
+ var only: [16]u8 = undefined;
+ try readExact(fds[0], &only);
+ try testing.expectEqual(@as(u32, 16), std.mem.bytesToValue(OutHeader, &only).len);
+
+ var too_many: [max_payloads + 1][]const u8 = undefined;
+ for (&too_many) |*p| p.* = "x";
+ try testing.expectError(error.TooManyPayloads, reply(fds[1], 1, &too_many));
+}
+
+test "replyError writes a negative errno" {
+ const fds = try makePipe();
+ defer _ = linux.close(fds[0]);
+ defer _ = linux.close(fds[1]);
+
+ try replyError(fds[1], 0xdead_beef, .NOENT);
+ var out: [16]u8 = undefined;
+ try readExact(fds[0], &out);
+ const h = std.mem.bytesToValue(OutHeader, &out);
+ try testing.expectEqual(@as(u32, 16), h.len);
+ try testing.expectEqual(@as(i32, -2), h.@"error");
+ try testing.expectEqual(@as(u64, 0xdead_beef), h.unique);
+
+ try replyError(fds[1], 1, .NOSYS);
+ try readExact(fds[0], &out);
+ try testing.expectEqual(-@as(i32, @intCast(@intFromEnum(linux.E.NOSYS))), std.mem.bytesToValue(OutHeader, &out).@"error");
+}
+
+test "readRequest parses one request from a pipe and rejects bad lengths" {
+ const fds = try makePipe();
+ defer _ = linux.close(fds[0]);
+ defer _ = linux.close(fds[1]);
+
+ var wire: [48]u8 align(8) = undefined;
+ const hdr = InHeader{ .len = 48, .opcode = @intFromEnum(Opcode.forget), .unique = 3, .nodeid = 2, .uid = 0, .gid = 0, .pid = 0, .total_extlen = 0, .padding = 0 };
+ @memcpy(wire[0..40], std.mem.asBytes(&hdr));
+ @memcpy(wire[40..48], std.mem.asBytes(&ForgetIn{ .nlookup = 11 }));
+ try testing.expectEqual(@as(usize, 48), linux.write(fds[1], &wire, wire.len));
+
+ var buf: [4096]u8 align(8) = undefined;
+ const req = (try readRequest(fds[0], &buf)) orelse return error.Io;
+ try testing.expectEqual(Opcode.forget, req.header.op());
+ try testing.expectEqual(@as(u64, 2), req.header.nodeid);
+ try testing.expectEqual(@as(u64, 11), (try body(ForgetIn, req)).nlookup);
+
+ // Header length disagreeing with what was read is a protocol error.
+ var bad = wire;
+ std.mem.bytesAsValue(InHeader, bad[0..40]).len = 40;
+ try testing.expectEqual(@as(usize, 48), linux.write(fds[1], &bad, bad.len));
+ try testing.expectError(error.Protocol, readRequest(fds[0], &buf));
+}
diff --git a/9ns/src/main.zig b/9ns/src/main.zig
new file mode 100644
index 0000000..26bd699
--- /dev/null
+++ b/9ns/src/main.zig
@@ -0,0 +1,444 @@
+//! 9ns: mount a 9P2000 tree into a fresh user+mount namespace via FUSE
+//! and run a program inside it.
+//!
+//! Exit codes: the child's status (128+sig if signalled); 125 for 9ns's
+//! own failures (usage, connect, attach, namespace/mount); 126/127 for exec
+//! failures.
+
+const std = @import("std");
+const linux = std.os.linux;
+const ns = @import("ns.zig");
+const nine = @import("nine.zig");
+const bridge = @import("bridge.zig");
+
+const version_string = "9ns 0.1.0";
+
+const usage_text =
+ \\Usage: 9ns [options] -- PROGRAM [ARGS...]
+ \\Transport (exactly one):
+ \\ --unix PATH Unix stream socket
+ \\ --tcp IP:PORT TCP (IPv4/IPv6 literal)
+ \\ --fd N already-connected inherited descriptor
+ \\ --spawn CMD run CMD (via /bin/sh -c) with a socketpair on its stdin/stdout
+ \\Options:
+ \\ --mount PATH mountpoint inside the new namespace (default /mnt/9p)
+ \\ --uname NAME 9P user name (default $USER, else "none")
+ \\ --aname NAME 9P tree to attach (default "")
+ \\ --msize BYTES maximum 9P message size to request (default 131072)
+ \\ --cache SECONDS attr/entry cache validity, may be fractional (default 1)
+ \\ --no-direct-io let the kernel cache file pages (trusts stat length)
+ \\ --debug trace FUSE and 9P operations on stderr
+ \\ --help, --version
+ \\PROGRAM defaults to $SHELL (else /bin/sh). The mountpoint is exported as $NINE_MOUNT.
+ \\
+;
+
+const own_failure: u8 = 125;
+/// Largest 9P message size we agree to request: the session allocates two
+/// buffers of this size up front, before the server negotiates it down.
+const max_msize: u32 = 16 * 1024 * 1024;
+
+/// Write `text` to stdout (informational output such as --help); errors are
+/// ignored, there is nowhere better to report them.
+fn printStdout(text: []const u8) void {
+ var off: usize = 0;
+ while (off < text.len) {
+ const rc = linux.write(1, text[off..].ptr, text.len - off);
+ switch (linux.errno(rc)) {
+ .SUCCESS => off += rc,
+ .INTR => continue,
+ else => return,
+ }
+ }
+}
+
+const Config = struct {
+ address: ?nine.Address = null,
+ spawn_cmd: ?[]const u8 = null,
+ mount: []const u8 = "/mnt/9p",
+ uname: ?[]const u8 = null,
+ aname: []const u8 = "",
+ msize: u32 = 131072,
+ cache_ns: u64 = 1_000_000_000,
+ direct_io: bool = true,
+ debug: bool = false,
+ /// Empty means "default program".
+ program: []const []const u8 = &.{},
+};
+
+const ParseResult = union(enum) {
+ run: Config,
+ /// Usage error, already reported on stderr; exit with this status.
+ exit: u8,
+ /// --help/--version: text for stdout, then exit 0. Printing is left to
+ /// `main` so that no test path writes to fd 1 (under `zig build test`
+ /// that is the test runner's protocol pipe).
+ info: []const u8,
+};
+
+fn usageError(comptime fmt: []const u8, args: anytype) ParseResult {
+ std.debug.print("9ns: " ++ fmt ++ "\n(try 9ns --help)\n", args);
+ return .{ .exit = own_failure };
+}
+
+fn parseArgs(arena: std.mem.Allocator, args: []const [:0]const u8) !ParseResult {
+ var cfg = Config{};
+ var transports: usize = 0;
+ var i: usize = 1;
+ var program_start: ?usize = null;
+ while (i < args.len) : (i += 1) {
+ const arg: []const u8 = args[i];
+ if (std.mem.eql(u8, arg, "--")) {
+ program_start = i + 1;
+ break;
+ }
+ if (!std.mem.startsWith(u8, arg, "--")) {
+ // A single-dash word is a typo for an option, not a program.
+ if (arg.len > 1 and arg[0] == '-') return usageError("unknown option {s} (options start with --)", .{arg});
+ // A bare word starts PROGRAM, as if "--" were given.
+ program_start = i;
+ break;
+ }
+ // Split "--opt=value".
+ var name = arg;
+ var inline_value: ?[]const u8 = null;
+ if (std.mem.indexOfScalar(u8, arg, '=')) |eq| {
+ name = arg[0..eq];
+ inline_value = arg[eq + 1 ..];
+ }
+ const Opt = enum { unix, tcp, fd, spawn, mount, uname, aname, msize, cache, @"no-direct-io", debug, help, version, unknown };
+ const opt = std.meta.stringToEnum(Opt, name[2..]) orelse .unknown;
+ switch (opt) {
+ .@"no-direct-io", .debug, .help, .version => if (inline_value != null) return usageError("{s} takes no value", .{name}),
+ .unknown => return usageError("unknown option {s}", .{name}),
+ else => {},
+ }
+ const value: []const u8 = switch (opt) {
+ .@"no-direct-io", .debug, .help, .version, .unknown => "",
+ else => inline_value orelse blk: {
+ i += 1;
+ if (i >= args.len) return usageError("{s} needs a value", .{name});
+ break :blk args[i];
+ },
+ };
+ switch (opt) {
+ .unix => {
+ if (value.len == 0) return usageError("--unix wants a socket path", .{});
+ cfg.address = .{ .unix = value };
+ transports += 1;
+ },
+ .tcp => {
+ cfg.address = parseTcp(value) orelse return usageError("--tcp wants IP:PORT (IPv6 as [ADDR]:PORT), got '{s}'", .{value});
+ transports += 1;
+ },
+ .fd => {
+ const n = std.fmt.parseInt(i32, value, 10) catch return usageError("--fd wants a number, got '{s}'", .{value});
+ if (n < 0) return usageError("--fd wants a non-negative number", .{});
+ cfg.address = .{ .fd = n };
+ transports += 1;
+ },
+ .spawn => {
+ if (value.len == 0) return usageError("--spawn wants a command", .{});
+ cfg.spawn_cmd = value;
+ transports += 1;
+ },
+ .mount => {
+ if (value.len == 0) return usageError("--mount wants a path", .{});
+ cfg.mount = value;
+ },
+ .uname => cfg.uname = value,
+ .aname => cfg.aname = value,
+ .msize => {
+ cfg.msize = std.fmt.parseInt(u32, value, 10) catch return usageError("--msize wants a number, got '{s}'", .{value});
+ if (cfg.msize < 4096 or cfg.msize > max_msize) return usageError("--msize must be between 4096 and {d}", .{max_msize});
+ },
+ .cache => {
+ const secs = std.fmt.parseFloat(f64, value) catch return usageError("--cache wants seconds, got '{s}'", .{value});
+ if (!(secs >= 0) or secs > 1e9) return usageError("--cache out of range", .{});
+ cfg.cache_ns = @intFromFloat(secs * 1e9);
+ },
+ .@"no-direct-io" => cfg.direct_io = false,
+ .debug => cfg.debug = true,
+ .help => return .{ .info = usage_text },
+ .version => return .{ .info = version_string ++ "\n" },
+ .unknown => unreachable,
+ }
+ }
+ if (transports == 0) return usageError("one transport is required (--unix, --tcp, --fd or --spawn)", .{});
+ if (transports > 1) return usageError("exactly one transport is allowed", .{});
+ if (program_start) |start| {
+ const prog = try arena.alloc([]const u8, args.len - start);
+ for (args[start..], 0..) |a, j| prog[j] = a;
+ cfg.program = prog;
+ }
+ return .{ .run = cfg };
+}
+
+fn parseTcp(spec: []const u8) ?nine.Address {
+ const colon = std.mem.lastIndexOfScalar(u8, spec, ':') orelse return null;
+ var host = spec[0..colon];
+ if (host.len >= 2 and host[0] == '[' and host[host.len - 1] == ']') host = host[1 .. host.len - 1];
+ if (host.len == 0) return null;
+ const port = std.fmt.parseInt(u16, spec[colon + 1 ..], 10) catch return null;
+ return .{ .tcp = .{ .host = host, .port = port } };
+}
+
+/// `--spawn`: run CMD under /bin/sh with one end of a socketpair as its
+/// stdin/stdout; the other end is the 9P transport.
+const Server = struct { pid: i32, fd: i32 };
+
+fn spawnServer(cmd: [:0]const u8, envp: [*:null]const ?[*:0]const u8) !Server {
+ var sv: [2]i32 = undefined;
+ switch (linux.errno(linux.socketpair(linux.AF.UNIX, linux.SOCK.STREAM | linux.SOCK.CLOEXEC, 0, &sv))) {
+ .SUCCESS => {},
+ else => |e| {
+ std.debug.print("9ns: socketpair: E{t}\n", .{e});
+ return error.SystemResources;
+ },
+ }
+ const rc = linux.fork();
+ switch (linux.errno(rc)) {
+ .SUCCESS => {},
+ else => |e| {
+ _ = linux.close(sv[0]);
+ _ = linux.close(sv[1]);
+ std.debug.print("9ns: fork: E{t}\n", .{e});
+ return error.SystemResources;
+ },
+ }
+ if (rc == 0) {
+ // Child: dup2 clears CLOEXEC on 0 and 1; everything else is CLOEXEC.
+ if (linux.errno(linux.dup2(sv[1], 0)) != .SUCCESS or linux.errno(linux.dup2(sv[1], 1)) != .SUCCESS) linux.exit_group(125);
+ // The server shares our process group, so a Ctrl-C meant for the
+ // program would kill it and take the mount down with it: ignore the
+ // tty signals (inherited across exec). SIGPIPE goes back to its
+ // default, we only ignore it for ourselves.
+ ignoreSignal(.INT);
+ ignoreSignal(.QUIT);
+ defaultSignal(.PIPE);
+ const argv = [_:null]?[*:0]const u8{ "sh", "-c", cmd.ptr };
+ const e = linux.errno(linux.execve("/bin/sh", &argv, envp));
+ std.debug.print("9ns: --spawn: exec /bin/sh: E{t}\n", .{e});
+ linux.exit_group(127);
+ }
+ _ = linux.close(sv[1]);
+ return .{ .pid = @intCast(rc), .fd = sv[0] };
+}
+
+fn stopServer(server: ?Server) void {
+ const s = server orelse return;
+ _ = linux.kill(s.pid, .TERM);
+ ns.reapAny(s.pid);
+}
+
+/// Fail early (before spawning servers or forking) if /dev/fuse is unusable.
+fn probeFuseDevice() bool {
+ const rc = linux.open("/dev/fuse", .{ .ACCMODE = .RDWR, .CLOEXEC = true }, 0);
+ switch (linux.errno(rc)) {
+ .SUCCESS => {
+ _ = linux.close(@intCast(rc));
+ return true;
+ },
+ .NOENT => std.debug.print("9ns: /dev/fuse: ENOENT (is the fuse module loaded? try: modprobe fuse)\n", .{}),
+ else => |e| std.debug.print("9ns: open /dev/fuse: E{t}\n", .{e}),
+ }
+ return false;
+}
+
+fn ignoreSignal(sig: linux.SIG) void {
+ const ign = linux.Sigaction{ .handler = .{ .handler = linux.SIG.IGN }, .mask = linux.sigemptyset(), .flags = 0 };
+ std.posix.sigaction(sig, &ign, null);
+}
+
+fn defaultSignal(sig: linux.SIG) void {
+ const dfl = linux.Sigaction{ .handler = .{ .handler = linux.SIG.DFL }, .mask = linux.sigemptyset(), .flags = 0 };
+ std.posix.sigaction(sig, &dfl, null);
+}
+
+/// `--fd N`: the descriptor is ours from now on; it must not leak into the
+/// program (which could otherwise read 9P replies meant for us). Fails on a
+/// bad descriptor, which is the earliest place to report it.
+fn adoptFd(fd: i32) bool {
+ switch (linux.errno(linux.fcntl(fd, linux.F.SETFD, linux.FD_CLOEXEC))) {
+ .SUCCESS => return true,
+ else => |e| {
+ std.debug.print("9ns: --fd {d}: E{t}\n", .{ fd, e });
+ return false;
+ },
+ }
+}
+
+fn describeAddress(a: nine.Address, buf: []u8) []const u8 {
+ return switch (a) {
+ .unix => |p| std.fmt.bufPrint(buf, "unix socket {s}", .{p}) catch "unix socket",
+ .tcp => |t| std.fmt.bufPrint(buf, "tcp {s}:{d}", .{ t.host, t.port }) catch "tcp",
+ .fd => |fd| std.fmt.bufPrint(buf, "fd {d}", .{fd}) catch "fd",
+ };
+}
+
+pub fn main(init: std.process.Init) !u8 {
+ const gpa = init.gpa;
+ const arena = init.arena.allocator();
+ const args = try init.minimal.args.toSlice(arena);
+ const envp: [*:null]const ?[*:0]const u8 = init.minimal.environ.block.slice.ptr;
+
+ var cfg = switch (try parseArgs(arena, args)) {
+ .exit => |code| return code,
+ .info => |text| {
+ printStdout(text);
+ return 0;
+ },
+ .run => |c| c,
+ };
+
+ // Defaults that come from the environment.
+ if (cfg.program.len == 0) {
+ const env_shell = ns.getenv(envp, "SHELL") orelse "";
+ const shell = if (env_shell.len == 0) "/bin/sh" else env_shell;
+ cfg.program = try arena.dupe([]const u8, &.{shell});
+ }
+ const uname = cfg.uname orelse ns.getenv(envp, "USER") orelse "none";
+ const mountpoint = ns.resolveMountpoint(gpa, cfg.mount) catch |err| {
+ std.debug.print("9ns: --mount {s}: {t}\n", .{ cfg.mount, err });
+ return own_failure;
+ };
+ defer gpa.free(mountpoint);
+
+ if (!probeFuseDevice()) return own_failure;
+
+ // Writes to a dead server socket must not kill us.
+ ignoreSignal(.PIPE);
+
+ var server: ?Server = null;
+ var address: nine.Address = undefined;
+ if (cfg.spawn_cmd) |cmd| {
+ const cmd_z = try arena.dupeZ(u8, cmd);
+ server = spawnServer(cmd_z, envp) catch return own_failure;
+ ns.watchServer(server.?.pid);
+ address = .{ .fd = server.?.fd };
+ } else {
+ address = cfg.address.?;
+ if (address == .fd and !adoptFd(address.fd)) return own_failure;
+ }
+
+ var addr_buf: [256]u8 = undefined;
+ var session = nine.Session.connect(gpa, address, cfg.msize) catch |err| {
+ std.debug.print("9ns: connect to {s}: {t}\n", .{ describeAddress(address, &addr_buf), err });
+ stopServer(server);
+ return own_failure;
+ };
+ defer session.deinit();
+ defer stopServer(server);
+
+ _ = session.attach(0, uname, cfg.aname) catch |err| {
+ switch (err) {
+ error.Nine => std.debug.print("9ns: attach (uname={s}, aname='{s}'): {s}\n", .{ uname, cfg.aname, session.ename[0..session.ename_len] }),
+ else => std.debug.print("9ns: attach: {t}\n", .{err}),
+ }
+ return own_failure;
+ };
+ if (cfg.debug) std.debug.print("9ns: attached to {s} (msize {d}), mounting on {s}\n", .{ describeAddress(address, &addr_buf), session.msize, mountpoint });
+
+ var child_pid: i32 = 0;
+ const stop_fd = ns.installSignals(&child_pid) catch return own_failure;
+
+ const uid = linux.getuid();
+ const gid = linux.getgid();
+ const child = ns.spawn(gpa, .{
+ .argv = cfg.program,
+ .envp = envp,
+ .mountpoint = mountpoint,
+ .uid = uid,
+ .gid = gid,
+ .max_read = bridge.max_write,
+ }) catch return own_failure;
+
+ bridge.serve(gpa, child.fuse_fd, &session, 0, stop_fd, .{
+ .uid = uid,
+ .gid = gid,
+ .attr_timeout_ns = cfg.cache_ns,
+ .direct_io = cfg.direct_io,
+ .debug = cfg.debug,
+ }) catch |err| switch (err) {
+ error.Closed => std.debug.print("9ns: 9P server connection closed\n", .{}),
+ else => std.debug.print("9ns: fuse: {t}\n", .{err}),
+ };
+
+ // Closing the device aborts the FUSE connection: anything still using
+ // the mount gets ENOTCONN instead of hanging on an unserved request.
+ _ = linux.close(child.fuse_fd);
+
+ const status = ns.reapIfExited(child.pid) orelse ns.waitChild(child.pid) catch own_failure;
+ // An exec failure (126/127) is already in `status`; this prints its message.
+ _ = ns.reportExecFailure(child);
+ return status;
+}
+
+test "parseTcp" {
+ const a = parseTcp("127.0.0.1:564").?;
+ try std.testing.expectEqualStrings("127.0.0.1", a.tcp.host);
+ try std.testing.expectEqual(@as(u16, 564), a.tcp.port);
+ const b = parseTcp("[::1]:9999").?;
+ try std.testing.expectEqualStrings("::1", b.tcp.host);
+ try std.testing.expectEqual(@as(u16, 9999), b.tcp.port);
+ try std.testing.expect(parseTcp("nohost") == null);
+ try std.testing.expect(parseTcp(":564") == null);
+ try std.testing.expect(parseTcp("1.2.3.4:") == null);
+ try std.testing.expect(parseTcp("1.2.3.4:70000") == null);
+}
+
+test "parseArgs" {
+ const arena = std.testing.allocator;
+ {
+ const args = [_][:0]const u8{ "9ns", "--unix", "/s", "--cache", "0.5", "--msize=8192", "--no-direct-io", "--", "sh", "-c", "x" };
+ const r = try parseArgs(arena, &args);
+ defer arena.free(r.run.program);
+ try std.testing.expectEqualStrings("/s", r.run.address.?.unix);
+ try std.testing.expectEqual(@as(u64, 500_000_000), r.run.cache_ns);
+ try std.testing.expectEqual(@as(u32, 8192), r.run.msize);
+ try std.testing.expect(!r.run.direct_io);
+ try std.testing.expectEqual(@as(usize, 3), r.run.program.len);
+ try std.testing.expectEqualStrings("x", r.run.program[2]);
+ }
+ {
+ const args = [_][:0]const u8{ "9ns", "--fd", "3" };
+ const r = try parseArgs(arena, &args);
+ try std.testing.expectEqual(@as(i32, 3), r.run.address.?.fd);
+ try std.testing.expectEqual(@as(usize, 0), r.run.program.len);
+ try std.testing.expectEqualStrings("/mnt/9p", r.run.mount);
+ }
+ {
+ // Two transports, no transport, unknown option, missing value: all 125.
+ const two = [_][:0]const u8{ "9ns", "--fd", "3", "--unix", "/s" };
+ try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &two)).exit);
+ const none = [_][:0]const u8{ "9ns", "--", "sh" };
+ try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &none)).exit);
+ const unknown = [_][:0]const u8{ "9ns", "--bogus" };
+ try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &unknown)).exit);
+ const missing = [_][:0]const u8{ "9ns", "--unix" };
+ try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &missing)).exit);
+ const badcache = [_][:0]const u8{ "9ns", "--fd", "3", "--cache", "abc" };
+ try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &badcache)).exit);
+ // Empty values, a single-dash typo, and an msize that would allocate gigabytes.
+ const emptyunix = [_][:0]const u8{ "9ns", "--unix=", "--", "sh" };
+ try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &emptyunix)).exit);
+ const emptymount = [_][:0]const u8{ "9ns", "--fd", "3", "--mount", "" };
+ try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &emptymount)).exit);
+ const singledash = [_][:0]const u8{ "9ns", "--fd", "3", "-mount", "/x" };
+ try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &singledash)).exit);
+ const hugemsize = [_][:0]const u8{ "9ns", "--fd", "3", "--msize", "4294967295" };
+ try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &hugemsize)).exit);
+ const okmsize = [_][:0]const u8{ "9ns", "--fd", "3", "--msize", "16777216" };
+ try std.testing.expectEqual(@as(u32, 16777216), (try parseArgs(arena, &okmsize)).run.msize);
+ }
+ {
+ const ver = [_][:0]const u8{ "9ns", "--version" };
+ try std.testing.expectEqualStrings(version_string ++ "\n", (try parseArgs(arena, &ver)).info);
+ const help = [_][:0]const u8{ "9ns", "--help" };
+ try std.testing.expect(std.mem.startsWith(u8, (try parseArgs(arena, &help)).info, "Usage: 9ns"));
+ }
+}
+
+test {
+ _ = ns;
+}
diff --git a/9ns/src/nine.zig b/9ns/src/nine.zig
new file mode 100644
index 0000000..70633e6
--- /dev/null
+++ b/9ns/src/nine.zig
@@ -0,0 +1,756 @@
+//! Synchronous 9P2000 session over a blocking file descriptor.
+//!
+//! A thin RPC layer over `cloud9.Client` (push/take, allocation-free). One request
+//! is outstanding at a time: the FUSE loop that drives this is single-threaded, so
+//! every call here blocks until its reply (or the connection's death) arrives.
+//! Fids are handed out from a free list; fid 0 is reserved for the root.
+const std = @import("std");
+const cloud9 = @import("cloud9");
+const linux = std.os.linux;
+
+pub const Address = union(enum) {
+ unix: []const u8,
+ tcp: struct { host: []const u8, port: u16 },
+ fd: i32,
+};
+
+/// A Stat whose every field means "leave unchanged" in a Twstat.
+pub const dontcare = cloud9.Stat{
+ .type = 0xFFFF,
+ .dev = 0xFFFF_FFFF,
+ .qid = .{ .type = 0xFF, .version = 0xFFFF_FFFF, .path = 0xFFFF_FFFF_FFFF_FFFF },
+ .mode = 0xFFFF_FFFF,
+ .atime = 0xFFFF_FFFF,
+ .mtime = 0xFFFF_FFFF,
+ .length = 0xFFFF_FFFF_FFFF_FFFF,
+ .name = "",
+ .uid = "",
+ .gid = "",
+ .muid = "",
+};
+
+pub const Session = struct {
+ pub const Error = error{ Nine, Protocol, Io, Closed, Stopped, TooLarge, OutOfMemory };
+
+ pub const Walk = struct { nwqid: u16, wqid: [cloud9.max_welem]cloud9.Qid };
+ pub const Open = struct { qid: cloud9.Qid, iounit: u32 };
+
+ gpa: std.mem.Allocator,
+ fd: i32,
+ client: cloud9.Client,
+ in_buf: []u8,
+ out_buf: []u8,
+ /// After `error.Nine`, the server's Rerror text (copied, bounded).
+ ename: [256]u8 = undefined,
+ ename_len: usize = 0,
+ /// Negotiated maximum message size.
+ msize: u32,
+ next_fid: u32 = 1,
+ free_fids: std.ArrayList(u32) = .empty,
+ /// Per-fid iounit learned from open/create (0 = none); used to chunk read/write.
+ iounits: std.AutoHashMapUnmanaged(u32, u32) = .empty,
+ /// Optional descriptor watched while waiting for a reply: when it becomes
+ /// readable (the bridge's "child exited" pipe) the pending rpc fails with
+ /// `error.Stopped` instead of blocking on a server that never answers.
+ stop_fd: i32 = -1,
+
+ /// Connect to `address`, then negotiate the protocol version.
+ /// `msize` is the maximum message size to ask for (0 = the buffers' size).
+ pub fn connect(gpa: std.mem.Allocator, address: Address, msize: u32) !Session {
+ const want: u32 = if (msize == 0) 8192 else @max(msize, 24);
+ const fd = try openTransport(address);
+ errdefer if (address != .fd) {
+ _ = linux.close(fd);
+ };
+
+ const in_buf = try gpa.alloc(u8, want);
+ errdefer gpa.free(in_buf);
+ const out_buf = try gpa.alloc(u8, want);
+ errdefer gpa.free(out_buf);
+
+ var s: Session = .{
+ .gpa = gpa,
+ .fd = fd,
+ .client = .init(.{ .in = in_buf, .out = out_buf }),
+ .in_buf = in_buf,
+ .out_buf = out_buf,
+ .msize = want,
+ };
+ const r = try s.rpc(.{ .version = .{ .msize = want } });
+ if (!std.mem.eql(u8, r.version.version, "9P2000")) return error.Protocol;
+ s.msize = r.version.msize;
+ return s;
+ }
+
+ /// Closes the descriptor and frees the buffers. Fids are not clunked.
+ pub fn deinit(s: *Session) void {
+ _ = linux.close(s.fd);
+ s.free_fids.deinit(s.gpa);
+ s.iounits.deinit(s.gpa);
+ s.gpa.free(s.in_buf);
+ s.gpa.free(s.out_buf);
+ s.* = undefined;
+ }
+
+ pub fn attach(s: *Session, fid: u32, uname: []const u8, aname: []const u8) Error!cloud9.Qid {
+ const r = try s.rpc(.{ .attach = .{ .fid = fid, .uname = uname, .aname = aname } });
+ return r.attach;
+ }
+
+ /// Fid 0 is never handed out: it belongs to the root attach.
+ pub fn allocFid(s: *Session) u32 {
+ if (s.free_fids.pop()) |fid| return fid;
+ const fid = s.next_fid;
+ s.next_fid += 1;
+ return fid;
+ }
+
+ /// Fids currently bound (excluding fid 0); a debugging aid for leak hunting.
+ pub fn fidsInUse(s: *const Session) usize {
+ return (s.next_fid - 1) - s.free_fids.items.len;
+ }
+
+ pub fn freeFid(s: *Session, fid: u32) void {
+ _ = s.iounits.remove(fid);
+ // If the free list cannot grow the fid is simply leaked; the counter keeps going.
+ s.free_fids.append(s.gpa, fid) catch {};
+ }
+
+ /// Generic RPC. Result slices borrow the input buffer until the next call.
+ pub fn rpc(s: *Session, req: cloud9.Client.Request) Error!cloud9.Client.Result {
+ s.ename_len = 0;
+ _ = s.client.submit(req) catch |e| switch (e) {
+ error.NoTags, error.Handshake, error.Dead => return error.Protocol,
+ error.NoSpace, error.TooLarge => return error.TooLarge,
+ error.BadRequest => {
+ s.setEname("bad request");
+ return error.Nine;
+ },
+ };
+ try s.flush();
+ var tmp: [64 * 1024]u8 = undefined;
+ while (true) {
+ if (s.client.take()) |done| {
+ switch (done.result) {
+ .fail => |ename| {
+ s.setEname(ename);
+ return error.Nine;
+ },
+ else => return done.result,
+ }
+ }
+ if (s.client.dead) return error.Protocol;
+ // After take() returned null the previous frame is gone, so the free
+ // space is at least what the pending frame still needs.
+ const room = s.client.in.len - s.client.in_len;
+ if (room == 0) return error.Protocol;
+ const n = try readSome(s.fd, s.stop_fd, tmp[0..@min(room, tmp.len)]);
+ if (n == 0) return error.Closed;
+ const pushed = s.client.push(tmp[0..n]);
+ if (pushed != n) return error.Protocol;
+ }
+ }
+
+ /// Walk `names` from `fid` to `newfid`. A partial walk leaves `newfid` unbound
+ /// (9P semantics) and reports `error.Nine` with ename "file does not exist".
+ pub fn walk(s: *Session, fid: u32, newfid: u32, names: []const []const u8) Error!Walk {
+ const r = try s.rpc(.{ .walk = .{ .fid = fid, .newfid = newfid, .names = names } });
+ if (r.walk.nwqid < names.len) {
+ s.setEname("file does not exist");
+ return error.Nine;
+ }
+ return .{ .nwqid = r.walk.nwqid, .wqid = r.walk.wqid };
+ }
+
+ /// allocFid + zero-element walk. The fid is released again on failure.
+ pub fn clone(s: *Session, fid: u32) Error!u32 {
+ const newfid = s.allocFid();
+ errdefer s.freeFid(newfid);
+ _ = try s.walk(fid, newfid, &.{});
+ return newfid;
+ }
+
+ pub fn open(s: *Session, fid: u32, mode: u8) Error!Open {
+ const r = try s.rpc(.{ .open = .{ .fid = fid, .mode = mode } });
+ s.noteIounit(fid, r.open.iounit);
+ return .{ .qid = r.open.qid, .iounit = r.open.iounit };
+ }
+
+ pub fn create(s: *Session, fid: u32, name: []const u8, perm: u32, mode: u8) Error!Open {
+ const r = try s.rpc(.{ .create = .{ .fid = fid, .name = name, .perm = perm, .mode = mode } });
+ s.noteIounit(fid, r.create.iounit);
+ return .{ .qid = r.create.qid, .iounit = r.create.iounit };
+ }
+
+ /// Reads into `buf`, chunking by min(maxRead, iounit) and stopping at the first
+ /// short read. Returns the number of bytes read (0 at end of file).
+ pub fn read(s: *Session, fid: u32, offset: u64, buf: []u8) Error!usize {
+ return readWith(s, rpc, fid, offset, buf, s.chunk(fid));
+ }
+
+ /// Writes `data`, chunking like `read` and stopping at the first short write.
+ pub fn write(s: *Session, fid: u32, offset: u64, data: []const u8) Error!usize {
+ return writeWith(s, rpc, fid, offset, data, s.chunkWrite(fid));
+ }
+
+ /// The returned Stat's strings (name/uid/gid/muid) borrow the session's input
+ /// buffer: they are valid only until the next rpc. Copy what must outlive it.
+ pub fn stat(s: *Session, fid: u32) Error!cloud9.Stat {
+ const r = try s.rpc(.{ .stat = .{ .fid = fid } });
+ return r.stat;
+ }
+
+ pub fn wstat(s: *Session, fid: u32, st: cloud9.Stat) Error!void {
+ _ = try s.rpc(.{ .wstat = .{ .fid = fid, .stat = st } });
+ }
+
+ /// Frees the fid locally even when the server reports an error.
+ pub fn clunk(s: *Session, fid: u32) Error!void {
+ defer s.freeFid(fid);
+ _ = try s.rpc(.{ .clunk = .{ .fid = fid } });
+ }
+
+ /// Frees the fid locally even when the server reports an error.
+ pub fn remove(s: *Session, fid: u32) Error!void {
+ defer s.freeFid(fid);
+ _ = try s.rpc(.{ .remove = .{ .fid = fid } });
+ }
+
+ /// Maps the last Rerror text to an errno (case-insensitive substring match).
+ pub fn errno(s: *const Session) linux.E {
+ return enameToErrno(s.ename[0..s.ename_len]);
+ }
+
+ // -- internals --------------------------------------------------------------
+
+ fn setEname(s: *Session, text: []const u8) void {
+ const n = @min(text.len, 255);
+ @memcpy(s.ename[0..n], text[0..n]);
+ s.ename_len = n;
+ }
+
+ fn noteIounit(s: *Session, fid: u32, iounit: u32) void {
+ if (iounit == 0) {
+ _ = s.iounits.remove(fid);
+ } else {
+ s.iounits.put(s.gpa, fid, iounit) catch {};
+ }
+ }
+
+ fn chunk(s: *Session, fid: u32) u32 {
+ return chunkSize(s.client.maxRead(), s.iounits.get(fid) orelse 0);
+ }
+
+ fn chunkWrite(s: *Session, fid: u32) u32 {
+ return chunkSize(s.client.maxWrite(), s.iounits.get(fid) orelse 0);
+ }
+
+ /// Writes everything in the client's output buffer to the socket.
+ fn flush(s: *Session) Error!void {
+ while (s.client.output().len != 0) {
+ const out = s.client.output();
+ const rc = linux.write(s.fd, out.ptr, out.len);
+ switch (linux.errno(rc)) {
+ .SUCCESS => {
+ if (rc == 0) return error.Closed;
+ s.client.wrote(rc);
+ },
+ .INTR, .AGAIN => continue,
+ .PIPE, .CONNRESET => return error.Closed,
+ else => return error.Io,
+ }
+ }
+ }
+};
+
+fn chunkSize(max: u32, iounit: u32) u32 {
+ if (iounit != 0 and iounit < max) return iounit;
+ return max;
+}
+
+/// Chunked read over any rpc-shaped function (injected so the loop is testable).
+fn readWith(
+ s: anytype,
+ comptime rpcFn: anytype,
+ fid: u32,
+ offset: u64,
+ buf: []u8,
+ max_chunk: u32,
+) Session.Error!usize {
+ if (max_chunk == 0) return error.Protocol;
+ var done: usize = 0;
+ while (done < buf.len) {
+ const want: u32 = @intCast(@min(buf.len - done, max_chunk));
+ const r = try rpcFn(s, .{ .read = .{ .fid = fid, .offset = offset + done, .count = want } });
+ const data = r.read;
+ @memcpy(buf[done..][0..data.len], data);
+ done += data.len;
+ if (data.len < want) break;
+ }
+ return done;
+}
+
+/// Chunked write over any rpc-shaped function.
+fn writeWith(
+ s: anytype,
+ comptime rpcFn: anytype,
+ fid: u32,
+ offset: u64,
+ data: []const u8,
+ max_chunk: u32,
+) Session.Error!usize {
+ if (max_chunk == 0) return error.Protocol;
+ var done: usize = 0;
+ while (done < data.len) {
+ const want: usize = @min(data.len - done, max_chunk);
+ const r = try rpcFn(s, .{ .write = .{ .fid = fid, .offset = offset + done, .data = data[done..][0..want] } });
+ done += r.write;
+ if (r.write < want) break;
+ }
+ return done;
+}
+
+fn readSome(fd: i32, stop_fd: i32, buf: []u8) Session.Error!usize {
+ while (true) {
+ if (stop_fd >= 0) {
+ var pfds = [_]linux.pollfd{
+ .{ .fd = fd, .events = linux.POLL.IN, .revents = 0 },
+ .{ .fd = stop_fd, .events = linux.POLL.IN, .revents = 0 },
+ };
+ const prc = linux.poll(&pfds, pfds.len, -1);
+ switch (linux.errno(prc)) {
+ .SUCCESS => {},
+ .INTR, .AGAIN => continue,
+ else => return error.Io,
+ }
+ if (pfds[1].revents != 0 and pfds[0].revents == 0) return error.Stopped;
+ }
+ const rc = linux.read(fd, buf.ptr, buf.len);
+ switch (linux.errno(rc)) {
+ .SUCCESS => return rc,
+ .INTR, .AGAIN => continue,
+ .CONNRESET => return error.Closed,
+ else => return error.Io,
+ }
+ }
+}
+
+/// Rerror text → errno, per docs/DESIGN.md (first match wins).
+pub fn enameToErrno(ename: []const u8) linux.E {
+ const Rule = struct { needle: []const u8, err: linux.E };
+ const rules = [_]Rule{
+ .{ .needle = "not exist", .err = .NOENT },
+ .{ .needle = "not found", .err = .NOENT },
+ .{ .needle = "no such", .err = .NOENT },
+ .{ .needle = "exists", .err = .EXIST },
+ .{ .needle = "not empty", .err = .NOTEMPTY },
+ .{ .needle = "not a dir", .err = .NOTDIR },
+ .{ .needle = "is a dir", .err = .ISDIR },
+ .{ .needle = "permission", .err = .ACCES },
+ .{ .needle = "denied", .err = .ACCES },
+ .{ .needle = "read-only", .err = .ROFS },
+ .{ .needle = "read only", .err = .ROFS },
+ .{ .needle = "readonly", .err = .ROFS },
+ .{ .needle = "no space", .err = .NOSPC },
+ .{ .needle = "not allowed", .err = .PERM },
+ .{ .needle = "not permitted", .err = .PERM },
+ .{ .needle = "cannot", .err = .PERM },
+ .{ .needle = "fid", .err = .BADF },
+ .{ .needle = "bad offset", .err = .INVAL },
+ .{ .needle = "invalid", .err = .INVAL },
+ .{ .needle = "bad ", .err = .INVAL },
+ .{ .needle = "busy", .err = .BUSY },
+ .{ .needle = "in use", .err = .BUSY },
+ .{ .needle = "too long", .err = .NAMETOOLONG },
+ .{ .needle = "not supported", .err = .OPNOTSUPP },
+ .{ .needle = "unsupported", .err = .OPNOTSUPP },
+ };
+ for (rules) |rule| {
+ if (std.ascii.findIgnoreCase(ename, rule.needle) != null) return rule.err;
+ }
+ return .IO;
+}
+
+// -- transport ------------------------------------------------------------------
+
+fn openTransport(address: Address) !i32 {
+ switch (address) {
+ .fd => |fd| return fd,
+ .unix => |path| {
+ if (path.len == 0 or path.len >= 108) return error.NameTooLong;
+ var sa: linux.sockaddr.un = .{ .path = @splat(0) };
+ @memcpy(sa.path[0..path.len], path);
+ const fd = try newSocket(linux.AF.UNIX, 0);
+ errdefer _ = linux.close(fd);
+ try doConnect(fd, @ptrCast(&sa), @sizeOf(linux.sockaddr.un));
+ return fd;
+ },
+ .tcp => |t| {
+ const ip = std.Io.net.IpAddress.parse(t.host, t.port) catch return error.InvalidAddress;
+ switch (ip) {
+ .ip4 => |a| {
+ const sa: linux.sockaddr.in = .{
+ .port = std.mem.nativeToBig(u16, t.port),
+ .addr = @bitCast(a.bytes),
+ };
+ const fd = try newSocket(linux.AF.INET, linux.IPPROTO.TCP);
+ errdefer _ = linux.close(fd);
+ setNodelay(fd);
+ try doConnect(fd, @ptrCast(&sa), @sizeOf(linux.sockaddr.in));
+ return fd;
+ },
+ .ip6 => |a| {
+ const sa: linux.sockaddr.in6 = .{
+ .port = std.mem.nativeToBig(u16, t.port),
+ .flowinfo = 0,
+ .addr = a.bytes,
+ .scope_id = 0,
+ };
+ const fd = try newSocket(linux.AF.INET6, linux.IPPROTO.TCP);
+ errdefer _ = linux.close(fd);
+ setNodelay(fd);
+ try doConnect(fd, @ptrCast(&sa), @sizeOf(linux.sockaddr.in6));
+ return fd;
+ },
+ }
+ },
+ }
+}
+
+fn newSocket(domain: u32, protocol: u32) !i32 {
+ const rc = linux.socket(domain, linux.SOCK.STREAM | linux.SOCK.CLOEXEC, protocol);
+ switch (linux.errno(rc)) {
+ .SUCCESS => return @intCast(rc),
+ .MFILE, .NFILE => return error.ProcessFdQuotaExceeded,
+ .AFNOSUPPORT, .PROTONOSUPPORT => return error.AddressFamilyNotSupported,
+ .ACCES => return error.AccessDenied,
+ .NOMEM, .NOBUFS => return error.SystemResources,
+ else => return error.Unexpected,
+ }
+}
+
+fn setNodelay(fd: i32) void {
+ const one: u32 = 1;
+ _ = linux.setsockopt(fd, linux.IPPROTO.TCP, linux.TCP.NODELAY, @ptrCast(&one), @sizeOf(u32));
+}
+
+fn doConnect(fd: i32, addr: *const linux.sockaddr, len: linux.socklen_t) !void {
+ while (true) {
+ const rc = linux.connect(fd, addr, len);
+ switch (linux.errno(rc)) {
+ .SUCCESS => return,
+ .INTR => continue,
+ .CONNREFUSED => return error.ConnectionRefused,
+ .NOENT, .NOTDIR => return error.FileNotFound,
+ .ACCES, .PERM => return error.AccessDenied,
+ .TIMEDOUT => return error.ConnectionTimedOut,
+ .NETUNREACH, .HOSTUNREACH => return error.NetworkUnreachable,
+ .ADDRNOTAVAIL => return error.AddressNotAvailable,
+ .AGAIN, .INPROGRESS => return error.WouldBlock,
+ else => return error.Unexpected,
+ }
+ }
+}
+
+// -- tests ----------------------------------------------------------------------
+
+const testing = std.testing;
+
+test {
+ testing.refAllDecls(@This());
+}
+
+test "ename → errno mapping" {
+ try testing.expectEqual(linux.E.NOENT, enameToErrno("file does not exist"));
+ try testing.expectEqual(linux.E.NOENT, enameToErrno("No Such File"));
+ try testing.expectEqual(linux.E.NOENT, enameToErrno("directory entry not found"));
+ try testing.expectEqual(linux.E.EXIST, enameToErrno("file already exists"));
+ try testing.expectEqual(linux.E.NOTEMPTY, enameToErrno("directory not empty"));
+ try testing.expectEqual(linux.E.NOTDIR, enameToErrno("not a directory"));
+ try testing.expectEqual(linux.E.ISDIR, enameToErrno("is a directory"));
+ try testing.expectEqual(linux.E.ACCES, enameToErrno("permission denied"));
+ try testing.expectEqual(linux.E.ACCES, enameToErrno("access denied"));
+ try testing.expectEqual(linux.E.ROFS, enameToErrno("read-only file system"));
+ try testing.expectEqual(linux.E.NOSPC, enameToErrno("no space left"));
+ try testing.expectEqual(linux.E.PERM, enameToErrno("operation not permitted"));
+ try testing.expectEqual(linux.E.PERM, enameToErrno("cannot remove root"));
+ try testing.expectEqual(linux.E.BADF, enameToErrno("unknown fid"));
+ try testing.expectEqual(linux.E.BADF, enameToErrno("fid in use")); // "fid" precedes "in use"
+ try testing.expectEqual(linux.E.INVAL, enameToErrno("bad offset"));
+ try testing.expectEqual(linux.E.INVAL, enameToErrno("invalid argument"));
+ try testing.expectEqual(linux.E.INVAL, enameToErrno("bad request"));
+ try testing.expectEqual(linux.E.BUSY, enameToErrno("device busy"));
+ try testing.expectEqual(linux.E.NAMETOOLONG, enameToErrno("name too long"));
+ try testing.expectEqual(linux.E.OPNOTSUPP, enameToErrno("operation not supported"));
+ try testing.expectEqual(linux.E.IO, enameToErrno("something odd happened"));
+ try testing.expectEqual(linux.E.IO, enameToErrno(""));
+}
+
+test "fid allocator recycles and never hands out 0" {
+ var s: Session = undefined;
+ s.gpa = testing.allocator;
+ s.next_fid = 1;
+ s.free_fids = .empty;
+ s.iounits = .empty;
+ defer s.free_fids.deinit(s.gpa);
+ defer s.iounits.deinit(s.gpa);
+
+ const a = s.allocFid();
+ const b = s.allocFid();
+ const c = s.allocFid();
+ try testing.expectEqual(@as(u32, 1), a);
+ try testing.expectEqual(@as(u32, 2), b);
+ try testing.expectEqual(@as(u32, 3), c);
+ s.freeFid(b);
+ try testing.expectEqual(b, s.allocFid());
+ s.freeFid(a);
+ s.freeFid(c);
+ const x = s.allocFid();
+ const y = s.allocFid();
+ try testing.expect((x == a and y == c) or (x == c and y == a));
+ try testing.expectEqual(@as(u32, 4), s.allocFid());
+ try testing.expect(a != 0 and b != 0 and c != 0);
+}
+
+test "chunkSize honours iounit only when smaller" {
+ try testing.expectEqual(@as(u32, 100), chunkSize(100, 0));
+ try testing.expectEqual(@as(u32, 40), chunkSize(100, 40));
+ try testing.expectEqual(@as(u32, 100), chunkSize(100, 400));
+}
+
+/// Fake rpc for the chunked read/write loops: a file of `len` bytes where byte i == i & 0xff.
+const FakeFile = struct {
+ len: usize,
+ calls: usize = 0,
+ max_count: u32 = 0,
+ short_write_at: ?usize = null,
+ scratch: [4096]u8 = undefined,
+
+ fn rpc(f: *FakeFile, req: cloud9.Client.Request) Session.Error!cloud9.Client.Result {
+ f.calls += 1;
+ switch (req) {
+ .read => |r| {
+ f.max_count = @max(f.max_count, r.count);
+ if (r.offset >= f.len) return .{ .read = "" };
+ const n: usize = @min(@as(usize, r.count), f.len - @as(usize, @intCast(r.offset)));
+ for (f.scratch[0..n], 0..) |*b, i| b.* = @truncate(r.offset + i);
+ return .{ .read = f.scratch[0..n] };
+ },
+ .write => |w| {
+ f.max_count = @max(f.max_count, @as(u32, @intCast(w.data.len)));
+ if (f.short_write_at) |at| {
+ if (w.offset + w.data.len > at) {
+ const n: usize = if (w.offset >= at) 0 else @intCast(at - w.offset);
+ return .{ .write = @intCast(n) };
+ }
+ }
+ return .{ .write = @intCast(w.data.len) };
+ },
+ else => unreachable,
+ }
+ }
+};
+
+test "read chunks by max_chunk and stops at a short read" {
+ var f: FakeFile = .{ .len = 2500 };
+ var buf: [4000]u8 = undefined;
+ const n = try readWith(&f, FakeFile.rpc, 7, 0, &buf, 1000);
+ try testing.expectEqual(@as(usize, 2500), n);
+ try testing.expectEqual(@as(usize, 3), f.calls); // 1000, 1000, 500 (short → stop)
+ try testing.expectEqual(@as(u32, 1000), f.max_count);
+ for (buf[0..n], 0..) |b, i| try testing.expectEqual(@as(u8, @truncate(i)), b);
+
+ // Reading exactly up to a chunk boundary uses one call per chunk and no more.
+ f = .{ .len = 2000 };
+ try testing.expectEqual(@as(usize, 2000), try readWith(&f, FakeFile.rpc, 7, 0, buf[0..2000], 1000));
+ try testing.expectEqual(@as(usize, 2), f.calls);
+
+ // Offset past EOF → 0.
+ f = .{ .len = 10 };
+ try testing.expectEqual(@as(usize, 0), try readWith(&f, FakeFile.rpc, 7, 50, &buf, 1000));
+}
+
+test "write chunks and stops at a short write" {
+ var f: FakeFile = .{ .len = 0 };
+ var data: [2500]u8 = undefined;
+ for (&data, 0..) |*b, i| b.* = @truncate(i);
+ try testing.expectEqual(@as(usize, 2500), try writeWith(&f, FakeFile.rpc, 7, 0, &data, 1000));
+ try testing.expectEqual(@as(usize, 3), f.calls);
+ try testing.expectEqual(@as(u32, 1000), f.max_count);
+
+ f = .{ .len = 0, .short_write_at = 1500 };
+ try testing.expectEqual(@as(usize, 1500), try writeWith(&f, FakeFile.rpc, 7, 0, &data, 1000));
+ try testing.expectEqual(@as(usize, 2), f.calls);
+}
+
+// -- in-process server test ---------------------------------------------------------
+
+/// A tiny 9P2000 backend on a cloud9.Server: answers version/attach/walk/stat/open/
+/// read/clunk/remove with canned data. Runs in its own thread over a socketpair.
+const FakeServer = struct {
+ fd: i32,
+ msize: u32,
+ max_read_count: u32 = 0,
+ file_len: usize,
+
+ const file_qid: cloud9.Qid = .{ .type = 0, .version = 3, .path = 0x1234 };
+ const dir_qid: cloud9.Qid = .{ .type = cloud9.qtdir, .version = 1, .path = 0x1 };
+
+ fn run(fs: *FakeServer) void {
+ fs.loop() catch |e| std.debug.print("fake server: {s}\n", .{@errorName(e)});
+ _ = linux.close(fs.fd);
+ }
+
+ fn loop(fs: *FakeServer) !void {
+ const gpa = testing.allocator;
+ const in = try gpa.alloc(u8, fs.msize);
+ defer gpa.free(in);
+ const out = try gpa.alloc(u8, fs.msize * 2);
+ defer gpa.free(out);
+ var srv: cloud9.Server = .init(.{ .in = in, .out = out });
+ var tmp: [4096]u8 = undefined;
+ var data: [8192]u8 = undefined;
+ while (true) {
+ while (try srv.receive()) |req| {
+ const tag = req.tag;
+ switch (req.msg) {
+ .tversion => |m| try srv.negotiate(m.msize, m.version),
+ .tattach => try srv.reply(tag, .{ .rattach = .{ .qid = dir_qid } }),
+ .twalk => |m| {
+ var wq: [cloud9.max_welem]cloud9.Qid = @splat(dir_qid);
+ var n: u16 = 0;
+ for (m.wname[0..m.nwname]) |name| {
+ if (std.mem.eql(u8, name, "file")) {
+ wq[n] = file_qid;
+ } else if (std.mem.eql(u8, name, "dir")) {
+ wq[n] = dir_qid;
+ } else break;
+ n += 1;
+ }
+ if (n == 0 and m.nwname != 0) {
+ try srv.reply(tag, .{ .rerror = .{ .ename = "file does not exist" } });
+ } else {
+ try srv.reply(tag, .{ .rwalk = .{ .nwqid = n, .wqid = wq } });
+ }
+ },
+ .tstat => try srv.reply(tag, .{ .rstat = .{ .stat = .{
+ .type = 0,
+ .dev = 0,
+ .qid = file_qid,
+ .mode = 0o644,
+ .atime = 1,
+ .mtime = 2,
+ .length = fs.file_len,
+ .name = "file",
+ .uid = "u",
+ .gid = "g",
+ .muid = "u",
+ } } }),
+ .topen => |m| try srv.reply(tag, .{ .ropen = .{ .qid = file_qid, .iounit = if (m.mode == cloud9.owrite) 700 else 0 } }),
+ .tread => |m| {
+ fs.max_read_count = @max(fs.max_read_count, m.count);
+ var n: usize = 0;
+ if (m.offset < fs.file_len) n = @min(@as(usize, m.count), fs.file_len - @as(usize, @intCast(m.offset)));
+ n = @min(n, data.len);
+ for (data[0..n], 0..) |*b, i| b.* = @truncate(m.offset + i);
+ try srv.reply(tag, .{ .rread = .{ .data = data[0..n] } });
+ },
+ .twrite => |m| try srv.reply(tag, .{ .rwrite = .{ .count = @intCast(m.data.len) } }),
+ .tclunk => try srv.reply(tag, .rclunk),
+ .tremove => try srv.reply(tag, .{ .rerror = .{ .ename = "permission denied" } }),
+ .twstat => try srv.reply(tag, .rwstat),
+ // A flush is the test's "hang up now" signal.
+ .tflush => return,
+ else => try srv.reply(tag, .{ .rerror = .{ .ename = "not supported" } }),
+ }
+ srv.release();
+ }
+ while (srv.output().len != 0) {
+ const o = srv.output();
+ const rc = linux.write(fs.fd, o.ptr, o.len);
+ if (linux.errno(rc) != .SUCCESS) return error.Write;
+ srv.wrote(rc);
+ }
+ const rc = linux.read(fs.fd, &tmp, tmp.len);
+ if (linux.errno(rc) != .SUCCESS) return error.Read;
+ if (rc == 0) return;
+ if (srv.push(tmp[0..rc]) != rc) return error.Overflow;
+ }
+ }
+};
+
+test "session against an in-process cloud9.Server" {
+ var fds: [2]i32 = undefined;
+ try testing.expectEqual(linux.E.SUCCESS, linux.errno(linux.socketpair(linux.AF.UNIX, linux.SOCK.STREAM | linux.SOCK.CLOEXEC, 0, &fds)));
+
+ var fs: FakeServer = .{ .fd = fds[1], .msize = 8192, .file_len = 20_000 };
+ const th = try std.Thread.spawn(.{}, FakeServer.run, .{&fs});
+
+ var s = try Session.connect(testing.allocator, .{ .fd = fds[0] }, 8192);
+ defer {
+ s.deinit();
+ th.join();
+ }
+ try testing.expectEqual(@as(u32, 8192), s.msize);
+
+ const root = try s.attach(0, "me", "");
+ try testing.expectEqual(FakeServer.dir_qid.path, root.path);
+
+ // Plain rpc + stat borrowing the input buffer.
+ const fid = s.allocFid();
+ const w = try s.walk(0, fid, &.{"file"});
+ try testing.expectEqual(@as(u16, 1), w.nwqid);
+ try testing.expectEqual(FakeServer.file_qid.path, w.wqid[0].path);
+ const st = try s.stat(fid);
+ try testing.expectEqualStrings("file", st.name);
+ try testing.expectEqual(@as(u64, 20_000), st.length);
+
+ // Chunked read: 20000 bytes at maxRead = msize - 11 = 8181 per chunk.
+ _ = try s.open(fid, cloud9.oread);
+ const buf = try testing.allocator.alloc(u8, 30_000);
+ defer testing.allocator.free(buf);
+ const n = try s.read(fid, 0, buf);
+ try testing.expectEqual(@as(usize, 20_000), n);
+ for (buf[0..n], 0..) |b, i| try testing.expectEqual(@as(u8, @truncate(i)), b);
+ try testing.expectEqual(@as(u32, 8181), fs.max_read_count);
+ try testing.expectEqual(@as(usize, 0), try s.read(fid, 20_000, buf));
+
+ // iounit from open bounds the chunk.
+ const wfid = try s.clone(fid);
+ _ = try s.open(wfid, cloud9.owrite);
+ fs.max_read_count = 0;
+ _ = try s.read(wfid, 0, buf[0..3000]);
+ try testing.expectEqual(@as(u32, 700), fs.max_read_count);
+ try testing.expectEqual(@as(usize, 3000), try s.write(wfid, 0, buf[0..3000]));
+
+ // Partial walk → error.Nine with a "not exist" ename → ENOENT.
+ const pfid = s.allocFid();
+ try testing.expectError(error.Nine, s.walk(0, pfid, &.{ "dir", "nope" }));
+ try testing.expectEqual(linux.E.NOENT, s.errno());
+ try testing.expectEqualStrings("file does not exist", s.ename[0..s.ename_len]);
+ s.freeFid(pfid);
+
+ // Server Rerror → error.Nine, ename copied, fid freed by remove even on error.
+ try testing.expectError(error.Nine, s.remove(wfid));
+ try testing.expectEqual(linux.E.ACCES, s.errno());
+ try testing.expectEqual(wfid, s.allocFid()); // recycled
+ s.freeFid(wfid);
+
+ // Unsupported op → "not supported" → ENOTSUP; a plain wstat succeeds.
+ try testing.expectError(error.Nine, s.rpc(.{ .auth = .{ .afid = 5, .uname = "me" } }));
+ try testing.expectEqual(linux.E.OPNOTSUPP, s.errno());
+ try s.wstat(fid, dontcare);
+ try s.clunk(fid);
+ try testing.expectEqual(fid, s.allocFid());
+ s.freeFid(fid);
+
+ // A clone bound to a fid that then fails to walk must release the fid.
+ const before = s.next_fid;
+ const cfid = s.allocFid();
+ s.freeFid(cfid);
+ try testing.expectError(error.Nine, s.walk(0, cfid, &.{"nope"}));
+ try testing.expectEqual(before, s.next_fid);
+
+ // The server hanging up makes the pending rpc fail with error.Closed.
+ try testing.expectError(error.Closed, s.rpc(.{ .flush = .{ .oldtag = 0 } }));
+}
diff --git a/9ns/src/ns.zig b/9ns/src/ns.zig
new file mode 100644
index 0000000..6da2c7f
--- /dev/null
+++ b/9ns/src/ns.zig
@@ -0,0 +1,1082 @@
+//! Namespace and process plumbing for 9ns.
+//!
+//! Everything here is raw `std.os.linux` syscalls (no libc). The child side
+//! of `spawn` runs between `fork` and `execve`; it does not allocate except
+//! inside `ensureMountpoint` (the process is single-threaded by then, so the
+//! inherited allocator is safe to use).
+//!
+//! Exit codes produced by the child before exec: 125 for namespace/mount
+//! setup failures, 126 when the program was found but is not executable,
+//! 127 when it was not found.
+
+const std = @import("std");
+const builtin = @import("builtin");
+const linux = std.os.linux;
+const Allocator = std.mem.Allocator;
+const E = linux.E;
+
+pub const Spawn = struct {
+ /// argv[0] is PATH-searched unless it contains '/'.
+ argv: []const []const u8,
+ /// Inherited environment; `NINE_MOUNT` is added or replaced.
+ envp: [*:null]const ?[*:0]const u8,
+ /// Absolute mountpoint (see `resolveMountpoint`).
+ mountpoint: []const u8,
+ uid: u32,
+ gid: u32,
+ max_read: u32,
+ /// When false the namespace is set up (including mountpoint shadowing)
+ /// but `/dev/fuse` is not opened and nothing is mounted; `Child.fuse_fd`
+ /// is then -1. Only for smoke tests.
+ mount_fuse: bool = true,
+};
+
+pub const Child = struct {
+ pid: i32,
+ /// The `/dev/fuse` connection backing the mount, opened by the child
+ /// inside its user namespace (the kernel refuses to mount a fuse fd that
+ /// was opened from another user namespace) and handed back over the
+ /// status socket with SCM_RIGHTS. Owned by the caller; CLOEXEC.
+ fuse_fd: i32,
+ /// Parent end of the status socket. The child reports an exec failure
+ /// on it (see `reportExecFailure`); it reads EOF once exec succeeded.
+ status_fd: i32,
+};
+
+/// Exit status used by the child for setup failures (matches 9ns's own).
+pub const setup_failure_status: u8 = 125;
+/// Refuse to shadow a directory with more entries than this.
+pub const max_shadow_entries: usize = 4096;
+
+const default_path = "/usr/local/bin:/bin:/usr/bin";
+const path_max = 4096;
+
+// ---------------------------------------------------------------------------
+// Mountpoint resolution
+// ---------------------------------------------------------------------------
+
+/// Absolute path (relative paths resolved against cwd), duplicate slashes
+/// collapsed, `.` and `..` components resolved lexically, no trailing slash.
+/// `/` itself is rejected.
+pub fn resolveMountpoint(gpa: Allocator, path: []const u8) ![:0]u8 {
+ var cwd_buf: [path_max]u8 = undefined;
+ var cwd: []const u8 = "/";
+ if (path.len == 0 or path[0] != '/') {
+ const rc = linux.getcwd(&cwd_buf, cwd_buf.len);
+ switch (linux.errno(rc)) {
+ .SUCCESS => {},
+ else => |e| {
+ std.debug.print("9ns: getcwd: E{t}\n", .{e});
+ return error.Cwd;
+ },
+ }
+ // rc counts the terminating NUL.
+ cwd = cwd_buf[0 .. rc - 1];
+ }
+ return normalizePath(gpa, cwd, path);
+}
+
+/// Pure part of `resolveMountpoint`: `cwd` is only used when `path` is relative.
+fn normalizePath(gpa: Allocator, cwd: []const u8, path: []const u8) ![:0]u8 {
+ if (path.len == 0) return error.InvalidMountpoint;
+ var out: std.ArrayList(u8) = .empty;
+ defer out.deinit(gpa);
+ if (path[0] != '/') try appendComponents(gpa, &out, cwd);
+ try appendComponents(gpa, &out, path);
+ if (out.items.len == 0) return error.InvalidMountpoint; // "/" or equivalent
+ return out.toOwnedSliceSentinel(gpa, 0);
+}
+
+fn appendComponents(gpa: Allocator, out: *std.ArrayList(u8), path: []const u8) !void {
+ var it = std.mem.tokenizeScalar(u8, path, '/');
+ while (it.next()) |comp| {
+ if (std.mem.eql(u8, comp, ".")) continue;
+ if (std.mem.eql(u8, comp, "..")) {
+ // Pop the last component (lexically; "/.." stays "/").
+ const idx = std.mem.lastIndexOfScalar(u8, out.items, '/') orelse 0;
+ out.shrinkRetainingCapacity(idx);
+ continue;
+ }
+ try out.append(gpa, '/');
+ try out.appendSlice(gpa, comp);
+ }
+}
+
+// ---------------------------------------------------------------------------
+// Environment helpers
+// ---------------------------------------------------------------------------
+
+/// Look a variable up in a raw envp block.
+pub fn getenv(envp: [*:null]const ?[*:0]const u8, name: []const u8) ?[]const u8 {
+ var i: usize = 0;
+ while (envp[i]) |entry| : (i += 1) {
+ const kv = std.mem.span(entry);
+ if (kv.len > name.len and kv[name.len] == '=' and std.mem.eql(u8, kv[0..name.len], name)) {
+ return kv[name.len + 1 ..];
+ }
+ }
+ return null;
+}
+
+/// Every path `execve` should try for `name`, in order: just `name` if it
+/// contains a '/', else `<dir>/<name>` for each `$PATH` element (an empty
+/// element means the current directory; `$PATH` unset falls back to
+/// `/usr/local/bin:/bin:/usr/bin`).
+pub fn pathCandidates(gpa: Allocator, envp: [*:null]const ?[*:0]const u8, name: []const u8) ![]const [:0]const u8 {
+ if (name.len == 0) return error.EmptyProgramName;
+ var list: std.ArrayList([:0]const u8) = .empty;
+ errdefer {
+ for (list.items) |c| gpa.free(c);
+ list.deinit(gpa);
+ }
+ if (std.mem.indexOfScalar(u8, name, '/') != null) {
+ try list.append(gpa, try gpa.dupeZ(u8, name));
+ return list.toOwnedSlice(gpa);
+ }
+ const path = getenv(envp, "PATH") orelse default_path;
+ var it = std.mem.splitScalar(u8, path, ':');
+ while (it.next()) |dir| {
+ const d = if (dir.len == 0) "." else dir;
+ try list.append(gpa, try std.fmt.allocPrintSentinel(gpa, "{s}/{s}", .{ d, name }, 0));
+ }
+ return list.toOwnedSlice(gpa);
+}
+
+/// First PATH candidate that is an executable regular file, or the name
+/// itself when it contains a '/'. Provided for completeness; `spawn` simply
+/// tries `execve` on every candidate instead.
+pub fn findInPath(gpa: Allocator, envp: [*:null]const ?[*:0]const u8, name: []const u8) ![:0]u8 {
+ const cands = try pathCandidates(gpa, envp, name);
+ defer {
+ for (cands) |c| gpa.free(c);
+ gpa.free(cands);
+ }
+ for (cands) |c| {
+ var stx: linux.Statx = undefined;
+ const rc = linux.statx(linux.AT.FDCWD, c.ptr, 0, .{ .TYPE = true, .MODE = true }, &stx);
+ if (linux.errno(rc) != .SUCCESS) continue;
+ if (stx.mode & linux.S.IFMT != linux.S.IFREG) continue;
+ if (stx.mode & 0o111 == 0) continue;
+ return gpa.dupeZ(u8, c);
+ }
+ return error.FileNotFound;
+}
+
+/// New envp block: every entry of `envp` except `NINE_MOUNT=...`,
+/// followed by `NINE_MOUNT=<mountpoint>`.
+fn buildEnvp(gpa: Allocator, envp: [*:null]const ?[*:0]const u8, mountpoint: []const u8) ![:null]?[*:0]const u8 {
+ const key = "NINE_MOUNT=";
+ var keep: usize = 0;
+ var i: usize = 0;
+ while (envp[i]) |entry| : (i += 1) {
+ if (!std.mem.startsWith(u8, std.mem.span(entry), key)) keep += 1;
+ }
+ const out = try gpa.allocSentinel(?[*:0]const u8, keep + 1, null);
+ errdefer gpa.free(out);
+ var j: usize = 0;
+ i = 0;
+ while (envp[i]) |entry| : (i += 1) {
+ if (std.mem.startsWith(u8, std.mem.span(entry), key)) continue;
+ out[j] = entry;
+ j += 1;
+ }
+ const mount_entry = try std.fmt.allocPrintSentinel(gpa, key ++ "{s}", .{mountpoint}, 0);
+ out[j] = mount_entry.ptr;
+ return out;
+}
+
+fn buildArgv(gpa: Allocator, argv: []const []const u8) ![:null]?[*:0]const u8 {
+ const out = try gpa.allocSentinel(?[*:0]const u8, argv.len, null);
+ for (argv, 0..) |a, i| out[i] = (try gpa.dupeZ(u8, a)).ptr;
+ return out;
+}
+
+// ---------------------------------------------------------------------------
+// Mountpoint policy
+// ---------------------------------------------------------------------------
+
+/// Make sure `path` is a directory, inside the *current* mount namespace:
+///
+/// * already a directory → done;
+/// * else `mkdir`; on `EACCES`/`EPERM`/`EROFS` shadow the parent directory
+/// with a tmpfs that re-exposes every existing entry (bind mounts for
+/// directories and files, recreated symlinks) and `mkdir` inside it;
+/// * anything else fails with the errno and a hint.
+///
+/// Every failure prints `9ns: <step> <path>: E<errno>` to stderr before
+/// returning. Meant to be called in the child of `spawn` (or from a
+/// throwaway namespace: `unshare -Urm`).
+pub fn ensureMountpoint(gpa: Allocator, path: [:0]const u8) !void {
+ if (fileType(linux.AT.FDCWD, path, false)) |ft| {
+ if (ft == .dir) return;
+ std.debug.print("9ns: mountpoint {s}: exists but is not a directory\n", .{path});
+ return error.Mountpoint;
+ }
+ if (fileType(linux.AT.FDCWD, path, true) == .symlink) {
+ std.debug.print("9ns: mountpoint {s}: dangling symlink\n", .{path});
+ return error.Mountpoint;
+ }
+ const mk = linux.errno(linux.mkdirat(linux.AT.FDCWD, path, 0o755));
+ switch (mk) {
+ .SUCCESS => return,
+ .ACCES, .PERM, .ROFS => {},
+ else => |e| {
+ std.debug.print("9ns: mkdir {s}: E{t} (pass --mount an existing directory)\n", .{ path, e });
+ return error.Mountpoint;
+ },
+ }
+ const parent = std.fs.path.dirname(path) orelse "/";
+ if (std.mem.eql(u8, parent, "/") or isSameDirectory(parent, "/")) {
+ std.debug.print("9ns: mkdir {s}: E{t}; refusing to shadow / (pass --mount an existing directory)\n", .{ path, mk });
+ return error.Mountpoint;
+ }
+ // The shadow rebuilds entries from /proc/self/fd/<fd>/<name>; a tmpfs
+ // over /proc (or a subtree of it) would take that away from itself.
+ if (std.mem.eql(u8, parent, "/proc") or std.mem.startsWith(u8, parent, "/proc/")) {
+ std.debug.print("9ns: mkdir {s}: E{t}; refusing to shadow {s} (pass --mount an existing directory)\n", .{ path, mk, parent });
+ return error.Mountpoint;
+ }
+ const parent_z = try gpa.dupeZ(u8, parent);
+ defer gpa.free(parent_z);
+ try shadowDirectory(gpa, parent_z);
+ switch (linux.errno(linux.mkdirat(linux.AT.FDCWD, path, 0o755))) {
+ .SUCCESS => {},
+ else => |e| {
+ std.debug.print("9ns: mkdir {s} (in shadow tmpfs): E{t}\n", .{ path, e });
+ return error.Mountpoint;
+ },
+ }
+}
+
+const FileType = enum { dir, symlink, other };
+
+/// True when both paths resolve (following symlinks, including magic ones
+/// such as /proc/self/root) to the same inode.
+fn isSameDirectory(a: []const u8, b: [*:0]const u8) bool {
+ var a_buf: [path_max]u8 = undefined;
+ const a_z = std.fmt.bufPrintZ(&a_buf, "{s}", .{a}) catch return false;
+ var sa: linux.Statx = undefined;
+ var sb: linux.Statx = undefined;
+ if (linux.errno(linux.statx(linux.AT.FDCWD, a_z, 0, .{ .INO = true }, &sa)) != .SUCCESS) return false;
+ if (linux.errno(linux.statx(linux.AT.FDCWD, b, 0, .{ .INO = true }, &sb)) != .SUCCESS) return false;
+ return sa.ino == sb.ino and sa.dev_major == sb.dev_major and sa.dev_minor == sb.dev_minor;
+}
+
+fn fileType(dirfd: i32, name: [*:0]const u8, nofollow: bool) ?FileType {
+ var stx: linux.Statx = undefined;
+ const flags: u32 = if (nofollow) linux.AT.SYMLINK_NOFOLLOW else 0;
+ const rc = linux.statx(dirfd, name, flags, .{ .TYPE = true }, &stx);
+ if (linux.errno(rc) != .SUCCESS) return null;
+ return switch (stx.mode & linux.S.IFMT) {
+ linux.S.IFDIR => .dir,
+ linux.S.IFLNK => .symlink,
+ else => .other,
+ };
+}
+
+const Entry = struct { name: [:0]u8, kind: FileType };
+
+/// Read every entry of the directory open at `fd` (excluding `.` and `..`).
+fn listDir(gpa: Allocator, fd: i32, dirpath: []const u8) ![]Entry {
+ var list: std.ArrayList(Entry) = .empty;
+ errdefer {
+ for (list.items) |e| gpa.free(e.name);
+ list.deinit(gpa);
+ }
+ var buf: [32 * 1024]u8 align(@alignOf(linux.dirent64)) = undefined;
+ while (true) {
+ const rc = linux.getdents64(fd, &buf, buf.len);
+ switch (linux.errno(rc)) {
+ .SUCCESS => {},
+ else => |e| {
+ std.debug.print("9ns: getdents64 {s}: E{t}\n", .{ dirpath, e });
+ return error.Mountpoint;
+ },
+ }
+ if (rc == 0) break;
+ var off: usize = 0;
+ while (off < rc) {
+ const d: *align(1) const linux.dirent64 = @ptrCast(&buf[off]);
+ const name_ptr: [*:0]const u8 = @ptrCast(&buf[off + @offsetOf(linux.dirent64, "name")]);
+ const name = std.mem.span(name_ptr);
+ const dtype = d.type;
+ off += d.reclen;
+ if (std.mem.eql(u8, name, ".") or std.mem.eql(u8, name, "..")) continue;
+ if (list.items.len >= max_shadow_entries) {
+ std.debug.print("9ns: refusing to shadow {s}: more than {d} entries\n", .{ dirpath, max_shadow_entries });
+ return error.TooManyEntries;
+ }
+ const kind: FileType = switch (dtype) {
+ linux.DT.DIR => .dir,
+ linux.DT.LNK => .symlink,
+ linux.DT.UNKNOWN => fileType(fd, name_ptr, true) orelse .other,
+ else => .other,
+ };
+ try list.append(gpa, .{ .name = try gpa.dupeZ(u8, name), .kind = kind });
+ }
+ }
+ return list.toOwnedSlice(gpa);
+}
+
+fn shadowDirectory(gpa: Allocator, parent: [:0]const u8) !void {
+ const open_rc = linux.open(parent, .{ .ACCMODE = .RDONLY, .DIRECTORY = true, .CLOEXEC = true }, 0);
+ switch (linux.errno(open_rc)) {
+ .SUCCESS => {},
+ else => |e| {
+ std.debug.print("9ns: open {s}: E{t}\n", .{ parent, e });
+ return error.Mountpoint;
+ },
+ }
+ const pfd: i32 = @intCast(open_rc);
+ defer _ = linux.close(pfd);
+
+ const entries = try listDir(gpa, pfd, parent);
+ defer {
+ for (entries) |e| gpa.free(e.name);
+ gpa.free(entries);
+ }
+
+ const tmpfs_opts: [*:0]const u8 = "mode=755";
+ switch (linux.errno(linux.mount("tmpfs", parent, "tmpfs", linux.MS.NOSUID | linux.MS.NODEV, @intFromPtr(tmpfs_opts)))) {
+ .SUCCESS => {},
+ else => |e| {
+ std.debug.print("9ns: mount tmpfs on {s}: E{t}\n", .{ parent, e });
+ return error.Mountpoint;
+ },
+ }
+
+ // `pfd` still refers to the original directory underneath the tmpfs, so
+ // `/proc/self/fd/<pfd>/<name>` reaches the hidden entries.
+ var src_buf: [path_max]u8 = undefined;
+ var dst_buf: [path_max]u8 = undefined;
+ var link_buf: [path_max]u8 = undefined;
+ for (entries) |e| {
+ const src = std.fmt.bufPrintZ(&src_buf, "/proc/self/fd/{d}/{s}", .{ pfd, e.name }) catch {
+ std.debug.print("9ns: shadow {s}/{s}: name too long (skipped)\n", .{ parent, e.name });
+ continue;
+ };
+ const dst = std.fmt.bufPrintZ(&dst_buf, "{s}/{s}", .{ parent, e.name }) catch {
+ std.debug.print("9ns: shadow {s}/{s}: name too long (skipped)\n", .{ parent, e.name });
+ continue;
+ };
+ switch (e.kind) {
+ .dir => {
+ if (!check("mkdir", dst, linux.mkdirat(linux.AT.FDCWD, dst, 0o755))) continue;
+ _ = check("bind", dst, linux.mount(src, dst, null, linux.MS.BIND | linux.MS.REC, 0));
+ },
+ .symlink => {
+ const rc = linux.readlinkat(pfd, e.name, &link_buf, link_buf.len - 1);
+ if (!check("readlink", dst, rc)) continue;
+ link_buf[rc] = 0;
+ const target: [*:0]const u8 = @ptrCast(&link_buf);
+ _ = check("symlink", dst, linux.symlinkat(target, linux.AT.FDCWD, dst));
+ },
+ .other => {
+ const rc = linux.openat(linux.AT.FDCWD, dst, .{ .ACCMODE = .WRONLY, .CREAT = true, .CLOEXEC = true }, 0o644);
+ if (!check("create", dst, rc)) continue;
+ _ = linux.close(@intCast(rc));
+ _ = check("bind", dst, linux.mount(src, dst, null, linux.MS.BIND | linux.MS.REC, 0));
+ },
+ }
+ }
+}
+
+/// Report a failed per-entry step as a warning (the entry is skipped; the
+/// rest of the shadow is still useful). Returns true on success.
+fn check(step: []const u8, path: [*:0]const u8, rc: usize) bool {
+ switch (linux.errno(rc)) {
+ .SUCCESS => return true,
+ else => |e| {
+ std.debug.print("9ns: shadow: {s} {s}: E{t} (skipped)\n", .{ step, std.mem.span(path), e });
+ return false;
+ },
+ }
+}
+
+// ---------------------------------------------------------------------------
+// spawn
+// ---------------------------------------------------------------------------
+
+const ChildArgs = struct {
+ gpa: Allocator,
+ status_sock: i32,
+ mountpoint: [:0]const u8,
+ fuse_opts_prefix: [:0]const u8, // everything after "fd=<n>,"
+ mount_fuse: bool,
+ uid_map: []const u8,
+ gid_map: []const u8,
+ argv: [:null]?[*:0]const u8,
+ envp: [:null]?[*:0]const u8,
+ candidates: []const [:0]const u8,
+ name: []const u8,
+};
+
+/// Status channel protocol (child → parent, over a CLOEXEC socketpair):
+/// a 0 byte means "namespace and mount are up" and carries the fuse fd as
+/// SCM_RIGHTS; a non-zero byte is an exit status followed by a message.
+/// EOF ends the conversation (exec succeeded, or the child died).
+const ok_byte: u8 = 0;
+
+/// fork; the child unshares user+mount namespaces, maps its uid/gid,
+/// makes `/` private, ensures the mountpoint, opens `/dev/fuse`, mounts it
+/// on the mountpoint and sends the fd back. `spawn` returns at that point
+/// (with `error.ChildFailed` and a message on stderr if any step failed).
+/// The child then stats the mountpoint, which makes the kernel fetch the
+/// root's attributes once the parent serves (the kernel seeds the fuse root
+/// with uid 0, unmapped in the new user namespace, so nothing could be
+/// created in the root until then), sets `NINE_MOUNT` and execs
+/// `argv`. An exec failure is reported on `Child.status_fd` and ends the
+/// child with 126/127; collect it with `reportExecFailure` after
+/// `bridge.serve` returns.
+///
+/// If `installSignals` was called, the pid is stored into the registered
+/// variable as soon as fork returns so no SIGCHLD can be missed.
+pub fn spawn(gpa: Allocator, s: Spawn) !Child {
+ if (s.argv.len == 0 or s.argv[0].len == 0) {
+ std.debug.print("9ns: empty program name\n", .{});
+ return error.EmptyProgramName;
+ }
+
+ const mountpoint = try gpa.dupeZ(u8, s.mountpoint);
+ defer gpa.free(mountpoint);
+ const fuse_opts_prefix = try std.fmt.allocPrintSentinel(gpa, "rootmode=40000,user_id={d},group_id={d},max_read={d}", .{ s.uid, s.gid, s.max_read }, 0);
+ defer gpa.free(fuse_opts_prefix);
+ var uid_buf: [64]u8 = undefined;
+ var gid_buf: [64]u8 = undefined;
+ const uid_map = try std.fmt.bufPrint(&uid_buf, "{d} {d} 1\n", .{ s.uid, s.uid });
+ const gid_map = try std.fmt.bufPrint(&gid_buf, "{d} {d} 1\n", .{ s.gid, s.gid });
+ const argv = try buildArgv(gpa, s.argv);
+ defer {
+ for (argv) |a| gpa.free(std.mem.span(a.?));
+ gpa.free(argv);
+ }
+ const envp = try buildEnvp(gpa, s.envp, s.mountpoint);
+ defer {
+ gpa.free(std.mem.span(envp[envp.len - 1].?)); // the NINE_MOUNT entry we created
+ gpa.free(envp);
+ }
+ const candidates = try pathCandidates(gpa, s.envp, s.argv[0]);
+ defer {
+ for (candidates) |c| gpa.free(c);
+ gpa.free(candidates);
+ }
+
+ var sv: [2]i32 = undefined;
+ switch (linux.errno(linux.socketpair(linux.AF.UNIX, linux.SOCK.STREAM | linux.SOCK.CLOEXEC, 0, &sv))) {
+ .SUCCESS => {},
+ else => |e| {
+ std.debug.print("9ns: socketpair: E{t}\n", .{e});
+ return error.SystemResources;
+ },
+ }
+
+ const child_args = ChildArgs{
+ .gpa = gpa,
+ .status_sock = sv[1],
+ .mountpoint = mountpoint,
+ .fuse_opts_prefix = fuse_opts_prefix,
+ .mount_fuse = s.mount_fuse,
+ .uid_map = uid_map,
+ .gid_map = gid_map,
+ .argv = argv,
+ .envp = envp,
+ .candidates = candidates,
+ .name = s.argv[0],
+ };
+
+ const fork_rc = linux.fork();
+ switch (linux.errno(fork_rc)) {
+ .SUCCESS => {},
+ else => |e| {
+ _ = linux.close(sv[0]);
+ _ = linux.close(sv[1]);
+ std.debug.print("9ns: fork: E{t}\n", .{e});
+ return error.SystemResources;
+ },
+ }
+ if (fork_rc == 0) childMain(&child_args);
+
+ const pid: i32 = @intCast(fork_rc);
+ if (child_pid_ptr) |p| @atomicStore(i32, p, pid, .seq_cst);
+ _ = linux.close(sv[1]);
+
+ // First byte: ok (with the fuse fd attached) or a failure status.
+ var first: [1]u8 = .{ok_byte}; // defined even if recvmsg stores nothing
+ var fuse_fd: i32 = -1;
+ var n: usize = 0;
+ while (true) {
+ const rc = recvWithFd(sv[0], &first, &fuse_fd, 0);
+ switch (linux.errno(rc)) {
+ .SUCCESS => {},
+ .INTR => continue,
+ else => break,
+ }
+ n = rc;
+ break;
+ }
+ if (n == 1 and first[0] == ok_byte and (fuse_fd >= 0 or !s.mount_fuse)) {
+ return .{ .pid = pid, .fuse_fd = fuse_fd, .status_fd = sv[0] };
+ }
+
+ // Failure. A status byte means the child is exiting on its own and a
+ // message follows. Anything else (EOF: the child died before reporting;
+ // an ok byte without the fd: the SCM_RIGHTS transfer was truncated, e.g.
+ // EMFILE) is a protocol violation: the child may be about to exec with a
+ // dead mount, so kill it before waiting rather than reading the status
+ // socket until an exec'd program eventually exits.
+ const reported = n == 1 and first[0] != ok_byte;
+ if (!reported) _ = linux.kill(pid, .KILL);
+ if (fuse_fd >= 0) _ = linux.close(fuse_fd);
+ var msg: [512]u8 = undefined;
+ var len: usize = 0;
+ while (reported and len < msg.len) {
+ const rc = linux.read(sv[0], msg[len..].ptr, msg.len - len);
+ switch (linux.errno(rc)) {
+ .SUCCESS => {},
+ .INTR => continue,
+ else => break,
+ }
+ if (rc == 0) break;
+ len += rc;
+ }
+ _ = linux.close(sv[0]);
+ if (reported) {
+ std.debug.print("9ns: {s}\n", .{msg[0..len]});
+ } else if (n == 1) {
+ std.debug.print("9ns: child handshake failed: no fuse fd received (out of file descriptors?)\n", .{});
+ } else {
+ std.debug.print("9ns: child exited before reporting\n", .{});
+ }
+ _ = waitChild(pid) catch {};
+ if (child_pid_ptr) |p| @atomicStore(i32, p, 0, .seq_cst);
+ const status: u8 = if (reported) first[0] else setup_failure_status;
+ return switch (status) {
+ 126 => error.ExecPermission,
+ 127 => error.ExecNotFound,
+ else => error.ChildFailed,
+ };
+}
+
+/// After the child is gone (or the mount is dead): print the exec failure
+/// the child reported on `status_fd`, if any, and close it. Returns the
+/// status byte the child announced, or null when exec succeeded / nothing
+/// was reported. Never blocks.
+pub fn reportExecFailure(child: Child) ?u8 {
+ defer _ = linux.close(child.status_fd);
+ var msg: [512]u8 = undefined;
+ var len: usize = 0;
+ while (len < msg.len) {
+ var iov = [_]std.posix.iovec{.{ .base = msg[len..].ptr, .len = msg.len - len }};
+ var hdr = linux.msghdr{
+ .name = null,
+ .namelen = 0,
+ .iov = &iov,
+ .iovlen = 1,
+ .control = null,
+ .controllen = 0,
+ .flags = 0,
+ };
+ const rc = linux.recvmsg(child.status_fd, &hdr, linux.MSG.DONTWAIT);
+ switch (linux.errno(rc)) {
+ .SUCCESS => {},
+ .INTR => continue,
+ else => break,
+ }
+ if (rc == 0) break;
+ len += rc;
+ }
+ if (len == 0) return null;
+ std.debug.print("9ns: {s}\n", .{msg[1..len]});
+ return msg[0];
+}
+
+const cmsg_fd_len = @sizeOf(linux.cmsghdr) + @sizeOf(i32);
+const cmsg_fd_space = std.mem.alignForward(usize, cmsg_fd_len, @sizeOf(usize));
+
+/// sendmsg one data byte, optionally with `fd` attached as SCM_RIGHTS.
+fn sendWithFd(sock: i32, byte: u8, fd: ?i32) usize {
+ const data = [_]u8{byte};
+ const iov = [_]std.posix.iovec_const{.{ .base = &data, .len = 1 }};
+ var cbuf: [cmsg_fd_space]u8 align(@alignOf(linux.cmsghdr)) = @splat(0);
+ var msg = linux.msghdr_const{
+ .name = null,
+ .namelen = 0,
+ .iov = &iov,
+ .iovlen = 1,
+ .control = null,
+ .controllen = 0,
+ .flags = 0,
+ };
+ if (fd) |f| {
+ const hdr: *linux.cmsghdr = @ptrCast(&cbuf);
+ hdr.* = .{ .len = cmsg_fd_len, .level = linux.SOL.SOCKET, .type = linux.SCM.RIGHTS };
+ @memcpy(cbuf[@sizeOf(linux.cmsghdr)..][0..@sizeOf(i32)], std.mem.asBytes(&f));
+ msg.control = &cbuf;
+ msg.controllen = cmsg_fd_space;
+ }
+ return linux.sendmsg(sock, &msg, linux.MSG.NOSIGNAL);
+}
+
+/// recvmsg into `buf`; an SCM_RIGHTS fd, if any, is stored in `fd_out`.
+fn recvWithFd(sock: i32, buf: []u8, fd_out: *i32, flags: u32) usize {
+ var iov = [_]std.posix.iovec{.{ .base = buf.ptr, .len = buf.len }};
+ var cbuf: [cmsg_fd_space]u8 align(@alignOf(linux.cmsghdr)) = @splat(0);
+ var msg = linux.msghdr{
+ .name = null,
+ .namelen = 0,
+ .iov = &iov,
+ .iovlen = 1,
+ .control = &cbuf,
+ .controllen = cbuf.len,
+ .flags = 0,
+ };
+ const rc = linux.recvmsg(sock, &msg, linux.MSG.CMSG_CLOEXEC | flags);
+ if (linux.errno(rc) != .SUCCESS) return rc;
+ if (msg.controllen >= cmsg_fd_len) {
+ const hdr: *const linux.cmsghdr = @ptrCast(&cbuf);
+ if (hdr.level == linux.SOL.SOCKET and hdr.type == linux.SCM.RIGHTS and hdr.len >= cmsg_fd_len) {
+ var fd: i32 = undefined;
+ @memcpy(std.mem.asBytes(&fd), cbuf[@sizeOf(linux.cmsghdr)..][0..@sizeOf(i32)]);
+ fd_out.* = fd;
+ }
+ }
+ return rc;
+}
+
+/// Child side of `spawn`. Never returns.
+fn childMain(c: *const ChildArgs) noreturn {
+ resetSignals();
+
+ const rc_unshare = linux.errno(linux.unshare(linux.CLONE.NEWUSER | linux.CLONE.NEWNS));
+ if (rc_unshare != .SUCCESS) childFail(c, setup_failure_status, "unshare(CLONE_NEWUSER|CLONE_NEWNS)", rc_unshare, true);
+ writeProcFile(c, "/proc/self/setgroups", "deny", true);
+ writeProcFile(c, "/proc/self/uid_map", c.uid_map, false);
+ writeProcFile(c, "/proc/self/gid_map", c.gid_map, false);
+
+ const root: [*:0]const u8 = "/";
+ const rc_priv = linux.mount(null, root, null, linux.MS.REC | linux.MS.PRIVATE, 0);
+ if (linux.errno(rc_priv) != .SUCCESS) childFail(c, setup_failure_status, "mount(/, MS_REC|MS_PRIVATE)", linux.errno(rc_priv), true);
+
+ ensureMountpoint(c.gpa, c.mountpoint) catch {
+ childFail(c, setup_failure_status, "mountpoint setup failed (pass --mount an existing directory)", .SUCCESS, false);
+ };
+
+ var fuse_fd: ?i32 = null;
+ if (c.mount_fuse) {
+ // Must be opened here, after unshare: the kernel only mounts a fuse
+ // device opened from the mount's own user namespace.
+ const rc_open = linux.open("/dev/fuse", .{ .ACCMODE = .RDWR, .CLOEXEC = true }, 0);
+ switch (linux.errno(rc_open)) {
+ .SUCCESS => {},
+ .NOENT => childFail(c, setup_failure_status, "open /dev/fuse: ENOENT (is the fuse module loaded? try: modprobe fuse)", .SUCCESS, false),
+ else => |e| childFail(c, setup_failure_status, "open /dev/fuse", e, true),
+ }
+ const fd: i32 = @intCast(rc_open);
+ var opts_buf: [256]u8 = undefined;
+ const opts = std.fmt.bufPrintZ(&opts_buf, "fd={d},{s}", .{ fd, c.fuse_opts_prefix }) catch unreachable;
+ const rc = linux.mount("9ns", c.mountpoint, "fuse", linux.MS.NOSUID | linux.MS.NODEV, @intFromPtr(opts.ptr));
+ if (linux.errno(rc) != .SUCCESS) childFail(c, setup_failure_status, "mount fuse", linux.errno(rc), true);
+ fuse_fd = fd;
+ }
+ const sent = sendWithFd(c.status_sock, ok_byte, fuse_fd);
+ if (linux.errno(sent) != .SUCCESS) linux.exit_group(setup_failure_status);
+ if (fuse_fd) |fd| {
+ _ = linux.close(fd); // the parent holds the connection now
+ // Force one GETATTR of the root (served by the parent, which is
+ // entering its serve loop now); see `spawn`. Errors don't matter.
+ var stx: linux.Statx = undefined;
+ _ = linux.statx(linux.AT.FDCWD, c.mountpoint, 0, .{ .TYPE = true }, &stx);
+ }
+
+ var last: E = .NOENT;
+ var saw_acces = false;
+ for (c.candidates) |cand| {
+ const rc = linux.execve(cand.ptr, c.argv.ptr, c.envp.ptr);
+ last = linux.errno(rc);
+ switch (last) {
+ .NOENT, .NOTDIR, .LOOP, .NAMETOOLONG => continue,
+ .ACCES => {
+ saw_acces = true;
+ continue;
+ },
+ else => break,
+ }
+ }
+ var buf: [512]u8 = undefined;
+ // "Not found" covers every candidate that could not even be resolved
+ // (a PATH element that is a file gives ENOTDIR, a symlink loop ELOOP);
+ // a candidate that existed but was not executable wins over those.
+ const not_found = switch (last) {
+ .NOENT, .NOTDIR, .LOOP, .NAMETOOLONG => true,
+ else => false,
+ };
+ if (not_found and saw_acces) last = .ACCES;
+ const status: u8 = if (not_found and !saw_acces) 127 else 126;
+ const text = std.fmt.bufPrint(&buf, "exec {s}", .{c.name}) catch "exec";
+ childFail(c, status, text, last, true);
+}
+
+fn writeProcFile(c: *const ChildArgs, path: [*:0]const u8, data: []const u8, ignore_missing: bool) void {
+ const rc = linux.open(path, .{ .ACCMODE = .WRONLY, .CLOEXEC = true }, 0);
+ switch (linux.errno(rc)) {
+ .SUCCESS => {},
+ .NOENT => if (ignore_missing) return else childFail(c, setup_failure_status, std.mem.span(path), .NOENT, true),
+ else => |e| childFail(c, setup_failure_status, std.mem.span(path), e, true),
+ }
+ const fd: i32 = @intCast(rc);
+ const w = linux.write(fd, data.ptr, data.len);
+ const we = linux.errno(w);
+ _ = linux.close(fd);
+ if (we != .SUCCESS) childFail(c, setup_failure_status, std.mem.span(path), we, true);
+ if (w != data.len) childFail(c, setup_failure_status, std.mem.span(path), .IO, true);
+}
+
+/// Write `<status byte><step>[: E<errno>]` to the status socket and exit.
+fn childFail(c: *const ChildArgs, status: u8, step: []const u8, e: E, with_errno: bool) noreturn {
+ var buf: [600]u8 = undefined;
+ buf[0] = status;
+ const rest = if (with_errno)
+ std.fmt.bufPrint(buf[1..], "{s}: E{t}", .{ step, e }) catch buf[1..1]
+ else
+ std.fmt.bufPrint(buf[1..], "{s}", .{step}) catch buf[1..1];
+ const msg = buf[0 .. 1 + rest.len];
+ var off: usize = 0;
+ while (off < msg.len) {
+ const rc = linux.write(c.status_sock, msg[off..].ptr, msg.len - off);
+ if (linux.errno(rc) == .INTR) continue;
+ if (linux.errno(rc) != .SUCCESS) break;
+ off += rc;
+ }
+ linux.exit_group(status);
+}
+
+// ---------------------------------------------------------------------------
+// Signals
+// ---------------------------------------------------------------------------
+
+var child_pid_ptr: ?*i32 = null;
+var chld_pipe_w: i32 = -1;
+var reaped = std.atomic.Value(bool).init(false);
+var reaped_status = std.atomic.Value(u32).init(0);
+/// A second child (the `--spawn` server) that the SIGCHLD handler reaps so
+/// it does not linger as a zombie when it dies mid-session. Its exit does
+/// not stop the serve loop. 0 = none.
+var server_pid = std.atomic.Value(i32).init(0);
+
+/// Register the `--spawn` server for reaping by the SIGCHLD handler.
+pub fn watchServer(pid: i32) void {
+ server_pid.store(pid, .seq_cst);
+}
+
+/// Seconds the serve loop gets to come back after the child died before
+/// the watchdog ends the process anyway.
+pub const exit_grace_seconds: isize = 3;
+
+/// The watched child is already dead but the serve loop has not come back
+/// (it is stuck in a 9P request the server never answers): a terminal
+/// signal, or the watchdog armed by `onChld`, then ends 9ns with the
+/// child's status instead of hanging. Nothing is lost: the mount is torn
+/// down when the process exits.
+fn bailIfChildGone() void {
+ if (!reaped.load(.acquire)) return;
+ const srv = server_pid.load(.seq_cst);
+ if (srv > 0) _ = linux.kill(srv, .TERM);
+ linux.exit_group(decodeStatus(reaped_status.load(.acquire)));
+}
+
+fn armWatchdog() void {
+ // setitimer takes an itimerval; std declares it with itimerspec, which
+ // has the same layout on 64-bit targets (the sub-second field is 0).
+ const t = linux.itimerspec{
+ .it_interval = .{ .sec = 0, .nsec = 0 },
+ .it_value = .{ .sec = exit_grace_seconds, .nsec = 0 },
+ };
+ _ = linux.setitimer(@intFromEnum(linux.ITIMER.REAL), &t, null);
+}
+
+fn onAlarm(_: linux.SIG) callconv(.c) void {
+ bailIfChildGone();
+}
+
+fn onForward(sig: linux.SIG) callconv(.c) void {
+ const p = child_pid_ptr orelse return;
+ const pid = @atomicLoad(i32, p, .seq_cst);
+ if (pid > 0) _ = linux.kill(pid, sig);
+ bailIfChildGone();
+}
+
+/// SIGINT/SIGQUIT: the child owns the tty and gets them itself; we only
+/// react when the child is already gone (see `bailIfChildGone`).
+fn onTerminal(_: linux.SIG) callconv(.c) void {
+ bailIfChildGone();
+}
+
+/// Only the watched child counts: reap it here (WNOHANG), remember its
+/// status, forget its pid (so a later SIGTERM cannot hit a recycled pid)
+/// and poke the self-pipe. The `--spawn` server is reaped too but does not
+/// interrupt `bridge.serve`; SIGCHLD from anything else is ignored.
+fn onChld(_: linux.SIG) callconv(.c) void {
+ const srv = server_pid.load(.seq_cst);
+ if (srv > 0) {
+ var sst: u32 = 0;
+ const src = linux.waitpid(srv, &sst, linux.W.NOHANG);
+ if (linux.errno(src) == .SUCCESS and src != 0) server_pid.store(0, .seq_cst);
+ }
+ const p = child_pid_ptr orelse return;
+ const pid = @atomicLoad(i32, p, .seq_cst);
+ if (pid <= 0) return;
+ var st: u32 = 0;
+ const rc = linux.waitpid(pid, &st, linux.W.NOHANG);
+ if (linux.errno(rc) != .SUCCESS or rc == 0) return;
+ reaped_status.store(st, .release);
+ reaped.store(true, .release);
+ @atomicStore(i32, p, 0, .seq_cst);
+ const b = [_]u8{'c'};
+ _ = linux.write(chld_pipe_w, &b, 1);
+ armWatchdog();
+}
+
+/// SIGPIPE ignored; SIGINT/SIGQUIT effectively ignored (the child owns the
+/// tty) unless the child is already dead; SIGTERM/SIGHUP forwarded to
+/// `*child_pid`; SIGCHLD for `*child_pid` reaps it, writes a byte to a
+/// nonblocking self-pipe whose read end is returned (use it as `stop_fd`)
+/// and arms a watchdog (`exit_grace_seconds`, SIGALRM) that ends the
+/// process with the child's status should the serve loop stay blocked.
+/// `*child_pid` is filled in by `spawn`.
+pub fn installSignals(child_pid: *i32) !i32 {
+ child_pid_ptr = child_pid;
+ var fds: [2]i32 = undefined;
+ switch (linux.errno(linux.pipe2(&fds, .{ .CLOEXEC = true, .NONBLOCK = true }))) {
+ .SUCCESS => {},
+ else => |e| {
+ std.debug.print("9ns: pipe2: E{t}\n", .{e});
+ return error.SystemResources;
+ },
+ }
+ chld_pipe_w = fds[1];
+
+ const ign = linux.Sigaction{ .handler = .{ .handler = linux.SIG.IGN }, .mask = linux.sigemptyset(), .flags = 0 };
+ const term = linux.Sigaction{ .handler = .{ .handler = &onTerminal }, .mask = linux.sigemptyset(), .flags = linux.SA.RESTART };
+ const fwd = linux.Sigaction{ .handler = .{ .handler = &onForward }, .mask = linux.sigemptyset(), .flags = linux.SA.RESTART };
+ const chld = linux.Sigaction{ .handler = .{ .handler = &onChld }, .mask = linux.sigemptyset(), .flags = linux.SA.RESTART | linux.SA.NOCLDSTOP };
+ const alrm = linux.Sigaction{ .handler = .{ .handler = &onAlarm }, .mask = linux.sigemptyset(), .flags = linux.SA.RESTART };
+ std.posix.sigaction(.INT, &term, null);
+ std.posix.sigaction(.QUIT, &term, null);
+ std.posix.sigaction(.ALRM, &alrm, null);
+ std.posix.sigaction(.PIPE, &ign, null);
+ std.posix.sigaction(.TERM, &fwd, null);
+ std.posix.sigaction(.HUP, &fwd, null);
+ std.posix.sigaction(.CHLD, &chld, null);
+ return fds[0];
+}
+
+/// Restore default dispositions in the child before exec (ignored signals
+/// would otherwise survive execve).
+fn resetSignals() void {
+ const dfl = linux.Sigaction{ .handler = .{ .handler = linux.SIG.DFL }, .mask = linux.sigemptyset(), .flags = 0 };
+ inline for (.{ linux.SIG.INT, linux.SIG.QUIT, linux.SIG.PIPE, linux.SIG.TERM, linux.SIG.HUP, linux.SIG.CHLD, linux.SIG.ALRM }) |sig| {
+ _ = linux.sigaction(sig, &dfl, null);
+ }
+}
+
+// ---------------------------------------------------------------------------
+// Waiting
+// ---------------------------------------------------------------------------
+
+fn takeReaped() ?u32 {
+ if (!reaped.load(.acquire)) return null;
+ return reaped_status.load(.acquire);
+}
+
+/// waitpid status → exit code (`128+sig` when killed by a signal).
+pub fn decodeStatus(st: u32) u8 {
+ if (linux.W.IFEXITED(st)) return linux.W.EXITSTATUS(st);
+ if (linux.W.IFSIGNALED(st)) return 128 +% @as(u8, @truncate(@intFromEnum(linux.W.TERMSIG(st))));
+ return 1;
+}
+
+/// Block until `pid` exits (the SIGCHLD handler may have reaped it already).
+pub fn waitChild(pid: i32) !u8 {
+ while (true) {
+ if (takeReaped()) |st| return decodeStatus(st);
+ var st: u32 = 0;
+ const rc = linux.waitpid(pid, &st, 0);
+ switch (linux.errno(rc)) {
+ .SUCCESS => return decodeStatus(st),
+ .INTR => continue,
+ .CHILD => {
+ if (takeReaped()) |s| return decodeStatus(s);
+ return error.NoChild;
+ },
+ else => |e| {
+ std.debug.print("9ns: waitpid: E{t}\n", .{e});
+ return error.Wait;
+ },
+ }
+ }
+}
+
+/// Non-blocking: the exit status of `pid` if it has exited, else null.
+pub fn reapIfExited(pid: i32) ?u8 {
+ if (takeReaped()) |st| return decodeStatus(st);
+ var st: u32 = 0;
+ const rc = linux.waitpid(pid, &st, linux.W.NOHANG);
+ switch (linux.errno(rc)) {
+ .SUCCESS => return if (rc == 0) null else decodeStatus(st),
+ .CHILD => return if (takeReaped()) |s| decodeStatus(s) else null,
+ else => return null,
+ }
+}
+
+/// Reap any child (used for the `--spawn` server at exit). Non-blocking.
+pub fn reapAny(pid: i32) void {
+ var st: u32 = 0;
+ _ = linux.waitpid(pid, &st, linux.W.NOHANG);
+}
+
+// ---------------------------------------------------------------------------
+// Tests (no namespaces needed; `ensureMountpoint` is exercised by
+// test/integration.sh through the 9ns binary)
+// ---------------------------------------------------------------------------
+
+const testing = std.testing;
+
+test "normalizePath: absolute paths" {
+ const gpa = testing.allocator;
+ const cases = [_]struct { in: []const u8, out: []const u8 }{
+ .{ .in = "/mnt/9p", .out = "/mnt/9p" },
+ .{ .in = "/mnt/9p/", .out = "/mnt/9p" },
+ .{ .in = "//mnt///9p//", .out = "/mnt/9p" },
+ .{ .in = "/mnt/./9p/.", .out = "/mnt/9p" },
+ .{ .in = "/mnt/x/../9p", .out = "/mnt/9p" },
+ .{ .in = "/../mnt/9p", .out = "/mnt/9p" },
+ .{ .in = "/a/b/c/../..", .out = "/a" },
+ };
+ for (cases) |c| {
+ const got = try normalizePath(gpa, "/cwd", c.in);
+ defer gpa.free(got);
+ try testing.expectEqualStrings(c.out, got);
+ try testing.expectEqual(@as(u8, 0), got[got.len]);
+ }
+}
+
+test "normalizePath: relative paths use cwd" {
+ const gpa = testing.allocator;
+ const cases = [_]struct { cwd: []const u8, in: []const u8, out: []const u8 }{
+ .{ .cwd = "/home/me", .in = "mnt", .out = "/home/me/mnt" },
+ .{ .cwd = "/home/me", .in = "./mnt/", .out = "/home/me/mnt" },
+ .{ .cwd = "/home/me", .in = "../mnt", .out = "/home/mnt" },
+ .{ .cwd = "/home/me/", .in = ".", .out = "/home/me" },
+ .{ .cwd = "/", .in = "x", .out = "/x" },
+ };
+ for (cases) |c| {
+ const got = try normalizePath(gpa, c.cwd, c.in);
+ defer gpa.free(got);
+ try testing.expectEqualStrings(c.out, got);
+ }
+}
+
+test "normalizePath: rejects root and empty" {
+ const gpa = testing.allocator;
+ try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/cwd", "/"));
+ try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/cwd", "///"));
+ try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/cwd", "/mnt/.."));
+ try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/cwd", ""));
+ try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/", ".."));
+}
+
+test "resolveMountpoint: relative resolves against the real cwd" {
+ const gpa = testing.allocator;
+ const got = try resolveMountpoint(gpa, "sub/dir");
+ defer gpa.free(got);
+ try testing.expect(got[0] == '/');
+ try testing.expect(std.mem.endsWith(u8, got, "/sub/dir"));
+}
+
+test "getenv" {
+ const env = [_:null]?[*:0]const u8{ "PATH=/a:/b", "X=", "PATHX=no", "NINE_MOUNT=/m" };
+ const envp: [*:null]const ?[*:0]const u8 = &env;
+ try testing.expectEqualStrings("/a:/b", getenv(envp, "PATH").?);
+ try testing.expectEqualStrings("", getenv(envp, "X").?);
+ try testing.expectEqualStrings("/m", getenv(envp, "NINE_MOUNT").?);
+ try testing.expect(getenv(envp, "NOPE") == null);
+ try testing.expect(getenv(envp, "PAT") == null);
+}
+
+test "pathCandidates: PATH search" {
+ const gpa = testing.allocator;
+ const env = [_:null]?[*:0]const u8{ "PATH=/usr/local/bin::/usr/bin", "HOME=/h" };
+ const cands = try pathCandidates(gpa, &env, "fish");
+ defer {
+ for (cands) |c| gpa.free(c);
+ gpa.free(cands);
+ }
+ try testing.expectEqual(@as(usize, 3), cands.len);
+ try testing.expectEqualStrings("/usr/local/bin/fish", cands[0]);
+ try testing.expectEqualStrings("./fish", cands[1]);
+ try testing.expectEqualStrings("/usr/bin/fish", cands[2]);
+}
+
+test "pathCandidates: slash means no search; default PATH" {
+ const gpa = testing.allocator;
+ const env = [_:null]?[*:0]const u8{"HOME=/h"};
+ {
+ const cands = try pathCandidates(gpa, &env, "./bin/x");
+ defer {
+ for (cands) |c| gpa.free(c);
+ gpa.free(cands);
+ }
+ try testing.expectEqual(@as(usize, 1), cands.len);
+ try testing.expectEqualStrings("./bin/x", cands[0]);
+ }
+ {
+ const cands = try pathCandidates(gpa, &env, "sh");
+ defer {
+ for (cands) |c| gpa.free(c);
+ gpa.free(cands);
+ }
+ try testing.expectEqual(@as(usize, 3), cands.len);
+ try testing.expectEqualStrings("/usr/local/bin/sh", cands[0]);
+ try testing.expectEqualStrings("/bin/sh", cands[1]);
+ }
+ try testing.expectError(error.EmptyProgramName, pathCandidates(gpa, &env, ""));
+}
+
+test "findInPath finds sh" {
+ const gpa = testing.allocator;
+ const env = [_:null]?[*:0]const u8{"PATH=/nonexistent:/bin:/usr/bin"};
+ const p = try findInPath(gpa, &env, "sh");
+ defer gpa.free(p);
+ try testing.expect(std.mem.endsWith(u8, p, "/sh"));
+ try testing.expectError(error.FileNotFound, findInPath(gpa, &env, "definitely-not-a-program-9ns"));
+}
+
+test "buildEnvp replaces NINE_MOUNT" {
+ const gpa = testing.allocator;
+ const env = [_:null]?[*:0]const u8{ "A=1", "NINE_MOUNT=/old", "B=2" };
+ const out = try buildEnvp(gpa, &env, "/mnt/9p");
+ defer {
+ gpa.free(std.mem.span(out[out.len - 1].?));
+ gpa.free(out);
+ }
+ try testing.expectEqual(@as(usize, 3), out.len);
+ try testing.expectEqualStrings("A=1", std.mem.span(out[0].?));
+ try testing.expectEqualStrings("B=2", std.mem.span(out[1].?));
+ try testing.expectEqualStrings("NINE_MOUNT=/mnt/9p", std.mem.span(out[2].?));
+ try testing.expect(out[3] == null);
+ try testing.expectEqualStrings("/mnt/9p", getenv(out.ptr, "NINE_MOUNT").?);
+}
+
+test "decodeStatus" {
+ try testing.expectEqual(@as(u8, 0), decodeStatus(0));
+ try testing.expectEqual(@as(u8, 7), decodeStatus(7 << 8));
+ try testing.expectEqual(@as(u8, 255), decodeStatus(255 << 8));
+ try testing.expectEqual(@as(u8, 128 + 9), decodeStatus(9)); // SIGKILL
+ try testing.expectEqual(@as(u8, 128 + 15), decodeStatus(15)); // SIGTERM
+}
+
+test "ensureMountpoint: existing directory is accepted, plain file rejected" {
+ const gpa = testing.allocator;
+ try ensureMountpoint(gpa, "/tmp");
+ try testing.expectError(error.Mountpoint, ensureMountpoint(gpa, "/proc/self/status"));
+}