diff options
| author | Gabriel Schneider <[email protected]> | 2026-09-19 23:28:22 -0300 |
|---|---|---|
| committer | Gabriel Schneider <[email protected]> | 2026-09-19 23:28:22 -0300 |
| commit | ba996acfcad1698adbf4a1834fe50e73b1c6cab9 (patch) | |
| tree | 282ba00ce5b10d7416aecb9f2f0f0a439340a57d /9ns/src | |
| parent | b05abcba3ea09ea106ad28364c6e40a3ec31b890 (diff) | |
| download | cloud9-ba996acfcad1698adbf4a1834fe50e73b1c6cab9.tar.gz cloud9-ba996acfcad1698adbf4a1834fe50e73b1c6cab9.zip | |
Rename programs: 9player -> 9ns, introspect -> 9proc, app -> web (9web)
Directories, binaries, build options (-D9ns, -D9proc), step names, module
name (9proc), thread and fs names, env var NINEPLAYER_MOUNT -> NINE_MOUNT,
docs and test scripts. Browser assets move to web/static.
Co-Authored-By: Claude Fable 5.1 <[email protected]>
Diffstat (limited to '9ns/src')
| -rw-r--r-- | 9ns/src/bridge.zig | 971 | ||||
| -rw-r--r-- | 9ns/src/fuse.zig | 653 | ||||
| -rw-r--r-- | 9ns/src/main.zig | 444 | ||||
| -rw-r--r-- | 9ns/src/nine.zig | 756 | ||||
| -rw-r--r-- | 9ns/src/ns.zig | 1082 |
5 files changed, 3906 insertions, 0 deletions
diff --git a/9ns/src/bridge.zig b/9ns/src/bridge.zig new file mode 100644 index 0000000..3a072ec --- /dev/null +++ b/9ns/src/bridge.zig @@ -0,0 +1,971 @@ +//! FUSE ↔ 9P2000 translation: the request loop that turns kernel FUSE requests +//! into synchronous 9P calls on a `nine.Session` and sends the replies back. +//! +//! Everything here is single-threaded and one request at a time. State is three +//! tables: inodes (nodeid → fid/qid, deduplicated by qid.path), open handles +//! (fh → fid plus a cached directory listing), and the reverse qid map. +const std = @import("std"); +const cloud9 = @import("cloud9"); +const fuse = @import("fuse.zig"); +const nine = @import("nine.zig"); +const linux = std.os.linux; + +pub const Options = struct { + /// Reported owner of every file. + uid: u32, + gid: u32, + /// attr/entry cache validity (0 = none). + attr_timeout_ns: u64 = 1_000_000_000, + /// FOPEN_DIRECT_IO on every regular file. + direct_io: bool = true, + /// Trace every request, reply and 9P call to stderr. + debug: bool = false, +}; + +/// Largest single READ/WRITE payload we accept from the kernel. +pub const max_write: u32 = 1 << 20; +/// Upper bound on the raw bytes of one directory listing (about a million entries); +/// past it the listing fails with EIO instead of eating memory. +pub const max_dir_bytes: u64 = 64 << 20; +/// The kernel refuses dirents longer than this (FUSE_NAME_MAX) with EIO. +pub const max_name_len: usize = 1024; +/// Request buffer: `max_write` plus room for the header and the largest in-struct. +pub const request_buf_len: usize = max_write + 4096; + +const Inode = struct { + fid: u32, + qid: cloud9.Qid, + nlookup: u64, + /// nodeid of the directory this inode was looked up in (root: itself). Used for "..". + parent: u64, +}; + +pub const Entry = struct { name: []u8, ino: u64, dtype: u32 }; + +pub const DirList = struct { + entries: std.ArrayList(Entry) = .empty, + + pub fn deinit(d: *DirList, gpa: std.mem.Allocator) void { + for (d.entries.items) |e| gpa.free(e.name); + d.entries.deinit(gpa); + } +}; + +const Handle = struct { + fid: u32, + nodeid: u64, + dir: ?DirList, +}; + +/// Errors a request handler may surface. Policy failures are ordinary errors +/// that the dispatcher maps to an errno; `FuseIo` means the kernel side is broken. +const HandlerError = nine.Session.Error || error{ + BadRequest, + NoEntry, + BadHandle, + Exdev, + Perm, + NotSup, + /// A directory listing the server sent could not be parsed (EIO, not fatal). + BadDir, + FuseIo, +}; + +/// Runs until the FUSE fd reports ENODEV, a DESTROY arrives, or `stop_fd` +/// becomes readable (also while a 9P reply is outstanding). Returns +/// `error.Closed` if the 9P server went away. +pub fn serve(gpa: std.mem.Allocator, fuse_fd: i32, session: *nine.Session, root_fid: u32, stop_fd: i32, opts: Options) !void { + var effective = opts; + // With page caching on, a nonzero attr cache lets the kernel trust a stale + // (often zero) size and truncate reads: 9P sizes are authoritative and change + // under us. direct_io ignores the cached size, so the cache is safe only there. + if (!effective.direct_io) effective.attr_timeout_ns = 0; + var b: Bridge = .{ + .gpa = gpa, + .fuse_fd = fuse_fd, + .nine = session, + .opts = effective, + }; + defer b.deinit(); + + b.req_buf = try gpa.alignedAlloc(u8, .@"8", request_buf_len); + b.data_buf = try gpa.alloc(u8, max_write); + + // Abandon any pending 9P reply once the child is gone (stop_fd readable), + // including the initial root stat below: a silent server must not pin us. + session.stop_fd = stop_fd; + defer session.stop_fd = -1; + + // Node 1 is the root; its qid comes from a stat so lookups resolving back to + // it (e.g. via a walk) dedupe onto node 1. + var root_qid: cloud9.Qid = .{ .type = cloud9.qtdir, .version = 0, .path = 0 }; + if (b.stat(root_fid)) |st| { + root_qid = st.qid; + b.root_path = st.qid.path; + try b.by_qid.put(gpa, st.qid.path, fuse.root_id); + } else |e| switch (e) { + error.Nine => {}, + error.Stopped => return, + else => return error.Closed, + } + try b.inodes.put(gpa, fuse.root_id, .{ .fid = root_fid, .qid = root_qid, .nlookup = 1, .parent = fuse.root_id }); + + var pfds = [_]linux.pollfd{ + .{ .fd = fuse_fd, .events = linux.POLL.IN, .revents = 0 }, + .{ .fd = stop_fd, .events = linux.POLL.IN, .revents = 0 }, + }; + while (true) { + pfds[0].revents = 0; + pfds[1].revents = 0; + const rc = linux.poll(&pfds, pfds.len, -1); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .INTR, .AGAIN => continue, + else => return error.Io, + } + if (pfds[1].revents != 0) { + b.trace("stop_fd readable; leaving serve loop", .{}); + return; + } + if (pfds[0].revents == 0) continue; + const req = (fuse.readRequest(fuse_fd, b.req_buf) catch |e| switch (e) { + error.Protocol => return error.FuseProtocol, + else => return error.FuseIo, + }) orelse { + b.trace("fuse fd reports ENODEV; unmounted", .{}); + return; + }; + if (!try b.dispatch(req)) return; + } +} + +const Bridge = struct { + gpa: std.mem.Allocator, + fuse_fd: i32, + nine: *nine.Session, + opts: Options, + req_buf: []align(8) u8 = &.{}, + data_buf: []u8 = &.{}, + inodes: std.AutoHashMapUnmanaged(u64, Inode) = .empty, + by_qid: std.AutoHashMapUnmanaged(u64, u64) = .empty, + handles: std.AutoHashMapUnmanaged(u64, Handle) = .empty, + next_node: u64 = 2, + next_fh: u64 = 1, + /// qid.path of the root, reported as ino 1 wherever it shows up. + root_path: u64 = 0, + /// errno of the most recent Rerror that a handler did not swallow. Kept here + /// because `Session.rpc` clears its ename on every call, and error paths + /// clunk (an rpc) before the dispatcher maps the failure to an errno. + last_err: linux.E = .IO, + + fn deinit(b: *Bridge) void { + var it = b.handles.valueIterator(); + while (it.next()) |h| if (h.dir) |*d| d.deinit(b.gpa); + b.handles.deinit(b.gpa); + b.inodes.deinit(b.gpa); + b.by_qid.deinit(b.gpa); + if (b.req_buf.len != 0) b.gpa.free(b.req_buf); + if (b.data_buf.len != 0) b.gpa.free(b.data_buf); + } + + fn trace(b: *const Bridge, comptime fmt: []const u8, args: anytype) void { + if (b.opts.debug) std.debug.print("9ns: " ++ fmt ++ "\n", args); + } + + // -- dispatch -------------------------------------------------------------------- + + /// Handles one request. Returns false when the loop should stop (DESTROY). + /// Fatal errors (dead 9P session, broken FUSE fd) propagate. + fn dispatch(b: *Bridge, req: fuse.Request) !bool { + const h = req.header; + const op = h.op(); + b.trace("<- {s} unique={d} nodeid={d} len={d} (fids={d} inodes={d} handles={d})", .{ opName(op), h.unique, h.nodeid, h.len, b.nine.fidsInUse(), b.inodes.count(), b.handles.count() }); + const wants_reply = switch (op) { + .forget, .batch_forget, .interrupt => false, + else => true, + }; + if (op == .destroy) { + b.reply(h.unique, &.{}) catch {}; + return false; + } + b.handle(req) catch |e| { + const code: linux.E = switch (e) { + error.Nine => b.last_err, + error.BadRequest => .INVAL, + error.NoEntry => .NOENT, + error.BadHandle => .BADF, + error.Exdev => .XDEV, + error.Perm => .PERM, + error.NotSup => .NOSYS, + error.OutOfMemory => .NOMEM, + error.TooLarge => .NAMETOOLONG, + error.BadDir => .IO, + error.Closed, error.Protocol, error.Io, error.Stopped => .IO, + error.FuseIo => return error.FuseIo, + }; + if (wants_reply) try b.replyError(h.unique, code); + switch (e) { + error.Closed, error.Protocol, error.Io => return error.Closed, + error.Stopped => return false, // the child is gone; the mount is being torn down + else => {}, + } + }; + return true; + } + + fn handle(b: *Bridge, req: fuse.Request) HandlerError!void { + const u = req.header.unique; + switch (req.header.op()) { + .init => { + const in = try body(fuse.InitIn, req); + const out = fuse.initReply(in, max_write); + try b.reply(u, &.{std.mem.asBytes(&out)}); + }, + .lookup => { + const name = try nameAfter(void, req); + const entry = try b.lookupEntry(req.header.nodeid, name); + try b.reply(u, &.{std.mem.asBytes(&entry)}); + }, + .forget => { + const in = try body(fuse.ForgetIn, req); + try b.forget(req.header.nodeid, in.nlookup); + }, + .batch_forget => { + const in = try body(fuse.BatchForgetIn, req); + const rest = req.body[@sizeOf(fuse.BatchForgetIn)..]; + const count: usize = in.count; + if (rest.len < count * @sizeOf(fuse.ForgetOne)) return error.BadRequest; + for (0..count) |i| { + const one = std.mem.bytesToValue(fuse.ForgetOne, rest[i * @sizeOf(fuse.ForgetOne) ..][0..@sizeOf(fuse.ForgetOne)]); + try b.forget(one.nodeid, one.nlookup); + } + }, + .getattr => { + const ino = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; + const st = try b.stat(ino.fid); + const out = b.attrOut(st, b.inoOf(req.header.nodeid, ino.qid)); + try b.reply(u, &.{std.mem.asBytes(&out)}); + }, + .setattr => try b.setattr(req), + .open => try b.openFile(req, false), + .opendir => try b.openFile(req, true), + .read => { + const in = try body(fuse.ReadIn, req); + const h = b.handles.get(in.fh) orelse return error.BadHandle; + const want: usize = @min(in.size, max_write); + const n = try b.read(h.fid, in.offset, b.data_buf[0..want]); + try b.reply(u, &.{b.data_buf[0..n]}); + }, + .write => { + const in = try body(fuse.WriteIn, req); + const h = b.handles.get(in.fh) orelse return error.BadHandle; + const rest = req.body[@sizeOf(fuse.WriteIn)..]; + if (rest.len < in.size) return error.BadRequest; + const n = try b.write(h.fid, in.offset, rest[0..in.size]); + const out = fuse.WriteOut{ .size = @intCast(n) }; + try b.reply(u, &.{std.mem.asBytes(&out)}); + }, + .readdir => try b.readdir(req), + .release, .releasedir => { + const in = try body(fuse.ReleaseIn, req); + const kv = b.handles.fetchRemove(in.fh) orelse return error.BadHandle; + var h = kv.value; + if (h.dir) |*d| d.deinit(b.gpa); + try b.clunk(h.fid); + try b.reply(u, &.{}); + }, + .flush, .fsync, .fsyncdir => try b.reply(u, &.{}), + .create => try b.create(req), + .mkdir => { + const in = try body(fuse.MkdirIn, req); + const name = try nameAfter(fuse.MkdirIn, req); + const parent = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; + const fid = try b.clone(parent.fid); + _ = b.create9(fid, name, cloud9.dmdir | (in.mode & 0o777), cloud9.oread) catch |e| { + b.clunkQuiet(fid); + return e; + }; + try b.clunk(fid); + const entry = try b.lookupEntry(req.header.nodeid, name); + try b.reply(u, &.{std.mem.asBytes(&entry)}); + }, + .unlink, .rmdir => { + const name = try nameAfter(void, req); + const parent = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; + const tmp = try b.walkName(parent.fid, name); + try b.remove(tmp); + try b.reply(u, &.{}); + }, + .rename => { + const in = try body(fuse.RenameIn, req); + const old = try nameAfter(fuse.RenameIn, req); + const new = try secondName(req, old, @sizeOf(fuse.RenameIn)); + try b.rename(req.header.nodeid, in.newdir, old, new, 0); + try b.reply(u, &.{}); + }, + .rename2 => { + const in = try body(fuse.Rename2In, req); + const old = try nameAfter(fuse.Rename2In, req); + const new = try secondName(req, old, @sizeOf(fuse.Rename2In)); + try b.rename(req.header.nodeid, in.newdir, old, new, in.flags); + try b.reply(u, &.{}); + }, + .statfs => { + const out = fuse.StatfsOut{ .st = .{ .bsize = 4096, .namelen = 255, .frsize = 4096 } }; + try b.reply(u, &.{std.mem.asBytes(&out)}); + }, + .interrupt => {}, + .destroy => unreachable, // handled in dispatch + .access => return error.NotSup, + else => return error.NotSup, + } + } + + // -- handlers ---------------------------------------------------------------------- + + /// walk(parent → new fid, [name]) + stat, deduplicated by qid.path. Bumps nlookup. + fn lookupEntry(b: *Bridge, parent_id: u64, name: []const u8) HandlerError!fuse.EntryOut { + const parent = b.inodes.get(parent_id) orelse return error.NoEntry; + const newfid = try b.walkName(parent.fid, name); + const st = b.stat(newfid) catch |e| { + b.clunkQuiet(newfid); + return e; + }; + const qid = st.qid; + var nodeid: u64 = undefined; + if (b.by_qid.get(qid.path)) |existing| { + // A directory and a file sharing a qid.path (a server bug) must not + // share a node: the kernel would mark the inode bad, and for the + // root that is fatal for the whole mount. + const merge = if (b.inodes.getPtr(existing)) |ino| (ino.qid.type & cloud9.qtdir) == (qid.type & cloud9.qtdir) else false; + if (merge) { + const ino = b.inodes.getPtr(existing).?; + ino.nlookup += 1; + ino.qid = qid; + nodeid = existing; + if (existing == fuse.root_id) { + b.clunkQuiet(newfid); + } else { + // Keep the fresh fid (it is bound to the current file at this + // name) and retire the older one. + const stale = ino.fid; + ino.fid = newfid; + b.clunkQuiet(stale); + } + } else { + // Stale reverse entry, or a type clash: bind a fresh node to it. + nodeid = try b.newInode(newfid, qid, parent_id); + } + } else { + nodeid = try b.newInode(newfid, qid, parent_id); + } + var out = fuse.EntryOut{ + .nodeid = nodeid, + .generation = 0, + .attr = b.attrFrom(st, b.inoOf(nodeid, qid)), + }; + out.entry_valid = b.opts.attr_timeout_ns / 1_000_000_000; + out.entry_valid_nsec = @intCast(b.opts.attr_timeout_ns % 1_000_000_000); + out.attr_valid = out.entry_valid; + out.attr_valid_nsec = out.entry_valid_nsec; + return out; + } + + fn newInode(b: *Bridge, fid: u32, qid: cloud9.Qid, parent: u64) HandlerError!u64 { + const nodeid = b.next_node; + b.inodes.put(b.gpa, nodeid, .{ .fid = fid, .qid = qid, .nlookup = 1, .parent = parent }) catch |e| { + b.clunkQuiet(fid); + return e; + }; + b.by_qid.put(b.gpa, qid.path, nodeid) catch |e| { + _ = b.inodes.remove(nodeid); + b.clunkQuiet(fid); + return e; + }; + b.next_node += 1; + return nodeid; + } + + fn forget(b: *Bridge, nodeid: u64, n: u64) HandlerError!void { + if (nodeid == fuse.root_id) return; + const ino = b.inodes.getPtr(nodeid) orelse return; + if (ino.nlookup > n) { + ino.nlookup -= n; + return; + } + const fid = ino.fid; + const path = ino.qid.path; + _ = b.inodes.remove(nodeid); + if (b.by_qid.get(path)) |mapped| { + if (mapped == nodeid) _ = b.by_qid.remove(path); + } + b.clunk(fid) catch |e| switch (e) { + error.Nine => {}, + else => return e, + }; + } + + fn setattr(b: *Bridge, req: fuse.Request) HandlerError!void { + const in = try body(fuse.SetattrIn, req); + const ino = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; + const old = try b.stat(ino.fid); + const old_mode = old.mode; + + var st = nine.dontcare; + var changed = false; + if (in.valid & fuse.FATTR_UID != 0 and in.uid != b.opts.uid) return error.Perm; + if (in.valid & fuse.FATTR_GID != 0 and in.gid != b.opts.gid) return error.Perm; + if (in.valid & fuse.FATTR_SIZE != 0) { + st.length = in.size; + changed = true; + } + if (in.valid & fuse.FATTR_MODE != 0) { + st.mode = (old_mode & ~@as(u32, 0o777)) | (in.mode & 0o777); + changed = true; + } + if (in.valid & fuse.FATTR_MTIME_NOW != 0) { + st.mtime = nowSeconds(); + changed = true; + } else if (in.valid & fuse.FATTR_MTIME != 0) { + st.mtime = @truncate(in.mtime); + changed = true; + } + if (changed) try b.wstat(ino.fid, st); + const fresh = try b.stat(ino.fid); + const out = b.attrOut(fresh, b.inoOf(req.header.nodeid, ino.qid)); + try b.reply(req.header.unique, &.{std.mem.asBytes(&out)}); + } + + fn openFile(b: *Bridge, req: fuse.Request, is_dir: bool) HandlerError!void { + const in = try body(fuse.OpenIn, req); + const ino = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; + const mode: u8 = if (is_dir) cloud9.oread else openMode(in.flags); + const fid = try b.clone(ino.fid); + _ = b.open9(fid, mode) catch |e| { + b.clunkQuiet(fid); + return e; + }; + const fh = try b.newHandle(fid, req.header.nodeid); + const out = fuse.OpenOut{ + .fh = fh, + .open_flags = if (!is_dir and b.opts.direct_io) fuse.FOPEN_DIRECT_IO else 0, + }; + try b.reply(req.header.unique, &.{std.mem.asBytes(&out)}); + } + + fn newHandle(b: *Bridge, fid: u32, nodeid: u64) HandlerError!u64 { + const fh = b.next_fh; + b.handles.put(b.gpa, fh, .{ .fid = fid, .nodeid = nodeid, .dir = null }) catch |e| { + b.clunkQuiet(fid); + return e; + }; + b.next_fh += 1; + return fh; + } + + fn create(b: *Bridge, req: fuse.Request) HandlerError!void { + const in = try body(fuse.CreateIn, req); + const name = try nameAfter(fuse.CreateIn, req); + const parent = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; + // The created fid becomes the open file. + const fid = try b.clone(parent.fid); + _ = b.create9(fid, name, in.mode & 0o777, openMode(in.flags)) catch |e| { + b.clunkQuiet(fid); + return e; + }; + const entry = b.lookupEntry(req.header.nodeid, name) catch |e| { + b.clunkQuiet(fid); + return e; + }; + const fh = try b.newHandle(fid, entry.nodeid); + const oo = fuse.OpenOut{ + .fh = fh, + .open_flags = if (b.opts.direct_io) fuse.FOPEN_DIRECT_IO else 0, + }; + try b.reply(req.header.unique, &.{ std.mem.asBytes(&entry), std.mem.asBytes(&oo) }); + } + + fn rename(b: *Bridge, parent_id: u64, newdir: u64, old: []const u8, new: []const u8, flags: u32) HandlerError!void { + if (newdir != parent_id) return error.Exdev; + const rf: linux.RENAME = @bitCast(flags); + if (rf.EXCHANGE or rf.WHITEOUT) return error.BadRequest; + const parent = b.inodes.get(parent_id) orelse return error.NoEntry; + const tmp = try b.walkName(parent.fid, old); + defer b.clunkQuiet(tmp); + var st = nine.dontcare; + st.name = new; + b.wstat(tmp, st) catch |e| { + // 9P2000 rename never replaces an existing name; POSIX rename does. + if (e != error.Nine or rf.NOREPLACE or b.nine.errno() != .EXIST) return e; + try b.renameOver(parent.fid, tmp, new); + }; + } + + /// Replace `new` with the file behind `src`. An (empty) directory target is + /// removed first: it holds no data and the VFS already ruled out mismatched + /// types. A file target is parked under a temporary name so that a failing + /// second rename can put it back instead of having destroyed it. + fn renameOver(b: *Bridge, parent_fid: u32, src: u32, new: []const u8) HandlerError!void { + const victim = try b.walkName(parent_fid, new); + const vst = b.stat(victim) catch |e| { + b.clunkQuiet(victim); + return e; + }; + var st = nine.dontcare; + st.name = new; + if (vst.mode & cloud9.dmdir != 0) { + b.trace(" rename target is a directory; removing it and retrying", .{}); + try b.remove(victim); + return b.wstat(src, st); + } + var park_buf: [48]u8 = undefined; + const park = std.fmt.bufPrint(&park_buf, ".9ns-rename-{x}", .{randomU64()}) catch unreachable; + b.trace(" rename target exists; parking it as {s} and retrying", .{park}); + var pst = nine.dontcare; + pst.name = park; + b.wstat(victim, pst) catch |e| { + b.clunkQuiet(victim); + return e; + }; + b.wstat(src, st) catch |e| { + b.trace(" rename still failed; restoring the target", .{}); + const saved = b.last_err; + b.wstat(victim, st) catch {}; + b.last_err = saved; + b.clunkQuiet(victim); + return e; + }; + b.remove(victim) catch b.trace(" could not remove the parked target {s}", .{park}); + } + + fn readdir(b: *Bridge, req: fuse.Request) HandlerError!void { + const in = try body(fuse.ReadIn, req); + const h = b.handles.getPtr(in.fh) orelse return error.BadHandle; + if (in.offset == 0 or h.dir == null) { + if (h.dir) |*d| d.deinit(b.gpa); + h.dir = null; + h.dir = try b.loadDir(h.fid, h.nodeid); + } + const dir = &h.dir.?; + const size: usize = @min(in.size, max_write); + const used = packDirents(dir.entries.items, in.offset, b.data_buf[0..size]); + try b.reply(req.header.unique, &.{b.data_buf[0..used]}); + } + + /// Reads the whole directory and builds its listing, "." and ".." first. + fn loadDir(b: *Bridge, fid: u32, nodeid: u64) HandlerError!DirList { + var list: DirList = .{}; + errdefer list.deinit(b.gpa); + const self_ino = b.inoOfNode(nodeid); + const parent_ino = if (b.inodes.get(nodeid)) |ino| b.inoOfNode(ino.parent) else self_ino; + try list.entries.append(b.gpa, .{ .name = try b.gpa.dupe(u8, "."), .ino = self_ino, .dtype = fuse.DT_DIR }); + try list.entries.append(b.gpa, .{ .name = try b.gpa.dupe(u8, ".."), .ino = parent_ino, .dtype = fuse.DT_DIR }); + + var offset: u64 = 0; + while (true) { + // A server that ignores the offset would otherwise feed us forever. + if (offset >= max_dir_bytes) return error.BadDir; + const n = try b.read(fid, offset, b.data_buf); + if (n == 0) break; + try parseDirRecords(b.gpa, b.data_buf[0..n], &list); + offset += n; + } + // Entries carrying the root's own qid.path get the root's ino (1), as GETATTR would report it. + for (list.entries.items[2..]) |*e| if (e.ino == b.root_path) { + e.ino = fuse.root_id; + }; + return list; + } + + // -- 9P wrappers (tracing) ----------------------------------------------------------- + + fn stat(b: *Bridge, fid: u32) nine.Session.Error!cloud9.Stat { + const st = b.nine.stat(fid) catch |e| return b.nineErr("stat", fid, e); + b.trace(" 9p stat fid={d} -> name={s} mode={o} len={d} qid={x}", .{ fid, st.name, st.mode, st.length, st.qid.path }); + return st; + } + + fn walkName(b: *Bridge, fid: u32, name: []const u8) nine.Session.Error!u32 { + const newfid = b.nine.allocFid(); + _ = b.nine.walk(fid, newfid, &.{name}) catch |e| { + b.nine.freeFid(newfid); + return b.nineErr("walk", fid, e); + }; + b.trace(" 9p walk fid={d} newfid={d} name={s} -> ok", .{ fid, newfid, name }); + return newfid; + } + + fn clone(b: *Bridge, fid: u32) nine.Session.Error!u32 { + const newfid = b.nine.clone(fid) catch |e| return b.nineErr("clone", fid, e); + b.trace(" 9p walk fid={d} newfid={d} (clone) -> ok", .{ fid, newfid }); + return newfid; + } + + fn open9(b: *Bridge, fid: u32, mode: u8) nine.Session.Error!nine.Session.Open { + const o = b.nine.open(fid, mode) catch |e| return b.nineErr("open", fid, e); + b.trace(" 9p open fid={d} mode={d} -> iounit={d}", .{ fid, mode, o.iounit }); + return o; + } + + fn create9(b: *Bridge, fid: u32, name: []const u8, perm: u32, mode: u8) nine.Session.Error!nine.Session.Open { + const o = b.nine.create(fid, name, perm, mode) catch |e| return b.nineErr("create", fid, e); + b.trace(" 9p create fid={d} name={s} perm={o} mode={d} -> iounit={d}", .{ fid, name, perm, mode, o.iounit }); + return o; + } + + fn read(b: *Bridge, fid: u32, offset: u64, buf: []u8) nine.Session.Error!usize { + const n = b.nine.read(fid, offset, buf) catch |e| return b.nineErr("read", fid, e); + b.trace(" 9p read fid={d} offset={d} count={d} -> {d}", .{ fid, offset, buf.len, n }); + return n; + } + + fn write(b: *Bridge, fid: u32, offset: u64, data: []const u8) nine.Session.Error!usize { + const n = b.nine.write(fid, offset, data) catch |e| return b.nineErr("write", fid, e); + b.trace(" 9p write fid={d} offset={d} count={d} -> {d}", .{ fid, offset, data.len, n }); + return n; + } + + fn wstat(b: *Bridge, fid: u32, st: cloud9.Stat) nine.Session.Error!void { + b.nine.wstat(fid, st) catch |e| return b.nineErr("wstat", fid, e); + b.trace(" 9p wstat fid={d} name={s} mode={x} len={x} mtime={x} -> ok", .{ fid, st.name, st.mode, st.length, st.mtime }); + } + + fn clunk(b: *Bridge, fid: u32) nine.Session.Error!void { + b.nine.clunk(fid) catch |e| return b.nineErr("clunk", fid, e); + b.trace(" 9p clunk fid={d} -> ok", .{fid}); + } + + /// Best-effort clunk during error unwinding; a dead session surfaces on the + /// next call. Does not disturb the errno of the failure being unwound. + fn clunkQuiet(b: *Bridge, fid: u32) void { + const saved = b.last_err; + defer b.last_err = saved; + b.clunk(fid) catch {}; + } + + fn remove(b: *Bridge, fid: u32) nine.Session.Error!void { + b.nine.remove(fid) catch |e| return b.nineErr("remove", fid, e); + b.trace(" 9p remove fid={d} -> ok", .{fid}); + } + + fn nineErr(b: *Bridge, what: []const u8, fid: u32, e: nine.Session.Error) nine.Session.Error { + if (e == error.Nine) { + b.last_err = b.nine.errno(); + b.trace(" 9p {s} fid={d} -> Rerror \"{s}\" ({s})", .{ what, fid, b.nine.ename[0..b.nine.ename_len], @tagName(b.nine.errno()) }); + } else { + b.trace(" 9p {s} fid={d} -> {s}", .{ what, fid, @errorName(e) }); + } + return e; + } + + // -- FUSE wrappers (tracing) -------------------------------------------------------- + + fn reply(b: *Bridge, unique: u64, payloads: []const []const u8) error{FuseIo}!void { + var total: usize = 0; + for (payloads) |p| total += p.len; + b.trace("-> unique={d} ok ({d} bytes)", .{ unique, total }); + fuse.reply(b.fuse_fd, unique, payloads) catch return error.FuseIo; + } + + fn replyError(b: *Bridge, unique: u64, code: linux.E) error{FuseIo}!void { + b.trace("-> unique={d} error E{s}", .{ unique, @tagName(code) }); + fuse.replyError(b.fuse_fd, unique, code) catch return error.FuseIo; + } + + // -- attrs --------------------------------------------------------------------------- + + fn inoOf(b: *const Bridge, nodeid: u64, qid: cloud9.Qid) u64 { + return if (nodeid == fuse.root_id or qid.path == b.root_path) fuse.root_id else qid.path; + } + + fn inoOfNode(b: *const Bridge, nodeid: u64) u64 { + if (nodeid == fuse.root_id) return fuse.root_id; + const ino = b.inodes.get(nodeid) orelse return nodeid; + return b.inoOf(nodeid, ino.qid); + } + + fn attrFrom(b: *const Bridge, st: cloud9.Stat, ino: u64) fuse.Attr { + return attrFromStat(st, ino, b.opts.uid, b.opts.gid); + } + + fn attrOut(b: *const Bridge, st: cloud9.Stat, ino: u64) fuse.AttrOut { + return .{ + .attr_valid = b.opts.attr_timeout_ns / 1_000_000_000, + .attr_valid_nsec = @intCast(b.opts.attr_timeout_ns % 1_000_000_000), + .attr = b.attrFrom(st, ino), + }; + } +}; + +// -- pure helpers (unit-tested) ------------------------------------------------------------ + +/// Attr from a 9P Stat: DMDIR → S_IFDIR else S_IFREG, low 9 permission bits kept. +pub fn attrFromStat(st: cloud9.Stat, ino: u64, uid: u32, gid: u32) fuse.Attr { + const ftype: u32 = if (st.mode & cloud9.dmdir != 0) fuse.S_IFDIR else fuse.S_IFREG; + return .{ + .ino = ino, + // The kernel marks an inode bad when size > LLONG_MAX; clamp hostile lengths. + .size = @min(st.length, std.math.maxInt(i64)), + // Saturating: a hostile length of 2^64-1 must not overflow. + .blocks = st.length / 512 + @intFromBool(st.length % 512 != 0), + .atime = st.atime, + .mtime = st.mtime, + .ctime = st.mtime, + .mode = ftype | (st.mode & 0o777), + .nlink = 1, + .uid = uid, + .gid = gid, + .blksize = 4096, + }; +} + +/// Kernel open(2) flags → 9P open mode. O_APPEND has no 9P equivalent and is ignored. +pub fn openMode(flags: u32) u8 { + const o: linux.O = @bitCast(flags); + var mode: u8 = switch (o.ACCMODE) { + .RDONLY => cloud9.oread, + .WRONLY => cloud9.owrite, + .RDWR => cloud9.ordwr, + }; + if (o.TRUNC) mode |= cloud9.otrunc; + return mode; +} + +/// Parses consecutive 9P directory records (2-byte size + Stat) and appends entries. +pub fn parseDirRecords(gpa: std.mem.Allocator, bytes: []const u8, list: *DirList) error{ OutOfMemory, BadDir }!void { + var pos: usize = 0; + while (pos < bytes.len) { + if (bytes.len - pos < 2) return error.BadDir; + const size: usize = std.mem.readInt(u16, bytes[pos..][0..2], .little); + if (bytes.len - pos < 2 + size) return error.BadDir; + const st = cloud9.Stat.decode(bytes[pos..][0 .. 2 + size]) catch return error.BadDir; + pos += 2 + size; + // The kernel rejects a whole READDIR reply (EIO) over one bad name, and + // "." and ".." are synthesised by loadDir: drop such records instead. + if (!validDirentName(st.name)) continue; + const name = try gpa.dupe(u8, st.name); + errdefer gpa.free(name); + try list.entries.append(gpa, .{ + .name = name, + .ino = st.qid.path, + .dtype = if (st.mode & cloud9.dmdir != 0) fuse.DT_DIR else fuse.DT_REG, + }); + } +} + +/// A name the kernel will accept in a dirent and that does not duplicate the synthetic "." / "..". +pub fn validDirentName(name: []const u8) bool { + if (name.len == 0 or name.len > max_name_len) return false; + if (std.mem.indexOfAny(u8, name, "/\x00") != null) return false; + if (std.mem.eql(u8, name, ".") or std.mem.eql(u8, name, "..")) return false; + return true; +} + +/// Packs dirents from `entries[offset..]` into `buf`; each record's `off` is its index + 1. +/// Returns the number of bytes used. +pub fn packDirents(entries: []const Entry, offset: u64, buf: []u8) usize { + var used: usize = 0; + var i: usize = @intCast(@min(offset, entries.len)); + while (i < entries.len) : (i += 1) { + const e = entries[i]; + if (!fuse.addDirent(buf, &used, e.ino, @as(u64, i) + 1, e.dtype, e.name)) break; + } + return used; +} + +fn randomU64() u64 { + var bytes: [8]u8 = undefined; + if (linux.errno(linux.getrandom(&bytes, bytes.len, 0)) == .SUCCESS) return std.mem.readInt(u64, &bytes, .little); + var ts: linux.timespec = undefined; + _ = linux.clock_gettime(.MONOTONIC, &ts); + return @as(u64, @bitCast(ts.nsec)) ^ (@as(u64, @bitCast(ts.sec)) << 32); +} + +fn nowSeconds() u32 { + var ts: linux.timespec = undefined; + if (linux.errno(linux.clock_gettime(.REALTIME, &ts)) != .SUCCESS) return 0; + return @intCast(@as(u64, @intCast(ts.sec)) & 0xFFFF_FFFF); +} + +fn opName(op: fuse.Opcode) []const u8 { + return switch (op) { + _ => "unknown", + else => @tagName(op), + }; +} + +// Thin adapters so fuse.zig's parse errors become HandlerError.BadRequest. +fn body(comptime T: type, req: fuse.Request) error{BadRequest}!*const T { + return fuse.body(T, req) catch error.BadRequest; +} + +fn nameAfter(comptime T: type, req: fuse.Request) error{BadRequest}![]const u8 { + return fuse.nameAfter(T, req) catch error.BadRequest; +} + +fn secondName(req: fuse.Request, first: []const u8, offset: usize) error{BadRequest}![]const u8 { + return fuse.secondName(req, first, offset) catch error.BadRequest; +} + +// -- tests ------------------------------------------------------------------------------ + +const testing = std.testing; + +test { + // Force semantic analysis of `serve` and the whole dispatch path, which no + // unit test can exercise without a FUSE mount. + testing.refAllDecls(@This()); +} + +fn testStat(name: []const u8, mode: u32, length: u64, path: u64) cloud9.Stat { + return .{ + .type = 0, + .dev = 0, + .qid = .{ .type = if (mode & cloud9.dmdir != 0) cloud9.qtdir else 0, .version = 0, .path = path }, + .mode = mode, + .atime = 100, + .mtime = 200, + .length = length, + .name = name, + .uid = "u", + .gid = "g", + .muid = "u", + }; +} + +test "attr mapping: DMDIR → S_IFDIR|perm, length → size/blocks" { + const d = attrFromStat(testStat("d", cloud9.dmdir | 0o755, 0, 9), 9, 1000, 1001); + try testing.expectEqual(fuse.S_IFDIR | 0o755, d.mode); + try testing.expectEqual(@as(u64, 9), d.ino); + try testing.expectEqual(@as(u64, 0), d.size); + try testing.expectEqual(@as(u64, 0), d.blocks); + try testing.expectEqual(@as(u32, 1000), d.uid); + try testing.expectEqual(@as(u32, 1001), d.gid); + try testing.expectEqual(@as(u32, 1), d.nlink); + + const f = attrFromStat(testStat("f", 0o640 | cloud9.dmappend, 1025, 4), 4, 0, 0); + try testing.expectEqual(fuse.S_IFREG | 0o640, f.mode); // dmappend bit not leaked + try testing.expectEqual(@as(u64, 1025), f.size); + try testing.expectEqual(@as(u64, 3), f.blocks); + try testing.expectEqual(@as(u32, 4096), f.blksize); + try testing.expectEqual(@as(u64, 100), f.atime); + try testing.expectEqual(@as(u64, 200), f.mtime); + try testing.expectEqual(@as(u64, 200), f.ctime); + + try testing.expectEqual(@as(u64, 1), attrFromStat(testStat("f", 0o600, 512, 4), 4, 0, 0).blocks); + try testing.expectEqual(@as(u64, 2), attrFromStat(testStat("f", 0o600, 513, 4), 4, 0, 0).blocks); +} + +test "open flag → 9P mode mapping" { + const rdonly: u32 = @bitCast(linux.O{ .ACCMODE = .RDONLY }); + const wronly: u32 = @bitCast(linux.O{ .ACCMODE = .WRONLY }); + const rdwr: u32 = @bitCast(linux.O{ .ACCMODE = .RDWR }); + const trunc: u32 = @bitCast(linux.O{ .TRUNC = true }); + const append: u32 = @bitCast(linux.O{ .APPEND = true }); + const creat: u32 = @bitCast(linux.O{ .CREAT = true }); + try testing.expectEqual(cloud9.oread, openMode(rdonly)); + try testing.expectEqual(cloud9.owrite, openMode(wronly)); + try testing.expectEqual(cloud9.ordwr, openMode(rdwr)); + try testing.expectEqual(cloud9.owrite | cloud9.otrunc, openMode(wronly | trunc)); + try testing.expectEqual(cloud9.ordwr | cloud9.otrunc, openMode(rdwr | trunc | creat)); + try testing.expectEqual(cloud9.owrite, openMode(wronly | append)); // O_APPEND ignored +} + +test "dirlist parsing from two hand-encoded Stat records" { + var buf: [512]u8 = undefined; + const a = try cloud9.Stat.encode(testStat("alpha", 0o644, 10, 0x11), &buf); + const bb = try cloud9.Stat.encode(testStat("beta", cloud9.dmdir | 0o755, 0, 0x22), buf[a.len..]); + const bytes = buf[0 .. a.len + bb.len]; + // Sanity: the record is prefixed by its own 2-byte size. + try testing.expectEqual(a.len - 2, std.mem.readInt(u16, bytes[0..2], .little)); + + var list: DirList = .{}; + defer list.deinit(testing.allocator); + try parseDirRecords(testing.allocator, bytes, &list); + try testing.expectEqual(@as(usize, 2), list.entries.items.len); + try testing.expectEqualStrings("alpha", list.entries.items[0].name); + try testing.expectEqual(@as(u64, 0x11), list.entries.items[0].ino); + try testing.expectEqual(fuse.DT_REG, list.entries.items[0].dtype); + try testing.expectEqualStrings("beta", list.entries.items[1].name); + try testing.expectEqual(@as(u64, 0x22), list.entries.items[1].ino); + try testing.expectEqual(fuse.DT_DIR, list.entries.items[1].dtype); + + // Truncated input is a protocol error and leaves earlier entries intact. + try testing.expectError(error.BadDir, parseDirRecords(testing.allocator, bytes[0 .. bytes.len - 1], &list)); + try testing.expectEqual(@as(usize, 3), list.entries.items.len); +} + +test "readdir packing and offset resumption" { + const names = [_][]const u8{ ".", "..", "one", "two", "three" }; + var entries: [names.len]Entry = undefined; + for (&entries, names, 0..) |*e, n, i| e.* = .{ .name = @constCast(n), .ino = 100 + i, .dtype = if (i < 2) fuse.DT_DIR else fuse.DT_REG }; + + // Everything fits: five records, off = index + 1. + var big: [1024]u8 = undefined; + const used = packDirents(&entries, 0, &big); + var pos: usize = 0; + var idx: usize = 0; + while (pos < used) : (idx += 1) { + const d = std.mem.bytesToValue(fuse.Dirent, big[pos..][0..@sizeOf(fuse.Dirent)]); + try testing.expectEqual(@as(u64, 100 + idx), d.ino); + try testing.expectEqual(@as(u64, idx + 1), d.off); + try testing.expectEqualStrings(names[idx], big[pos + @sizeOf(fuse.Dirent) ..][0..d.namelen]); + pos += (@sizeOf(fuse.Dirent) + d.namelen + 7) & ~@as(usize, 7); + } + try testing.expectEqual(names.len, idx); + + // A buffer that fits exactly two records ("." = 32, ".." = 32) stops there… + var small: [64]u8 = undefined; + const first_used = packDirents(&entries, 0, &small); + try testing.expectEqual(@as(usize, 64), first_used); + const last = std.mem.bytesToValue(fuse.Dirent, small[32..][0..@sizeOf(fuse.Dirent)]); + try testing.expectEqual(@as(u64, 2), last.off); + // …and resuming at the last `off` yields "one" next. + const second_used = packDirents(&entries, last.off, &small); + const next = std.mem.bytesToValue(fuse.Dirent, small[0..@sizeOf(fuse.Dirent)]); + try testing.expectEqualStrings("one", small[@sizeOf(fuse.Dirent)..][0..next.namelen]); + try testing.expectEqual(@as(u64, 3), next.off); + try testing.expect(second_used > 0); + + // Past the end: nothing (EOF for the kernel). + try testing.expectEqual(@as(usize, 0), packDirents(&entries, names.len, &big)); + try testing.expectEqual(@as(usize, 0), packDirents(&entries, 1000, &big)); +} + +test "attr mapping saturates hostile lengths instead of overflowing" { + const a = attrFromStat(testStat("f", 0o600, std.math.maxInt(u64), 4), 4, 0, 0); + try testing.expectEqual(@as(u64, std.math.maxInt(i64)), a.size); + try testing.expectEqual(@as(u64, std.math.maxInt(u64) / 512 + 1), a.blocks); + const b = attrFromStat(testStat("f", 0o600, 1024, 4), 4, 0, 0); + try testing.expectEqual(@as(u64, 2), b.blocks); + try testing.expectEqual(@as(u64, 1024), b.size); +} + +test "dirent names the kernel would reject are dropped from listings" { + try testing.expect(validDirentName("a")); + try testing.expect(validDirentName("x" ** 1024)); + try testing.expect(!validDirentName("")); + try testing.expect(!validDirentName("a/b")); + try testing.expect(!validDirentName("a\x00b")); + try testing.expect(!validDirentName(".")); + try testing.expect(!validDirentName("..")); + try testing.expect(!validDirentName("x" ** 1025)); + + var buf: [4096]u8 = undefined; + var n: usize = 0; + for ([_][]const u8{ ".", "..", "", "a/b", "keep", "x" ** 1025, "also" }) |name| { + n += (try cloud9.Stat.encode(testStat(name, 0o644, 1, 0x30), buf[n..])).len; + } + var list: DirList = .{}; + defer list.deinit(testing.allocator); + try parseDirRecords(testing.allocator, buf[0..n], &list); + try testing.expectEqual(@as(usize, 2), list.entries.items.len); + try testing.expectEqualStrings("keep", list.entries.items[0].name); + try testing.expectEqualStrings("also", list.entries.items[1].name); +} + +test "DirList frees its names" { + var list: DirList = .{}; + try list.entries.append(testing.allocator, .{ .name = try testing.allocator.dupe(u8, "x"), .ino = 1, .dtype = fuse.DT_REG }); + list.deinit(testing.allocator); +} diff --git a/9ns/src/fuse.zig b/9ns/src/fuse.zig new file mode 100644 index 0000000..682b3d5 --- /dev/null +++ b/9ns/src/fuse.zig @@ -0,0 +1,653 @@ +//! Kernel FUSE protocol subset (no libfuse, no libc, no policy). +//! +//! Extern structs mirror `/usr/include/linux/fuse.h` (kernel header 7.45); +//! every layout is checked against the header's size at comptime. Only the +//! opcodes and structs 9ns needs are here. The I/O helpers are blocking +//! and allocation-free: the caller owns a single request buffer. +//! +//! Wire rules worth remembering: +//! * The kernel delivers exactly one request per `read(2)` on `/dev/fuse`, +//! and a reply must be exactly one `write(2)`/`writev(2)`. +//! * Request bodies start right after the 40-byte `InHeader`; since every +//! in-struct is 8-byte aligned in the header, `body()` requires the caller's +//! buffer to be 8-byte aligned (`std.heap` page allocations and +//! `align(8)` arrays both qualify). +//! * A write that fails with `ENOENT` means the request was interrupted and +//! the kernel already forgot it: the reply is silently dropped. +//! * `ENODEV` on read means the filesystem was unmounted. + +const std = @import("std"); +const linux = std.os.linux; + +pub const kernel_version: u32 = 7; +/// The minor we answer; the kernel adapts to the lower of the two. +pub const kernel_minor: u32 = 31; +pub const root_id: u64 = 1; + +pub const FOPEN_DIRECT_IO: u32 = 1 << 0; +pub const FOPEN_KEEP_CACHE: u32 = 1 << 1; +pub const FOPEN_NONSEEKABLE: u32 = 1 << 2; + +pub const FUSE_ASYNC_READ: u32 = 1 << 0; +/// The kernel passes O_TRUNC in OPEN instead of a separate SETATTR(size=0); 9P has OTRUNC for exactly this. +pub const FUSE_ATOMIC_O_TRUNC: u32 = 1 << 3; +/// Without this the kernel's cached-write path (`--no-direct-io`) sends one 4 KiB WRITE per page. +pub const FUSE_BIG_WRITES: u32 = 1 << 5; +/// Re-fetch a cached inode's size/mtime and drop stale pages when they change. +/// Required for `--no-direct-io` correctness: 9P sizes change under us, and +/// without this the kernel trusts a stale cached size and truncates reads. +pub const FUSE_AUTO_INVAL_DATA: u32 = 1 << 12; +pub const FUSE_MAX_PAGES: u32 = 1 << 22; + +pub const FATTR_MODE: u32 = 1 << 0; +pub const FATTR_UID: u32 = 1 << 1; +pub const FATTR_GID: u32 = 1 << 2; +pub const FATTR_SIZE: u32 = 1 << 3; +pub const FATTR_ATIME: u32 = 1 << 4; +pub const FATTR_MTIME: u32 = 1 << 5; +pub const FATTR_FH: u32 = 1 << 6; +pub const FATTR_ATIME_NOW: u32 = 1 << 7; +pub const FATTR_MTIME_NOW: u32 = 1 << 8; +pub const FATTR_LOCKOWNER: u32 = 1 << 9; +pub const FATTR_CTIME: u32 = 1 << 10; + +/// File type bits for `Attr.mode` and `Dirent.type` (used by the bridge). +pub const S_IFDIR: u32 = linux.S.IFDIR; +pub const S_IFREG: u32 = linux.S.IFREG; +pub const DT_DIR: u32 = linux.DT.DIR; +pub const DT_REG: u32 = linux.DT.REG; + +pub const Opcode = enum(u32) { + lookup = 1, + forget = 2, + getattr = 3, + setattr = 4, + readlink = 5, + symlink = 6, + mknod = 8, + mkdir = 9, + unlink = 10, + rmdir = 11, + rename = 12, + link = 13, + open = 14, + read = 15, + write = 16, + statfs = 17, + release = 18, + fsync = 20, + setxattr = 21, + getxattr = 22, + listxattr = 23, + removexattr = 24, + flush = 25, + init = 26, + opendir = 27, + readdir = 28, + releasedir = 29, + fsyncdir = 30, + getlk = 31, + setlk = 32, + setlkw = 33, + access = 34, + create = 35, + interrupt = 36, + bmap = 37, + destroy = 38, + ioctl = 39, + poll = 40, + notify_reply = 41, + batch_forget = 42, + fallocate = 43, + readdirplus = 44, + rename2 = 45, + lseek = 46, + copy_file_range = 47, + setupmapping = 48, + removemapping = 49, + syncfs = 50, + tmpfile = 51, + statx = 52, + _, +}; + +// --------------------------------------------------------------------------- +// Structs (field order and widths follow linux/fuse.h exactly) +// --------------------------------------------------------------------------- + +pub const InHeader = extern struct { + len: u32, + opcode: u32, + unique: u64, + nodeid: u64, + uid: u32, + gid: u32, + pid: u32, + total_extlen: u16, + padding: u16, + + pub fn op(h: InHeader) Opcode { + return @enumFromInt(h.opcode); + } +}; + +pub const OutHeader = extern struct { + len: u32, + @"error": i32, + unique: u64, +}; + +pub const Attr = extern struct { + ino: u64 = 0, + size: u64 = 0, + blocks: u64 = 0, + atime: u64 = 0, + mtime: u64 = 0, + ctime: u64 = 0, + atimensec: u32 = 0, + mtimensec: u32 = 0, + ctimensec: u32 = 0, + mode: u32 = 0, + nlink: u32 = 0, + uid: u32 = 0, + gid: u32 = 0, + rdev: u32 = 0, + blksize: u32 = 0, + flags: u32 = 0, +}; + +pub const EntryOut = extern struct { + nodeid: u64 = 0, + generation: u64 = 0, + entry_valid: u64 = 0, + attr_valid: u64 = 0, + entry_valid_nsec: u32 = 0, + attr_valid_nsec: u32 = 0, + attr: Attr = .{}, +}; + +pub const AttrOut = extern struct { + attr_valid: u64 = 0, + attr_valid_nsec: u32 = 0, + dummy: u32 = 0, + attr: Attr = .{}, +}; + +pub const GetattrIn = extern struct { getattr_flags: u32, dummy: u32, fh: u64 }; + +pub const SetattrIn = extern struct { + valid: u32, + padding: u32, + fh: u64, + size: u64, + lock_owner: u64, + atime: u64, + mtime: u64, + ctime: u64, + atimensec: u32, + mtimensec: u32, + ctimensec: u32, + mode: u32, + unused4: u32, + uid: u32, + gid: u32, + unused5: u32, +}; + +pub const OpenIn = extern struct { flags: u32, open_flags: u32 }; +pub const OpenOut = extern struct { fh: u64 = 0, open_flags: u32 = 0, backing_id: i32 = 0 }; +pub const ReleaseIn = extern struct { fh: u64, flags: u32, release_flags: u32, lock_owner: u64 }; +pub const FlushIn = extern struct { fh: u64, unused: u32, padding: u32, lock_owner: u64 }; + +pub const ReadIn = extern struct { + fh: u64, + offset: u64, + size: u32, + read_flags: u32, + lock_owner: u64, + flags: u32, + padding: u32, +}; + +pub const WriteIn = extern struct { + fh: u64, + offset: u64, + size: u32, + write_flags: u32, + lock_owner: u64, + flags: u32, + padding: u32, +}; + +pub const WriteOut = extern struct { size: u32, padding: u32 = 0 }; +pub const CreateIn = extern struct { flags: u32, mode: u32, umask: u32, open_flags: u32 }; +pub const MkdirIn = extern struct { mode: u32, umask: u32 }; +pub const RenameIn = extern struct { newdir: u64 }; +pub const Rename2In = extern struct { newdir: u64, flags: u32, padding: u32 }; +pub const ForgetIn = extern struct { nlookup: u64 }; +pub const BatchForgetIn = extern struct { count: u32, dummy: u32 }; +pub const ForgetOne = extern struct { nodeid: u64, nlookup: u64 }; +pub const FsyncIn = extern struct { fh: u64, fsync_flags: u32, padding: u32 }; +pub const AccessIn = extern struct { mask: u32, padding: u32 }; +pub const InterruptIn = extern struct { unique: u64 }; +pub const LseekIn = extern struct { fh: u64, offset: u64, whence: u32, padding: u32 }; + +pub const Kstatfs = extern struct { + blocks: u64 = 0, + bfree: u64 = 0, + bavail: u64 = 0, + files: u64 = 0, + ffree: u64 = 0, + bsize: u32 = 0, + namelen: u32 = 0, + frsize: u32 = 0, + padding: u32 = 0, + spare: [6]u32 = [_]u32{0} ** 6, +}; + +pub const StatfsOut = extern struct { st: Kstatfs = .{} }; + +pub const InitIn = extern struct { + major: u32, + minor: u32, + max_readahead: u32, + flags: u32, + flags2: u32, + unused: [11]u32, +}; + +/// 64 bytes; the kernel accepts this size whenever the answered minor >= 23. +pub const InitOut = extern struct { + major: u32 = kernel_version, + minor: u32 = kernel_minor, + max_readahead: u32 = 0, + flags: u32 = 0, + max_background: u16 = 0, + congestion_threshold: u16 = 0, + max_write: u32 = 0, + time_gran: u32 = 0, + max_pages: u16 = 0, + map_alignment: u16 = 0, + flags2: u32 = 0, + max_stack_depth: u32 = 0, + request_timeout: u16 = 0, + unused: [11]u16 = [_]u16{0} ** 11, +}; + +/// Fixed 24-byte head of `fuse_dirent`; the name follows, padded to 8 bytes. +pub const Dirent = extern struct { ino: u64, off: u64, namelen: u32, type: u32 }; + +comptime { + std.debug.assert(@sizeOf(InHeader) == 40); + std.debug.assert(@sizeOf(OutHeader) == 16); + std.debug.assert(@sizeOf(Attr) == 88); + std.debug.assert(@sizeOf(EntryOut) == 128); + std.debug.assert(@sizeOf(AttrOut) == 104); + std.debug.assert(@sizeOf(GetattrIn) == 16); + std.debug.assert(@sizeOf(SetattrIn) == 88); + std.debug.assert(@sizeOf(OpenIn) == 8); + std.debug.assert(@sizeOf(OpenOut) == 16); + std.debug.assert(@sizeOf(ReleaseIn) == 24); + std.debug.assert(@sizeOf(FlushIn) == 24); + std.debug.assert(@sizeOf(ReadIn) == 40); + std.debug.assert(@sizeOf(WriteIn) == 40); + std.debug.assert(@sizeOf(WriteOut) == 8); + std.debug.assert(@sizeOf(CreateIn) == 16); + std.debug.assert(@sizeOf(MkdirIn) == 8); + std.debug.assert(@sizeOf(RenameIn) == 8); + std.debug.assert(@sizeOf(Rename2In) == 16); + std.debug.assert(@sizeOf(ForgetIn) == 8); + std.debug.assert(@sizeOf(BatchForgetIn) == 8); + std.debug.assert(@sizeOf(ForgetOne) == 16); + std.debug.assert(@sizeOf(FsyncIn) == 16); + std.debug.assert(@sizeOf(AccessIn) == 8); + std.debug.assert(@sizeOf(InterruptIn) == 8); + std.debug.assert(@sizeOf(Kstatfs) == 80); + std.debug.assert(@sizeOf(StatfsOut) == 80); + std.debug.assert(@sizeOf(InitIn) == 64); + std.debug.assert(@sizeOf(InitOut) == 64); + std.debug.assert(@sizeOf(Dirent) == 24); + std.debug.assert(@sizeOf(LseekIn) == 24); +} + +// --------------------------------------------------------------------------- +// Request / reply helpers +// --------------------------------------------------------------------------- + +pub const Error = error{ Protocol, Io, TooManyPayloads }; + +pub const Request = struct { + header: InHeader, + /// Bytes after the header; a slice into the caller's buffer. + body: []const u8, +}; + +/// Reads one kernel request with a single `read(2)`. Returns null on ENODEV +/// (unmounted). Retries EINTR/EAGAIN/ENOENT. `buf` should be at least +/// `max_write + 4096` bytes and 8-byte aligned so `body()` can view it. +pub fn readRequest(fd: i32, buf: []u8) Error!?Request { + while (true) { + const rc = linux.read(fd, buf.ptr, buf.len); + switch (linux.errno(rc)) { + .SUCCESS => { + const n: usize = rc; + if (n < @sizeOf(InHeader)) return error.Protocol; + const header = std.mem.bytesToValue(InHeader, buf[0..@sizeOf(InHeader)]); + if (header.len != n) return error.Protocol; + return .{ .header = header, .body = buf[@sizeOf(InHeader)..n] }; + }, + .INTR, .AGAIN, .NOENT => continue, + .NODEV => return null, + else => return error.Io, + } + } +} + +/// Maximum number of payload slices a single `reply` can carry. +pub const max_payloads = 7; + +/// Success reply: `OutHeader` followed by the concatenated `payloads`, sent in +/// one `writev(2)`. An ENOENT from the kernel means the request was +/// interrupted; the reply is dropped and this returns normally. +pub fn reply(fd: i32, unique: u64, payloads: []const []const u8) Error!void { + if (payloads.len > max_payloads) return error.TooManyPayloads; + var total: usize = @sizeOf(OutHeader); + for (payloads) |p| total += p.len; + if (total > std.math.maxInt(u32)) return error.Protocol; + const header = OutHeader{ .len = @intCast(total), .@"error" = 0, .unique = unique }; + var iov: [max_payloads + 1]std.posix.iovec_const = undefined; + iov[0] = .{ .base = @ptrCast(&header), .len = @sizeOf(OutHeader) }; + for (payloads, 1..) |p, i| iov[i] = .{ .base = p.ptr, .len = p.len }; + return writeAll(fd, &iov, payloads.len + 1, total); +} + +/// Error reply: an `OutHeader` carrying `-errno` and no payload. +pub fn replyError(fd: i32, unique: u64, err: linux.E) Error!void { + const code: i32 = @intCast(@intFromEnum(err)); + const header = OutHeader{ .len = @sizeOf(OutHeader), .@"error" = -code, .unique = unique }; + var iov = [_]std.posix.iovec_const{.{ .base = @ptrCast(&header), .len = @sizeOf(OutHeader) }}; + return writeAll(fd, &iov, 1, @sizeOf(OutHeader)); +} + +fn writeAll(fd: i32, iov: [*]const std.posix.iovec_const, count: usize, total: usize) Error!void { + while (true) { + const rc = linux.writev(fd, iov, count); + switch (linux.errno(rc)) { + .SUCCESS => return if (rc == total) {} else error.Protocol, + .INTR => continue, + .NOENT => return, // request was interrupted; reply dropped + else => return error.Io, + } + } +} + +/// Appends a `fuse_dirent` (head + name, padded to a multiple of 8) at +/// `buf[used.*..]`. Returns false and leaves `buf`/`used` unchanged if the +/// record does not fit. +pub fn addDirent(buf: []u8, used: *usize, ino: u64, off: u64, dtype: u32, name: []const u8) bool { + const raw = @sizeOf(Dirent) + name.len; + const rec = (raw + 7) & ~@as(usize, 7); + if (used.* > buf.len or buf.len - used.* < rec) return false; + const dst = buf[used.*..][0..rec]; + const head = Dirent{ .ino = ino, .off = off, .namelen = @intCast(name.len), .type = dtype }; + @memcpy(dst[0..@sizeOf(Dirent)], std.mem.asBytes(&head)); + @memcpy(dst[@sizeOf(Dirent)..raw], name); + @memset(dst[raw..rec], 0); + used.* += rec; + return true; +} + +/// Views the first `@sizeOf(T)` bytes of `req.body` as `T` (copy-free). +/// Fails with `error.Protocol` if the body is too short or misaligned. +pub fn body(comptime T: type, req: Request) Error!*const T { + if (req.body.len < @sizeOf(T)) return error.Protocol; + if (@intFromPtr(req.body.ptr) % @alignOf(T) != 0) return error.Protocol; + return @ptrCast(@alignCast(req.body.ptr)); +} + +/// The NUL-terminated string at `req.body[offset..]`, without the NUL. +pub fn nameAt(req: Request, offset: usize) Error![]const u8 { + if (offset > req.body.len) return error.Protocol; + const rest = req.body[offset..]; + const end = std.mem.indexOfScalar(u8, rest, 0) orelse return error.Protocol; + return rest[0..end]; +} + +/// The NUL-terminated string following a `T` body (or at offset 0 when +/// `T == void`), e.g. LOOKUP's name (`void`) or MKDIR's name (`MkdirIn`). +pub fn nameAfter(comptime T: type, req: Request) Error![]const u8 { + const offset = if (T == void) 0 else @sizeOf(T); + return nameAt(req, offset); +} + +/// The string that follows `first` (obtained via `nameAt(req, offset)`), +/// for "old\0new\0" pairs such as RENAME's. +pub fn secondName(req: Request, first: []const u8, offset: usize) Error![]const u8 { + return nameAt(req, offset + first.len + 1); +} + +/// Builds the INIT reply per docs/DESIGN.md. +pub fn initReply(in: *const InitIn, max_write: u32) InitOut { + var out = InitOut{ + .major = kernel_version, + .minor = @min(kernel_minor, in.minor), + .max_readahead = in.max_readahead, + .flags = FUSE_ASYNC_READ | FUSE_ATOMIC_O_TRUNC | FUSE_BIG_WRITES | FUSE_AUTO_INVAL_DATA, + .max_background = 16, + .congestion_threshold = 12, + .max_write = max_write, + .time_gran = 1, + }; + if (in.flags & FUSE_MAX_PAGES != 0) { + out.flags |= FUSE_MAX_PAGES; + out.max_pages = 256; + } + return out; +} + +// --------------------------------------------------------------------------- +// Tests +// --------------------------------------------------------------------------- + +const testing = std.testing; + +test "struct sizes match linux/fuse.h" { + // The comptime block above is the real check; this makes it run under + // `zig test` even if the module is otherwise unreferenced. + try testing.expectEqual(@as(usize, 40), @sizeOf(InHeader)); + try testing.expectEqual(@as(usize, 64), @sizeOf(InitOut)); + try testing.expectEqual(@as(usize, 24), @sizeOf(Dirent)); + try testing.expectEqual(@as(u32, 26), @intFromEnum(Opcode.init)); + try testing.expectEqual(Opcode.statx, @as(Opcode, @enumFromInt(52))); +} + +test "addDirent pads records to 8 bytes and refuses when full" { + var buf: [1024]u8 = undefined; + var used: usize = 0; + const name = "abcdefghijklmnopq"; // 17 chars + var expect_total: usize = 0; + var n: usize = 1; + while (n <= 17) : (n += 1) { + const before = used; + try testing.expect(addDirent(&buf, &used, n, n, DT_REG, name[0..n])); + const rec = used - before; + try testing.expectEqual(@as(usize, 0), rec % 8); + try testing.expectEqual((24 + n + 7) & ~@as(usize, 7), rec); + // check head fields and NUL padding + const head = std.mem.bytesToValue(Dirent, buf[before..][0..24]); + try testing.expectEqual(n, head.ino); + try testing.expectEqual(@as(u32, @intCast(n)), head.namelen); + try testing.expectEqualStrings(name[0..n], buf[before + 24 ..][0..n]); + for (buf[before + 24 + n .. used]) |b| try testing.expectEqual(@as(u8, 0), b); + expect_total += rec; + } + try testing.expectEqual(expect_total, used); + + // A record that does not fit leaves everything untouched. + var small: [40]u8 = undefined; + var used2: usize = 0; + try testing.expect(addDirent(&small, &used2, 1, 1, DT_DIR, "0123456789abcdef")); // 24+16 = 40 + try testing.expectEqual(@as(usize, 40), used2); + try testing.expect(!addDirent(&small, &used2, 2, 2, DT_DIR, "x")); + try testing.expectEqual(@as(usize, 40), used2); + var tight: [31]u8 = undefined; + var used3: usize = 0; + try testing.expect(!addDirent(&tight, &used3, 1, 1, DT_REG, "a")); // needs 32 + try testing.expectEqual(@as(usize, 0), used3); +} + +test "body/nameAfter/secondName on hand-built requests" { + var buf: [128]u8 align(8) = undefined; + // LOOKUP(parent=1, "hello") + const name = "hello"; + const hdr = InHeader{ + .len = @intCast(@sizeOf(InHeader) + name.len + 1), + .opcode = @intFromEnum(Opcode.lookup), + .unique = 7, + .nodeid = root_id, + .uid = 1000, + .gid = 1000, + .pid = 42, + .total_extlen = 0, + .padding = 0, + }; + @memcpy(buf[0..40], std.mem.asBytes(&hdr)); + @memcpy(buf[40..45], name); + buf[45] = 0; + const req = Request{ .header = hdr, .body = buf[40..hdr.len] }; + try testing.expectEqual(Opcode.lookup, req.header.op()); + try testing.expectEqualStrings("hello", try nameAfter(void, req)); + try testing.expectError(error.Protocol, body(MkdirIn, Request{ .header = hdr, .body = buf[40..44] })); + + // MKDIR(mode=0o755) + "dir" + const mk = MkdirIn{ .mode = 0o755, .umask = 0o22 }; + @memcpy(buf[40..48], std.mem.asBytes(&mk)); + @memcpy(buf[48..51], "dir"); + buf[51] = 0; + const mreq = Request{ .header = hdr, .body = buf[40..52] }; + const got = try body(MkdirIn, mreq); + try testing.expectEqual(@as(u32, 0o755), got.mode); + try testing.expectEqualStrings("dir", try nameAfter(MkdirIn, mreq)); + + // RENAME(newdir) + "old\0new\0" + const rn = RenameIn{ .newdir = 9 }; + @memcpy(buf[40..48], std.mem.asBytes(&rn)); + @memcpy(buf[48..56], "old\x00new\x00"); + const rreq = Request{ .header = hdr, .body = buf[40..56] }; + try testing.expectEqual(@as(u64, 9), (try body(RenameIn, rreq)).newdir); + const old = try nameAfter(RenameIn, rreq); + try testing.expectEqualStrings("old", old); + try testing.expectEqualStrings("new", try secondName(rreq, old, @sizeOf(RenameIn))); + try testing.expectError(error.Protocol, secondName(rreq, "new", @sizeOf(RenameIn) + 4)); + + // Missing NUL and misalignment are protocol errors. + try testing.expectError(error.Protocol, nameAt(Request{ .header = hdr, .body = buf[48..51] }, 0)); + try testing.expectError(error.Protocol, body(MkdirIn, Request{ .header = hdr, .body = buf[41..57] })); +} + +test "initReply fields" { + var in = InitIn{ .major = 7, .minor = 45, .max_readahead = 131072, .flags = 0, .flags2 = 0, .unused = [_]u32{0} ** 11 }; + const a = initReply(&in, 1 << 20); + try testing.expectEqual(@as(u32, 7), a.major); + try testing.expectEqual(@as(u32, 31), a.minor); + try testing.expectEqual(@as(u32, 131072), a.max_readahead); + try testing.expectEqual(FUSE_ASYNC_READ | FUSE_ATOMIC_O_TRUNC | FUSE_BIG_WRITES | FUSE_AUTO_INVAL_DATA, a.flags); + try testing.expectEqual(@as(u16, 0), a.max_pages); + try testing.expectEqual(@as(u16, 16), a.max_background); + try testing.expectEqual(@as(u16, 12), a.congestion_threshold); + try testing.expectEqual(@as(u32, 1 << 20), a.max_write); + try testing.expectEqual(@as(u32, 1), a.time_gran); + + in.flags = FUSE_MAX_PAGES | FUSE_ASYNC_READ; + in.minor = 27; + const b = initReply(&in, 4096); + try testing.expectEqual(@as(u32, 27), b.minor); + try testing.expectEqual(FUSE_ASYNC_READ | FUSE_ATOMIC_O_TRUNC | FUSE_BIG_WRITES | FUSE_AUTO_INVAL_DATA | FUSE_MAX_PAGES, b.flags); + try testing.expectEqual(@as(u16, 256), b.max_pages); + try testing.expectEqual(@as(u32, 4096), b.max_write); +} + +fn makePipe() ![2]i32 { + var fds: [2]i32 = undefined; + if (linux.errno(linux.pipe2(&fds, .{ .CLOEXEC = true })) != .SUCCESS) return error.Io; + return fds; +} + +fn readExact(fd: i32, out: []u8) !void { + var got: usize = 0; + while (got < out.len) { + const rc = linux.read(fd, out[got..].ptr, out.len - got); + if (linux.errno(rc) != .SUCCESS or rc == 0) return error.Io; + got += rc; + } +} + +test "reply writes header + payloads through a pipe" { + const fds = try makePipe(); + defer _ = linux.close(fds[0]); + defer _ = linux.close(fds[1]); + + const oo = OpenOut{ .fh = 0x1234, .open_flags = FOPEN_DIRECT_IO }; + try reply(fds[1], 99, &.{ std.mem.asBytes(&oo), "tail" }); + + var out: [16 + 16 + 4]u8 = undefined; + try readExact(fds[0], &out); + const h = std.mem.bytesToValue(OutHeader, out[0..16]); + try testing.expectEqual(@as(u32, 36), h.len); + try testing.expectEqual(@as(i32, 0), h.@"error"); + try testing.expectEqual(@as(u64, 99), h.unique); + try testing.expectEqualSlices(u8, std.mem.asBytes(&oo), out[16..32]); + try testing.expectEqualStrings("tail", out[32..36]); + + // Empty payload list: header only. + try reply(fds[1], 5, &.{}); + var only: [16]u8 = undefined; + try readExact(fds[0], &only); + try testing.expectEqual(@as(u32, 16), std.mem.bytesToValue(OutHeader, &only).len); + + var too_many: [max_payloads + 1][]const u8 = undefined; + for (&too_many) |*p| p.* = "x"; + try testing.expectError(error.TooManyPayloads, reply(fds[1], 1, &too_many)); +} + +test "replyError writes a negative errno" { + const fds = try makePipe(); + defer _ = linux.close(fds[0]); + defer _ = linux.close(fds[1]); + + try replyError(fds[1], 0xdead_beef, .NOENT); + var out: [16]u8 = undefined; + try readExact(fds[0], &out); + const h = std.mem.bytesToValue(OutHeader, &out); + try testing.expectEqual(@as(u32, 16), h.len); + try testing.expectEqual(@as(i32, -2), h.@"error"); + try testing.expectEqual(@as(u64, 0xdead_beef), h.unique); + + try replyError(fds[1], 1, .NOSYS); + try readExact(fds[0], &out); + try testing.expectEqual(-@as(i32, @intCast(@intFromEnum(linux.E.NOSYS))), std.mem.bytesToValue(OutHeader, &out).@"error"); +} + +test "readRequest parses one request from a pipe and rejects bad lengths" { + const fds = try makePipe(); + defer _ = linux.close(fds[0]); + defer _ = linux.close(fds[1]); + + var wire: [48]u8 align(8) = undefined; + const hdr = InHeader{ .len = 48, .opcode = @intFromEnum(Opcode.forget), .unique = 3, .nodeid = 2, .uid = 0, .gid = 0, .pid = 0, .total_extlen = 0, .padding = 0 }; + @memcpy(wire[0..40], std.mem.asBytes(&hdr)); + @memcpy(wire[40..48], std.mem.asBytes(&ForgetIn{ .nlookup = 11 })); + try testing.expectEqual(@as(usize, 48), linux.write(fds[1], &wire, wire.len)); + + var buf: [4096]u8 align(8) = undefined; + const req = (try readRequest(fds[0], &buf)) orelse return error.Io; + try testing.expectEqual(Opcode.forget, req.header.op()); + try testing.expectEqual(@as(u64, 2), req.header.nodeid); + try testing.expectEqual(@as(u64, 11), (try body(ForgetIn, req)).nlookup); + + // Header length disagreeing with what was read is a protocol error. + var bad = wire; + std.mem.bytesAsValue(InHeader, bad[0..40]).len = 40; + try testing.expectEqual(@as(usize, 48), linux.write(fds[1], &bad, bad.len)); + try testing.expectError(error.Protocol, readRequest(fds[0], &buf)); +} diff --git a/9ns/src/main.zig b/9ns/src/main.zig new file mode 100644 index 0000000..26bd699 --- /dev/null +++ b/9ns/src/main.zig @@ -0,0 +1,444 @@ +//! 9ns: mount a 9P2000 tree into a fresh user+mount namespace via FUSE +//! and run a program inside it. +//! +//! Exit codes: the child's status (128+sig if signalled); 125 for 9ns's +//! own failures (usage, connect, attach, namespace/mount); 126/127 for exec +//! failures. + +const std = @import("std"); +const linux = std.os.linux; +const ns = @import("ns.zig"); +const nine = @import("nine.zig"); +const bridge = @import("bridge.zig"); + +const version_string = "9ns 0.1.0"; + +const usage_text = + \\Usage: 9ns [options] -- PROGRAM [ARGS...] + \\Transport (exactly one): + \\ --unix PATH Unix stream socket + \\ --tcp IP:PORT TCP (IPv4/IPv6 literal) + \\ --fd N already-connected inherited descriptor + \\ --spawn CMD run CMD (via /bin/sh -c) with a socketpair on its stdin/stdout + \\Options: + \\ --mount PATH mountpoint inside the new namespace (default /mnt/9p) + \\ --uname NAME 9P user name (default $USER, else "none") + \\ --aname NAME 9P tree to attach (default "") + \\ --msize BYTES maximum 9P message size to request (default 131072) + \\ --cache SECONDS attr/entry cache validity, may be fractional (default 1) + \\ --no-direct-io let the kernel cache file pages (trusts stat length) + \\ --debug trace FUSE and 9P operations on stderr + \\ --help, --version + \\PROGRAM defaults to $SHELL (else /bin/sh). The mountpoint is exported as $NINE_MOUNT. + \\ +; + +const own_failure: u8 = 125; +/// Largest 9P message size we agree to request: the session allocates two +/// buffers of this size up front, before the server negotiates it down. +const max_msize: u32 = 16 * 1024 * 1024; + +/// Write `text` to stdout (informational output such as --help); errors are +/// ignored, there is nowhere better to report them. +fn printStdout(text: []const u8) void { + var off: usize = 0; + while (off < text.len) { + const rc = linux.write(1, text[off..].ptr, text.len - off); + switch (linux.errno(rc)) { + .SUCCESS => off += rc, + .INTR => continue, + else => return, + } + } +} + +const Config = struct { + address: ?nine.Address = null, + spawn_cmd: ?[]const u8 = null, + mount: []const u8 = "/mnt/9p", + uname: ?[]const u8 = null, + aname: []const u8 = "", + msize: u32 = 131072, + cache_ns: u64 = 1_000_000_000, + direct_io: bool = true, + debug: bool = false, + /// Empty means "default program". + program: []const []const u8 = &.{}, +}; + +const ParseResult = union(enum) { + run: Config, + /// Usage error, already reported on stderr; exit with this status. + exit: u8, + /// --help/--version: text for stdout, then exit 0. Printing is left to + /// `main` so that no test path writes to fd 1 (under `zig build test` + /// that is the test runner's protocol pipe). + info: []const u8, +}; + +fn usageError(comptime fmt: []const u8, args: anytype) ParseResult { + std.debug.print("9ns: " ++ fmt ++ "\n(try 9ns --help)\n", args); + return .{ .exit = own_failure }; +} + +fn parseArgs(arena: std.mem.Allocator, args: []const [:0]const u8) !ParseResult { + var cfg = Config{}; + var transports: usize = 0; + var i: usize = 1; + var program_start: ?usize = null; + while (i < args.len) : (i += 1) { + const arg: []const u8 = args[i]; + if (std.mem.eql(u8, arg, "--")) { + program_start = i + 1; + break; + } + if (!std.mem.startsWith(u8, arg, "--")) { + // A single-dash word is a typo for an option, not a program. + if (arg.len > 1 and arg[0] == '-') return usageError("unknown option {s} (options start with --)", .{arg}); + // A bare word starts PROGRAM, as if "--" were given. + program_start = i; + break; + } + // Split "--opt=value". + var name = arg; + var inline_value: ?[]const u8 = null; + if (std.mem.indexOfScalar(u8, arg, '=')) |eq| { + name = arg[0..eq]; + inline_value = arg[eq + 1 ..]; + } + const Opt = enum { unix, tcp, fd, spawn, mount, uname, aname, msize, cache, @"no-direct-io", debug, help, version, unknown }; + const opt = std.meta.stringToEnum(Opt, name[2..]) orelse .unknown; + switch (opt) { + .@"no-direct-io", .debug, .help, .version => if (inline_value != null) return usageError("{s} takes no value", .{name}), + .unknown => return usageError("unknown option {s}", .{name}), + else => {}, + } + const value: []const u8 = switch (opt) { + .@"no-direct-io", .debug, .help, .version, .unknown => "", + else => inline_value orelse blk: { + i += 1; + if (i >= args.len) return usageError("{s} needs a value", .{name}); + break :blk args[i]; + }, + }; + switch (opt) { + .unix => { + if (value.len == 0) return usageError("--unix wants a socket path", .{}); + cfg.address = .{ .unix = value }; + transports += 1; + }, + .tcp => { + cfg.address = parseTcp(value) orelse return usageError("--tcp wants IP:PORT (IPv6 as [ADDR]:PORT), got '{s}'", .{value}); + transports += 1; + }, + .fd => { + const n = std.fmt.parseInt(i32, value, 10) catch return usageError("--fd wants a number, got '{s}'", .{value}); + if (n < 0) return usageError("--fd wants a non-negative number", .{}); + cfg.address = .{ .fd = n }; + transports += 1; + }, + .spawn => { + if (value.len == 0) return usageError("--spawn wants a command", .{}); + cfg.spawn_cmd = value; + transports += 1; + }, + .mount => { + if (value.len == 0) return usageError("--mount wants a path", .{}); + cfg.mount = value; + }, + .uname => cfg.uname = value, + .aname => cfg.aname = value, + .msize => { + cfg.msize = std.fmt.parseInt(u32, value, 10) catch return usageError("--msize wants a number, got '{s}'", .{value}); + if (cfg.msize < 4096 or cfg.msize > max_msize) return usageError("--msize must be between 4096 and {d}", .{max_msize}); + }, + .cache => { + const secs = std.fmt.parseFloat(f64, value) catch return usageError("--cache wants seconds, got '{s}'", .{value}); + if (!(secs >= 0) or secs > 1e9) return usageError("--cache out of range", .{}); + cfg.cache_ns = @intFromFloat(secs * 1e9); + }, + .@"no-direct-io" => cfg.direct_io = false, + .debug => cfg.debug = true, + .help => return .{ .info = usage_text }, + .version => return .{ .info = version_string ++ "\n" }, + .unknown => unreachable, + } + } + if (transports == 0) return usageError("one transport is required (--unix, --tcp, --fd or --spawn)", .{}); + if (transports > 1) return usageError("exactly one transport is allowed", .{}); + if (program_start) |start| { + const prog = try arena.alloc([]const u8, args.len - start); + for (args[start..], 0..) |a, j| prog[j] = a; + cfg.program = prog; + } + return .{ .run = cfg }; +} + +fn parseTcp(spec: []const u8) ?nine.Address { + const colon = std.mem.lastIndexOfScalar(u8, spec, ':') orelse return null; + var host = spec[0..colon]; + if (host.len >= 2 and host[0] == '[' and host[host.len - 1] == ']') host = host[1 .. host.len - 1]; + if (host.len == 0) return null; + const port = std.fmt.parseInt(u16, spec[colon + 1 ..], 10) catch return null; + return .{ .tcp = .{ .host = host, .port = port } }; +} + +/// `--spawn`: run CMD under /bin/sh with one end of a socketpair as its +/// stdin/stdout; the other end is the 9P transport. +const Server = struct { pid: i32, fd: i32 }; + +fn spawnServer(cmd: [:0]const u8, envp: [*:null]const ?[*:0]const u8) !Server { + var sv: [2]i32 = undefined; + switch (linux.errno(linux.socketpair(linux.AF.UNIX, linux.SOCK.STREAM | linux.SOCK.CLOEXEC, 0, &sv))) { + .SUCCESS => {}, + else => |e| { + std.debug.print("9ns: socketpair: E{t}\n", .{e}); + return error.SystemResources; + }, + } + const rc = linux.fork(); + switch (linux.errno(rc)) { + .SUCCESS => {}, + else => |e| { + _ = linux.close(sv[0]); + _ = linux.close(sv[1]); + std.debug.print("9ns: fork: E{t}\n", .{e}); + return error.SystemResources; + }, + } + if (rc == 0) { + // Child: dup2 clears CLOEXEC on 0 and 1; everything else is CLOEXEC. + if (linux.errno(linux.dup2(sv[1], 0)) != .SUCCESS or linux.errno(linux.dup2(sv[1], 1)) != .SUCCESS) linux.exit_group(125); + // The server shares our process group, so a Ctrl-C meant for the + // program would kill it and take the mount down with it: ignore the + // tty signals (inherited across exec). SIGPIPE goes back to its + // default, we only ignore it for ourselves. + ignoreSignal(.INT); + ignoreSignal(.QUIT); + defaultSignal(.PIPE); + const argv = [_:null]?[*:0]const u8{ "sh", "-c", cmd.ptr }; + const e = linux.errno(linux.execve("/bin/sh", &argv, envp)); + std.debug.print("9ns: --spawn: exec /bin/sh: E{t}\n", .{e}); + linux.exit_group(127); + } + _ = linux.close(sv[1]); + return .{ .pid = @intCast(rc), .fd = sv[0] }; +} + +fn stopServer(server: ?Server) void { + const s = server orelse return; + _ = linux.kill(s.pid, .TERM); + ns.reapAny(s.pid); +} + +/// Fail early (before spawning servers or forking) if /dev/fuse is unusable. +fn probeFuseDevice() bool { + const rc = linux.open("/dev/fuse", .{ .ACCMODE = .RDWR, .CLOEXEC = true }, 0); + switch (linux.errno(rc)) { + .SUCCESS => { + _ = linux.close(@intCast(rc)); + return true; + }, + .NOENT => std.debug.print("9ns: /dev/fuse: ENOENT (is the fuse module loaded? try: modprobe fuse)\n", .{}), + else => |e| std.debug.print("9ns: open /dev/fuse: E{t}\n", .{e}), + } + return false; +} + +fn ignoreSignal(sig: linux.SIG) void { + const ign = linux.Sigaction{ .handler = .{ .handler = linux.SIG.IGN }, .mask = linux.sigemptyset(), .flags = 0 }; + std.posix.sigaction(sig, &ign, null); +} + +fn defaultSignal(sig: linux.SIG) void { + const dfl = linux.Sigaction{ .handler = .{ .handler = linux.SIG.DFL }, .mask = linux.sigemptyset(), .flags = 0 }; + std.posix.sigaction(sig, &dfl, null); +} + +/// `--fd N`: the descriptor is ours from now on; it must not leak into the +/// program (which could otherwise read 9P replies meant for us). Fails on a +/// bad descriptor, which is the earliest place to report it. +fn adoptFd(fd: i32) bool { + switch (linux.errno(linux.fcntl(fd, linux.F.SETFD, linux.FD_CLOEXEC))) { + .SUCCESS => return true, + else => |e| { + std.debug.print("9ns: --fd {d}: E{t}\n", .{ fd, e }); + return false; + }, + } +} + +fn describeAddress(a: nine.Address, buf: []u8) []const u8 { + return switch (a) { + .unix => |p| std.fmt.bufPrint(buf, "unix socket {s}", .{p}) catch "unix socket", + .tcp => |t| std.fmt.bufPrint(buf, "tcp {s}:{d}", .{ t.host, t.port }) catch "tcp", + .fd => |fd| std.fmt.bufPrint(buf, "fd {d}", .{fd}) catch "fd", + }; +} + +pub fn main(init: std.process.Init) !u8 { + const gpa = init.gpa; + const arena = init.arena.allocator(); + const args = try init.minimal.args.toSlice(arena); + const envp: [*:null]const ?[*:0]const u8 = init.minimal.environ.block.slice.ptr; + + var cfg = switch (try parseArgs(arena, args)) { + .exit => |code| return code, + .info => |text| { + printStdout(text); + return 0; + }, + .run => |c| c, + }; + + // Defaults that come from the environment. + if (cfg.program.len == 0) { + const env_shell = ns.getenv(envp, "SHELL") orelse ""; + const shell = if (env_shell.len == 0) "/bin/sh" else env_shell; + cfg.program = try arena.dupe([]const u8, &.{shell}); + } + const uname = cfg.uname orelse ns.getenv(envp, "USER") orelse "none"; + const mountpoint = ns.resolveMountpoint(gpa, cfg.mount) catch |err| { + std.debug.print("9ns: --mount {s}: {t}\n", .{ cfg.mount, err }); + return own_failure; + }; + defer gpa.free(mountpoint); + + if (!probeFuseDevice()) return own_failure; + + // Writes to a dead server socket must not kill us. + ignoreSignal(.PIPE); + + var server: ?Server = null; + var address: nine.Address = undefined; + if (cfg.spawn_cmd) |cmd| { + const cmd_z = try arena.dupeZ(u8, cmd); + server = spawnServer(cmd_z, envp) catch return own_failure; + ns.watchServer(server.?.pid); + address = .{ .fd = server.?.fd }; + } else { + address = cfg.address.?; + if (address == .fd and !adoptFd(address.fd)) return own_failure; + } + + var addr_buf: [256]u8 = undefined; + var session = nine.Session.connect(gpa, address, cfg.msize) catch |err| { + std.debug.print("9ns: connect to {s}: {t}\n", .{ describeAddress(address, &addr_buf), err }); + stopServer(server); + return own_failure; + }; + defer session.deinit(); + defer stopServer(server); + + _ = session.attach(0, uname, cfg.aname) catch |err| { + switch (err) { + error.Nine => std.debug.print("9ns: attach (uname={s}, aname='{s}'): {s}\n", .{ uname, cfg.aname, session.ename[0..session.ename_len] }), + else => std.debug.print("9ns: attach: {t}\n", .{err}), + } + return own_failure; + }; + if (cfg.debug) std.debug.print("9ns: attached to {s} (msize {d}), mounting on {s}\n", .{ describeAddress(address, &addr_buf), session.msize, mountpoint }); + + var child_pid: i32 = 0; + const stop_fd = ns.installSignals(&child_pid) catch return own_failure; + + const uid = linux.getuid(); + const gid = linux.getgid(); + const child = ns.spawn(gpa, .{ + .argv = cfg.program, + .envp = envp, + .mountpoint = mountpoint, + .uid = uid, + .gid = gid, + .max_read = bridge.max_write, + }) catch return own_failure; + + bridge.serve(gpa, child.fuse_fd, &session, 0, stop_fd, .{ + .uid = uid, + .gid = gid, + .attr_timeout_ns = cfg.cache_ns, + .direct_io = cfg.direct_io, + .debug = cfg.debug, + }) catch |err| switch (err) { + error.Closed => std.debug.print("9ns: 9P server connection closed\n", .{}), + else => std.debug.print("9ns: fuse: {t}\n", .{err}), + }; + + // Closing the device aborts the FUSE connection: anything still using + // the mount gets ENOTCONN instead of hanging on an unserved request. + _ = linux.close(child.fuse_fd); + + const status = ns.reapIfExited(child.pid) orelse ns.waitChild(child.pid) catch own_failure; + // An exec failure (126/127) is already in `status`; this prints its message. + _ = ns.reportExecFailure(child); + return status; +} + +test "parseTcp" { + const a = parseTcp("127.0.0.1:564").?; + try std.testing.expectEqualStrings("127.0.0.1", a.tcp.host); + try std.testing.expectEqual(@as(u16, 564), a.tcp.port); + const b = parseTcp("[::1]:9999").?; + try std.testing.expectEqualStrings("::1", b.tcp.host); + try std.testing.expectEqual(@as(u16, 9999), b.tcp.port); + try std.testing.expect(parseTcp("nohost") == null); + try std.testing.expect(parseTcp(":564") == null); + try std.testing.expect(parseTcp("1.2.3.4:") == null); + try std.testing.expect(parseTcp("1.2.3.4:70000") == null); +} + +test "parseArgs" { + const arena = std.testing.allocator; + { + const args = [_][:0]const u8{ "9ns", "--unix", "/s", "--cache", "0.5", "--msize=8192", "--no-direct-io", "--", "sh", "-c", "x" }; + const r = try parseArgs(arena, &args); + defer arena.free(r.run.program); + try std.testing.expectEqualStrings("/s", r.run.address.?.unix); + try std.testing.expectEqual(@as(u64, 500_000_000), r.run.cache_ns); + try std.testing.expectEqual(@as(u32, 8192), r.run.msize); + try std.testing.expect(!r.run.direct_io); + try std.testing.expectEqual(@as(usize, 3), r.run.program.len); + try std.testing.expectEqualStrings("x", r.run.program[2]); + } + { + const args = [_][:0]const u8{ "9ns", "--fd", "3" }; + const r = try parseArgs(arena, &args); + try std.testing.expectEqual(@as(i32, 3), r.run.address.?.fd); + try std.testing.expectEqual(@as(usize, 0), r.run.program.len); + try std.testing.expectEqualStrings("/mnt/9p", r.run.mount); + } + { + // Two transports, no transport, unknown option, missing value: all 125. + const two = [_][:0]const u8{ "9ns", "--fd", "3", "--unix", "/s" }; + try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &two)).exit); + const none = [_][:0]const u8{ "9ns", "--", "sh" }; + try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &none)).exit); + const unknown = [_][:0]const u8{ "9ns", "--bogus" }; + try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &unknown)).exit); + const missing = [_][:0]const u8{ "9ns", "--unix" }; + try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &missing)).exit); + const badcache = [_][:0]const u8{ "9ns", "--fd", "3", "--cache", "abc" }; + try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &badcache)).exit); + // Empty values, a single-dash typo, and an msize that would allocate gigabytes. + const emptyunix = [_][:0]const u8{ "9ns", "--unix=", "--", "sh" }; + try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &emptyunix)).exit); + const emptymount = [_][:0]const u8{ "9ns", "--fd", "3", "--mount", "" }; + try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &emptymount)).exit); + const singledash = [_][:0]const u8{ "9ns", "--fd", "3", "-mount", "/x" }; + try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &singledash)).exit); + const hugemsize = [_][:0]const u8{ "9ns", "--fd", "3", "--msize", "4294967295" }; + try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &hugemsize)).exit); + const okmsize = [_][:0]const u8{ "9ns", "--fd", "3", "--msize", "16777216" }; + try std.testing.expectEqual(@as(u32, 16777216), (try parseArgs(arena, &okmsize)).run.msize); + } + { + const ver = [_][:0]const u8{ "9ns", "--version" }; + try std.testing.expectEqualStrings(version_string ++ "\n", (try parseArgs(arena, &ver)).info); + const help = [_][:0]const u8{ "9ns", "--help" }; + try std.testing.expect(std.mem.startsWith(u8, (try parseArgs(arena, &help)).info, "Usage: 9ns")); + } +} + +test { + _ = ns; +} diff --git a/9ns/src/nine.zig b/9ns/src/nine.zig new file mode 100644 index 0000000..70633e6 --- /dev/null +++ b/9ns/src/nine.zig @@ -0,0 +1,756 @@ +//! Synchronous 9P2000 session over a blocking file descriptor. +//! +//! A thin RPC layer over `cloud9.Client` (push/take, allocation-free). One request +//! is outstanding at a time: the FUSE loop that drives this is single-threaded, so +//! every call here blocks until its reply (or the connection's death) arrives. +//! Fids are handed out from a free list; fid 0 is reserved for the root. +const std = @import("std"); +const cloud9 = @import("cloud9"); +const linux = std.os.linux; + +pub const Address = union(enum) { + unix: []const u8, + tcp: struct { host: []const u8, port: u16 }, + fd: i32, +}; + +/// A Stat whose every field means "leave unchanged" in a Twstat. +pub const dontcare = cloud9.Stat{ + .type = 0xFFFF, + .dev = 0xFFFF_FFFF, + .qid = .{ .type = 0xFF, .version = 0xFFFF_FFFF, .path = 0xFFFF_FFFF_FFFF_FFFF }, + .mode = 0xFFFF_FFFF, + .atime = 0xFFFF_FFFF, + .mtime = 0xFFFF_FFFF, + .length = 0xFFFF_FFFF_FFFF_FFFF, + .name = "", + .uid = "", + .gid = "", + .muid = "", +}; + +pub const Session = struct { + pub const Error = error{ Nine, Protocol, Io, Closed, Stopped, TooLarge, OutOfMemory }; + + pub const Walk = struct { nwqid: u16, wqid: [cloud9.max_welem]cloud9.Qid }; + pub const Open = struct { qid: cloud9.Qid, iounit: u32 }; + + gpa: std.mem.Allocator, + fd: i32, + client: cloud9.Client, + in_buf: []u8, + out_buf: []u8, + /// After `error.Nine`, the server's Rerror text (copied, bounded). + ename: [256]u8 = undefined, + ename_len: usize = 0, + /// Negotiated maximum message size. + msize: u32, + next_fid: u32 = 1, + free_fids: std.ArrayList(u32) = .empty, + /// Per-fid iounit learned from open/create (0 = none); used to chunk read/write. + iounits: std.AutoHashMapUnmanaged(u32, u32) = .empty, + /// Optional descriptor watched while waiting for a reply: when it becomes + /// readable (the bridge's "child exited" pipe) the pending rpc fails with + /// `error.Stopped` instead of blocking on a server that never answers. + stop_fd: i32 = -1, + + /// Connect to `address`, then negotiate the protocol version. + /// `msize` is the maximum message size to ask for (0 = the buffers' size). + pub fn connect(gpa: std.mem.Allocator, address: Address, msize: u32) !Session { + const want: u32 = if (msize == 0) 8192 else @max(msize, 24); + const fd = try openTransport(address); + errdefer if (address != .fd) { + _ = linux.close(fd); + }; + + const in_buf = try gpa.alloc(u8, want); + errdefer gpa.free(in_buf); + const out_buf = try gpa.alloc(u8, want); + errdefer gpa.free(out_buf); + + var s: Session = .{ + .gpa = gpa, + .fd = fd, + .client = .init(.{ .in = in_buf, .out = out_buf }), + .in_buf = in_buf, + .out_buf = out_buf, + .msize = want, + }; + const r = try s.rpc(.{ .version = .{ .msize = want } }); + if (!std.mem.eql(u8, r.version.version, "9P2000")) return error.Protocol; + s.msize = r.version.msize; + return s; + } + + /// Closes the descriptor and frees the buffers. Fids are not clunked. + pub fn deinit(s: *Session) void { + _ = linux.close(s.fd); + s.free_fids.deinit(s.gpa); + s.iounits.deinit(s.gpa); + s.gpa.free(s.in_buf); + s.gpa.free(s.out_buf); + s.* = undefined; + } + + pub fn attach(s: *Session, fid: u32, uname: []const u8, aname: []const u8) Error!cloud9.Qid { + const r = try s.rpc(.{ .attach = .{ .fid = fid, .uname = uname, .aname = aname } }); + return r.attach; + } + + /// Fid 0 is never handed out: it belongs to the root attach. + pub fn allocFid(s: *Session) u32 { + if (s.free_fids.pop()) |fid| return fid; + const fid = s.next_fid; + s.next_fid += 1; + return fid; + } + + /// Fids currently bound (excluding fid 0); a debugging aid for leak hunting. + pub fn fidsInUse(s: *const Session) usize { + return (s.next_fid - 1) - s.free_fids.items.len; + } + + pub fn freeFid(s: *Session, fid: u32) void { + _ = s.iounits.remove(fid); + // If the free list cannot grow the fid is simply leaked; the counter keeps going. + s.free_fids.append(s.gpa, fid) catch {}; + } + + /// Generic RPC. Result slices borrow the input buffer until the next call. + pub fn rpc(s: *Session, req: cloud9.Client.Request) Error!cloud9.Client.Result { + s.ename_len = 0; + _ = s.client.submit(req) catch |e| switch (e) { + error.NoTags, error.Handshake, error.Dead => return error.Protocol, + error.NoSpace, error.TooLarge => return error.TooLarge, + error.BadRequest => { + s.setEname("bad request"); + return error.Nine; + }, + }; + try s.flush(); + var tmp: [64 * 1024]u8 = undefined; + while (true) { + if (s.client.take()) |done| { + switch (done.result) { + .fail => |ename| { + s.setEname(ename); + return error.Nine; + }, + else => return done.result, + } + } + if (s.client.dead) return error.Protocol; + // After take() returned null the previous frame is gone, so the free + // space is at least what the pending frame still needs. + const room = s.client.in.len - s.client.in_len; + if (room == 0) return error.Protocol; + const n = try readSome(s.fd, s.stop_fd, tmp[0..@min(room, tmp.len)]); + if (n == 0) return error.Closed; + const pushed = s.client.push(tmp[0..n]); + if (pushed != n) return error.Protocol; + } + } + + /// Walk `names` from `fid` to `newfid`. A partial walk leaves `newfid` unbound + /// (9P semantics) and reports `error.Nine` with ename "file does not exist". + pub fn walk(s: *Session, fid: u32, newfid: u32, names: []const []const u8) Error!Walk { + const r = try s.rpc(.{ .walk = .{ .fid = fid, .newfid = newfid, .names = names } }); + if (r.walk.nwqid < names.len) { + s.setEname("file does not exist"); + return error.Nine; + } + return .{ .nwqid = r.walk.nwqid, .wqid = r.walk.wqid }; + } + + /// allocFid + zero-element walk. The fid is released again on failure. + pub fn clone(s: *Session, fid: u32) Error!u32 { + const newfid = s.allocFid(); + errdefer s.freeFid(newfid); + _ = try s.walk(fid, newfid, &.{}); + return newfid; + } + + pub fn open(s: *Session, fid: u32, mode: u8) Error!Open { + const r = try s.rpc(.{ .open = .{ .fid = fid, .mode = mode } }); + s.noteIounit(fid, r.open.iounit); + return .{ .qid = r.open.qid, .iounit = r.open.iounit }; + } + + pub fn create(s: *Session, fid: u32, name: []const u8, perm: u32, mode: u8) Error!Open { + const r = try s.rpc(.{ .create = .{ .fid = fid, .name = name, .perm = perm, .mode = mode } }); + s.noteIounit(fid, r.create.iounit); + return .{ .qid = r.create.qid, .iounit = r.create.iounit }; + } + + /// Reads into `buf`, chunking by min(maxRead, iounit) and stopping at the first + /// short read. Returns the number of bytes read (0 at end of file). + pub fn read(s: *Session, fid: u32, offset: u64, buf: []u8) Error!usize { + return readWith(s, rpc, fid, offset, buf, s.chunk(fid)); + } + + /// Writes `data`, chunking like `read` and stopping at the first short write. + pub fn write(s: *Session, fid: u32, offset: u64, data: []const u8) Error!usize { + return writeWith(s, rpc, fid, offset, data, s.chunkWrite(fid)); + } + + /// The returned Stat's strings (name/uid/gid/muid) borrow the session's input + /// buffer: they are valid only until the next rpc. Copy what must outlive it. + pub fn stat(s: *Session, fid: u32) Error!cloud9.Stat { + const r = try s.rpc(.{ .stat = .{ .fid = fid } }); + return r.stat; + } + + pub fn wstat(s: *Session, fid: u32, st: cloud9.Stat) Error!void { + _ = try s.rpc(.{ .wstat = .{ .fid = fid, .stat = st } }); + } + + /// Frees the fid locally even when the server reports an error. + pub fn clunk(s: *Session, fid: u32) Error!void { + defer s.freeFid(fid); + _ = try s.rpc(.{ .clunk = .{ .fid = fid } }); + } + + /// Frees the fid locally even when the server reports an error. + pub fn remove(s: *Session, fid: u32) Error!void { + defer s.freeFid(fid); + _ = try s.rpc(.{ .remove = .{ .fid = fid } }); + } + + /// Maps the last Rerror text to an errno (case-insensitive substring match). + pub fn errno(s: *const Session) linux.E { + return enameToErrno(s.ename[0..s.ename_len]); + } + + // -- internals -------------------------------------------------------------- + + fn setEname(s: *Session, text: []const u8) void { + const n = @min(text.len, 255); + @memcpy(s.ename[0..n], text[0..n]); + s.ename_len = n; + } + + fn noteIounit(s: *Session, fid: u32, iounit: u32) void { + if (iounit == 0) { + _ = s.iounits.remove(fid); + } else { + s.iounits.put(s.gpa, fid, iounit) catch {}; + } + } + + fn chunk(s: *Session, fid: u32) u32 { + return chunkSize(s.client.maxRead(), s.iounits.get(fid) orelse 0); + } + + fn chunkWrite(s: *Session, fid: u32) u32 { + return chunkSize(s.client.maxWrite(), s.iounits.get(fid) orelse 0); + } + + /// Writes everything in the client's output buffer to the socket. + fn flush(s: *Session) Error!void { + while (s.client.output().len != 0) { + const out = s.client.output(); + const rc = linux.write(s.fd, out.ptr, out.len); + switch (linux.errno(rc)) { + .SUCCESS => { + if (rc == 0) return error.Closed; + s.client.wrote(rc); + }, + .INTR, .AGAIN => continue, + .PIPE, .CONNRESET => return error.Closed, + else => return error.Io, + } + } + } +}; + +fn chunkSize(max: u32, iounit: u32) u32 { + if (iounit != 0 and iounit < max) return iounit; + return max; +} + +/// Chunked read over any rpc-shaped function (injected so the loop is testable). +fn readWith( + s: anytype, + comptime rpcFn: anytype, + fid: u32, + offset: u64, + buf: []u8, + max_chunk: u32, +) Session.Error!usize { + if (max_chunk == 0) return error.Protocol; + var done: usize = 0; + while (done < buf.len) { + const want: u32 = @intCast(@min(buf.len - done, max_chunk)); + const r = try rpcFn(s, .{ .read = .{ .fid = fid, .offset = offset + done, .count = want } }); + const data = r.read; + @memcpy(buf[done..][0..data.len], data); + done += data.len; + if (data.len < want) break; + } + return done; +} + +/// Chunked write over any rpc-shaped function. +fn writeWith( + s: anytype, + comptime rpcFn: anytype, + fid: u32, + offset: u64, + data: []const u8, + max_chunk: u32, +) Session.Error!usize { + if (max_chunk == 0) return error.Protocol; + var done: usize = 0; + while (done < data.len) { + const want: usize = @min(data.len - done, max_chunk); + const r = try rpcFn(s, .{ .write = .{ .fid = fid, .offset = offset + done, .data = data[done..][0..want] } }); + done += r.write; + if (r.write < want) break; + } + return done; +} + +fn readSome(fd: i32, stop_fd: i32, buf: []u8) Session.Error!usize { + while (true) { + if (stop_fd >= 0) { + var pfds = [_]linux.pollfd{ + .{ .fd = fd, .events = linux.POLL.IN, .revents = 0 }, + .{ .fd = stop_fd, .events = linux.POLL.IN, .revents = 0 }, + }; + const prc = linux.poll(&pfds, pfds.len, -1); + switch (linux.errno(prc)) { + .SUCCESS => {}, + .INTR, .AGAIN => continue, + else => return error.Io, + } + if (pfds[1].revents != 0 and pfds[0].revents == 0) return error.Stopped; + } + const rc = linux.read(fd, buf.ptr, buf.len); + switch (linux.errno(rc)) { + .SUCCESS => return rc, + .INTR, .AGAIN => continue, + .CONNRESET => return error.Closed, + else => return error.Io, + } + } +} + +/// Rerror text → errno, per docs/DESIGN.md (first match wins). +pub fn enameToErrno(ename: []const u8) linux.E { + const Rule = struct { needle: []const u8, err: linux.E }; + const rules = [_]Rule{ + .{ .needle = "not exist", .err = .NOENT }, + .{ .needle = "not found", .err = .NOENT }, + .{ .needle = "no such", .err = .NOENT }, + .{ .needle = "exists", .err = .EXIST }, + .{ .needle = "not empty", .err = .NOTEMPTY }, + .{ .needle = "not a dir", .err = .NOTDIR }, + .{ .needle = "is a dir", .err = .ISDIR }, + .{ .needle = "permission", .err = .ACCES }, + .{ .needle = "denied", .err = .ACCES }, + .{ .needle = "read-only", .err = .ROFS }, + .{ .needle = "read only", .err = .ROFS }, + .{ .needle = "readonly", .err = .ROFS }, + .{ .needle = "no space", .err = .NOSPC }, + .{ .needle = "not allowed", .err = .PERM }, + .{ .needle = "not permitted", .err = .PERM }, + .{ .needle = "cannot", .err = .PERM }, + .{ .needle = "fid", .err = .BADF }, + .{ .needle = "bad offset", .err = .INVAL }, + .{ .needle = "invalid", .err = .INVAL }, + .{ .needle = "bad ", .err = .INVAL }, + .{ .needle = "busy", .err = .BUSY }, + .{ .needle = "in use", .err = .BUSY }, + .{ .needle = "too long", .err = .NAMETOOLONG }, + .{ .needle = "not supported", .err = .OPNOTSUPP }, + .{ .needle = "unsupported", .err = .OPNOTSUPP }, + }; + for (rules) |rule| { + if (std.ascii.findIgnoreCase(ename, rule.needle) != null) return rule.err; + } + return .IO; +} + +// -- transport ------------------------------------------------------------------ + +fn openTransport(address: Address) !i32 { + switch (address) { + .fd => |fd| return fd, + .unix => |path| { + if (path.len == 0 or path.len >= 108) return error.NameTooLong; + var sa: linux.sockaddr.un = .{ .path = @splat(0) }; + @memcpy(sa.path[0..path.len], path); + const fd = try newSocket(linux.AF.UNIX, 0); + errdefer _ = linux.close(fd); + try doConnect(fd, @ptrCast(&sa), @sizeOf(linux.sockaddr.un)); + return fd; + }, + .tcp => |t| { + const ip = std.Io.net.IpAddress.parse(t.host, t.port) catch return error.InvalidAddress; + switch (ip) { + .ip4 => |a| { + const sa: linux.sockaddr.in = .{ + .port = std.mem.nativeToBig(u16, t.port), + .addr = @bitCast(a.bytes), + }; + const fd = try newSocket(linux.AF.INET, linux.IPPROTO.TCP); + errdefer _ = linux.close(fd); + setNodelay(fd); + try doConnect(fd, @ptrCast(&sa), @sizeOf(linux.sockaddr.in)); + return fd; + }, + .ip6 => |a| { + const sa: linux.sockaddr.in6 = .{ + .port = std.mem.nativeToBig(u16, t.port), + .flowinfo = 0, + .addr = a.bytes, + .scope_id = 0, + }; + const fd = try newSocket(linux.AF.INET6, linux.IPPROTO.TCP); + errdefer _ = linux.close(fd); + setNodelay(fd); + try doConnect(fd, @ptrCast(&sa), @sizeOf(linux.sockaddr.in6)); + return fd; + }, + } + }, + } +} + +fn newSocket(domain: u32, protocol: u32) !i32 { + const rc = linux.socket(domain, linux.SOCK.STREAM | linux.SOCK.CLOEXEC, protocol); + switch (linux.errno(rc)) { + .SUCCESS => return @intCast(rc), + .MFILE, .NFILE => return error.ProcessFdQuotaExceeded, + .AFNOSUPPORT, .PROTONOSUPPORT => return error.AddressFamilyNotSupported, + .ACCES => return error.AccessDenied, + .NOMEM, .NOBUFS => return error.SystemResources, + else => return error.Unexpected, + } +} + +fn setNodelay(fd: i32) void { + const one: u32 = 1; + _ = linux.setsockopt(fd, linux.IPPROTO.TCP, linux.TCP.NODELAY, @ptrCast(&one), @sizeOf(u32)); +} + +fn doConnect(fd: i32, addr: *const linux.sockaddr, len: linux.socklen_t) !void { + while (true) { + const rc = linux.connect(fd, addr, len); + switch (linux.errno(rc)) { + .SUCCESS => return, + .INTR => continue, + .CONNREFUSED => return error.ConnectionRefused, + .NOENT, .NOTDIR => return error.FileNotFound, + .ACCES, .PERM => return error.AccessDenied, + .TIMEDOUT => return error.ConnectionTimedOut, + .NETUNREACH, .HOSTUNREACH => return error.NetworkUnreachable, + .ADDRNOTAVAIL => return error.AddressNotAvailable, + .AGAIN, .INPROGRESS => return error.WouldBlock, + else => return error.Unexpected, + } + } +} + +// -- tests ---------------------------------------------------------------------- + +const testing = std.testing; + +test { + testing.refAllDecls(@This()); +} + +test "ename → errno mapping" { + try testing.expectEqual(linux.E.NOENT, enameToErrno("file does not exist")); + try testing.expectEqual(linux.E.NOENT, enameToErrno("No Such File")); + try testing.expectEqual(linux.E.NOENT, enameToErrno("directory entry not found")); + try testing.expectEqual(linux.E.EXIST, enameToErrno("file already exists")); + try testing.expectEqual(linux.E.NOTEMPTY, enameToErrno("directory not empty")); + try testing.expectEqual(linux.E.NOTDIR, enameToErrno("not a directory")); + try testing.expectEqual(linux.E.ISDIR, enameToErrno("is a directory")); + try testing.expectEqual(linux.E.ACCES, enameToErrno("permission denied")); + try testing.expectEqual(linux.E.ACCES, enameToErrno("access denied")); + try testing.expectEqual(linux.E.ROFS, enameToErrno("read-only file system")); + try testing.expectEqual(linux.E.NOSPC, enameToErrno("no space left")); + try testing.expectEqual(linux.E.PERM, enameToErrno("operation not permitted")); + try testing.expectEqual(linux.E.PERM, enameToErrno("cannot remove root")); + try testing.expectEqual(linux.E.BADF, enameToErrno("unknown fid")); + try testing.expectEqual(linux.E.BADF, enameToErrno("fid in use")); // "fid" precedes "in use" + try testing.expectEqual(linux.E.INVAL, enameToErrno("bad offset")); + try testing.expectEqual(linux.E.INVAL, enameToErrno("invalid argument")); + try testing.expectEqual(linux.E.INVAL, enameToErrno("bad request")); + try testing.expectEqual(linux.E.BUSY, enameToErrno("device busy")); + try testing.expectEqual(linux.E.NAMETOOLONG, enameToErrno("name too long")); + try testing.expectEqual(linux.E.OPNOTSUPP, enameToErrno("operation not supported")); + try testing.expectEqual(linux.E.IO, enameToErrno("something odd happened")); + try testing.expectEqual(linux.E.IO, enameToErrno("")); +} + +test "fid allocator recycles and never hands out 0" { + var s: Session = undefined; + s.gpa = testing.allocator; + s.next_fid = 1; + s.free_fids = .empty; + s.iounits = .empty; + defer s.free_fids.deinit(s.gpa); + defer s.iounits.deinit(s.gpa); + + const a = s.allocFid(); + const b = s.allocFid(); + const c = s.allocFid(); + try testing.expectEqual(@as(u32, 1), a); + try testing.expectEqual(@as(u32, 2), b); + try testing.expectEqual(@as(u32, 3), c); + s.freeFid(b); + try testing.expectEqual(b, s.allocFid()); + s.freeFid(a); + s.freeFid(c); + const x = s.allocFid(); + const y = s.allocFid(); + try testing.expect((x == a and y == c) or (x == c and y == a)); + try testing.expectEqual(@as(u32, 4), s.allocFid()); + try testing.expect(a != 0 and b != 0 and c != 0); +} + +test "chunkSize honours iounit only when smaller" { + try testing.expectEqual(@as(u32, 100), chunkSize(100, 0)); + try testing.expectEqual(@as(u32, 40), chunkSize(100, 40)); + try testing.expectEqual(@as(u32, 100), chunkSize(100, 400)); +} + +/// Fake rpc for the chunked read/write loops: a file of `len` bytes where byte i == i & 0xff. +const FakeFile = struct { + len: usize, + calls: usize = 0, + max_count: u32 = 0, + short_write_at: ?usize = null, + scratch: [4096]u8 = undefined, + + fn rpc(f: *FakeFile, req: cloud9.Client.Request) Session.Error!cloud9.Client.Result { + f.calls += 1; + switch (req) { + .read => |r| { + f.max_count = @max(f.max_count, r.count); + if (r.offset >= f.len) return .{ .read = "" }; + const n: usize = @min(@as(usize, r.count), f.len - @as(usize, @intCast(r.offset))); + for (f.scratch[0..n], 0..) |*b, i| b.* = @truncate(r.offset + i); + return .{ .read = f.scratch[0..n] }; + }, + .write => |w| { + f.max_count = @max(f.max_count, @as(u32, @intCast(w.data.len))); + if (f.short_write_at) |at| { + if (w.offset + w.data.len > at) { + const n: usize = if (w.offset >= at) 0 else @intCast(at - w.offset); + return .{ .write = @intCast(n) }; + } + } + return .{ .write = @intCast(w.data.len) }; + }, + else => unreachable, + } + } +}; + +test "read chunks by max_chunk and stops at a short read" { + var f: FakeFile = .{ .len = 2500 }; + var buf: [4000]u8 = undefined; + const n = try readWith(&f, FakeFile.rpc, 7, 0, &buf, 1000); + try testing.expectEqual(@as(usize, 2500), n); + try testing.expectEqual(@as(usize, 3), f.calls); // 1000, 1000, 500 (short → stop) + try testing.expectEqual(@as(u32, 1000), f.max_count); + for (buf[0..n], 0..) |b, i| try testing.expectEqual(@as(u8, @truncate(i)), b); + + // Reading exactly up to a chunk boundary uses one call per chunk and no more. + f = .{ .len = 2000 }; + try testing.expectEqual(@as(usize, 2000), try readWith(&f, FakeFile.rpc, 7, 0, buf[0..2000], 1000)); + try testing.expectEqual(@as(usize, 2), f.calls); + + // Offset past EOF → 0. + f = .{ .len = 10 }; + try testing.expectEqual(@as(usize, 0), try readWith(&f, FakeFile.rpc, 7, 50, &buf, 1000)); +} + +test "write chunks and stops at a short write" { + var f: FakeFile = .{ .len = 0 }; + var data: [2500]u8 = undefined; + for (&data, 0..) |*b, i| b.* = @truncate(i); + try testing.expectEqual(@as(usize, 2500), try writeWith(&f, FakeFile.rpc, 7, 0, &data, 1000)); + try testing.expectEqual(@as(usize, 3), f.calls); + try testing.expectEqual(@as(u32, 1000), f.max_count); + + f = .{ .len = 0, .short_write_at = 1500 }; + try testing.expectEqual(@as(usize, 1500), try writeWith(&f, FakeFile.rpc, 7, 0, &data, 1000)); + try testing.expectEqual(@as(usize, 2), f.calls); +} + +// -- in-process server test --------------------------------------------------------- + +/// A tiny 9P2000 backend on a cloud9.Server: answers version/attach/walk/stat/open/ +/// read/clunk/remove with canned data. Runs in its own thread over a socketpair. +const FakeServer = struct { + fd: i32, + msize: u32, + max_read_count: u32 = 0, + file_len: usize, + + const file_qid: cloud9.Qid = .{ .type = 0, .version = 3, .path = 0x1234 }; + const dir_qid: cloud9.Qid = .{ .type = cloud9.qtdir, .version = 1, .path = 0x1 }; + + fn run(fs: *FakeServer) void { + fs.loop() catch |e| std.debug.print("fake server: {s}\n", .{@errorName(e)}); + _ = linux.close(fs.fd); + } + + fn loop(fs: *FakeServer) !void { + const gpa = testing.allocator; + const in = try gpa.alloc(u8, fs.msize); + defer gpa.free(in); + const out = try gpa.alloc(u8, fs.msize * 2); + defer gpa.free(out); + var srv: cloud9.Server = .init(.{ .in = in, .out = out }); + var tmp: [4096]u8 = undefined; + var data: [8192]u8 = undefined; + while (true) { + while (try srv.receive()) |req| { + const tag = req.tag; + switch (req.msg) { + .tversion => |m| try srv.negotiate(m.msize, m.version), + .tattach => try srv.reply(tag, .{ .rattach = .{ .qid = dir_qid } }), + .twalk => |m| { + var wq: [cloud9.max_welem]cloud9.Qid = @splat(dir_qid); + var n: u16 = 0; + for (m.wname[0..m.nwname]) |name| { + if (std.mem.eql(u8, name, "file")) { + wq[n] = file_qid; + } else if (std.mem.eql(u8, name, "dir")) { + wq[n] = dir_qid; + } else break; + n += 1; + } + if (n == 0 and m.nwname != 0) { + try srv.reply(tag, .{ .rerror = .{ .ename = "file does not exist" } }); + } else { + try srv.reply(tag, .{ .rwalk = .{ .nwqid = n, .wqid = wq } }); + } + }, + .tstat => try srv.reply(tag, .{ .rstat = .{ .stat = .{ + .type = 0, + .dev = 0, + .qid = file_qid, + .mode = 0o644, + .atime = 1, + .mtime = 2, + .length = fs.file_len, + .name = "file", + .uid = "u", + .gid = "g", + .muid = "u", + } } }), + .topen => |m| try srv.reply(tag, .{ .ropen = .{ .qid = file_qid, .iounit = if (m.mode == cloud9.owrite) 700 else 0 } }), + .tread => |m| { + fs.max_read_count = @max(fs.max_read_count, m.count); + var n: usize = 0; + if (m.offset < fs.file_len) n = @min(@as(usize, m.count), fs.file_len - @as(usize, @intCast(m.offset))); + n = @min(n, data.len); + for (data[0..n], 0..) |*b, i| b.* = @truncate(m.offset + i); + try srv.reply(tag, .{ .rread = .{ .data = data[0..n] } }); + }, + .twrite => |m| try srv.reply(tag, .{ .rwrite = .{ .count = @intCast(m.data.len) } }), + .tclunk => try srv.reply(tag, .rclunk), + .tremove => try srv.reply(tag, .{ .rerror = .{ .ename = "permission denied" } }), + .twstat => try srv.reply(tag, .rwstat), + // A flush is the test's "hang up now" signal. + .tflush => return, + else => try srv.reply(tag, .{ .rerror = .{ .ename = "not supported" } }), + } + srv.release(); + } + while (srv.output().len != 0) { + const o = srv.output(); + const rc = linux.write(fs.fd, o.ptr, o.len); + if (linux.errno(rc) != .SUCCESS) return error.Write; + srv.wrote(rc); + } + const rc = linux.read(fs.fd, &tmp, tmp.len); + if (linux.errno(rc) != .SUCCESS) return error.Read; + if (rc == 0) return; + if (srv.push(tmp[0..rc]) != rc) return error.Overflow; + } + } +}; + +test "session against an in-process cloud9.Server" { + var fds: [2]i32 = undefined; + try testing.expectEqual(linux.E.SUCCESS, linux.errno(linux.socketpair(linux.AF.UNIX, linux.SOCK.STREAM | linux.SOCK.CLOEXEC, 0, &fds))); + + var fs: FakeServer = .{ .fd = fds[1], .msize = 8192, .file_len = 20_000 }; + const th = try std.Thread.spawn(.{}, FakeServer.run, .{&fs}); + + var s = try Session.connect(testing.allocator, .{ .fd = fds[0] }, 8192); + defer { + s.deinit(); + th.join(); + } + try testing.expectEqual(@as(u32, 8192), s.msize); + + const root = try s.attach(0, "me", ""); + try testing.expectEqual(FakeServer.dir_qid.path, root.path); + + // Plain rpc + stat borrowing the input buffer. + const fid = s.allocFid(); + const w = try s.walk(0, fid, &.{"file"}); + try testing.expectEqual(@as(u16, 1), w.nwqid); + try testing.expectEqual(FakeServer.file_qid.path, w.wqid[0].path); + const st = try s.stat(fid); + try testing.expectEqualStrings("file", st.name); + try testing.expectEqual(@as(u64, 20_000), st.length); + + // Chunked read: 20000 bytes at maxRead = msize - 11 = 8181 per chunk. + _ = try s.open(fid, cloud9.oread); + const buf = try testing.allocator.alloc(u8, 30_000); + defer testing.allocator.free(buf); + const n = try s.read(fid, 0, buf); + try testing.expectEqual(@as(usize, 20_000), n); + for (buf[0..n], 0..) |b, i| try testing.expectEqual(@as(u8, @truncate(i)), b); + try testing.expectEqual(@as(u32, 8181), fs.max_read_count); + try testing.expectEqual(@as(usize, 0), try s.read(fid, 20_000, buf)); + + // iounit from open bounds the chunk. + const wfid = try s.clone(fid); + _ = try s.open(wfid, cloud9.owrite); + fs.max_read_count = 0; + _ = try s.read(wfid, 0, buf[0..3000]); + try testing.expectEqual(@as(u32, 700), fs.max_read_count); + try testing.expectEqual(@as(usize, 3000), try s.write(wfid, 0, buf[0..3000])); + + // Partial walk → error.Nine with a "not exist" ename → ENOENT. + const pfid = s.allocFid(); + try testing.expectError(error.Nine, s.walk(0, pfid, &.{ "dir", "nope" })); + try testing.expectEqual(linux.E.NOENT, s.errno()); + try testing.expectEqualStrings("file does not exist", s.ename[0..s.ename_len]); + s.freeFid(pfid); + + // Server Rerror → error.Nine, ename copied, fid freed by remove even on error. + try testing.expectError(error.Nine, s.remove(wfid)); + try testing.expectEqual(linux.E.ACCES, s.errno()); + try testing.expectEqual(wfid, s.allocFid()); // recycled + s.freeFid(wfid); + + // Unsupported op → "not supported" → ENOTSUP; a plain wstat succeeds. + try testing.expectError(error.Nine, s.rpc(.{ .auth = .{ .afid = 5, .uname = "me" } })); + try testing.expectEqual(linux.E.OPNOTSUPP, s.errno()); + try s.wstat(fid, dontcare); + try s.clunk(fid); + try testing.expectEqual(fid, s.allocFid()); + s.freeFid(fid); + + // A clone bound to a fid that then fails to walk must release the fid. + const before = s.next_fid; + const cfid = s.allocFid(); + s.freeFid(cfid); + try testing.expectError(error.Nine, s.walk(0, cfid, &.{"nope"})); + try testing.expectEqual(before, s.next_fid); + + // The server hanging up makes the pending rpc fail with error.Closed. + try testing.expectError(error.Closed, s.rpc(.{ .flush = .{ .oldtag = 0 } })); +} diff --git a/9ns/src/ns.zig b/9ns/src/ns.zig new file mode 100644 index 0000000..6da2c7f --- /dev/null +++ b/9ns/src/ns.zig @@ -0,0 +1,1082 @@ +//! Namespace and process plumbing for 9ns. +//! +//! Everything here is raw `std.os.linux` syscalls (no libc). The child side +//! of `spawn` runs between `fork` and `execve`; it does not allocate except +//! inside `ensureMountpoint` (the process is single-threaded by then, so the +//! inherited allocator is safe to use). +//! +//! Exit codes produced by the child before exec: 125 for namespace/mount +//! setup failures, 126 when the program was found but is not executable, +//! 127 when it was not found. + +const std = @import("std"); +const builtin = @import("builtin"); +const linux = std.os.linux; +const Allocator = std.mem.Allocator; +const E = linux.E; + +pub const Spawn = struct { + /// argv[0] is PATH-searched unless it contains '/'. + argv: []const []const u8, + /// Inherited environment; `NINE_MOUNT` is added or replaced. + envp: [*:null]const ?[*:0]const u8, + /// Absolute mountpoint (see `resolveMountpoint`). + mountpoint: []const u8, + uid: u32, + gid: u32, + max_read: u32, + /// When false the namespace is set up (including mountpoint shadowing) + /// but `/dev/fuse` is not opened and nothing is mounted; `Child.fuse_fd` + /// is then -1. Only for smoke tests. + mount_fuse: bool = true, +}; + +pub const Child = struct { + pid: i32, + /// The `/dev/fuse` connection backing the mount, opened by the child + /// inside its user namespace (the kernel refuses to mount a fuse fd that + /// was opened from another user namespace) and handed back over the + /// status socket with SCM_RIGHTS. Owned by the caller; CLOEXEC. + fuse_fd: i32, + /// Parent end of the status socket. The child reports an exec failure + /// on it (see `reportExecFailure`); it reads EOF once exec succeeded. + status_fd: i32, +}; + +/// Exit status used by the child for setup failures (matches 9ns's own). +pub const setup_failure_status: u8 = 125; +/// Refuse to shadow a directory with more entries than this. +pub const max_shadow_entries: usize = 4096; + +const default_path = "/usr/local/bin:/bin:/usr/bin"; +const path_max = 4096; + +// --------------------------------------------------------------------------- +// Mountpoint resolution +// --------------------------------------------------------------------------- + +/// Absolute path (relative paths resolved against cwd), duplicate slashes +/// collapsed, `.` and `..` components resolved lexically, no trailing slash. +/// `/` itself is rejected. +pub fn resolveMountpoint(gpa: Allocator, path: []const u8) ![:0]u8 { + var cwd_buf: [path_max]u8 = undefined; + var cwd: []const u8 = "/"; + if (path.len == 0 or path[0] != '/') { + const rc = linux.getcwd(&cwd_buf, cwd_buf.len); + switch (linux.errno(rc)) { + .SUCCESS => {}, + else => |e| { + std.debug.print("9ns: getcwd: E{t}\n", .{e}); + return error.Cwd; + }, + } + // rc counts the terminating NUL. + cwd = cwd_buf[0 .. rc - 1]; + } + return normalizePath(gpa, cwd, path); +} + +/// Pure part of `resolveMountpoint`: `cwd` is only used when `path` is relative. +fn normalizePath(gpa: Allocator, cwd: []const u8, path: []const u8) ![:0]u8 { + if (path.len == 0) return error.InvalidMountpoint; + var out: std.ArrayList(u8) = .empty; + defer out.deinit(gpa); + if (path[0] != '/') try appendComponents(gpa, &out, cwd); + try appendComponents(gpa, &out, path); + if (out.items.len == 0) return error.InvalidMountpoint; // "/" or equivalent + return out.toOwnedSliceSentinel(gpa, 0); +} + +fn appendComponents(gpa: Allocator, out: *std.ArrayList(u8), path: []const u8) !void { + var it = std.mem.tokenizeScalar(u8, path, '/'); + while (it.next()) |comp| { + if (std.mem.eql(u8, comp, ".")) continue; + if (std.mem.eql(u8, comp, "..")) { + // Pop the last component (lexically; "/.." stays "/"). + const idx = std.mem.lastIndexOfScalar(u8, out.items, '/') orelse 0; + out.shrinkRetainingCapacity(idx); + continue; + } + try out.append(gpa, '/'); + try out.appendSlice(gpa, comp); + } +} + +// --------------------------------------------------------------------------- +// Environment helpers +// --------------------------------------------------------------------------- + +/// Look a variable up in a raw envp block. +pub fn getenv(envp: [*:null]const ?[*:0]const u8, name: []const u8) ?[]const u8 { + var i: usize = 0; + while (envp[i]) |entry| : (i += 1) { + const kv = std.mem.span(entry); + if (kv.len > name.len and kv[name.len] == '=' and std.mem.eql(u8, kv[0..name.len], name)) { + return kv[name.len + 1 ..]; + } + } + return null; +} + +/// Every path `execve` should try for `name`, in order: just `name` if it +/// contains a '/', else `<dir>/<name>` for each `$PATH` element (an empty +/// element means the current directory; `$PATH` unset falls back to +/// `/usr/local/bin:/bin:/usr/bin`). +pub fn pathCandidates(gpa: Allocator, envp: [*:null]const ?[*:0]const u8, name: []const u8) ![]const [:0]const u8 { + if (name.len == 0) return error.EmptyProgramName; + var list: std.ArrayList([:0]const u8) = .empty; + errdefer { + for (list.items) |c| gpa.free(c); + list.deinit(gpa); + } + if (std.mem.indexOfScalar(u8, name, '/') != null) { + try list.append(gpa, try gpa.dupeZ(u8, name)); + return list.toOwnedSlice(gpa); + } + const path = getenv(envp, "PATH") orelse default_path; + var it = std.mem.splitScalar(u8, path, ':'); + while (it.next()) |dir| { + const d = if (dir.len == 0) "." else dir; + try list.append(gpa, try std.fmt.allocPrintSentinel(gpa, "{s}/{s}", .{ d, name }, 0)); + } + return list.toOwnedSlice(gpa); +} + +/// First PATH candidate that is an executable regular file, or the name +/// itself when it contains a '/'. Provided for completeness; `spawn` simply +/// tries `execve` on every candidate instead. +pub fn findInPath(gpa: Allocator, envp: [*:null]const ?[*:0]const u8, name: []const u8) ![:0]u8 { + const cands = try pathCandidates(gpa, envp, name); + defer { + for (cands) |c| gpa.free(c); + gpa.free(cands); + } + for (cands) |c| { + var stx: linux.Statx = undefined; + const rc = linux.statx(linux.AT.FDCWD, c.ptr, 0, .{ .TYPE = true, .MODE = true }, &stx); + if (linux.errno(rc) != .SUCCESS) continue; + if (stx.mode & linux.S.IFMT != linux.S.IFREG) continue; + if (stx.mode & 0o111 == 0) continue; + return gpa.dupeZ(u8, c); + } + return error.FileNotFound; +} + +/// New envp block: every entry of `envp` except `NINE_MOUNT=...`, +/// followed by `NINE_MOUNT=<mountpoint>`. +fn buildEnvp(gpa: Allocator, envp: [*:null]const ?[*:0]const u8, mountpoint: []const u8) ![:null]?[*:0]const u8 { + const key = "NINE_MOUNT="; + var keep: usize = 0; + var i: usize = 0; + while (envp[i]) |entry| : (i += 1) { + if (!std.mem.startsWith(u8, std.mem.span(entry), key)) keep += 1; + } + const out = try gpa.allocSentinel(?[*:0]const u8, keep + 1, null); + errdefer gpa.free(out); + var j: usize = 0; + i = 0; + while (envp[i]) |entry| : (i += 1) { + if (std.mem.startsWith(u8, std.mem.span(entry), key)) continue; + out[j] = entry; + j += 1; + } + const mount_entry = try std.fmt.allocPrintSentinel(gpa, key ++ "{s}", .{mountpoint}, 0); + out[j] = mount_entry.ptr; + return out; +} + +fn buildArgv(gpa: Allocator, argv: []const []const u8) ![:null]?[*:0]const u8 { + const out = try gpa.allocSentinel(?[*:0]const u8, argv.len, null); + for (argv, 0..) |a, i| out[i] = (try gpa.dupeZ(u8, a)).ptr; + return out; +} + +// --------------------------------------------------------------------------- +// Mountpoint policy +// --------------------------------------------------------------------------- + +/// Make sure `path` is a directory, inside the *current* mount namespace: +/// +/// * already a directory → done; +/// * else `mkdir`; on `EACCES`/`EPERM`/`EROFS` shadow the parent directory +/// with a tmpfs that re-exposes every existing entry (bind mounts for +/// directories and files, recreated symlinks) and `mkdir` inside it; +/// * anything else fails with the errno and a hint. +/// +/// Every failure prints `9ns: <step> <path>: E<errno>` to stderr before +/// returning. Meant to be called in the child of `spawn` (or from a +/// throwaway namespace: `unshare -Urm`). +pub fn ensureMountpoint(gpa: Allocator, path: [:0]const u8) !void { + if (fileType(linux.AT.FDCWD, path, false)) |ft| { + if (ft == .dir) return; + std.debug.print("9ns: mountpoint {s}: exists but is not a directory\n", .{path}); + return error.Mountpoint; + } + if (fileType(linux.AT.FDCWD, path, true) == .symlink) { + std.debug.print("9ns: mountpoint {s}: dangling symlink\n", .{path}); + return error.Mountpoint; + } + const mk = linux.errno(linux.mkdirat(linux.AT.FDCWD, path, 0o755)); + switch (mk) { + .SUCCESS => return, + .ACCES, .PERM, .ROFS => {}, + else => |e| { + std.debug.print("9ns: mkdir {s}: E{t} (pass --mount an existing directory)\n", .{ path, e }); + return error.Mountpoint; + }, + } + const parent = std.fs.path.dirname(path) orelse "/"; + if (std.mem.eql(u8, parent, "/") or isSameDirectory(parent, "/")) { + std.debug.print("9ns: mkdir {s}: E{t}; refusing to shadow / (pass --mount an existing directory)\n", .{ path, mk }); + return error.Mountpoint; + } + // The shadow rebuilds entries from /proc/self/fd/<fd>/<name>; a tmpfs + // over /proc (or a subtree of it) would take that away from itself. + if (std.mem.eql(u8, parent, "/proc") or std.mem.startsWith(u8, parent, "/proc/")) { + std.debug.print("9ns: mkdir {s}: E{t}; refusing to shadow {s} (pass --mount an existing directory)\n", .{ path, mk, parent }); + return error.Mountpoint; + } + const parent_z = try gpa.dupeZ(u8, parent); + defer gpa.free(parent_z); + try shadowDirectory(gpa, parent_z); + switch (linux.errno(linux.mkdirat(linux.AT.FDCWD, path, 0o755))) { + .SUCCESS => {}, + else => |e| { + std.debug.print("9ns: mkdir {s} (in shadow tmpfs): E{t}\n", .{ path, e }); + return error.Mountpoint; + }, + } +} + +const FileType = enum { dir, symlink, other }; + +/// True when both paths resolve (following symlinks, including magic ones +/// such as /proc/self/root) to the same inode. +fn isSameDirectory(a: []const u8, b: [*:0]const u8) bool { + var a_buf: [path_max]u8 = undefined; + const a_z = std.fmt.bufPrintZ(&a_buf, "{s}", .{a}) catch return false; + var sa: linux.Statx = undefined; + var sb: linux.Statx = undefined; + if (linux.errno(linux.statx(linux.AT.FDCWD, a_z, 0, .{ .INO = true }, &sa)) != .SUCCESS) return false; + if (linux.errno(linux.statx(linux.AT.FDCWD, b, 0, .{ .INO = true }, &sb)) != .SUCCESS) return false; + return sa.ino == sb.ino and sa.dev_major == sb.dev_major and sa.dev_minor == sb.dev_minor; +} + +fn fileType(dirfd: i32, name: [*:0]const u8, nofollow: bool) ?FileType { + var stx: linux.Statx = undefined; + const flags: u32 = if (nofollow) linux.AT.SYMLINK_NOFOLLOW else 0; + const rc = linux.statx(dirfd, name, flags, .{ .TYPE = true }, &stx); + if (linux.errno(rc) != .SUCCESS) return null; + return switch (stx.mode & linux.S.IFMT) { + linux.S.IFDIR => .dir, + linux.S.IFLNK => .symlink, + else => .other, + }; +} + +const Entry = struct { name: [:0]u8, kind: FileType }; + +/// Read every entry of the directory open at `fd` (excluding `.` and `..`). +fn listDir(gpa: Allocator, fd: i32, dirpath: []const u8) ![]Entry { + var list: std.ArrayList(Entry) = .empty; + errdefer { + for (list.items) |e| gpa.free(e.name); + list.deinit(gpa); + } + var buf: [32 * 1024]u8 align(@alignOf(linux.dirent64)) = undefined; + while (true) { + const rc = linux.getdents64(fd, &buf, buf.len); + switch (linux.errno(rc)) { + .SUCCESS => {}, + else => |e| { + std.debug.print("9ns: getdents64 {s}: E{t}\n", .{ dirpath, e }); + return error.Mountpoint; + }, + } + if (rc == 0) break; + var off: usize = 0; + while (off < rc) { + const d: *align(1) const linux.dirent64 = @ptrCast(&buf[off]); + const name_ptr: [*:0]const u8 = @ptrCast(&buf[off + @offsetOf(linux.dirent64, "name")]); + const name = std.mem.span(name_ptr); + const dtype = d.type; + off += d.reclen; + if (std.mem.eql(u8, name, ".") or std.mem.eql(u8, name, "..")) continue; + if (list.items.len >= max_shadow_entries) { + std.debug.print("9ns: refusing to shadow {s}: more than {d} entries\n", .{ dirpath, max_shadow_entries }); + return error.TooManyEntries; + } + const kind: FileType = switch (dtype) { + linux.DT.DIR => .dir, + linux.DT.LNK => .symlink, + linux.DT.UNKNOWN => fileType(fd, name_ptr, true) orelse .other, + else => .other, + }; + try list.append(gpa, .{ .name = try gpa.dupeZ(u8, name), .kind = kind }); + } + } + return list.toOwnedSlice(gpa); +} + +fn shadowDirectory(gpa: Allocator, parent: [:0]const u8) !void { + const open_rc = linux.open(parent, .{ .ACCMODE = .RDONLY, .DIRECTORY = true, .CLOEXEC = true }, 0); + switch (linux.errno(open_rc)) { + .SUCCESS => {}, + else => |e| { + std.debug.print("9ns: open {s}: E{t}\n", .{ parent, e }); + return error.Mountpoint; + }, + } + const pfd: i32 = @intCast(open_rc); + defer _ = linux.close(pfd); + + const entries = try listDir(gpa, pfd, parent); + defer { + for (entries) |e| gpa.free(e.name); + gpa.free(entries); + } + + const tmpfs_opts: [*:0]const u8 = "mode=755"; + switch (linux.errno(linux.mount("tmpfs", parent, "tmpfs", linux.MS.NOSUID | linux.MS.NODEV, @intFromPtr(tmpfs_opts)))) { + .SUCCESS => {}, + else => |e| { + std.debug.print("9ns: mount tmpfs on {s}: E{t}\n", .{ parent, e }); + return error.Mountpoint; + }, + } + + // `pfd` still refers to the original directory underneath the tmpfs, so + // `/proc/self/fd/<pfd>/<name>` reaches the hidden entries. + var src_buf: [path_max]u8 = undefined; + var dst_buf: [path_max]u8 = undefined; + var link_buf: [path_max]u8 = undefined; + for (entries) |e| { + const src = std.fmt.bufPrintZ(&src_buf, "/proc/self/fd/{d}/{s}", .{ pfd, e.name }) catch { + std.debug.print("9ns: shadow {s}/{s}: name too long (skipped)\n", .{ parent, e.name }); + continue; + }; + const dst = std.fmt.bufPrintZ(&dst_buf, "{s}/{s}", .{ parent, e.name }) catch { + std.debug.print("9ns: shadow {s}/{s}: name too long (skipped)\n", .{ parent, e.name }); + continue; + }; + switch (e.kind) { + .dir => { + if (!check("mkdir", dst, linux.mkdirat(linux.AT.FDCWD, dst, 0o755))) continue; + _ = check("bind", dst, linux.mount(src, dst, null, linux.MS.BIND | linux.MS.REC, 0)); + }, + .symlink => { + const rc = linux.readlinkat(pfd, e.name, &link_buf, link_buf.len - 1); + if (!check("readlink", dst, rc)) continue; + link_buf[rc] = 0; + const target: [*:0]const u8 = @ptrCast(&link_buf); + _ = check("symlink", dst, linux.symlinkat(target, linux.AT.FDCWD, dst)); + }, + .other => { + const rc = linux.openat(linux.AT.FDCWD, dst, .{ .ACCMODE = .WRONLY, .CREAT = true, .CLOEXEC = true }, 0o644); + if (!check("create", dst, rc)) continue; + _ = linux.close(@intCast(rc)); + _ = check("bind", dst, linux.mount(src, dst, null, linux.MS.BIND | linux.MS.REC, 0)); + }, + } + } +} + +/// Report a failed per-entry step as a warning (the entry is skipped; the +/// rest of the shadow is still useful). Returns true on success. +fn check(step: []const u8, path: [*:0]const u8, rc: usize) bool { + switch (linux.errno(rc)) { + .SUCCESS => return true, + else => |e| { + std.debug.print("9ns: shadow: {s} {s}: E{t} (skipped)\n", .{ step, std.mem.span(path), e }); + return false; + }, + } +} + +// --------------------------------------------------------------------------- +// spawn +// --------------------------------------------------------------------------- + +const ChildArgs = struct { + gpa: Allocator, + status_sock: i32, + mountpoint: [:0]const u8, + fuse_opts_prefix: [:0]const u8, // everything after "fd=<n>," + mount_fuse: bool, + uid_map: []const u8, + gid_map: []const u8, + argv: [:null]?[*:0]const u8, + envp: [:null]?[*:0]const u8, + candidates: []const [:0]const u8, + name: []const u8, +}; + +/// Status channel protocol (child → parent, over a CLOEXEC socketpair): +/// a 0 byte means "namespace and mount are up" and carries the fuse fd as +/// SCM_RIGHTS; a non-zero byte is an exit status followed by a message. +/// EOF ends the conversation (exec succeeded, or the child died). +const ok_byte: u8 = 0; + +/// fork; the child unshares user+mount namespaces, maps its uid/gid, +/// makes `/` private, ensures the mountpoint, opens `/dev/fuse`, mounts it +/// on the mountpoint and sends the fd back. `spawn` returns at that point +/// (with `error.ChildFailed` and a message on stderr if any step failed). +/// The child then stats the mountpoint, which makes the kernel fetch the +/// root's attributes once the parent serves (the kernel seeds the fuse root +/// with uid 0, unmapped in the new user namespace, so nothing could be +/// created in the root until then), sets `NINE_MOUNT` and execs +/// `argv`. An exec failure is reported on `Child.status_fd` and ends the +/// child with 126/127; collect it with `reportExecFailure` after +/// `bridge.serve` returns. +/// +/// If `installSignals` was called, the pid is stored into the registered +/// variable as soon as fork returns so no SIGCHLD can be missed. +pub fn spawn(gpa: Allocator, s: Spawn) !Child { + if (s.argv.len == 0 or s.argv[0].len == 0) { + std.debug.print("9ns: empty program name\n", .{}); + return error.EmptyProgramName; + } + + const mountpoint = try gpa.dupeZ(u8, s.mountpoint); + defer gpa.free(mountpoint); + const fuse_opts_prefix = try std.fmt.allocPrintSentinel(gpa, "rootmode=40000,user_id={d},group_id={d},max_read={d}", .{ s.uid, s.gid, s.max_read }, 0); + defer gpa.free(fuse_opts_prefix); + var uid_buf: [64]u8 = undefined; + var gid_buf: [64]u8 = undefined; + const uid_map = try std.fmt.bufPrint(&uid_buf, "{d} {d} 1\n", .{ s.uid, s.uid }); + const gid_map = try std.fmt.bufPrint(&gid_buf, "{d} {d} 1\n", .{ s.gid, s.gid }); + const argv = try buildArgv(gpa, s.argv); + defer { + for (argv) |a| gpa.free(std.mem.span(a.?)); + gpa.free(argv); + } + const envp = try buildEnvp(gpa, s.envp, s.mountpoint); + defer { + gpa.free(std.mem.span(envp[envp.len - 1].?)); // the NINE_MOUNT entry we created + gpa.free(envp); + } + const candidates = try pathCandidates(gpa, s.envp, s.argv[0]); + defer { + for (candidates) |c| gpa.free(c); + gpa.free(candidates); + } + + var sv: [2]i32 = undefined; + switch (linux.errno(linux.socketpair(linux.AF.UNIX, linux.SOCK.STREAM | linux.SOCK.CLOEXEC, 0, &sv))) { + .SUCCESS => {}, + else => |e| { + std.debug.print("9ns: socketpair: E{t}\n", .{e}); + return error.SystemResources; + }, + } + + const child_args = ChildArgs{ + .gpa = gpa, + .status_sock = sv[1], + .mountpoint = mountpoint, + .fuse_opts_prefix = fuse_opts_prefix, + .mount_fuse = s.mount_fuse, + .uid_map = uid_map, + .gid_map = gid_map, + .argv = argv, + .envp = envp, + .candidates = candidates, + .name = s.argv[0], + }; + + const fork_rc = linux.fork(); + switch (linux.errno(fork_rc)) { + .SUCCESS => {}, + else => |e| { + _ = linux.close(sv[0]); + _ = linux.close(sv[1]); + std.debug.print("9ns: fork: E{t}\n", .{e}); + return error.SystemResources; + }, + } + if (fork_rc == 0) childMain(&child_args); + + const pid: i32 = @intCast(fork_rc); + if (child_pid_ptr) |p| @atomicStore(i32, p, pid, .seq_cst); + _ = linux.close(sv[1]); + + // First byte: ok (with the fuse fd attached) or a failure status. + var first: [1]u8 = .{ok_byte}; // defined even if recvmsg stores nothing + var fuse_fd: i32 = -1; + var n: usize = 0; + while (true) { + const rc = recvWithFd(sv[0], &first, &fuse_fd, 0); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .INTR => continue, + else => break, + } + n = rc; + break; + } + if (n == 1 and first[0] == ok_byte and (fuse_fd >= 0 or !s.mount_fuse)) { + return .{ .pid = pid, .fuse_fd = fuse_fd, .status_fd = sv[0] }; + } + + // Failure. A status byte means the child is exiting on its own and a + // message follows. Anything else (EOF: the child died before reporting; + // an ok byte without the fd: the SCM_RIGHTS transfer was truncated, e.g. + // EMFILE) is a protocol violation: the child may be about to exec with a + // dead mount, so kill it before waiting rather than reading the status + // socket until an exec'd program eventually exits. + const reported = n == 1 and first[0] != ok_byte; + if (!reported) _ = linux.kill(pid, .KILL); + if (fuse_fd >= 0) _ = linux.close(fuse_fd); + var msg: [512]u8 = undefined; + var len: usize = 0; + while (reported and len < msg.len) { + const rc = linux.read(sv[0], msg[len..].ptr, msg.len - len); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .INTR => continue, + else => break, + } + if (rc == 0) break; + len += rc; + } + _ = linux.close(sv[0]); + if (reported) { + std.debug.print("9ns: {s}\n", .{msg[0..len]}); + } else if (n == 1) { + std.debug.print("9ns: child handshake failed: no fuse fd received (out of file descriptors?)\n", .{}); + } else { + std.debug.print("9ns: child exited before reporting\n", .{}); + } + _ = waitChild(pid) catch {}; + if (child_pid_ptr) |p| @atomicStore(i32, p, 0, .seq_cst); + const status: u8 = if (reported) first[0] else setup_failure_status; + return switch (status) { + 126 => error.ExecPermission, + 127 => error.ExecNotFound, + else => error.ChildFailed, + }; +} + +/// After the child is gone (or the mount is dead): print the exec failure +/// the child reported on `status_fd`, if any, and close it. Returns the +/// status byte the child announced, or null when exec succeeded / nothing +/// was reported. Never blocks. +pub fn reportExecFailure(child: Child) ?u8 { + defer _ = linux.close(child.status_fd); + var msg: [512]u8 = undefined; + var len: usize = 0; + while (len < msg.len) { + var iov = [_]std.posix.iovec{.{ .base = msg[len..].ptr, .len = msg.len - len }}; + var hdr = linux.msghdr{ + .name = null, + .namelen = 0, + .iov = &iov, + .iovlen = 1, + .control = null, + .controllen = 0, + .flags = 0, + }; + const rc = linux.recvmsg(child.status_fd, &hdr, linux.MSG.DONTWAIT); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .INTR => continue, + else => break, + } + if (rc == 0) break; + len += rc; + } + if (len == 0) return null; + std.debug.print("9ns: {s}\n", .{msg[1..len]}); + return msg[0]; +} + +const cmsg_fd_len = @sizeOf(linux.cmsghdr) + @sizeOf(i32); +const cmsg_fd_space = std.mem.alignForward(usize, cmsg_fd_len, @sizeOf(usize)); + +/// sendmsg one data byte, optionally with `fd` attached as SCM_RIGHTS. +fn sendWithFd(sock: i32, byte: u8, fd: ?i32) usize { + const data = [_]u8{byte}; + const iov = [_]std.posix.iovec_const{.{ .base = &data, .len = 1 }}; + var cbuf: [cmsg_fd_space]u8 align(@alignOf(linux.cmsghdr)) = @splat(0); + var msg = linux.msghdr_const{ + .name = null, + .namelen = 0, + .iov = &iov, + .iovlen = 1, + .control = null, + .controllen = 0, + .flags = 0, + }; + if (fd) |f| { + const hdr: *linux.cmsghdr = @ptrCast(&cbuf); + hdr.* = .{ .len = cmsg_fd_len, .level = linux.SOL.SOCKET, .type = linux.SCM.RIGHTS }; + @memcpy(cbuf[@sizeOf(linux.cmsghdr)..][0..@sizeOf(i32)], std.mem.asBytes(&f)); + msg.control = &cbuf; + msg.controllen = cmsg_fd_space; + } + return linux.sendmsg(sock, &msg, linux.MSG.NOSIGNAL); +} + +/// recvmsg into `buf`; an SCM_RIGHTS fd, if any, is stored in `fd_out`. +fn recvWithFd(sock: i32, buf: []u8, fd_out: *i32, flags: u32) usize { + var iov = [_]std.posix.iovec{.{ .base = buf.ptr, .len = buf.len }}; + var cbuf: [cmsg_fd_space]u8 align(@alignOf(linux.cmsghdr)) = @splat(0); + var msg = linux.msghdr{ + .name = null, + .namelen = 0, + .iov = &iov, + .iovlen = 1, + .control = &cbuf, + .controllen = cbuf.len, + .flags = 0, + }; + const rc = linux.recvmsg(sock, &msg, linux.MSG.CMSG_CLOEXEC | flags); + if (linux.errno(rc) != .SUCCESS) return rc; + if (msg.controllen >= cmsg_fd_len) { + const hdr: *const linux.cmsghdr = @ptrCast(&cbuf); + if (hdr.level == linux.SOL.SOCKET and hdr.type == linux.SCM.RIGHTS and hdr.len >= cmsg_fd_len) { + var fd: i32 = undefined; + @memcpy(std.mem.asBytes(&fd), cbuf[@sizeOf(linux.cmsghdr)..][0..@sizeOf(i32)]); + fd_out.* = fd; + } + } + return rc; +} + +/// Child side of `spawn`. Never returns. +fn childMain(c: *const ChildArgs) noreturn { + resetSignals(); + + const rc_unshare = linux.errno(linux.unshare(linux.CLONE.NEWUSER | linux.CLONE.NEWNS)); + if (rc_unshare != .SUCCESS) childFail(c, setup_failure_status, "unshare(CLONE_NEWUSER|CLONE_NEWNS)", rc_unshare, true); + writeProcFile(c, "/proc/self/setgroups", "deny", true); + writeProcFile(c, "/proc/self/uid_map", c.uid_map, false); + writeProcFile(c, "/proc/self/gid_map", c.gid_map, false); + + const root: [*:0]const u8 = "/"; + const rc_priv = linux.mount(null, root, null, linux.MS.REC | linux.MS.PRIVATE, 0); + if (linux.errno(rc_priv) != .SUCCESS) childFail(c, setup_failure_status, "mount(/, MS_REC|MS_PRIVATE)", linux.errno(rc_priv), true); + + ensureMountpoint(c.gpa, c.mountpoint) catch { + childFail(c, setup_failure_status, "mountpoint setup failed (pass --mount an existing directory)", .SUCCESS, false); + }; + + var fuse_fd: ?i32 = null; + if (c.mount_fuse) { + // Must be opened here, after unshare: the kernel only mounts a fuse + // device opened from the mount's own user namespace. + const rc_open = linux.open("/dev/fuse", .{ .ACCMODE = .RDWR, .CLOEXEC = true }, 0); + switch (linux.errno(rc_open)) { + .SUCCESS => {}, + .NOENT => childFail(c, setup_failure_status, "open /dev/fuse: ENOENT (is the fuse module loaded? try: modprobe fuse)", .SUCCESS, false), + else => |e| childFail(c, setup_failure_status, "open /dev/fuse", e, true), + } + const fd: i32 = @intCast(rc_open); + var opts_buf: [256]u8 = undefined; + const opts = std.fmt.bufPrintZ(&opts_buf, "fd={d},{s}", .{ fd, c.fuse_opts_prefix }) catch unreachable; + const rc = linux.mount("9ns", c.mountpoint, "fuse", linux.MS.NOSUID | linux.MS.NODEV, @intFromPtr(opts.ptr)); + if (linux.errno(rc) != .SUCCESS) childFail(c, setup_failure_status, "mount fuse", linux.errno(rc), true); + fuse_fd = fd; + } + const sent = sendWithFd(c.status_sock, ok_byte, fuse_fd); + if (linux.errno(sent) != .SUCCESS) linux.exit_group(setup_failure_status); + if (fuse_fd) |fd| { + _ = linux.close(fd); // the parent holds the connection now + // Force one GETATTR of the root (served by the parent, which is + // entering its serve loop now); see `spawn`. Errors don't matter. + var stx: linux.Statx = undefined; + _ = linux.statx(linux.AT.FDCWD, c.mountpoint, 0, .{ .TYPE = true }, &stx); + } + + var last: E = .NOENT; + var saw_acces = false; + for (c.candidates) |cand| { + const rc = linux.execve(cand.ptr, c.argv.ptr, c.envp.ptr); + last = linux.errno(rc); + switch (last) { + .NOENT, .NOTDIR, .LOOP, .NAMETOOLONG => continue, + .ACCES => { + saw_acces = true; + continue; + }, + else => break, + } + } + var buf: [512]u8 = undefined; + // "Not found" covers every candidate that could not even be resolved + // (a PATH element that is a file gives ENOTDIR, a symlink loop ELOOP); + // a candidate that existed but was not executable wins over those. + const not_found = switch (last) { + .NOENT, .NOTDIR, .LOOP, .NAMETOOLONG => true, + else => false, + }; + if (not_found and saw_acces) last = .ACCES; + const status: u8 = if (not_found and !saw_acces) 127 else 126; + const text = std.fmt.bufPrint(&buf, "exec {s}", .{c.name}) catch "exec"; + childFail(c, status, text, last, true); +} + +fn writeProcFile(c: *const ChildArgs, path: [*:0]const u8, data: []const u8, ignore_missing: bool) void { + const rc = linux.open(path, .{ .ACCMODE = .WRONLY, .CLOEXEC = true }, 0); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .NOENT => if (ignore_missing) return else childFail(c, setup_failure_status, std.mem.span(path), .NOENT, true), + else => |e| childFail(c, setup_failure_status, std.mem.span(path), e, true), + } + const fd: i32 = @intCast(rc); + const w = linux.write(fd, data.ptr, data.len); + const we = linux.errno(w); + _ = linux.close(fd); + if (we != .SUCCESS) childFail(c, setup_failure_status, std.mem.span(path), we, true); + if (w != data.len) childFail(c, setup_failure_status, std.mem.span(path), .IO, true); +} + +/// Write `<status byte><step>[: E<errno>]` to the status socket and exit. +fn childFail(c: *const ChildArgs, status: u8, step: []const u8, e: E, with_errno: bool) noreturn { + var buf: [600]u8 = undefined; + buf[0] = status; + const rest = if (with_errno) + std.fmt.bufPrint(buf[1..], "{s}: E{t}", .{ step, e }) catch buf[1..1] + else + std.fmt.bufPrint(buf[1..], "{s}", .{step}) catch buf[1..1]; + const msg = buf[0 .. 1 + rest.len]; + var off: usize = 0; + while (off < msg.len) { + const rc = linux.write(c.status_sock, msg[off..].ptr, msg.len - off); + if (linux.errno(rc) == .INTR) continue; + if (linux.errno(rc) != .SUCCESS) break; + off += rc; + } + linux.exit_group(status); +} + +// --------------------------------------------------------------------------- +// Signals +// --------------------------------------------------------------------------- + +var child_pid_ptr: ?*i32 = null; +var chld_pipe_w: i32 = -1; +var reaped = std.atomic.Value(bool).init(false); +var reaped_status = std.atomic.Value(u32).init(0); +/// A second child (the `--spawn` server) that the SIGCHLD handler reaps so +/// it does not linger as a zombie when it dies mid-session. Its exit does +/// not stop the serve loop. 0 = none. +var server_pid = std.atomic.Value(i32).init(0); + +/// Register the `--spawn` server for reaping by the SIGCHLD handler. +pub fn watchServer(pid: i32) void { + server_pid.store(pid, .seq_cst); +} + +/// Seconds the serve loop gets to come back after the child died before +/// the watchdog ends the process anyway. +pub const exit_grace_seconds: isize = 3; + +/// The watched child is already dead but the serve loop has not come back +/// (it is stuck in a 9P request the server never answers): a terminal +/// signal, or the watchdog armed by `onChld`, then ends 9ns with the +/// child's status instead of hanging. Nothing is lost: the mount is torn +/// down when the process exits. +fn bailIfChildGone() void { + if (!reaped.load(.acquire)) return; + const srv = server_pid.load(.seq_cst); + if (srv > 0) _ = linux.kill(srv, .TERM); + linux.exit_group(decodeStatus(reaped_status.load(.acquire))); +} + +fn armWatchdog() void { + // setitimer takes an itimerval; std declares it with itimerspec, which + // has the same layout on 64-bit targets (the sub-second field is 0). + const t = linux.itimerspec{ + .it_interval = .{ .sec = 0, .nsec = 0 }, + .it_value = .{ .sec = exit_grace_seconds, .nsec = 0 }, + }; + _ = linux.setitimer(@intFromEnum(linux.ITIMER.REAL), &t, null); +} + +fn onAlarm(_: linux.SIG) callconv(.c) void { + bailIfChildGone(); +} + +fn onForward(sig: linux.SIG) callconv(.c) void { + const p = child_pid_ptr orelse return; + const pid = @atomicLoad(i32, p, .seq_cst); + if (pid > 0) _ = linux.kill(pid, sig); + bailIfChildGone(); +} + +/// SIGINT/SIGQUIT: the child owns the tty and gets them itself; we only +/// react when the child is already gone (see `bailIfChildGone`). +fn onTerminal(_: linux.SIG) callconv(.c) void { + bailIfChildGone(); +} + +/// Only the watched child counts: reap it here (WNOHANG), remember its +/// status, forget its pid (so a later SIGTERM cannot hit a recycled pid) +/// and poke the self-pipe. The `--spawn` server is reaped too but does not +/// interrupt `bridge.serve`; SIGCHLD from anything else is ignored. +fn onChld(_: linux.SIG) callconv(.c) void { + const srv = server_pid.load(.seq_cst); + if (srv > 0) { + var sst: u32 = 0; + const src = linux.waitpid(srv, &sst, linux.W.NOHANG); + if (linux.errno(src) == .SUCCESS and src != 0) server_pid.store(0, .seq_cst); + } + const p = child_pid_ptr orelse return; + const pid = @atomicLoad(i32, p, .seq_cst); + if (pid <= 0) return; + var st: u32 = 0; + const rc = linux.waitpid(pid, &st, linux.W.NOHANG); + if (linux.errno(rc) != .SUCCESS or rc == 0) return; + reaped_status.store(st, .release); + reaped.store(true, .release); + @atomicStore(i32, p, 0, .seq_cst); + const b = [_]u8{'c'}; + _ = linux.write(chld_pipe_w, &b, 1); + armWatchdog(); +} + +/// SIGPIPE ignored; SIGINT/SIGQUIT effectively ignored (the child owns the +/// tty) unless the child is already dead; SIGTERM/SIGHUP forwarded to +/// `*child_pid`; SIGCHLD for `*child_pid` reaps it, writes a byte to a +/// nonblocking self-pipe whose read end is returned (use it as `stop_fd`) +/// and arms a watchdog (`exit_grace_seconds`, SIGALRM) that ends the +/// process with the child's status should the serve loop stay blocked. +/// `*child_pid` is filled in by `spawn`. +pub fn installSignals(child_pid: *i32) !i32 { + child_pid_ptr = child_pid; + var fds: [2]i32 = undefined; + switch (linux.errno(linux.pipe2(&fds, .{ .CLOEXEC = true, .NONBLOCK = true }))) { + .SUCCESS => {}, + else => |e| { + std.debug.print("9ns: pipe2: E{t}\n", .{e}); + return error.SystemResources; + }, + } + chld_pipe_w = fds[1]; + + const ign = linux.Sigaction{ .handler = .{ .handler = linux.SIG.IGN }, .mask = linux.sigemptyset(), .flags = 0 }; + const term = linux.Sigaction{ .handler = .{ .handler = &onTerminal }, .mask = linux.sigemptyset(), .flags = linux.SA.RESTART }; + const fwd = linux.Sigaction{ .handler = .{ .handler = &onForward }, .mask = linux.sigemptyset(), .flags = linux.SA.RESTART }; + const chld = linux.Sigaction{ .handler = .{ .handler = &onChld }, .mask = linux.sigemptyset(), .flags = linux.SA.RESTART | linux.SA.NOCLDSTOP }; + const alrm = linux.Sigaction{ .handler = .{ .handler = &onAlarm }, .mask = linux.sigemptyset(), .flags = linux.SA.RESTART }; + std.posix.sigaction(.INT, &term, null); + std.posix.sigaction(.QUIT, &term, null); + std.posix.sigaction(.ALRM, &alrm, null); + std.posix.sigaction(.PIPE, &ign, null); + std.posix.sigaction(.TERM, &fwd, null); + std.posix.sigaction(.HUP, &fwd, null); + std.posix.sigaction(.CHLD, &chld, null); + return fds[0]; +} + +/// Restore default dispositions in the child before exec (ignored signals +/// would otherwise survive execve). +fn resetSignals() void { + const dfl = linux.Sigaction{ .handler = .{ .handler = linux.SIG.DFL }, .mask = linux.sigemptyset(), .flags = 0 }; + inline for (.{ linux.SIG.INT, linux.SIG.QUIT, linux.SIG.PIPE, linux.SIG.TERM, linux.SIG.HUP, linux.SIG.CHLD, linux.SIG.ALRM }) |sig| { + _ = linux.sigaction(sig, &dfl, null); + } +} + +// --------------------------------------------------------------------------- +// Waiting +// --------------------------------------------------------------------------- + +fn takeReaped() ?u32 { + if (!reaped.load(.acquire)) return null; + return reaped_status.load(.acquire); +} + +/// waitpid status → exit code (`128+sig` when killed by a signal). +pub fn decodeStatus(st: u32) u8 { + if (linux.W.IFEXITED(st)) return linux.W.EXITSTATUS(st); + if (linux.W.IFSIGNALED(st)) return 128 +% @as(u8, @truncate(@intFromEnum(linux.W.TERMSIG(st)))); + return 1; +} + +/// Block until `pid` exits (the SIGCHLD handler may have reaped it already). +pub fn waitChild(pid: i32) !u8 { + while (true) { + if (takeReaped()) |st| return decodeStatus(st); + var st: u32 = 0; + const rc = linux.waitpid(pid, &st, 0); + switch (linux.errno(rc)) { + .SUCCESS => return decodeStatus(st), + .INTR => continue, + .CHILD => { + if (takeReaped()) |s| return decodeStatus(s); + return error.NoChild; + }, + else => |e| { + std.debug.print("9ns: waitpid: E{t}\n", .{e}); + return error.Wait; + }, + } + } +} + +/// Non-blocking: the exit status of `pid` if it has exited, else null. +pub fn reapIfExited(pid: i32) ?u8 { + if (takeReaped()) |st| return decodeStatus(st); + var st: u32 = 0; + const rc = linux.waitpid(pid, &st, linux.W.NOHANG); + switch (linux.errno(rc)) { + .SUCCESS => return if (rc == 0) null else decodeStatus(st), + .CHILD => return if (takeReaped()) |s| decodeStatus(s) else null, + else => return null, + } +} + +/// Reap any child (used for the `--spawn` server at exit). Non-blocking. +pub fn reapAny(pid: i32) void { + var st: u32 = 0; + _ = linux.waitpid(pid, &st, linux.W.NOHANG); +} + +// --------------------------------------------------------------------------- +// Tests (no namespaces needed; `ensureMountpoint` is exercised by +// test/integration.sh through the 9ns binary) +// --------------------------------------------------------------------------- + +const testing = std.testing; + +test "normalizePath: absolute paths" { + const gpa = testing.allocator; + const cases = [_]struct { in: []const u8, out: []const u8 }{ + .{ .in = "/mnt/9p", .out = "/mnt/9p" }, + .{ .in = "/mnt/9p/", .out = "/mnt/9p" }, + .{ .in = "//mnt///9p//", .out = "/mnt/9p" }, + .{ .in = "/mnt/./9p/.", .out = "/mnt/9p" }, + .{ .in = "/mnt/x/../9p", .out = "/mnt/9p" }, + .{ .in = "/../mnt/9p", .out = "/mnt/9p" }, + .{ .in = "/a/b/c/../..", .out = "/a" }, + }; + for (cases) |c| { + const got = try normalizePath(gpa, "/cwd", c.in); + defer gpa.free(got); + try testing.expectEqualStrings(c.out, got); + try testing.expectEqual(@as(u8, 0), got[got.len]); + } +} + +test "normalizePath: relative paths use cwd" { + const gpa = testing.allocator; + const cases = [_]struct { cwd: []const u8, in: []const u8, out: []const u8 }{ + .{ .cwd = "/home/me", .in = "mnt", .out = "/home/me/mnt" }, + .{ .cwd = "/home/me", .in = "./mnt/", .out = "/home/me/mnt" }, + .{ .cwd = "/home/me", .in = "../mnt", .out = "/home/mnt" }, + .{ .cwd = "/home/me/", .in = ".", .out = "/home/me" }, + .{ .cwd = "/", .in = "x", .out = "/x" }, + }; + for (cases) |c| { + const got = try normalizePath(gpa, c.cwd, c.in); + defer gpa.free(got); + try testing.expectEqualStrings(c.out, got); + } +} + +test "normalizePath: rejects root and empty" { + const gpa = testing.allocator; + try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/cwd", "/")); + try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/cwd", "///")); + try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/cwd", "/mnt/..")); + try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/cwd", "")); + try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/", "..")); +} + +test "resolveMountpoint: relative resolves against the real cwd" { + const gpa = testing.allocator; + const got = try resolveMountpoint(gpa, "sub/dir"); + defer gpa.free(got); + try testing.expect(got[0] == '/'); + try testing.expect(std.mem.endsWith(u8, got, "/sub/dir")); +} + +test "getenv" { + const env = [_:null]?[*:0]const u8{ "PATH=/a:/b", "X=", "PATHX=no", "NINE_MOUNT=/m" }; + const envp: [*:null]const ?[*:0]const u8 = &env; + try testing.expectEqualStrings("/a:/b", getenv(envp, "PATH").?); + try testing.expectEqualStrings("", getenv(envp, "X").?); + try testing.expectEqualStrings("/m", getenv(envp, "NINE_MOUNT").?); + try testing.expect(getenv(envp, "NOPE") == null); + try testing.expect(getenv(envp, "PAT") == null); +} + +test "pathCandidates: PATH search" { + const gpa = testing.allocator; + const env = [_:null]?[*:0]const u8{ "PATH=/usr/local/bin::/usr/bin", "HOME=/h" }; + const cands = try pathCandidates(gpa, &env, "fish"); + defer { + for (cands) |c| gpa.free(c); + gpa.free(cands); + } + try testing.expectEqual(@as(usize, 3), cands.len); + try testing.expectEqualStrings("/usr/local/bin/fish", cands[0]); + try testing.expectEqualStrings("./fish", cands[1]); + try testing.expectEqualStrings("/usr/bin/fish", cands[2]); +} + +test "pathCandidates: slash means no search; default PATH" { + const gpa = testing.allocator; + const env = [_:null]?[*:0]const u8{"HOME=/h"}; + { + const cands = try pathCandidates(gpa, &env, "./bin/x"); + defer { + for (cands) |c| gpa.free(c); + gpa.free(cands); + } + try testing.expectEqual(@as(usize, 1), cands.len); + try testing.expectEqualStrings("./bin/x", cands[0]); + } + { + const cands = try pathCandidates(gpa, &env, "sh"); + defer { + for (cands) |c| gpa.free(c); + gpa.free(cands); + } + try testing.expectEqual(@as(usize, 3), cands.len); + try testing.expectEqualStrings("/usr/local/bin/sh", cands[0]); + try testing.expectEqualStrings("/bin/sh", cands[1]); + } + try testing.expectError(error.EmptyProgramName, pathCandidates(gpa, &env, "")); +} + +test "findInPath finds sh" { + const gpa = testing.allocator; + const env = [_:null]?[*:0]const u8{"PATH=/nonexistent:/bin:/usr/bin"}; + const p = try findInPath(gpa, &env, "sh"); + defer gpa.free(p); + try testing.expect(std.mem.endsWith(u8, p, "/sh")); + try testing.expectError(error.FileNotFound, findInPath(gpa, &env, "definitely-not-a-program-9ns")); +} + +test "buildEnvp replaces NINE_MOUNT" { + const gpa = testing.allocator; + const env = [_:null]?[*:0]const u8{ "A=1", "NINE_MOUNT=/old", "B=2" }; + const out = try buildEnvp(gpa, &env, "/mnt/9p"); + defer { + gpa.free(std.mem.span(out[out.len - 1].?)); + gpa.free(out); + } + try testing.expectEqual(@as(usize, 3), out.len); + try testing.expectEqualStrings("A=1", std.mem.span(out[0].?)); + try testing.expectEqualStrings("B=2", std.mem.span(out[1].?)); + try testing.expectEqualStrings("NINE_MOUNT=/mnt/9p", std.mem.span(out[2].?)); + try testing.expect(out[3] == null); + try testing.expectEqualStrings("/mnt/9p", getenv(out.ptr, "NINE_MOUNT").?); +} + +test "decodeStatus" { + try testing.expectEqual(@as(u8, 0), decodeStatus(0)); + try testing.expectEqual(@as(u8, 7), decodeStatus(7 << 8)); + try testing.expectEqual(@as(u8, 255), decodeStatus(255 << 8)); + try testing.expectEqual(@as(u8, 128 + 9), decodeStatus(9)); // SIGKILL + try testing.expectEqual(@as(u8, 128 + 15), decodeStatus(15)); // SIGTERM +} + +test "ensureMountpoint: existing directory is accepted, plain file rejected" { + const gpa = testing.allocator; + try ensureMountpoint(gpa, "/tmp"); + try testing.expectError(error.Mountpoint, ensureMountpoint(gpa, "/proc/self/status")); +} |
