From b05abcba3ea09ea106ad28364c6e40a3ec31b890 Mon Sep 17 00:00:00 2001 From: Gabriel Schneider Date: Sat, 19 Sep 2026 21:26:05 -0300 Subject: Add 9player and introspect as programs beside the library 9player/: FUSE mount CLI that mounts a 9P2000 tree into a fresh user+mount namespace and runs a program in it (no root, no libfuse, no libc). introspect/: the 9P debug/introspection library (freestanding core, value renderers, Linux probe with threads/stacks/memory/breakpoints/panics) and its demo server. Each has its own build fragment; the root build.zig wires them behind -D9player/-Dintrospect with namespaced steps (9player-itest, introspect-check-freestanding, programs-test, ...) and exports the introspect module for dependents. This is the layout for related programs. Co-Authored-By: Claude Fable 5.1 --- 9player/src/bridge.zig | 971 +++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 971 insertions(+) create mode 100644 9player/src/bridge.zig (limited to '9player/src/bridge.zig') diff --git a/9player/src/bridge.zig b/9player/src/bridge.zig new file mode 100644 index 0000000..387e184 --- /dev/null +++ b/9player/src/bridge.zig @@ -0,0 +1,971 @@ +//! FUSE ↔ 9P2000 translation: the request loop that turns kernel FUSE requests +//! into synchronous 9P calls on a `nine.Session` and sends the replies back. +//! +//! Everything here is single-threaded and one request at a time. State is three +//! tables: inodes (nodeid → fid/qid, deduplicated by qid.path), open handles +//! (fh → fid plus a cached directory listing), and the reverse qid map. +const std = @import("std"); +const cloud9 = @import("cloud9"); +const fuse = @import("fuse.zig"); +const nine = @import("nine.zig"); +const linux = std.os.linux; + +pub const Options = struct { + /// Reported owner of every file. + uid: u32, + gid: u32, + /// attr/entry cache validity (0 = none). + attr_timeout_ns: u64 = 1_000_000_000, + /// FOPEN_DIRECT_IO on every regular file. + direct_io: bool = true, + /// Trace every request, reply and 9P call to stderr. + debug: bool = false, +}; + +/// Largest single READ/WRITE payload we accept from the kernel. +pub const max_write: u32 = 1 << 20; +/// Upper bound on the raw bytes of one directory listing (about a million entries); +/// past it the listing fails with EIO instead of eating memory. +pub const max_dir_bytes: u64 = 64 << 20; +/// The kernel refuses dirents longer than this (FUSE_NAME_MAX) with EIO. +pub const max_name_len: usize = 1024; +/// Request buffer: `max_write` plus room for the header and the largest in-struct. +pub const request_buf_len: usize = max_write + 4096; + +const Inode = struct { + fid: u32, + qid: cloud9.Qid, + nlookup: u64, + /// nodeid of the directory this inode was looked up in (root: itself). Used for "..". + parent: u64, +}; + +pub const Entry = struct { name: []u8, ino: u64, dtype: u32 }; + +pub const DirList = struct { + entries: std.ArrayList(Entry) = .empty, + + pub fn deinit(d: *DirList, gpa: std.mem.Allocator) void { + for (d.entries.items) |e| gpa.free(e.name); + d.entries.deinit(gpa); + } +}; + +const Handle = struct { + fid: u32, + nodeid: u64, + dir: ?DirList, +}; + +/// Errors a request handler may surface. Policy failures are ordinary errors +/// that the dispatcher maps to an errno; `FuseIo` means the kernel side is broken. +const HandlerError = nine.Session.Error || error{ + BadRequest, + NoEntry, + BadHandle, + Exdev, + Perm, + NotSup, + /// A directory listing the server sent could not be parsed (EIO, not fatal). + BadDir, + FuseIo, +}; + +/// Runs until the FUSE fd reports ENODEV, a DESTROY arrives, or `stop_fd` +/// becomes readable (also while a 9P reply is outstanding). Returns +/// `error.Closed` if the 9P server went away. +pub fn serve(gpa: std.mem.Allocator, fuse_fd: i32, session: *nine.Session, root_fid: u32, stop_fd: i32, opts: Options) !void { + var effective = opts; + // With page caching on, a nonzero attr cache lets the kernel trust a stale + // (often zero) size and truncate reads: 9P sizes are authoritative and change + // under us. direct_io ignores the cached size, so the cache is safe only there. + if (!effective.direct_io) effective.attr_timeout_ns = 0; + var b: Bridge = .{ + .gpa = gpa, + .fuse_fd = fuse_fd, + .nine = session, + .opts = effective, + }; + defer b.deinit(); + + b.req_buf = try gpa.alignedAlloc(u8, .@"8", request_buf_len); + b.data_buf = try gpa.alloc(u8, max_write); + + // Abandon any pending 9P reply once the child is gone (stop_fd readable), + // including the initial root stat below: a silent server must not pin us. + session.stop_fd = stop_fd; + defer session.stop_fd = -1; + + // Node 1 is the root; its qid comes from a stat so lookups resolving back to + // it (e.g. via a walk) dedupe onto node 1. + var root_qid: cloud9.Qid = .{ .type = cloud9.qtdir, .version = 0, .path = 0 }; + if (b.stat(root_fid)) |st| { + root_qid = st.qid; + b.root_path = st.qid.path; + try b.by_qid.put(gpa, st.qid.path, fuse.root_id); + } else |e| switch (e) { + error.Nine => {}, + error.Stopped => return, + else => return error.Closed, + } + try b.inodes.put(gpa, fuse.root_id, .{ .fid = root_fid, .qid = root_qid, .nlookup = 1, .parent = fuse.root_id }); + + var pfds = [_]linux.pollfd{ + .{ .fd = fuse_fd, .events = linux.POLL.IN, .revents = 0 }, + .{ .fd = stop_fd, .events = linux.POLL.IN, .revents = 0 }, + }; + while (true) { + pfds[0].revents = 0; + pfds[1].revents = 0; + const rc = linux.poll(&pfds, pfds.len, -1); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .INTR, .AGAIN => continue, + else => return error.Io, + } + if (pfds[1].revents != 0) { + b.trace("stop_fd readable; leaving serve loop", .{}); + return; + } + if (pfds[0].revents == 0) continue; + const req = (fuse.readRequest(fuse_fd, b.req_buf) catch |e| switch (e) { + error.Protocol => return error.FuseProtocol, + else => return error.FuseIo, + }) orelse { + b.trace("fuse fd reports ENODEV; unmounted", .{}); + return; + }; + if (!try b.dispatch(req)) return; + } +} + +const Bridge = struct { + gpa: std.mem.Allocator, + fuse_fd: i32, + nine: *nine.Session, + opts: Options, + req_buf: []align(8) u8 = &.{}, + data_buf: []u8 = &.{}, + inodes: std.AutoHashMapUnmanaged(u64, Inode) = .empty, + by_qid: std.AutoHashMapUnmanaged(u64, u64) = .empty, + handles: std.AutoHashMapUnmanaged(u64, Handle) = .empty, + next_node: u64 = 2, + next_fh: u64 = 1, + /// qid.path of the root, reported as ino 1 wherever it shows up. + root_path: u64 = 0, + /// errno of the most recent Rerror that a handler did not swallow. Kept here + /// because `Session.rpc` clears its ename on every call, and error paths + /// clunk (an rpc) before the dispatcher maps the failure to an errno. + last_err: linux.E = .IO, + + fn deinit(b: *Bridge) void { + var it = b.handles.valueIterator(); + while (it.next()) |h| if (h.dir) |*d| d.deinit(b.gpa); + b.handles.deinit(b.gpa); + b.inodes.deinit(b.gpa); + b.by_qid.deinit(b.gpa); + if (b.req_buf.len != 0) b.gpa.free(b.req_buf); + if (b.data_buf.len != 0) b.gpa.free(b.data_buf); + } + + fn trace(b: *const Bridge, comptime fmt: []const u8, args: anytype) void { + if (b.opts.debug) std.debug.print("9player: " ++ fmt ++ "\n", args); + } + + // -- dispatch -------------------------------------------------------------------- + + /// Handles one request. Returns false when the loop should stop (DESTROY). + /// Fatal errors (dead 9P session, broken FUSE fd) propagate. + fn dispatch(b: *Bridge, req: fuse.Request) !bool { + const h = req.header; + const op = h.op(); + b.trace("<- {s} unique={d} nodeid={d} len={d} (fids={d} inodes={d} handles={d})", .{ opName(op), h.unique, h.nodeid, h.len, b.nine.fidsInUse(), b.inodes.count(), b.handles.count() }); + const wants_reply = switch (op) { + .forget, .batch_forget, .interrupt => false, + else => true, + }; + if (op == .destroy) { + b.reply(h.unique, &.{}) catch {}; + return false; + } + b.handle(req) catch |e| { + const code: linux.E = switch (e) { + error.Nine => b.last_err, + error.BadRequest => .INVAL, + error.NoEntry => .NOENT, + error.BadHandle => .BADF, + error.Exdev => .XDEV, + error.Perm => .PERM, + error.NotSup => .NOSYS, + error.OutOfMemory => .NOMEM, + error.TooLarge => .NAMETOOLONG, + error.BadDir => .IO, + error.Closed, error.Protocol, error.Io, error.Stopped => .IO, + error.FuseIo => return error.FuseIo, + }; + if (wants_reply) try b.replyError(h.unique, code); + switch (e) { + error.Closed, error.Protocol, error.Io => return error.Closed, + error.Stopped => return false, // the child is gone; the mount is being torn down + else => {}, + } + }; + return true; + } + + fn handle(b: *Bridge, req: fuse.Request) HandlerError!void { + const u = req.header.unique; + switch (req.header.op()) { + .init => { + const in = try body(fuse.InitIn, req); + const out = fuse.initReply(in, max_write); + try b.reply(u, &.{std.mem.asBytes(&out)}); + }, + .lookup => { + const name = try nameAfter(void, req); + const entry = try b.lookupEntry(req.header.nodeid, name); + try b.reply(u, &.{std.mem.asBytes(&entry)}); + }, + .forget => { + const in = try body(fuse.ForgetIn, req); + try b.forget(req.header.nodeid, in.nlookup); + }, + .batch_forget => { + const in = try body(fuse.BatchForgetIn, req); + const rest = req.body[@sizeOf(fuse.BatchForgetIn)..]; + const count: usize = in.count; + if (rest.len < count * @sizeOf(fuse.ForgetOne)) return error.BadRequest; + for (0..count) |i| { + const one = std.mem.bytesToValue(fuse.ForgetOne, rest[i * @sizeOf(fuse.ForgetOne) ..][0..@sizeOf(fuse.ForgetOne)]); + try b.forget(one.nodeid, one.nlookup); + } + }, + .getattr => { + const ino = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; + const st = try b.stat(ino.fid); + const out = b.attrOut(st, b.inoOf(req.header.nodeid, ino.qid)); + try b.reply(u, &.{std.mem.asBytes(&out)}); + }, + .setattr => try b.setattr(req), + .open => try b.openFile(req, false), + .opendir => try b.openFile(req, true), + .read => { + const in = try body(fuse.ReadIn, req); + const h = b.handles.get(in.fh) orelse return error.BadHandle; + const want: usize = @min(in.size, max_write); + const n = try b.read(h.fid, in.offset, b.data_buf[0..want]); + try b.reply(u, &.{b.data_buf[0..n]}); + }, + .write => { + const in = try body(fuse.WriteIn, req); + const h = b.handles.get(in.fh) orelse return error.BadHandle; + const rest = req.body[@sizeOf(fuse.WriteIn)..]; + if (rest.len < in.size) return error.BadRequest; + const n = try b.write(h.fid, in.offset, rest[0..in.size]); + const out = fuse.WriteOut{ .size = @intCast(n) }; + try b.reply(u, &.{std.mem.asBytes(&out)}); + }, + .readdir => try b.readdir(req), + .release, .releasedir => { + const in = try body(fuse.ReleaseIn, req); + const kv = b.handles.fetchRemove(in.fh) orelse return error.BadHandle; + var h = kv.value; + if (h.dir) |*d| d.deinit(b.gpa); + try b.clunk(h.fid); + try b.reply(u, &.{}); + }, + .flush, .fsync, .fsyncdir => try b.reply(u, &.{}), + .create => try b.create(req), + .mkdir => { + const in = try body(fuse.MkdirIn, req); + const name = try nameAfter(fuse.MkdirIn, req); + const parent = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; + const fid = try b.clone(parent.fid); + _ = b.create9(fid, name, cloud9.dmdir | (in.mode & 0o777), cloud9.oread) catch |e| { + b.clunkQuiet(fid); + return e; + }; + try b.clunk(fid); + const entry = try b.lookupEntry(req.header.nodeid, name); + try b.reply(u, &.{std.mem.asBytes(&entry)}); + }, + .unlink, .rmdir => { + const name = try nameAfter(void, req); + const parent = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; + const tmp = try b.walkName(parent.fid, name); + try b.remove(tmp); + try b.reply(u, &.{}); + }, + .rename => { + const in = try body(fuse.RenameIn, req); + const old = try nameAfter(fuse.RenameIn, req); + const new = try secondName(req, old, @sizeOf(fuse.RenameIn)); + try b.rename(req.header.nodeid, in.newdir, old, new, 0); + try b.reply(u, &.{}); + }, + .rename2 => { + const in = try body(fuse.Rename2In, req); + const old = try nameAfter(fuse.Rename2In, req); + const new = try secondName(req, old, @sizeOf(fuse.Rename2In)); + try b.rename(req.header.nodeid, in.newdir, old, new, in.flags); + try b.reply(u, &.{}); + }, + .statfs => { + const out = fuse.StatfsOut{ .st = .{ .bsize = 4096, .namelen = 255, .frsize = 4096 } }; + try b.reply(u, &.{std.mem.asBytes(&out)}); + }, + .interrupt => {}, + .destroy => unreachable, // handled in dispatch + .access => return error.NotSup, + else => return error.NotSup, + } + } + + // -- handlers ---------------------------------------------------------------------- + + /// walk(parent → new fid, [name]) + stat, deduplicated by qid.path. Bumps nlookup. + fn lookupEntry(b: *Bridge, parent_id: u64, name: []const u8) HandlerError!fuse.EntryOut { + const parent = b.inodes.get(parent_id) orelse return error.NoEntry; + const newfid = try b.walkName(parent.fid, name); + const st = b.stat(newfid) catch |e| { + b.clunkQuiet(newfid); + return e; + }; + const qid = st.qid; + var nodeid: u64 = undefined; + if (b.by_qid.get(qid.path)) |existing| { + // A directory and a file sharing a qid.path (a server bug) must not + // share a node: the kernel would mark the inode bad, and for the + // root that is fatal for the whole mount. + const merge = if (b.inodes.getPtr(existing)) |ino| (ino.qid.type & cloud9.qtdir) == (qid.type & cloud9.qtdir) else false; + if (merge) { + const ino = b.inodes.getPtr(existing).?; + ino.nlookup += 1; + ino.qid = qid; + nodeid = existing; + if (existing == fuse.root_id) { + b.clunkQuiet(newfid); + } else { + // Keep the fresh fid (it is bound to the current file at this + // name) and retire the older one. + const stale = ino.fid; + ino.fid = newfid; + b.clunkQuiet(stale); + } + } else { + // Stale reverse entry, or a type clash: bind a fresh node to it. + nodeid = try b.newInode(newfid, qid, parent_id); + } + } else { + nodeid = try b.newInode(newfid, qid, parent_id); + } + var out = fuse.EntryOut{ + .nodeid = nodeid, + .generation = 0, + .attr = b.attrFrom(st, b.inoOf(nodeid, qid)), + }; + out.entry_valid = b.opts.attr_timeout_ns / 1_000_000_000; + out.entry_valid_nsec = @intCast(b.opts.attr_timeout_ns % 1_000_000_000); + out.attr_valid = out.entry_valid; + out.attr_valid_nsec = out.entry_valid_nsec; + return out; + } + + fn newInode(b: *Bridge, fid: u32, qid: cloud9.Qid, parent: u64) HandlerError!u64 { + const nodeid = b.next_node; + b.inodes.put(b.gpa, nodeid, .{ .fid = fid, .qid = qid, .nlookup = 1, .parent = parent }) catch |e| { + b.clunkQuiet(fid); + return e; + }; + b.by_qid.put(b.gpa, qid.path, nodeid) catch |e| { + _ = b.inodes.remove(nodeid); + b.clunkQuiet(fid); + return e; + }; + b.next_node += 1; + return nodeid; + } + + fn forget(b: *Bridge, nodeid: u64, n: u64) HandlerError!void { + if (nodeid == fuse.root_id) return; + const ino = b.inodes.getPtr(nodeid) orelse return; + if (ino.nlookup > n) { + ino.nlookup -= n; + return; + } + const fid = ino.fid; + const path = ino.qid.path; + _ = b.inodes.remove(nodeid); + if (b.by_qid.get(path)) |mapped| { + if (mapped == nodeid) _ = b.by_qid.remove(path); + } + b.clunk(fid) catch |e| switch (e) { + error.Nine => {}, + else => return e, + }; + } + + fn setattr(b: *Bridge, req: fuse.Request) HandlerError!void { + const in = try body(fuse.SetattrIn, req); + const ino = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; + const old = try b.stat(ino.fid); + const old_mode = old.mode; + + var st = nine.dontcare; + var changed = false; + if (in.valid & fuse.FATTR_UID != 0 and in.uid != b.opts.uid) return error.Perm; + if (in.valid & fuse.FATTR_GID != 0 and in.gid != b.opts.gid) return error.Perm; + if (in.valid & fuse.FATTR_SIZE != 0) { + st.length = in.size; + changed = true; + } + if (in.valid & fuse.FATTR_MODE != 0) { + st.mode = (old_mode & ~@as(u32, 0o777)) | (in.mode & 0o777); + changed = true; + } + if (in.valid & fuse.FATTR_MTIME_NOW != 0) { + st.mtime = nowSeconds(); + changed = true; + } else if (in.valid & fuse.FATTR_MTIME != 0) { + st.mtime = @truncate(in.mtime); + changed = true; + } + if (changed) try b.wstat(ino.fid, st); + const fresh = try b.stat(ino.fid); + const out = b.attrOut(fresh, b.inoOf(req.header.nodeid, ino.qid)); + try b.reply(req.header.unique, &.{std.mem.asBytes(&out)}); + } + + fn openFile(b: *Bridge, req: fuse.Request, is_dir: bool) HandlerError!void { + const in = try body(fuse.OpenIn, req); + const ino = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; + const mode: u8 = if (is_dir) cloud9.oread else openMode(in.flags); + const fid = try b.clone(ino.fid); + _ = b.open9(fid, mode) catch |e| { + b.clunkQuiet(fid); + return e; + }; + const fh = try b.newHandle(fid, req.header.nodeid); + const out = fuse.OpenOut{ + .fh = fh, + .open_flags = if (!is_dir and b.opts.direct_io) fuse.FOPEN_DIRECT_IO else 0, + }; + try b.reply(req.header.unique, &.{std.mem.asBytes(&out)}); + } + + fn newHandle(b: *Bridge, fid: u32, nodeid: u64) HandlerError!u64 { + const fh = b.next_fh; + b.handles.put(b.gpa, fh, .{ .fid = fid, .nodeid = nodeid, .dir = null }) catch |e| { + b.clunkQuiet(fid); + return e; + }; + b.next_fh += 1; + return fh; + } + + fn create(b: *Bridge, req: fuse.Request) HandlerError!void { + const in = try body(fuse.CreateIn, req); + const name = try nameAfter(fuse.CreateIn, req); + const parent = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; + // The created fid becomes the open file. + const fid = try b.clone(parent.fid); + _ = b.create9(fid, name, in.mode & 0o777, openMode(in.flags)) catch |e| { + b.clunkQuiet(fid); + return e; + }; + const entry = b.lookupEntry(req.header.nodeid, name) catch |e| { + b.clunkQuiet(fid); + return e; + }; + const fh = try b.newHandle(fid, entry.nodeid); + const oo = fuse.OpenOut{ + .fh = fh, + .open_flags = if (b.opts.direct_io) fuse.FOPEN_DIRECT_IO else 0, + }; + try b.reply(req.header.unique, &.{ std.mem.asBytes(&entry), std.mem.asBytes(&oo) }); + } + + fn rename(b: *Bridge, parent_id: u64, newdir: u64, old: []const u8, new: []const u8, flags: u32) HandlerError!void { + if (newdir != parent_id) return error.Exdev; + const rf: linux.RENAME = @bitCast(flags); + if (rf.EXCHANGE or rf.WHITEOUT) return error.BadRequest; + const parent = b.inodes.get(parent_id) orelse return error.NoEntry; + const tmp = try b.walkName(parent.fid, old); + defer b.clunkQuiet(tmp); + var st = nine.dontcare; + st.name = new; + b.wstat(tmp, st) catch |e| { + // 9P2000 rename never replaces an existing name; POSIX rename does. + if (e != error.Nine or rf.NOREPLACE or b.nine.errno() != .EXIST) return e; + try b.renameOver(parent.fid, tmp, new); + }; + } + + /// Replace `new` with the file behind `src`. An (empty) directory target is + /// removed first: it holds no data and the VFS already ruled out mismatched + /// types. A file target is parked under a temporary name so that a failing + /// second rename can put it back instead of having destroyed it. + fn renameOver(b: *Bridge, parent_fid: u32, src: u32, new: []const u8) HandlerError!void { + const victim = try b.walkName(parent_fid, new); + const vst = b.stat(victim) catch |e| { + b.clunkQuiet(victim); + return e; + }; + var st = nine.dontcare; + st.name = new; + if (vst.mode & cloud9.dmdir != 0) { + b.trace(" rename target is a directory; removing it and retrying", .{}); + try b.remove(victim); + return b.wstat(src, st); + } + var park_buf: [48]u8 = undefined; + const park = std.fmt.bufPrint(&park_buf, ".9player-rename-{x}", .{randomU64()}) catch unreachable; + b.trace(" rename target exists; parking it as {s} and retrying", .{park}); + var pst = nine.dontcare; + pst.name = park; + b.wstat(victim, pst) catch |e| { + b.clunkQuiet(victim); + return e; + }; + b.wstat(src, st) catch |e| { + b.trace(" rename still failed; restoring the target", .{}); + const saved = b.last_err; + b.wstat(victim, st) catch {}; + b.last_err = saved; + b.clunkQuiet(victim); + return e; + }; + b.remove(victim) catch b.trace(" could not remove the parked target {s}", .{park}); + } + + fn readdir(b: *Bridge, req: fuse.Request) HandlerError!void { + const in = try body(fuse.ReadIn, req); + const h = b.handles.getPtr(in.fh) orelse return error.BadHandle; + if (in.offset == 0 or h.dir == null) { + if (h.dir) |*d| d.deinit(b.gpa); + h.dir = null; + h.dir = try b.loadDir(h.fid, h.nodeid); + } + const dir = &h.dir.?; + const size: usize = @min(in.size, max_write); + const used = packDirents(dir.entries.items, in.offset, b.data_buf[0..size]); + try b.reply(req.header.unique, &.{b.data_buf[0..used]}); + } + + /// Reads the whole directory and builds its listing, "." and ".." first. + fn loadDir(b: *Bridge, fid: u32, nodeid: u64) HandlerError!DirList { + var list: DirList = .{}; + errdefer list.deinit(b.gpa); + const self_ino = b.inoOfNode(nodeid); + const parent_ino = if (b.inodes.get(nodeid)) |ino| b.inoOfNode(ino.parent) else self_ino; + try list.entries.append(b.gpa, .{ .name = try b.gpa.dupe(u8, "."), .ino = self_ino, .dtype = fuse.DT_DIR }); + try list.entries.append(b.gpa, .{ .name = try b.gpa.dupe(u8, ".."), .ino = parent_ino, .dtype = fuse.DT_DIR }); + + var offset: u64 = 0; + while (true) { + // A server that ignores the offset would otherwise feed us forever. + if (offset >= max_dir_bytes) return error.BadDir; + const n = try b.read(fid, offset, b.data_buf); + if (n == 0) break; + try parseDirRecords(b.gpa, b.data_buf[0..n], &list); + offset += n; + } + // Entries carrying the root's own qid.path get the root's ino (1), as GETATTR would report it. + for (list.entries.items[2..]) |*e| if (e.ino == b.root_path) { + e.ino = fuse.root_id; + }; + return list; + } + + // -- 9P wrappers (tracing) ----------------------------------------------------------- + + fn stat(b: *Bridge, fid: u32) nine.Session.Error!cloud9.Stat { + const st = b.nine.stat(fid) catch |e| return b.nineErr("stat", fid, e); + b.trace(" 9p stat fid={d} -> name={s} mode={o} len={d} qid={x}", .{ fid, st.name, st.mode, st.length, st.qid.path }); + return st; + } + + fn walkName(b: *Bridge, fid: u32, name: []const u8) nine.Session.Error!u32 { + const newfid = b.nine.allocFid(); + _ = b.nine.walk(fid, newfid, &.{name}) catch |e| { + b.nine.freeFid(newfid); + return b.nineErr("walk", fid, e); + }; + b.trace(" 9p walk fid={d} newfid={d} name={s} -> ok", .{ fid, newfid, name }); + return newfid; + } + + fn clone(b: *Bridge, fid: u32) nine.Session.Error!u32 { + const newfid = b.nine.clone(fid) catch |e| return b.nineErr("clone", fid, e); + b.trace(" 9p walk fid={d} newfid={d} (clone) -> ok", .{ fid, newfid }); + return newfid; + } + + fn open9(b: *Bridge, fid: u32, mode: u8) nine.Session.Error!nine.Session.Open { + const o = b.nine.open(fid, mode) catch |e| return b.nineErr("open", fid, e); + b.trace(" 9p open fid={d} mode={d} -> iounit={d}", .{ fid, mode, o.iounit }); + return o; + } + + fn create9(b: *Bridge, fid: u32, name: []const u8, perm: u32, mode: u8) nine.Session.Error!nine.Session.Open { + const o = b.nine.create(fid, name, perm, mode) catch |e| return b.nineErr("create", fid, e); + b.trace(" 9p create fid={d} name={s} perm={o} mode={d} -> iounit={d}", .{ fid, name, perm, mode, o.iounit }); + return o; + } + + fn read(b: *Bridge, fid: u32, offset: u64, buf: []u8) nine.Session.Error!usize { + const n = b.nine.read(fid, offset, buf) catch |e| return b.nineErr("read", fid, e); + b.trace(" 9p read fid={d} offset={d} count={d} -> {d}", .{ fid, offset, buf.len, n }); + return n; + } + + fn write(b: *Bridge, fid: u32, offset: u64, data: []const u8) nine.Session.Error!usize { + const n = b.nine.write(fid, offset, data) catch |e| return b.nineErr("write", fid, e); + b.trace(" 9p write fid={d} offset={d} count={d} -> {d}", .{ fid, offset, data.len, n }); + return n; + } + + fn wstat(b: *Bridge, fid: u32, st: cloud9.Stat) nine.Session.Error!void { + b.nine.wstat(fid, st) catch |e| return b.nineErr("wstat", fid, e); + b.trace(" 9p wstat fid={d} name={s} mode={x} len={x} mtime={x} -> ok", .{ fid, st.name, st.mode, st.length, st.mtime }); + } + + fn clunk(b: *Bridge, fid: u32) nine.Session.Error!void { + b.nine.clunk(fid) catch |e| return b.nineErr("clunk", fid, e); + b.trace(" 9p clunk fid={d} -> ok", .{fid}); + } + + /// Best-effort clunk during error unwinding; a dead session surfaces on the + /// next call. Does not disturb the errno of the failure being unwound. + fn clunkQuiet(b: *Bridge, fid: u32) void { + const saved = b.last_err; + defer b.last_err = saved; + b.clunk(fid) catch {}; + } + + fn remove(b: *Bridge, fid: u32) nine.Session.Error!void { + b.nine.remove(fid) catch |e| return b.nineErr("remove", fid, e); + b.trace(" 9p remove fid={d} -> ok", .{fid}); + } + + fn nineErr(b: *Bridge, what: []const u8, fid: u32, e: nine.Session.Error) nine.Session.Error { + if (e == error.Nine) { + b.last_err = b.nine.errno(); + b.trace(" 9p {s} fid={d} -> Rerror \"{s}\" ({s})", .{ what, fid, b.nine.ename[0..b.nine.ename_len], @tagName(b.nine.errno()) }); + } else { + b.trace(" 9p {s} fid={d} -> {s}", .{ what, fid, @errorName(e) }); + } + return e; + } + + // -- FUSE wrappers (tracing) -------------------------------------------------------- + + fn reply(b: *Bridge, unique: u64, payloads: []const []const u8) error{FuseIo}!void { + var total: usize = 0; + for (payloads) |p| total += p.len; + b.trace("-> unique={d} ok ({d} bytes)", .{ unique, total }); + fuse.reply(b.fuse_fd, unique, payloads) catch return error.FuseIo; + } + + fn replyError(b: *Bridge, unique: u64, code: linux.E) error{FuseIo}!void { + b.trace("-> unique={d} error E{s}", .{ unique, @tagName(code) }); + fuse.replyError(b.fuse_fd, unique, code) catch return error.FuseIo; + } + + // -- attrs --------------------------------------------------------------------------- + + fn inoOf(b: *const Bridge, nodeid: u64, qid: cloud9.Qid) u64 { + return if (nodeid == fuse.root_id or qid.path == b.root_path) fuse.root_id else qid.path; + } + + fn inoOfNode(b: *const Bridge, nodeid: u64) u64 { + if (nodeid == fuse.root_id) return fuse.root_id; + const ino = b.inodes.get(nodeid) orelse return nodeid; + return b.inoOf(nodeid, ino.qid); + } + + fn attrFrom(b: *const Bridge, st: cloud9.Stat, ino: u64) fuse.Attr { + return attrFromStat(st, ino, b.opts.uid, b.opts.gid); + } + + fn attrOut(b: *const Bridge, st: cloud9.Stat, ino: u64) fuse.AttrOut { + return .{ + .attr_valid = b.opts.attr_timeout_ns / 1_000_000_000, + .attr_valid_nsec = @intCast(b.opts.attr_timeout_ns % 1_000_000_000), + .attr = b.attrFrom(st, ino), + }; + } +}; + +// -- pure helpers (unit-tested) ------------------------------------------------------------ + +/// Attr from a 9P Stat: DMDIR → S_IFDIR else S_IFREG, low 9 permission bits kept. +pub fn attrFromStat(st: cloud9.Stat, ino: u64, uid: u32, gid: u32) fuse.Attr { + const ftype: u32 = if (st.mode & cloud9.dmdir != 0) fuse.S_IFDIR else fuse.S_IFREG; + return .{ + .ino = ino, + // The kernel marks an inode bad when size > LLONG_MAX; clamp hostile lengths. + .size = @min(st.length, std.math.maxInt(i64)), + // Saturating: a hostile length of 2^64-1 must not overflow. + .blocks = st.length / 512 + @intFromBool(st.length % 512 != 0), + .atime = st.atime, + .mtime = st.mtime, + .ctime = st.mtime, + .mode = ftype | (st.mode & 0o777), + .nlink = 1, + .uid = uid, + .gid = gid, + .blksize = 4096, + }; +} + +/// Kernel open(2) flags → 9P open mode. O_APPEND has no 9P equivalent and is ignored. +pub fn openMode(flags: u32) u8 { + const o: linux.O = @bitCast(flags); + var mode: u8 = switch (o.ACCMODE) { + .RDONLY => cloud9.oread, + .WRONLY => cloud9.owrite, + .RDWR => cloud9.ordwr, + }; + if (o.TRUNC) mode |= cloud9.otrunc; + return mode; +} + +/// Parses consecutive 9P directory records (2-byte size + Stat) and appends entries. +pub fn parseDirRecords(gpa: std.mem.Allocator, bytes: []const u8, list: *DirList) error{ OutOfMemory, BadDir }!void { + var pos: usize = 0; + while (pos < bytes.len) { + if (bytes.len - pos < 2) return error.BadDir; + const size: usize = std.mem.readInt(u16, bytes[pos..][0..2], .little); + if (bytes.len - pos < 2 + size) return error.BadDir; + const st = cloud9.Stat.decode(bytes[pos..][0 .. 2 + size]) catch return error.BadDir; + pos += 2 + size; + // The kernel rejects a whole READDIR reply (EIO) over one bad name, and + // "." and ".." are synthesised by loadDir: drop such records instead. + if (!validDirentName(st.name)) continue; + const name = try gpa.dupe(u8, st.name); + errdefer gpa.free(name); + try list.entries.append(gpa, .{ + .name = name, + .ino = st.qid.path, + .dtype = if (st.mode & cloud9.dmdir != 0) fuse.DT_DIR else fuse.DT_REG, + }); + } +} + +/// A name the kernel will accept in a dirent and that does not duplicate the synthetic "." / "..". +pub fn validDirentName(name: []const u8) bool { + if (name.len == 0 or name.len > max_name_len) return false; + if (std.mem.indexOfAny(u8, name, "/\x00") != null) return false; + if (std.mem.eql(u8, name, ".") or std.mem.eql(u8, name, "..")) return false; + return true; +} + +/// Packs dirents from `entries[offset..]` into `buf`; each record's `off` is its index + 1. +/// Returns the number of bytes used. +pub fn packDirents(entries: []const Entry, offset: u64, buf: []u8) usize { + var used: usize = 0; + var i: usize = @intCast(@min(offset, entries.len)); + while (i < entries.len) : (i += 1) { + const e = entries[i]; + if (!fuse.addDirent(buf, &used, e.ino, @as(u64, i) + 1, e.dtype, e.name)) break; + } + return used; +} + +fn randomU64() u64 { + var bytes: [8]u8 = undefined; + if (linux.errno(linux.getrandom(&bytes, bytes.len, 0)) == .SUCCESS) return std.mem.readInt(u64, &bytes, .little); + var ts: linux.timespec = undefined; + _ = linux.clock_gettime(.MONOTONIC, &ts); + return @as(u64, @bitCast(ts.nsec)) ^ (@as(u64, @bitCast(ts.sec)) << 32); +} + +fn nowSeconds() u32 { + var ts: linux.timespec = undefined; + if (linux.errno(linux.clock_gettime(.REALTIME, &ts)) != .SUCCESS) return 0; + return @intCast(@as(u64, @intCast(ts.sec)) & 0xFFFF_FFFF); +} + +fn opName(op: fuse.Opcode) []const u8 { + return switch (op) { + _ => "unknown", + else => @tagName(op), + }; +} + +// Thin adapters so fuse.zig's parse errors become HandlerError.BadRequest. +fn body(comptime T: type, req: fuse.Request) error{BadRequest}!*const T { + return fuse.body(T, req) catch error.BadRequest; +} + +fn nameAfter(comptime T: type, req: fuse.Request) error{BadRequest}![]const u8 { + return fuse.nameAfter(T, req) catch error.BadRequest; +} + +fn secondName(req: fuse.Request, first: []const u8, offset: usize) error{BadRequest}![]const u8 { + return fuse.secondName(req, first, offset) catch error.BadRequest; +} + +// -- tests ------------------------------------------------------------------------------ + +const testing = std.testing; + +test { + // Force semantic analysis of `serve` and the whole dispatch path, which no + // unit test can exercise without a FUSE mount. + testing.refAllDecls(@This()); +} + +fn testStat(name: []const u8, mode: u32, length: u64, path: u64) cloud9.Stat { + return .{ + .type = 0, + .dev = 0, + .qid = .{ .type = if (mode & cloud9.dmdir != 0) cloud9.qtdir else 0, .version = 0, .path = path }, + .mode = mode, + .atime = 100, + .mtime = 200, + .length = length, + .name = name, + .uid = "u", + .gid = "g", + .muid = "u", + }; +} + +test "attr mapping: DMDIR → S_IFDIR|perm, length → size/blocks" { + const d = attrFromStat(testStat("d", cloud9.dmdir | 0o755, 0, 9), 9, 1000, 1001); + try testing.expectEqual(fuse.S_IFDIR | 0o755, d.mode); + try testing.expectEqual(@as(u64, 9), d.ino); + try testing.expectEqual(@as(u64, 0), d.size); + try testing.expectEqual(@as(u64, 0), d.blocks); + try testing.expectEqual(@as(u32, 1000), d.uid); + try testing.expectEqual(@as(u32, 1001), d.gid); + try testing.expectEqual(@as(u32, 1), d.nlink); + + const f = attrFromStat(testStat("f", 0o640 | cloud9.dmappend, 1025, 4), 4, 0, 0); + try testing.expectEqual(fuse.S_IFREG | 0o640, f.mode); // dmappend bit not leaked + try testing.expectEqual(@as(u64, 1025), f.size); + try testing.expectEqual(@as(u64, 3), f.blocks); + try testing.expectEqual(@as(u32, 4096), f.blksize); + try testing.expectEqual(@as(u64, 100), f.atime); + try testing.expectEqual(@as(u64, 200), f.mtime); + try testing.expectEqual(@as(u64, 200), f.ctime); + + try testing.expectEqual(@as(u64, 1), attrFromStat(testStat("f", 0o600, 512, 4), 4, 0, 0).blocks); + try testing.expectEqual(@as(u64, 2), attrFromStat(testStat("f", 0o600, 513, 4), 4, 0, 0).blocks); +} + +test "open flag → 9P mode mapping" { + const rdonly: u32 = @bitCast(linux.O{ .ACCMODE = .RDONLY }); + const wronly: u32 = @bitCast(linux.O{ .ACCMODE = .WRONLY }); + const rdwr: u32 = @bitCast(linux.O{ .ACCMODE = .RDWR }); + const trunc: u32 = @bitCast(linux.O{ .TRUNC = true }); + const append: u32 = @bitCast(linux.O{ .APPEND = true }); + const creat: u32 = @bitCast(linux.O{ .CREAT = true }); + try testing.expectEqual(cloud9.oread, openMode(rdonly)); + try testing.expectEqual(cloud9.owrite, openMode(wronly)); + try testing.expectEqual(cloud9.ordwr, openMode(rdwr)); + try testing.expectEqual(cloud9.owrite | cloud9.otrunc, openMode(wronly | trunc)); + try testing.expectEqual(cloud9.ordwr | cloud9.otrunc, openMode(rdwr | trunc | creat)); + try testing.expectEqual(cloud9.owrite, openMode(wronly | append)); // O_APPEND ignored +} + +test "dirlist parsing from two hand-encoded Stat records" { + var buf: [512]u8 = undefined; + const a = try cloud9.Stat.encode(testStat("alpha", 0o644, 10, 0x11), &buf); + const bb = try cloud9.Stat.encode(testStat("beta", cloud9.dmdir | 0o755, 0, 0x22), buf[a.len..]); + const bytes = buf[0 .. a.len + bb.len]; + // Sanity: the record is prefixed by its own 2-byte size. + try testing.expectEqual(a.len - 2, std.mem.readInt(u16, bytes[0..2], .little)); + + var list: DirList = .{}; + defer list.deinit(testing.allocator); + try parseDirRecords(testing.allocator, bytes, &list); + try testing.expectEqual(@as(usize, 2), list.entries.items.len); + try testing.expectEqualStrings("alpha", list.entries.items[0].name); + try testing.expectEqual(@as(u64, 0x11), list.entries.items[0].ino); + try testing.expectEqual(fuse.DT_REG, list.entries.items[0].dtype); + try testing.expectEqualStrings("beta", list.entries.items[1].name); + try testing.expectEqual(@as(u64, 0x22), list.entries.items[1].ino); + try testing.expectEqual(fuse.DT_DIR, list.entries.items[1].dtype); + + // Truncated input is a protocol error and leaves earlier entries intact. + try testing.expectError(error.BadDir, parseDirRecords(testing.allocator, bytes[0 .. bytes.len - 1], &list)); + try testing.expectEqual(@as(usize, 3), list.entries.items.len); +} + +test "readdir packing and offset resumption" { + const names = [_][]const u8{ ".", "..", "one", "two", "three" }; + var entries: [names.len]Entry = undefined; + for (&entries, names, 0..) |*e, n, i| e.* = .{ .name = @constCast(n), .ino = 100 + i, .dtype = if (i < 2) fuse.DT_DIR else fuse.DT_REG }; + + // Everything fits: five records, off = index + 1. + var big: [1024]u8 = undefined; + const used = packDirents(&entries, 0, &big); + var pos: usize = 0; + var idx: usize = 0; + while (pos < used) : (idx += 1) { + const d = std.mem.bytesToValue(fuse.Dirent, big[pos..][0..@sizeOf(fuse.Dirent)]); + try testing.expectEqual(@as(u64, 100 + idx), d.ino); + try testing.expectEqual(@as(u64, idx + 1), d.off); + try testing.expectEqualStrings(names[idx], big[pos + @sizeOf(fuse.Dirent) ..][0..d.namelen]); + pos += (@sizeOf(fuse.Dirent) + d.namelen + 7) & ~@as(usize, 7); + } + try testing.expectEqual(names.len, idx); + + // A buffer that fits exactly two records ("." = 32, ".." = 32) stops there… + var small: [64]u8 = undefined; + const first_used = packDirents(&entries, 0, &small); + try testing.expectEqual(@as(usize, 64), first_used); + const last = std.mem.bytesToValue(fuse.Dirent, small[32..][0..@sizeOf(fuse.Dirent)]); + try testing.expectEqual(@as(u64, 2), last.off); + // …and resuming at the last `off` yields "one" next. + const second_used = packDirents(&entries, last.off, &small); + const next = std.mem.bytesToValue(fuse.Dirent, small[0..@sizeOf(fuse.Dirent)]); + try testing.expectEqualStrings("one", small[@sizeOf(fuse.Dirent)..][0..next.namelen]); + try testing.expectEqual(@as(u64, 3), next.off); + try testing.expect(second_used > 0); + + // Past the end: nothing (EOF for the kernel). + try testing.expectEqual(@as(usize, 0), packDirents(&entries, names.len, &big)); + try testing.expectEqual(@as(usize, 0), packDirents(&entries, 1000, &big)); +} + +test "attr mapping saturates hostile lengths instead of overflowing" { + const a = attrFromStat(testStat("f", 0o600, std.math.maxInt(u64), 4), 4, 0, 0); + try testing.expectEqual(@as(u64, std.math.maxInt(i64)), a.size); + try testing.expectEqual(@as(u64, std.math.maxInt(u64) / 512 + 1), a.blocks); + const b = attrFromStat(testStat("f", 0o600, 1024, 4), 4, 0, 0); + try testing.expectEqual(@as(u64, 2), b.blocks); + try testing.expectEqual(@as(u64, 1024), b.size); +} + +test "dirent names the kernel would reject are dropped from listings" { + try testing.expect(validDirentName("a")); + try testing.expect(validDirentName("x" ** 1024)); + try testing.expect(!validDirentName("")); + try testing.expect(!validDirentName("a/b")); + try testing.expect(!validDirentName("a\x00b")); + try testing.expect(!validDirentName(".")); + try testing.expect(!validDirentName("..")); + try testing.expect(!validDirentName("x" ** 1025)); + + var buf: [4096]u8 = undefined; + var n: usize = 0; + for ([_][]const u8{ ".", "..", "", "a/b", "keep", "x" ** 1025, "also" }) |name| { + n += (try cloud9.Stat.encode(testStat(name, 0o644, 1, 0x30), buf[n..])).len; + } + var list: DirList = .{}; + defer list.deinit(testing.allocator); + try parseDirRecords(testing.allocator, buf[0..n], &list); + try testing.expectEqual(@as(usize, 2), list.entries.items.len); + try testing.expectEqualStrings("keep", list.entries.items[0].name); + try testing.expectEqualStrings("also", list.entries.items[1].name); +} + +test "DirList frees its names" { + var list: DirList = .{}; + try list.entries.append(testing.allocator, .{ .name = try testing.allocator.dupe(u8, "x"), .ino = 1, .dtype = fuse.DT_REG }); + list.deinit(testing.allocator); +} -- cgit v1.3