//! FUSE ↔ 9P2000 translation: the request loop that turns kernel FUSE requests //! into synchronous 9P calls on a `nine.Session` and sends the replies back. //! //! Everything here is single-threaded and one request at a time. State is three //! tables: inodes (nodeid → fid/qid, deduplicated by qid.path), open handles //! (fh → fid plus a cached directory listing), and the reverse qid map. const std = @import("std"); const cloud9 = @import("cloud9"); const fuse = @import("fuse.zig"); const nine = @import("nine.zig"); const linux = std.os.linux; pub const Options = struct { /// Reported owner of every file. uid: u32, gid: u32, /// attr/entry cache validity (0 = none). attr_timeout_ns: u64 = 1_000_000_000, /// FOPEN_DIRECT_IO on every regular file. direct_io: bool = true, /// Trace every request, reply and 9P call to stderr. debug: bool = false, }; /// Largest single READ/WRITE payload we accept from the kernel. pub const max_write: u32 = 1 << 20; /// Upper bound on the raw bytes of one directory listing (about a million entries); /// past it the listing fails with EIO instead of eating memory. pub const max_dir_bytes: u64 = 64 << 20; /// The kernel refuses dirents longer than this (FUSE_NAME_MAX) with EIO. pub const max_name_len: usize = 1024; /// Request buffer: `max_write` plus room for the header and the largest in-struct. pub const request_buf_len: usize = max_write + 4096; const Inode = struct { fid: u32, qid: cloud9.Qid, nlookup: u64, /// nodeid of the directory this inode was looked up in (root: itself). Used for "..". parent: u64, }; pub const Entry = struct { name: []u8, ino: u64, dtype: u32 }; pub const DirList = struct { entries: std.ArrayList(Entry) = .empty, pub fn deinit(d: *DirList, gpa: std.mem.Allocator) void { for (d.entries.items) |e| gpa.free(e.name); d.entries.deinit(gpa); } }; const Handle = struct { fid: u32, nodeid: u64, dir: ?DirList, }; /// Errors a request handler may surface. Policy failures are ordinary errors /// that the dispatcher maps to an errno; `FuseIo` means the kernel side is broken. const HandlerError = nine.Session.Error || error{ BadRequest, NoEntry, BadHandle, Exdev, Perm, NotSup, /// A directory listing the server sent could not be parsed (EIO, not fatal). BadDir, FuseIo, }; /// Runs until the FUSE fd reports ENODEV, a DESTROY arrives, or `stop_fd` /// becomes readable (also while a 9P reply is outstanding). Returns /// `error.Closed` if the 9P server went away. pub fn serve(gpa: std.mem.Allocator, fuse_fd: i32, session: *nine.Session, root_fid: u32, stop_fd: i32, opts: Options) !void { var effective = opts; // With page caching on, a nonzero attr cache lets the kernel trust a stale // (often zero) size and truncate reads: 9P sizes are authoritative and change // under us. direct_io ignores the cached size, so the cache is safe only there. if (!effective.direct_io) effective.attr_timeout_ns = 0; var b: Bridge = .{ .gpa = gpa, .fuse_fd = fuse_fd, .nine = session, .opts = effective, }; defer b.deinit(); b.req_buf = try gpa.alignedAlloc(u8, .@"8", request_buf_len); b.data_buf = try gpa.alloc(u8, max_write); // Abandon any pending 9P reply once the child is gone (stop_fd readable), // including the initial root stat below: a silent server must not pin us. session.stop_fd = stop_fd; defer session.stop_fd = -1; // Node 1 is the root; its qid comes from a stat so lookups resolving back to // it (e.g. via a walk) dedupe onto node 1. var root_qid: cloud9.Qid = .{ .type = cloud9.qtdir, .version = 0, .path = 0 }; if (b.stat(root_fid)) |st| { root_qid = st.qid; b.root_path = st.qid.path; try b.by_qid.put(gpa, st.qid.path, fuse.root_id); } else |e| switch (e) { error.Nine => {}, error.Stopped => return, else => return error.Closed, } try b.inodes.put(gpa, fuse.root_id, .{ .fid = root_fid, .qid = root_qid, .nlookup = 1, .parent = fuse.root_id }); var pfds = [_]linux.pollfd{ .{ .fd = fuse_fd, .events = linux.POLL.IN, .revents = 0 }, .{ .fd = stop_fd, .events = linux.POLL.IN, .revents = 0 }, }; while (true) { pfds[0].revents = 0; pfds[1].revents = 0; const rc = linux.poll(&pfds, pfds.len, -1); switch (linux.errno(rc)) { .SUCCESS => {}, .INTR, .AGAIN => continue, else => return error.Io, } if (pfds[1].revents != 0) { b.trace("stop_fd readable; leaving serve loop", .{}); return; } if (pfds[0].revents == 0) continue; const req = (fuse.readRequest(fuse_fd, b.req_buf) catch |e| switch (e) { error.Protocol => return error.FuseProtocol, else => return error.FuseIo, }) orelse { b.trace("fuse fd reports ENODEV; unmounted", .{}); return; }; if (!try b.dispatch(req)) return; } } const Bridge = struct { gpa: std.mem.Allocator, fuse_fd: i32, nine: *nine.Session, opts: Options, req_buf: []align(8) u8 = &.{}, data_buf: []u8 = &.{}, inodes: std.AutoHashMapUnmanaged(u64, Inode) = .empty, by_qid: std.AutoHashMapUnmanaged(u64, u64) = .empty, handles: std.AutoHashMapUnmanaged(u64, Handle) = .empty, next_node: u64 = 2, next_fh: u64 = 1, /// qid.path of the root, reported as ino 1 wherever it shows up. root_path: u64 = 0, /// errno of the most recent Rerror that a handler did not swallow. Kept here /// because `Session.rpc` clears its ename on every call, and error paths /// clunk (an rpc) before the dispatcher maps the failure to an errno. last_err: linux.E = .IO, fn deinit(b: *Bridge) void { var it = b.handles.valueIterator(); while (it.next()) |h| if (h.dir) |*d| d.deinit(b.gpa); b.handles.deinit(b.gpa); b.inodes.deinit(b.gpa); b.by_qid.deinit(b.gpa); if (b.req_buf.len != 0) b.gpa.free(b.req_buf); if (b.data_buf.len != 0) b.gpa.free(b.data_buf); } fn trace(b: *const Bridge, comptime fmt: []const u8, args: anytype) void { if (b.opts.debug) std.debug.print("9player: " ++ fmt ++ "\n", args); } // -- dispatch -------------------------------------------------------------------- /// Handles one request. Returns false when the loop should stop (DESTROY). /// Fatal errors (dead 9P session, broken FUSE fd) propagate. fn dispatch(b: *Bridge, req: fuse.Request) !bool { const h = req.header; const op = h.op(); b.trace("<- {s} unique={d} nodeid={d} len={d} (fids={d} inodes={d} handles={d})", .{ opName(op), h.unique, h.nodeid, h.len, b.nine.fidsInUse(), b.inodes.count(), b.handles.count() }); const wants_reply = switch (op) { .forget, .batch_forget, .interrupt => false, else => true, }; if (op == .destroy) { b.reply(h.unique, &.{}) catch {}; return false; } b.handle(req) catch |e| { const code: linux.E = switch (e) { error.Nine => b.last_err, error.BadRequest => .INVAL, error.NoEntry => .NOENT, error.BadHandle => .BADF, error.Exdev => .XDEV, error.Perm => .PERM, error.NotSup => .NOSYS, error.OutOfMemory => .NOMEM, error.TooLarge => .NAMETOOLONG, error.BadDir => .IO, error.Closed, error.Protocol, error.Io, error.Stopped => .IO, error.FuseIo => return error.FuseIo, }; if (wants_reply) try b.replyError(h.unique, code); switch (e) { error.Closed, error.Protocol, error.Io => return error.Closed, error.Stopped => return false, // the child is gone; the mount is being torn down else => {}, } }; return true; } fn handle(b: *Bridge, req: fuse.Request) HandlerError!void { const u = req.header.unique; switch (req.header.op()) { .init => { const in = try body(fuse.InitIn, req); const out = fuse.initReply(in, max_write); try b.reply(u, &.{std.mem.asBytes(&out)}); }, .lookup => { const name = try nameAfter(void, req); const entry = try b.lookupEntry(req.header.nodeid, name); try b.reply(u, &.{std.mem.asBytes(&entry)}); }, .forget => { const in = try body(fuse.ForgetIn, req); try b.forget(req.header.nodeid, in.nlookup); }, .batch_forget => { const in = try body(fuse.BatchForgetIn, req); const rest = req.body[@sizeOf(fuse.BatchForgetIn)..]; const count: usize = in.count; if (rest.len < count * @sizeOf(fuse.ForgetOne)) return error.BadRequest; for (0..count) |i| { const one = std.mem.bytesToValue(fuse.ForgetOne, rest[i * @sizeOf(fuse.ForgetOne) ..][0..@sizeOf(fuse.ForgetOne)]); try b.forget(one.nodeid, one.nlookup); } }, .getattr => { const ino = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; const st = try b.stat(ino.fid); const out = b.attrOut(st, b.inoOf(req.header.nodeid, ino.qid)); try b.reply(u, &.{std.mem.asBytes(&out)}); }, .setattr => try b.setattr(req), .open => try b.openFile(req, false), .opendir => try b.openFile(req, true), .read => { const in = try body(fuse.ReadIn, req); const h = b.handles.get(in.fh) orelse return error.BadHandle; const want: usize = @min(in.size, max_write); const n = try b.read(h.fid, in.offset, b.data_buf[0..want]); try b.reply(u, &.{b.data_buf[0..n]}); }, .write => { const in = try body(fuse.WriteIn, req); const h = b.handles.get(in.fh) orelse return error.BadHandle; const rest = req.body[@sizeOf(fuse.WriteIn)..]; if (rest.len < in.size) return error.BadRequest; const n = try b.write(h.fid, in.offset, rest[0..in.size]); const out = fuse.WriteOut{ .size = @intCast(n) }; try b.reply(u, &.{std.mem.asBytes(&out)}); }, .readdir => try b.readdir(req), .release, .releasedir => { const in = try body(fuse.ReleaseIn, req); const kv = b.handles.fetchRemove(in.fh) orelse return error.BadHandle; var h = kv.value; if (h.dir) |*d| d.deinit(b.gpa); try b.clunk(h.fid); try b.reply(u, &.{}); }, .flush, .fsync, .fsyncdir => try b.reply(u, &.{}), .create => try b.create(req), .mkdir => { const in = try body(fuse.MkdirIn, req); const name = try nameAfter(fuse.MkdirIn, req); const parent = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; const fid = try b.clone(parent.fid); _ = b.create9(fid, name, cloud9.dmdir | (in.mode & 0o777), cloud9.oread) catch |e| { b.clunkQuiet(fid); return e; }; try b.clunk(fid); const entry = try b.lookupEntry(req.header.nodeid, name); try b.reply(u, &.{std.mem.asBytes(&entry)}); }, .unlink, .rmdir => { const name = try nameAfter(void, req); const parent = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; const tmp = try b.walkName(parent.fid, name); try b.remove(tmp); try b.reply(u, &.{}); }, .rename => { const in = try body(fuse.RenameIn, req); const old = try nameAfter(fuse.RenameIn, req); const new = try secondName(req, old, @sizeOf(fuse.RenameIn)); try b.rename(req.header.nodeid, in.newdir, old, new, 0); try b.reply(u, &.{}); }, .rename2 => { const in = try body(fuse.Rename2In, req); const old = try nameAfter(fuse.Rename2In, req); const new = try secondName(req, old, @sizeOf(fuse.Rename2In)); try b.rename(req.header.nodeid, in.newdir, old, new, in.flags); try b.reply(u, &.{}); }, .statfs => { const out = fuse.StatfsOut{ .st = .{ .bsize = 4096, .namelen = 255, .frsize = 4096 } }; try b.reply(u, &.{std.mem.asBytes(&out)}); }, .interrupt => {}, .destroy => unreachable, // handled in dispatch .access => return error.NotSup, else => return error.NotSup, } } // -- handlers ---------------------------------------------------------------------- /// walk(parent → new fid, [name]) + stat, deduplicated by qid.path. Bumps nlookup. fn lookupEntry(b: *Bridge, parent_id: u64, name: []const u8) HandlerError!fuse.EntryOut { const parent = b.inodes.get(parent_id) orelse return error.NoEntry; const newfid = try b.walkName(parent.fid, name); const st = b.stat(newfid) catch |e| { b.clunkQuiet(newfid); return e; }; const qid = st.qid; var nodeid: u64 = undefined; if (b.by_qid.get(qid.path)) |existing| { // A directory and a file sharing a qid.path (a server bug) must not // share a node: the kernel would mark the inode bad, and for the // root that is fatal for the whole mount. const merge = if (b.inodes.getPtr(existing)) |ino| (ino.qid.type & cloud9.qtdir) == (qid.type & cloud9.qtdir) else false; if (merge) { const ino = b.inodes.getPtr(existing).?; ino.nlookup += 1; ino.qid = qid; nodeid = existing; if (existing == fuse.root_id) { b.clunkQuiet(newfid); } else { // Keep the fresh fid (it is bound to the current file at this // name) and retire the older one. const stale = ino.fid; ino.fid = newfid; b.clunkQuiet(stale); } } else { // Stale reverse entry, or a type clash: bind a fresh node to it. nodeid = try b.newInode(newfid, qid, parent_id); } } else { nodeid = try b.newInode(newfid, qid, parent_id); } var out = fuse.EntryOut{ .nodeid = nodeid, .generation = 0, .attr = b.attrFrom(st, b.inoOf(nodeid, qid)), }; out.entry_valid = b.opts.attr_timeout_ns / 1_000_000_000; out.entry_valid_nsec = @intCast(b.opts.attr_timeout_ns % 1_000_000_000); out.attr_valid = out.entry_valid; out.attr_valid_nsec = out.entry_valid_nsec; return out; } fn newInode(b: *Bridge, fid: u32, qid: cloud9.Qid, parent: u64) HandlerError!u64 { const nodeid = b.next_node; b.inodes.put(b.gpa, nodeid, .{ .fid = fid, .qid = qid, .nlookup = 1, .parent = parent }) catch |e| { b.clunkQuiet(fid); return e; }; b.by_qid.put(b.gpa, qid.path, nodeid) catch |e| { _ = b.inodes.remove(nodeid); b.clunkQuiet(fid); return e; }; b.next_node += 1; return nodeid; } fn forget(b: *Bridge, nodeid: u64, n: u64) HandlerError!void { if (nodeid == fuse.root_id) return; const ino = b.inodes.getPtr(nodeid) orelse return; if (ino.nlookup > n) { ino.nlookup -= n; return; } const fid = ino.fid; const path = ino.qid.path; _ = b.inodes.remove(nodeid); if (b.by_qid.get(path)) |mapped| { if (mapped == nodeid) _ = b.by_qid.remove(path); } b.clunk(fid) catch |e| switch (e) { error.Nine => {}, else => return e, }; } fn setattr(b: *Bridge, req: fuse.Request) HandlerError!void { const in = try body(fuse.SetattrIn, req); const ino = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; const old = try b.stat(ino.fid); const old_mode = old.mode; var st = nine.dontcare; var changed = false; if (in.valid & fuse.FATTR_UID != 0 and in.uid != b.opts.uid) return error.Perm; if (in.valid & fuse.FATTR_GID != 0 and in.gid != b.opts.gid) return error.Perm; if (in.valid & fuse.FATTR_SIZE != 0) { st.length = in.size; changed = true; } if (in.valid & fuse.FATTR_MODE != 0) { st.mode = (old_mode & ~@as(u32, 0o777)) | (in.mode & 0o777); changed = true; } if (in.valid & fuse.FATTR_MTIME_NOW != 0) { st.mtime = nowSeconds(); changed = true; } else if (in.valid & fuse.FATTR_MTIME != 0) { st.mtime = @truncate(in.mtime); changed = true; } if (changed) try b.wstat(ino.fid, st); const fresh = try b.stat(ino.fid); const out = b.attrOut(fresh, b.inoOf(req.header.nodeid, ino.qid)); try b.reply(req.header.unique, &.{std.mem.asBytes(&out)}); } fn openFile(b: *Bridge, req: fuse.Request, is_dir: bool) HandlerError!void { const in = try body(fuse.OpenIn, req); const ino = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; const mode: u8 = if (is_dir) cloud9.oread else openMode(in.flags); const fid = try b.clone(ino.fid); _ = b.open9(fid, mode) catch |e| { b.clunkQuiet(fid); return e; }; const fh = try b.newHandle(fid, req.header.nodeid); const out = fuse.OpenOut{ .fh = fh, .open_flags = if (!is_dir and b.opts.direct_io) fuse.FOPEN_DIRECT_IO else 0, }; try b.reply(req.header.unique, &.{std.mem.asBytes(&out)}); } fn newHandle(b: *Bridge, fid: u32, nodeid: u64) HandlerError!u64 { const fh = b.next_fh; b.handles.put(b.gpa, fh, .{ .fid = fid, .nodeid = nodeid, .dir = null }) catch |e| { b.clunkQuiet(fid); return e; }; b.next_fh += 1; return fh; } fn create(b: *Bridge, req: fuse.Request) HandlerError!void { const in = try body(fuse.CreateIn, req); const name = try nameAfter(fuse.CreateIn, req); const parent = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; // The created fid becomes the open file. const fid = try b.clone(parent.fid); _ = b.create9(fid, name, in.mode & 0o777, openMode(in.flags)) catch |e| { b.clunkQuiet(fid); return e; }; const entry = b.lookupEntry(req.header.nodeid, name) catch |e| { b.clunkQuiet(fid); return e; }; const fh = try b.newHandle(fid, entry.nodeid); const oo = fuse.OpenOut{ .fh = fh, .open_flags = if (b.opts.direct_io) fuse.FOPEN_DIRECT_IO else 0, }; try b.reply(req.header.unique, &.{ std.mem.asBytes(&entry), std.mem.asBytes(&oo) }); } fn rename(b: *Bridge, parent_id: u64, newdir: u64, old: []const u8, new: []const u8, flags: u32) HandlerError!void { if (newdir != parent_id) return error.Exdev; const rf: linux.RENAME = @bitCast(flags); if (rf.EXCHANGE or rf.WHITEOUT) return error.BadRequest; const parent = b.inodes.get(parent_id) orelse return error.NoEntry; const tmp = try b.walkName(parent.fid, old); defer b.clunkQuiet(tmp); var st = nine.dontcare; st.name = new; b.wstat(tmp, st) catch |e| { // 9P2000 rename never replaces an existing name; POSIX rename does. if (e != error.Nine or rf.NOREPLACE or b.nine.errno() != .EXIST) return e; try b.renameOver(parent.fid, tmp, new); }; } /// Replace `new` with the file behind `src`. An (empty) directory target is /// removed first: it holds no data and the VFS already ruled out mismatched /// types. A file target is parked under a temporary name so that a failing /// second rename can put it back instead of having destroyed it. fn renameOver(b: *Bridge, parent_fid: u32, src: u32, new: []const u8) HandlerError!void { const victim = try b.walkName(parent_fid, new); const vst = b.stat(victim) catch |e| { b.clunkQuiet(victim); return e; }; var st = nine.dontcare; st.name = new; if (vst.mode & cloud9.dmdir != 0) { b.trace(" rename target is a directory; removing it and retrying", .{}); try b.remove(victim); return b.wstat(src, st); } var park_buf: [48]u8 = undefined; const park = std.fmt.bufPrint(&park_buf, ".9player-rename-{x}", .{randomU64()}) catch unreachable; b.trace(" rename target exists; parking it as {s} and retrying", .{park}); var pst = nine.dontcare; pst.name = park; b.wstat(victim, pst) catch |e| { b.clunkQuiet(victim); return e; }; b.wstat(src, st) catch |e| { b.trace(" rename still failed; restoring the target", .{}); const saved = b.last_err; b.wstat(victim, st) catch {}; b.last_err = saved; b.clunkQuiet(victim); return e; }; b.remove(victim) catch b.trace(" could not remove the parked target {s}", .{park}); } fn readdir(b: *Bridge, req: fuse.Request) HandlerError!void { const in = try body(fuse.ReadIn, req); const h = b.handles.getPtr(in.fh) orelse return error.BadHandle; if (in.offset == 0 or h.dir == null) { if (h.dir) |*d| d.deinit(b.gpa); h.dir = null; h.dir = try b.loadDir(h.fid, h.nodeid); } const dir = &h.dir.?; const size: usize = @min(in.size, max_write); const used = packDirents(dir.entries.items, in.offset, b.data_buf[0..size]); try b.reply(req.header.unique, &.{b.data_buf[0..used]}); } /// Reads the whole directory and builds its listing, "." and ".." first. fn loadDir(b: *Bridge, fid: u32, nodeid: u64) HandlerError!DirList { var list: DirList = .{}; errdefer list.deinit(b.gpa); const self_ino = b.inoOfNode(nodeid); const parent_ino = if (b.inodes.get(nodeid)) |ino| b.inoOfNode(ino.parent) else self_ino; try list.entries.append(b.gpa, .{ .name = try b.gpa.dupe(u8, "."), .ino = self_ino, .dtype = fuse.DT_DIR }); try list.entries.append(b.gpa, .{ .name = try b.gpa.dupe(u8, ".."), .ino = parent_ino, .dtype = fuse.DT_DIR }); var offset: u64 = 0; while (true) { // A server that ignores the offset would otherwise feed us forever. if (offset >= max_dir_bytes) return error.BadDir; const n = try b.read(fid, offset, b.data_buf); if (n == 0) break; try parseDirRecords(b.gpa, b.data_buf[0..n], &list); offset += n; } // Entries carrying the root's own qid.path get the root's ino (1), as GETATTR would report it. for (list.entries.items[2..]) |*e| if (e.ino == b.root_path) { e.ino = fuse.root_id; }; return list; } // -- 9P wrappers (tracing) ----------------------------------------------------------- fn stat(b: *Bridge, fid: u32) nine.Session.Error!cloud9.Stat { const st = b.nine.stat(fid) catch |e| return b.nineErr("stat", fid, e); b.trace(" 9p stat fid={d} -> name={s} mode={o} len={d} qid={x}", .{ fid, st.name, st.mode, st.length, st.qid.path }); return st; } fn walkName(b: *Bridge, fid: u32, name: []const u8) nine.Session.Error!u32 { const newfid = b.nine.allocFid(); _ = b.nine.walk(fid, newfid, &.{name}) catch |e| { b.nine.freeFid(newfid); return b.nineErr("walk", fid, e); }; b.trace(" 9p walk fid={d} newfid={d} name={s} -> ok", .{ fid, newfid, name }); return newfid; } fn clone(b: *Bridge, fid: u32) nine.Session.Error!u32 { const newfid = b.nine.clone(fid) catch |e| return b.nineErr("clone", fid, e); b.trace(" 9p walk fid={d} newfid={d} (clone) -> ok", .{ fid, newfid }); return newfid; } fn open9(b: *Bridge, fid: u32, mode: u8) nine.Session.Error!nine.Session.Open { const o = b.nine.open(fid, mode) catch |e| return b.nineErr("open", fid, e); b.trace(" 9p open fid={d} mode={d} -> iounit={d}", .{ fid, mode, o.iounit }); return o; } fn create9(b: *Bridge, fid: u32, name: []const u8, perm: u32, mode: u8) nine.Session.Error!nine.Session.Open { const o = b.nine.create(fid, name, perm, mode) catch |e| return b.nineErr("create", fid, e); b.trace(" 9p create fid={d} name={s} perm={o} mode={d} -> iounit={d}", .{ fid, name, perm, mode, o.iounit }); return o; } fn read(b: *Bridge, fid: u32, offset: u64, buf: []u8) nine.Session.Error!usize { const n = b.nine.read(fid, offset, buf) catch |e| return b.nineErr("read", fid, e); b.trace(" 9p read fid={d} offset={d} count={d} -> {d}", .{ fid, offset, buf.len, n }); return n; } fn write(b: *Bridge, fid: u32, offset: u64, data: []const u8) nine.Session.Error!usize { const n = b.nine.write(fid, offset, data) catch |e| return b.nineErr("write", fid, e); b.trace(" 9p write fid={d} offset={d} count={d} -> {d}", .{ fid, offset, data.len, n }); return n; } fn wstat(b: *Bridge, fid: u32, st: cloud9.Stat) nine.Session.Error!void { b.nine.wstat(fid, st) catch |e| return b.nineErr("wstat", fid, e); b.trace(" 9p wstat fid={d} name={s} mode={x} len={x} mtime={x} -> ok", .{ fid, st.name, st.mode, st.length, st.mtime }); } fn clunk(b: *Bridge, fid: u32) nine.Session.Error!void { b.nine.clunk(fid) catch |e| return b.nineErr("clunk", fid, e); b.trace(" 9p clunk fid={d} -> ok", .{fid}); } /// Best-effort clunk during error unwinding; a dead session surfaces on the /// next call. Does not disturb the errno of the failure being unwound. fn clunkQuiet(b: *Bridge, fid: u32) void { const saved = b.last_err; defer b.last_err = saved; b.clunk(fid) catch {}; } fn remove(b: *Bridge, fid: u32) nine.Session.Error!void { b.nine.remove(fid) catch |e| return b.nineErr("remove", fid, e); b.trace(" 9p remove fid={d} -> ok", .{fid}); } fn nineErr(b: *Bridge, what: []const u8, fid: u32, e: nine.Session.Error) nine.Session.Error { if (e == error.Nine) { b.last_err = b.nine.errno(); b.trace(" 9p {s} fid={d} -> Rerror \"{s}\" ({s})", .{ what, fid, b.nine.ename[0..b.nine.ename_len], @tagName(b.nine.errno()) }); } else { b.trace(" 9p {s} fid={d} -> {s}", .{ what, fid, @errorName(e) }); } return e; } // -- FUSE wrappers (tracing) -------------------------------------------------------- fn reply(b: *Bridge, unique: u64, payloads: []const []const u8) error{FuseIo}!void { var total: usize = 0; for (payloads) |p| total += p.len; b.trace("-> unique={d} ok ({d} bytes)", .{ unique, total }); fuse.reply(b.fuse_fd, unique, payloads) catch return error.FuseIo; } fn replyError(b: *Bridge, unique: u64, code: linux.E) error{FuseIo}!void { b.trace("-> unique={d} error E{s}", .{ unique, @tagName(code) }); fuse.replyError(b.fuse_fd, unique, code) catch return error.FuseIo; } // -- attrs --------------------------------------------------------------------------- fn inoOf(b: *const Bridge, nodeid: u64, qid: cloud9.Qid) u64 { return if (nodeid == fuse.root_id or qid.path == b.root_path) fuse.root_id else qid.path; } fn inoOfNode(b: *const Bridge, nodeid: u64) u64 { if (nodeid == fuse.root_id) return fuse.root_id; const ino = b.inodes.get(nodeid) orelse return nodeid; return b.inoOf(nodeid, ino.qid); } fn attrFrom(b: *const Bridge, st: cloud9.Stat, ino: u64) fuse.Attr { return attrFromStat(st, ino, b.opts.uid, b.opts.gid); } fn attrOut(b: *const Bridge, st: cloud9.Stat, ino: u64) fuse.AttrOut { return .{ .attr_valid = b.opts.attr_timeout_ns / 1_000_000_000, .attr_valid_nsec = @intCast(b.opts.attr_timeout_ns % 1_000_000_000), .attr = b.attrFrom(st, ino), }; } }; // -- pure helpers (unit-tested) ------------------------------------------------------------ /// Attr from a 9P Stat: DMDIR → S_IFDIR else S_IFREG, low 9 permission bits kept. pub fn attrFromStat(st: cloud9.Stat, ino: u64, uid: u32, gid: u32) fuse.Attr { const ftype: u32 = if (st.mode & cloud9.dmdir != 0) fuse.S_IFDIR else fuse.S_IFREG; return .{ .ino = ino, // The kernel marks an inode bad when size > LLONG_MAX; clamp hostile lengths. .size = @min(st.length, std.math.maxInt(i64)), // Saturating: a hostile length of 2^64-1 must not overflow. .blocks = st.length / 512 + @intFromBool(st.length % 512 != 0), .atime = st.atime, .mtime = st.mtime, .ctime = st.mtime, .mode = ftype | (st.mode & 0o777), .nlink = 1, .uid = uid, .gid = gid, .blksize = 4096, }; } /// Kernel open(2) flags → 9P open mode. O_APPEND has no 9P equivalent and is ignored. pub fn openMode(flags: u32) u8 { const o: linux.O = @bitCast(flags); var mode: u8 = switch (o.ACCMODE) { .RDONLY => cloud9.oread, .WRONLY => cloud9.owrite, .RDWR => cloud9.ordwr, }; if (o.TRUNC) mode |= cloud9.otrunc; return mode; } /// Parses consecutive 9P directory records (2-byte size + Stat) and appends entries. pub fn parseDirRecords(gpa: std.mem.Allocator, bytes: []const u8, list: *DirList) error{ OutOfMemory, BadDir }!void { var pos: usize = 0; while (pos < bytes.len) { if (bytes.len - pos < 2) return error.BadDir; const size: usize = std.mem.readInt(u16, bytes[pos..][0..2], .little); if (bytes.len - pos < 2 + size) return error.BadDir; const st = cloud9.Stat.decode(bytes[pos..][0 .. 2 + size]) catch return error.BadDir; pos += 2 + size; // The kernel rejects a whole READDIR reply (EIO) over one bad name, and // "." and ".." are synthesised by loadDir: drop such records instead. if (!validDirentName(st.name)) continue; const name = try gpa.dupe(u8, st.name); errdefer gpa.free(name); try list.entries.append(gpa, .{ .name = name, .ino = st.qid.path, .dtype = if (st.mode & cloud9.dmdir != 0) fuse.DT_DIR else fuse.DT_REG, }); } } /// A name the kernel will accept in a dirent and that does not duplicate the synthetic "." / "..". pub fn validDirentName(name: []const u8) bool { if (name.len == 0 or name.len > max_name_len) return false; if (std.mem.indexOfAny(u8, name, "/\x00") != null) return false; if (std.mem.eql(u8, name, ".") or std.mem.eql(u8, name, "..")) return false; return true; } /// Packs dirents from `entries[offset..]` into `buf`; each record's `off` is its index + 1. /// Returns the number of bytes used. pub fn packDirents(entries: []const Entry, offset: u64, buf: []u8) usize { var used: usize = 0; var i: usize = @intCast(@min(offset, entries.len)); while (i < entries.len) : (i += 1) { const e = entries[i]; if (!fuse.addDirent(buf, &used, e.ino, @as(u64, i) + 1, e.dtype, e.name)) break; } return used; } fn randomU64() u64 { var bytes: [8]u8 = undefined; if (linux.errno(linux.getrandom(&bytes, bytes.len, 0)) == .SUCCESS) return std.mem.readInt(u64, &bytes, .little); var ts: linux.timespec = undefined; _ = linux.clock_gettime(.MONOTONIC, &ts); return @as(u64, @bitCast(ts.nsec)) ^ (@as(u64, @bitCast(ts.sec)) << 32); } fn nowSeconds() u32 { var ts: linux.timespec = undefined; if (linux.errno(linux.clock_gettime(.REALTIME, &ts)) != .SUCCESS) return 0; return @intCast(@as(u64, @intCast(ts.sec)) & 0xFFFF_FFFF); } fn opName(op: fuse.Opcode) []const u8 { return switch (op) { _ => "unknown", else => @tagName(op), }; } // Thin adapters so fuse.zig's parse errors become HandlerError.BadRequest. fn body(comptime T: type, req: fuse.Request) error{BadRequest}!*const T { return fuse.body(T, req) catch error.BadRequest; } fn nameAfter(comptime T: type, req: fuse.Request) error{BadRequest}![]const u8 { return fuse.nameAfter(T, req) catch error.BadRequest; } fn secondName(req: fuse.Request, first: []const u8, offset: usize) error{BadRequest}![]const u8 { return fuse.secondName(req, first, offset) catch error.BadRequest; } // -- tests ------------------------------------------------------------------------------ const testing = std.testing; test { // Force semantic analysis of `serve` and the whole dispatch path, which no // unit test can exercise without a FUSE mount. testing.refAllDecls(@This()); } fn testStat(name: []const u8, mode: u32, length: u64, path: u64) cloud9.Stat { return .{ .type = 0, .dev = 0, .qid = .{ .type = if (mode & cloud9.dmdir != 0) cloud9.qtdir else 0, .version = 0, .path = path }, .mode = mode, .atime = 100, .mtime = 200, .length = length, .name = name, .uid = "u", .gid = "g", .muid = "u", }; } test "attr mapping: DMDIR → S_IFDIR|perm, length → size/blocks" { const d = attrFromStat(testStat("d", cloud9.dmdir | 0o755, 0, 9), 9, 1000, 1001); try testing.expectEqual(fuse.S_IFDIR | 0o755, d.mode); try testing.expectEqual(@as(u64, 9), d.ino); try testing.expectEqual(@as(u64, 0), d.size); try testing.expectEqual(@as(u64, 0), d.blocks); try testing.expectEqual(@as(u32, 1000), d.uid); try testing.expectEqual(@as(u32, 1001), d.gid); try testing.expectEqual(@as(u32, 1), d.nlink); const f = attrFromStat(testStat("f", 0o640 | cloud9.dmappend, 1025, 4), 4, 0, 0); try testing.expectEqual(fuse.S_IFREG | 0o640, f.mode); // dmappend bit not leaked try testing.expectEqual(@as(u64, 1025), f.size); try testing.expectEqual(@as(u64, 3), f.blocks); try testing.expectEqual(@as(u32, 4096), f.blksize); try testing.expectEqual(@as(u64, 100), f.atime); try testing.expectEqual(@as(u64, 200), f.mtime); try testing.expectEqual(@as(u64, 200), f.ctime); try testing.expectEqual(@as(u64, 1), attrFromStat(testStat("f", 0o600, 512, 4), 4, 0, 0).blocks); try testing.expectEqual(@as(u64, 2), attrFromStat(testStat("f", 0o600, 513, 4), 4, 0, 0).blocks); } test "open flag → 9P mode mapping" { const rdonly: u32 = @bitCast(linux.O{ .ACCMODE = .RDONLY }); const wronly: u32 = @bitCast(linux.O{ .ACCMODE = .WRONLY }); const rdwr: u32 = @bitCast(linux.O{ .ACCMODE = .RDWR }); const trunc: u32 = @bitCast(linux.O{ .TRUNC = true }); const append: u32 = @bitCast(linux.O{ .APPEND = true }); const creat: u32 = @bitCast(linux.O{ .CREAT = true }); try testing.expectEqual(cloud9.oread, openMode(rdonly)); try testing.expectEqual(cloud9.owrite, openMode(wronly)); try testing.expectEqual(cloud9.ordwr, openMode(rdwr)); try testing.expectEqual(cloud9.owrite | cloud9.otrunc, openMode(wronly | trunc)); try testing.expectEqual(cloud9.ordwr | cloud9.otrunc, openMode(rdwr | trunc | creat)); try testing.expectEqual(cloud9.owrite, openMode(wronly | append)); // O_APPEND ignored } test "dirlist parsing from two hand-encoded Stat records" { var buf: [512]u8 = undefined; const a = try cloud9.Stat.encode(testStat("alpha", 0o644, 10, 0x11), &buf); const bb = try cloud9.Stat.encode(testStat("beta", cloud9.dmdir | 0o755, 0, 0x22), buf[a.len..]); const bytes = buf[0 .. a.len + bb.len]; // Sanity: the record is prefixed by its own 2-byte size. try testing.expectEqual(a.len - 2, std.mem.readInt(u16, bytes[0..2], .little)); var list: DirList = .{}; defer list.deinit(testing.allocator); try parseDirRecords(testing.allocator, bytes, &list); try testing.expectEqual(@as(usize, 2), list.entries.items.len); try testing.expectEqualStrings("alpha", list.entries.items[0].name); try testing.expectEqual(@as(u64, 0x11), list.entries.items[0].ino); try testing.expectEqual(fuse.DT_REG, list.entries.items[0].dtype); try testing.expectEqualStrings("beta", list.entries.items[1].name); try testing.expectEqual(@as(u64, 0x22), list.entries.items[1].ino); try testing.expectEqual(fuse.DT_DIR, list.entries.items[1].dtype); // Truncated input is a protocol error and leaves earlier entries intact. try testing.expectError(error.BadDir, parseDirRecords(testing.allocator, bytes[0 .. bytes.len - 1], &list)); try testing.expectEqual(@as(usize, 3), list.entries.items.len); } test "readdir packing and offset resumption" { const names = [_][]const u8{ ".", "..", "one", "two", "three" }; var entries: [names.len]Entry = undefined; for (&entries, names, 0..) |*e, n, i| e.* = .{ .name = @constCast(n), .ino = 100 + i, .dtype = if (i < 2) fuse.DT_DIR else fuse.DT_REG }; // Everything fits: five records, off = index + 1. var big: [1024]u8 = undefined; const used = packDirents(&entries, 0, &big); var pos: usize = 0; var idx: usize = 0; while (pos < used) : (idx += 1) { const d = std.mem.bytesToValue(fuse.Dirent, big[pos..][0..@sizeOf(fuse.Dirent)]); try testing.expectEqual(@as(u64, 100 + idx), d.ino); try testing.expectEqual(@as(u64, idx + 1), d.off); try testing.expectEqualStrings(names[idx], big[pos + @sizeOf(fuse.Dirent) ..][0..d.namelen]); pos += (@sizeOf(fuse.Dirent) + d.namelen + 7) & ~@as(usize, 7); } try testing.expectEqual(names.len, idx); // A buffer that fits exactly two records ("." = 32, ".." = 32) stops there… var small: [64]u8 = undefined; const first_used = packDirents(&entries, 0, &small); try testing.expectEqual(@as(usize, 64), first_used); const last = std.mem.bytesToValue(fuse.Dirent, small[32..][0..@sizeOf(fuse.Dirent)]); try testing.expectEqual(@as(u64, 2), last.off); // …and resuming at the last `off` yields "one" next. const second_used = packDirents(&entries, last.off, &small); const next = std.mem.bytesToValue(fuse.Dirent, small[0..@sizeOf(fuse.Dirent)]); try testing.expectEqualStrings("one", small[@sizeOf(fuse.Dirent)..][0..next.namelen]); try testing.expectEqual(@as(u64, 3), next.off); try testing.expect(second_used > 0); // Past the end: nothing (EOF for the kernel). try testing.expectEqual(@as(usize, 0), packDirents(&entries, names.len, &big)); try testing.expectEqual(@as(usize, 0), packDirents(&entries, 1000, &big)); } test "attr mapping saturates hostile lengths instead of overflowing" { const a = attrFromStat(testStat("f", 0o600, std.math.maxInt(u64), 4), 4, 0, 0); try testing.expectEqual(@as(u64, std.math.maxInt(i64)), a.size); try testing.expectEqual(@as(u64, std.math.maxInt(u64) / 512 + 1), a.blocks); const b = attrFromStat(testStat("f", 0o600, 1024, 4), 4, 0, 0); try testing.expectEqual(@as(u64, 2), b.blocks); try testing.expectEqual(@as(u64, 1024), b.size); } test "dirent names the kernel would reject are dropped from listings" { try testing.expect(validDirentName("a")); try testing.expect(validDirentName("x" ** 1024)); try testing.expect(!validDirentName("")); try testing.expect(!validDirentName("a/b")); try testing.expect(!validDirentName("a\x00b")); try testing.expect(!validDirentName(".")); try testing.expect(!validDirentName("..")); try testing.expect(!validDirentName("x" ** 1025)); var buf: [4096]u8 = undefined; var n: usize = 0; for ([_][]const u8{ ".", "..", "", "a/b", "keep", "x" ** 1025, "also" }) |name| { n += (try cloud9.Stat.encode(testStat(name, 0o644, 1, 0x30), buf[n..])).len; } var list: DirList = .{}; defer list.deinit(testing.allocator); try parseDirRecords(testing.allocator, buf[0..n], &list); try testing.expectEqual(@as(usize, 2), list.entries.items.len); try testing.expectEqualStrings("keep", list.entries.items[0].name); try testing.expectEqualStrings("also", list.entries.items[1].name); } test "DirList frees its names" { var list: DirList = .{}; try list.entries.append(testing.allocator, .{ .name = try testing.allocator.dupe(u8, "x"), .ino = 1, .dtype = fuse.DT_REG }); list.deinit(testing.allocator); }