diff options
| author | Gabriel Schneider <[email protected]> | 2026-09-21 14:23:27 -0300 |
|---|---|---|
| committer | Gabriel Schneider <[email protected]> | 2026-09-21 15:20:27 -0300 |
| commit | 0d7e295efee1fca0935cf4a8bee9629c007dd2b6 (patch) | |
| tree | d371eb028c63da5ab5b506bd25ac6579cf8a93fc /9ns/src/bridge.zig | |
| parent | 3a23f6a29e47ace901bd4d82b9db4055fcc12bb9 (diff) | |
| download | cloud9-0d7e295efee1fca0935cf4a8bee9629c007dd2b6.tar.gz cloud9-0d7e295efee1fca0935cf4a8bee9629c007dd2b6.zip | |
9ns --mntgen: registry subdirectories are mount points too
A registry entry that is a directory is now served the way the root is:
a synthetic directory listing the real one, dialing the sockets inside
it on walk and recursing into further directories, to max_synth_depth
(8) levels across max_synth_dirs (64) synthetic nodes. That is the
plan9port mntgen shape and the layout zmx now posts under, so a live
session reads at /mnt/9p/zmx/<name>. Before this a directory in the
registry was dialed like a socket and answered EIO for good.
post gains the two entry points the traversal needs: postedDir (the
registry scan, against any directory) and dialPath (a dial by composed
path, no name validation).
Hardening, each from an attack that broke the code:
- BATCH_FORGET carries entries for many owners and puts 0 in the header
nodeid, so routing it by the header dropped all of them: 32 of 64
synthetic slots leaked in one close burst and the subdirectories that
held them answered EIO forever. distributeForgets unpacks the body and
hands each entry to its owner.
- probe() and connectBlocking() copied a caller's path into the kernel
address with no bound: a path past sun_path overran the 110-byte stack
sockaddr (a panic in Debug, silent corruption in ReleaseFast). Both
refuse it now, probe as `.live` so a claim never deletes what it could
not inspect.
- That bound then caught 9proc's own listener, which handed probe() the
whole 108-byte sun_path array instead of the path inside it. The probe
reads `.live` for anything it cannot ask about, so every stale socket
became AlreadyListening and no server could ever take a dead
predecessor's name back. It passes the path now.
Suites: 87/87 root (+7 post/serve attack regressions), 48/48 9ns,
51+88 9ns integration (+4 traversal and slot-recycling checks), 213/0
9ns adversarial, 60/60 9proc plus its adversarial suites with a new
stale-socket takeover check, freestanding green.
Diffstat (limited to '9ns/src/bridge.zig')
| -rw-r--r-- | 9ns/src/bridge.zig | 372 |
1 files changed, 365 insertions, 7 deletions
diff --git a/9ns/src/bridge.zig b/9ns/src/bridge.zig index 76327c3..457c883 100644 --- a/9ns/src/bridge.zig +++ b/9ns/src/bridge.zig @@ -855,6 +855,15 @@ pub const max_mounts: usize = 4096; pub const max_root_dirs: usize = 64; /// The staged buffer handed to `post.posted` for one root listing. pub const stage_len: usize = 8192; +/// Registry entries that are directories are served like the root itself +/// (a synthetic directory mirroring the real one, dialing the sockets +/// found inside); this bounds the synthetic directories one 9ns serves. +pub const max_synth_dirs: usize = 64; +/// How deep those registry subdirectories nest. +pub const max_synth_depth: u8 = 8; +/// The node-id index reserved for synthetic registry subdirectories; +/// mounts use ordinals below 4096, so this never collides with one. +pub const synth_index: u32 = 0xFFFF_FFFF; /// The node id a server's subtree lives under: `index` in the top bits, /// `local` (1 = that server's 9P root) below. @@ -891,6 +900,31 @@ pub const MntgenOptions = struct { msize: u32 = 131072, }; +/// One registry subdirectory served as a synthetic directory: the real +/// path it mirrors, the registry-relative key its children dial under, +/// its slot (which names its node id) and its depth from the registry. +const SynthDir = struct { + slot: usize = 0, + parent: u64 = 0, + depth: u8 = 0, + path_buf: [post.sun_path_len]u8 = undefined, + path_len: u16 = 0, + rel_buf: [post.sun_path_len]u8 = undefined, + rel_len: u16 = 0, + + fn path(sd: *const SynthDir) [:0]const u8 { + return sd.path_buf[0..sd.path_len :0]; + } + + fn rel(sd: *const SynthDir) []const u8 { + return sd.rel_buf[0..sd.rel_len]; + } + + fn nodeid(sd: *const SynthDir) u64 { + return mountNode(synth_index, sd.slot + 1); + } +}; + /// One dialed server: its 9P session, its bridge state and its worker /// thread, plus the queue the dispatcher feeds requests through. const Mount = struct { @@ -998,6 +1032,8 @@ const Mntgen = struct { next_index: u32 = 1, /// Open directory handles of the synthetic root (dispatcher-owned). root_dirs: [max_root_dirs]?*DirList = @splat(null), + /// Synthetic registry subdirectories (dispatcher-owned), by slot. + synths: [max_synth_dirs]?*SynthDir = @splat(null), fn deinit(mg: *Mntgen) void { // Wake every worker, then join: at this point the child is gone (or @@ -1031,6 +1067,12 @@ const Mntgen = struct { slot.* = null; } } + for (&mg.synths) |*slot| { + if (slot.*) |sd| { + mg.gpa.destroy(sd); + slot.* = null; + } + } if (mg.req_buf.len != 0) mg.gpa.free(mg.req_buf); if (mg.data_buf.len != 0) mg.gpa.free(mg.data_buf); if (mg.stage.len != 0) mg.gpa.free(mg.stage); @@ -1079,12 +1121,23 @@ const Mntgen = struct { mg.routeInterrupt(in.unique); return true; }, + // A BATCH_FORGET carries entries for many owners at once and + // cannot be routed by its header nodeid (the kernel sends 0); + // see `distributeForgets`. + .batch_forget => { + mg.distributeForgets(req); + return true; + }, else => {}, } if (h.nodeid == fuse.root_id) { try mg.handleRoot(req); return true; } + if (mountIndex(h.nodeid) == synth_index) { + try mg.handleSynthDir(req); + return true; + } const wants_reply = switch (op) { .forget, .batch_forget => false, else => true, @@ -1133,6 +1186,68 @@ const Mntgen = struct { } } + /// FUSE_BATCH_FORGET carries (nodeid, nlookup) entries for many owners + /// at once — synthetic-root, synthetic-subdirectory and per-mount + /// nodeids can all appear in one batch, and the kernel puts 0 in the + /// header nodeid. Routing such a batch like an ordinary request would + /// hand every entry to one wrong owner and drop the rest: a mount's + /// bridge would leak the fid behind each dropped entry forever, and + /// a dropped synthetic-subdirectory entry would leak its slot until + /// every subdirectory lookup answers EIO. So the dispatcher keeps + /// what it owns — the root holds nothing, a synth slot is freed here — + /// and forwards each mount-owned entry to its worker as a plain + /// single FUSE_FORGET. + fn distributeForgets(mg: *Mntgen, req: fuse.Request) void { + const in = fuse.body(fuse.BatchForgetIn, req) catch return; + const rest = req.body[@sizeOf(fuse.BatchForgetIn)..]; + const count: usize = in.count; + if (rest.len < count * @sizeOf(fuse.ForgetOne)) return; + for (0..count) |i| { + const one = std.mem.bytesToValue(fuse.ForgetOne, rest[i * @sizeOf(fuse.ForgetOne) ..][0..@sizeOf(fuse.ForgetOne)]); + mg.forgetOne(one.nodeid, one.nlookup, req.header.unique); + } + } + + /// Forgets one node by id, whatever owns it: the root holds nothing, + /// a synthetic subdirectory's slot goes back to the pool, and a + /// mount-owned node is forwarded to its worker (which clunks the fid + /// behind it). Unknown or dead mounts drop the entry, like a single + /// FORGET routed by `route`. + fn forgetOne(mg: *Mntgen, nodeid: u64, nlookup: u64, unique: u64) void { + if (nodeid == fuse.root_id) return; + const idx = mountIndex(nodeid); + if (idx == synth_index) { + const local = nodeid & mount_node_mask; + if (local == 0 or local > max_synth_dirs) return; + const slot: usize = @intCast(local - 1); + if (mg.synths[slot]) |sd| { + mg.trace(" synthetic directory slot {d} forgotten", .{slot}); + mg.gpa.destroy(sd); + mg.synths[slot] = null; + } + return; + } + const m = if (idx < max_mounts) mg.mounts[idx] else null; + if (m == null or m.?.dead.load(.seq_cst)) return; + // A synthesized single FORGET (no reply is expected for one, so + // the unique is only bookkeeping). + var buf: [@sizeOf(fuse.InHeader) + @sizeOf(fuse.ForgetIn)]u8 = undefined; + const hdr = fuse.InHeader{ + .len = @sizeOf(fuse.InHeader) + @sizeOf(fuse.ForgetIn), + .opcode = @intFromEnum(fuse.Opcode.forget), + .unique = unique, + .nodeid = nodeid, + .uid = 0, + .gid = 0, + .pid = 0, + .total_extlen = 0, + .padding = 0, + }; + @memcpy(buf[0..@sizeOf(fuse.InHeader)], std.mem.asBytes(&hdr)); + @memcpy(buf[@sizeOf(fuse.InHeader)..], std.mem.asBytes(&fuse.ForgetIn{ .nlookup = nlookup })); + mg.enqueue(m.?, &buf); + } + /// A FUSE_INTERRUPT names the request it wants cancelled; FUSE uniques /// are unique across the whole connection, so the mount whose bridge is /// currently serving that unique gets the packet and its session turns @@ -1257,11 +1372,34 @@ const Mntgen = struct { const out = mg.rootEntryOut(m.root_node, m.root_attr); return mg.reply(u, &.{std.mem.asBytes(&out)}); } - const m = mg.dialMount(name) catch |e| switch (e) { - error.NotPosted => { - mg.trace(" lookup '{s}': nothing posted under that name", .{name}); - return mg.replyError(u, .NOENT); + // What the entry is decides what a walk into it becomes: a socket + // dials (the original behavior), a directory is served like the + // root itself (its sockets dial on walk, its directories recurse), + // anything else answers EIO. + var path_buf: [post.sun_path_len]u8 = undefined; + const entry_path = post.registryPath(mg.mo.env, name, &path_buf) catch + return mg.replyError(u, .NOENT); + const st = std.Io.Dir.statFile(.cwd(), mg.mo.io, entry_path, .{}) catch { + mg.trace(" lookup '{s}': nothing posted under that name", .{name}); + return mg.replyError(u, .NOENT); + }; + switch (st.kind) { + .directory => { + const sd = mg.newSynth(entry_path, name, 1, fuse.root_id) catch |e| { + mg.trace(" lookup '{s}': no synthetic slot: {t}", .{ name, e }); + return mg.replyError(u, .IO); + }; + const node = sd.nodeid(); + const out = mg.rootEntryOut(node, mg.synthAttr(node)); + return mg.reply(u, &.{std.mem.asBytes(&out)}); }, + .unix_domain_socket => {}, + else => { + mg.trace(" lookup '{s}': registry entry is not a socket", .{name}); + return mg.replyError(u, .IO); + }, + } + const m = mg.dialMount(name) catch |e| switch (e) { error.Stale => { mg.trace(" lookup '{s}': registry entry is stale (no server behind it)", .{name}); return mg.replyError(u, .IO); @@ -1297,6 +1435,204 @@ const Mntgen = struct { return list; } + // -- synthetic registry subdirectories ------------------------------------ + + /// A directory entry in the registry (or in one of its + /// subdirectories) is served like the root: a synthetic directory + /// listing the real one, whose sockets dial on walk and whose + /// directories recurse. Served on the dispatcher thread, like the + /// root. + fn handleSynthDir(mg: *Mntgen, req: fuse.Request) error{FuseIo}!void { + const u = req.header.unique; + const local = req.header.nodeid & mount_node_mask; + if (local == 0 or local > max_synth_dirs) return mg.replyError(u, .IO); + const slot: usize = @intCast(local - 1); + const sd = mg.synths[slot] orelse return mg.replyError(u, .IO); + switch (req.header.op()) { + .forget, .batch_forget => { + // The kernel dropped the dentry; the slot goes with it. + mg.trace(" synthetic directory slot {d} forgotten", .{slot}); + mg.gpa.destroy(sd); + mg.synths[slot] = null; + }, + .getattr => { + const out = fuse.AttrOut{ .attr = mg.synthAttr(sd.nodeid()) }; + try mg.reply(u, &.{std.mem.asBytes(&out)}); + }, + .lookup => try mg.synthLookup(sd, req), + .opendir => { + var fh: ?usize = null; + for (&mg.root_dirs, 0..) |*dir_slot, i| { + if (dir_slot.* == null) { + fh = i; + break; + } + } + const dir_slot = fh orelse return mg.replyError(u, .MFILE); + const list = mg.synthListing(sd) catch { + return mg.replyError(u, .IO); + }; + mg.root_dirs[dir_slot] = list; + const out = fuse.OpenOut{ .fh = dir_slot }; + try mg.reply(u, &.{std.mem.asBytes(&out)}); + }, + .readdir => { + const in = fuse.body(fuse.ReadIn, req) catch return mg.replyError(u, .BADF); + if (in.fh >= max_root_dirs) return mg.replyError(u, .BADF); + const list = mg.root_dirs[@intCast(in.fh)] orelse return mg.replyError(u, .BADF); + const size: usize = @min(in.size, max_write); + const used = packDirents(list.entries.items, in.offset, mg.data_buf[0..size]); + try mg.reply(u, &.{mg.data_buf[0..used]}); + }, + .release, .releasedir => { + const in = fuse.body(fuse.ReleaseIn, req) catch return mg.replyError(u, .BADF); + if (in.fh < max_root_dirs) { + if (mg.root_dirs[@intCast(in.fh)]) |list| { + list.deinit(mg.gpa); + mg.gpa.destroy(list); + mg.root_dirs[@intCast(in.fh)] = null; + } + } + try mg.reply(u, &.{}); + }, + .statfs => { + const out = fuse.StatfsOut{ .st = .{ .bsize = 4096, .namelen = 255, .frsize = 4096 } }; + try mg.reply(u, &.{std.mem.asBytes(&out)}); + }, + .flush, .fsync, .fsyncdir => try mg.reply(u, &.{}), + // Capability probes read as "not supported", like the root's. + .access, .setxattr, .getxattr, .listxattr, .removexattr, .statx => try mg.replyError(u, .NOSYS), + else => try mg.replyError(u, .PERM), + } + } + + fn synthAttr(mg: *const Mntgen, nodeid: u64) fuse.Attr { + // Read-only like the root and like /srv. + return .{ + .ino = nodeid, + .mode = fuse.S_IFDIR | 0o555, + .nlink = 2, + .uid = mg.opts.uid, + .gid = mg.opts.gid, + .blksize = 4096, + }; + } + + fn synthLookup(mg: *Mntgen, sd: *SynthDir, req: fuse.Request) error{FuseIo}!void { + const u = req.header.unique; + const name = fuse.nameAfter(void, req) catch return mg.replyError(u, .INVAL); + if (std.mem.eql(u8, name, ".")) { + const out = mg.rootEntryOut(sd.nodeid(), mg.synthAttr(sd.nodeid())); + return mg.reply(u, &.{std.mem.asBytes(&out)}); + } + // The kernel resolves ".." from its own dentry tree and a LOOKUP of + // it has never been observed, but it must not alias the directory + // onto itself either: answer with the parent's node id (the root's + // for a top-level subdirectory — its attr is the same shape). + if (std.mem.eql(u8, name, "..")) { + const out = mg.rootEntryOut(sd.parent, mg.synthAttr(sd.parent)); + return mg.reply(u, &.{std.mem.asBytes(&out)}); + } + if (!post.legalName(name)) return mg.replyError(u, .NOENT); + // The registry-relative key this child dials under (a mount's + // name, for findMount). + var key_buf: [post.sun_path_len]u8 = undefined; + const key = std.fmt.bufPrint(&key_buf, "{s}/{s}", .{ sd.rel(), name }) catch + return mg.replyError(u, .NOTNAM); + var path_buf: [post.sun_path_len]u8 = undefined; + const child = std.fmt.bufPrintSentinel(&path_buf, "{s}/{s}", .{ sd.path(), name }, 0) catch + return mg.replyError(u, .NOTNAM); + if (findMount(mg.mounts, key)) |m| { + mg.trace(" lookup '{s}': mount {d} already live", .{ key, m.index }); + const out = mg.rootEntryOut(m.root_node, m.root_attr); + return mg.reply(u, &.{std.mem.asBytes(&out)}); + } + const st = std.Io.Dir.statFile(.cwd(), mg.mo.io, child, .{}) catch |e| switch (e) { + error.FileNotFound => { + mg.trace(" lookup '{s}': no entry", .{key}); + return mg.replyError(u, .NOENT); + }, + else => { + mg.trace(" lookup '{s}': stat failed: {t}", .{ key, e }); + return mg.replyError(u, .IO); + }, + }; + switch (st.kind) { + .directory => { + const child_sd = mg.newSynth(child, key, sd.depth + 1, sd.nodeid()) catch |e| { + mg.trace(" lookup '{s}': no synthetic slot: {t}", .{ key, e }); + return mg.replyError(u, .IO); + }; + const node = child_sd.nodeid(); + const out = mg.rootEntryOut(node, mg.synthAttr(node)); + return mg.reply(u, &.{std.mem.asBytes(&out)}); + }, + .unix_domain_socket => { + const m = mg.dialMountAt(child, key) catch |e| { + mg.trace(" lookup '{s}': dial failed: {t}", .{ key, e }); + return mg.replyError(u, .IO); + }; + mg.trace(" lookup '{s}': dialed as mount {d}", .{ key, m.index }); + const out = mg.rootEntryOut(m.root_node, m.root_attr); + try mg.reply(u, &.{std.mem.asBytes(&out)}); + }, + // Not a service and not a directory: the entry answers EIO on + // walk, like a plain file in the registry itself. + else => { + mg.trace(" lookup '{s}': entry is not a socket or directory", .{key}); + return mg.replyError(u, .IO); + }, + } + } + + /// One OPENDIR of a synthetic subdirectory: `.` and `..` plus the + /// real directory's entries, snapshotted for the life of the handle + /// (a fresh OPENDIR sees fresh entries), like the root. + fn synthListing(mg: *Mntgen, sd: *SynthDir) !*DirList { + const list = try mg.gpa.create(DirList); + errdefer mg.gpa.destroy(list); + list.* = .{}; + errdefer list.deinit(mg.gpa); + const node = sd.nodeid(); + try list.entries.append(mg.gpa, .{ .name = try mg.gpa.dupe(u8, "."), .ino = node, .dtype = fuse.DT_DIR }); + try list.entries.append(mg.gpa, .{ .name = try mg.gpa.dupe(u8, ".."), .ino = sd.parent, .dtype = fuse.DT_DIR }); + var names = post.postedDir(mg.mo.io, sd.path(), mg.stage) catch |e| { + mg.trace(" directory listing failed: {t}", .{e}); + return error.Registry; + }; + while (names.next()) |name| { + if (!validDirentName(name)) continue; + try list.entries.append(mg.gpa, .{ .name = try mg.gpa.dupe(u8, name), .ino = nameIno(name), .dtype = fuse.DT_DIR }); + } + return list; + } + + /// Allocates a synthetic directory node mirroring `path`, keyed by + /// the registry-relative `key`, at `depth` under `parent`'s node id. + fn newSynth(mg: *Mntgen, path: [:0]const u8, key: []const u8, depth: u8, parent: u64) !*SynthDir { + if (depth > max_synth_depth) return error.TooDeep; + var slot: ?usize = null; + for (&mg.synths, 0..) |*s, i| { + if (s.* == null) { + slot = i; + break; + } + } + const i = slot orelse return error.TooMany; + const sd = try mg.gpa.create(SynthDir); + errdefer mg.gpa.destroy(sd); + sd.* = .{ .slot = i, .parent = parent, .depth = depth }; + if (path.len + 1 > sd.path_buf.len) return error.NameTooLong; + @memcpy(sd.path_buf[0..path.len], path); + sd.path_buf[path.len] = 0; + sd.path_len = @intCast(path.len); + if (key.len > sd.rel_buf.len) return error.NameTooLong; + @memcpy(sd.rel_buf[0..key.len], key); + sd.rel_len = @intCast(key.len); + mg.synths[i] = sd; + return sd; + } + // -- dialing --------------------------------------------------------------- /// Dials `name` out of the registry, attaches, stats the server root and @@ -1304,10 +1640,22 @@ const Mntgen = struct { /// of the LOOKUP that triggered it (so a hung server delays that walk, /// like it would delay any 9P client). fn dialMount(mg: *Mntgen, name: []const u8) !*Mount { + const stream = try post.dial(mg.mo.io, mg.mo.env, name); + return mg.mountStream(stream, name); + } + + /// Dials the socket at `path` (a registry subdirectory entry) and + /// mounts it under `key`, the registry-relative path — the + /// subdirectory analogue of `dialMount`. + fn dialMountAt(mg: *Mntgen, path: [:0]const u8, key: []const u8) !*Mount { + const stream = try post.dialPath(mg.mo.io, path); + return mg.mountStream(stream, key); + } + + /// The shared dial tail: session, attach, stat, bridge and worker. + fn mountStream(mg: *Mntgen, stream: std.Io.net.Stream, key: []const u8) !*Mount { if (mg.next_index >= max_mounts) return error.TooManyMounts; const index: u32 = mg.next_index; - - const stream = try post.dial(mg.mo.io, mg.mo.env, name); const fd: i32 = @intCast(stream.socket.handle); // The session below does blocking I/O: make sure a dial that left // the descriptor nonblocking cannot spin its read loop on EAGAIN, @@ -1369,7 +1717,7 @@ const Mntgen = struct { .gpa = mg.gpa, .io = mg.io, .debug = mg.opts.debug, - .name = try mg.gpa.dupe(u8, name), + .name = try mg.gpa.dupe(u8, key), .index = index, .root_node = root_node, .root_attr = attrFromStat(st, inoFromPath(st.qid.path) ^ b.ino_xor, mg.opts.uid, mg.opts.gid), @@ -1848,6 +2196,16 @@ test "mntgen node id layout: index in the top bits, local ids below" { try testing.expect(mount_node_mask == (1 << 32) - 1); } +test "mntgen synthetic subdirectory node ids route apart from mounts" { + for (1..max_synth_dirs + 1) |i| { + const node = mountNode(synth_index, i); + try testing.expectEqual(synth_index, mountIndex(node)); + try testing.expectEqual(i, node & mount_node_mask); + } + // The reserved index can never be a mount ordinal. + try testing.expect(synth_index >= max_mounts); +} + test "mntgen synthetic-root inos are deterministic and name-derived" { const a = nameIno("alpha"); try testing.expectEqual(a, nameIno("alpha")); |
