summaryrefslogtreecommitdiff
path: root/9ns/src/bridge.zig
diff options
context:
space:
mode:
authorGabriel Schneider <[email protected]>2026-09-21 14:23:27 -0300
committerGabriel Schneider <[email protected]>2026-09-21 15:20:27 -0300
commit0d7e295efee1fca0935cf4a8bee9629c007dd2b6 (patch)
treed371eb028c63da5ab5b506bd25ac6579cf8a93fc /9ns/src/bridge.zig
parent3a23f6a29e47ace901bd4d82b9db4055fcc12bb9 (diff)
downloadcloud9-0d7e295efee1fca0935cf4a8bee9629c007dd2b6.tar.gz
cloud9-0d7e295efee1fca0935cf4a8bee9629c007dd2b6.zip
9ns --mntgen: registry subdirectories are mount points too
A registry entry that is a directory is now served the way the root is: a synthetic directory listing the real one, dialing the sockets inside it on walk and recursing into further directories, to max_synth_depth (8) levels across max_synth_dirs (64) synthetic nodes. That is the plan9port mntgen shape and the layout zmx now posts under, so a live session reads at /mnt/9p/zmx/<name>. Before this a directory in the registry was dialed like a socket and answered EIO for good. post gains the two entry points the traversal needs: postedDir (the registry scan, against any directory) and dialPath (a dial by composed path, no name validation). Hardening, each from an attack that broke the code: - BATCH_FORGET carries entries for many owners and puts 0 in the header nodeid, so routing it by the header dropped all of them: 32 of 64 synthetic slots leaked in one close burst and the subdirectories that held them answered EIO forever. distributeForgets unpacks the body and hands each entry to its owner. - probe() and connectBlocking() copied a caller's path into the kernel address with no bound: a path past sun_path overran the 110-byte stack sockaddr (a panic in Debug, silent corruption in ReleaseFast). Both refuse it now, probe as `.live` so a claim never deletes what it could not inspect. - That bound then caught 9proc's own listener, which handed probe() the whole 108-byte sun_path array instead of the path inside it. The probe reads `.live` for anything it cannot ask about, so every stale socket became AlreadyListening and no server could ever take a dead predecessor's name back. It passes the path now. Suites: 87/87 root (+7 post/serve attack regressions), 48/48 9ns, 51+88 9ns integration (+4 traversal and slot-recycling checks), 213/0 9ns adversarial, 60/60 9proc plus its adversarial suites with a new stale-socket takeover check, freestanding green.
Diffstat (limited to '9ns/src/bridge.zig')
-rw-r--r--9ns/src/bridge.zig372
1 files changed, 365 insertions, 7 deletions
diff --git a/9ns/src/bridge.zig b/9ns/src/bridge.zig
index 76327c3..457c883 100644
--- a/9ns/src/bridge.zig
+++ b/9ns/src/bridge.zig
@@ -855,6 +855,15 @@ pub const max_mounts: usize = 4096;
pub const max_root_dirs: usize = 64;
/// The staged buffer handed to `post.posted` for one root listing.
pub const stage_len: usize = 8192;
+/// Registry entries that are directories are served like the root itself
+/// (a synthetic directory mirroring the real one, dialing the sockets
+/// found inside); this bounds the synthetic directories one 9ns serves.
+pub const max_synth_dirs: usize = 64;
+/// How deep those registry subdirectories nest.
+pub const max_synth_depth: u8 = 8;
+/// The node-id index reserved for synthetic registry subdirectories;
+/// mounts use ordinals below 4096, so this never collides with one.
+pub const synth_index: u32 = 0xFFFF_FFFF;
/// The node id a server's subtree lives under: `index` in the top bits,
/// `local` (1 = that server's 9P root) below.
@@ -891,6 +900,31 @@ pub const MntgenOptions = struct {
msize: u32 = 131072,
};
+/// One registry subdirectory served as a synthetic directory: the real
+/// path it mirrors, the registry-relative key its children dial under,
+/// its slot (which names its node id) and its depth from the registry.
+const SynthDir = struct {
+ slot: usize = 0,
+ parent: u64 = 0,
+ depth: u8 = 0,
+ path_buf: [post.sun_path_len]u8 = undefined,
+ path_len: u16 = 0,
+ rel_buf: [post.sun_path_len]u8 = undefined,
+ rel_len: u16 = 0,
+
+ fn path(sd: *const SynthDir) [:0]const u8 {
+ return sd.path_buf[0..sd.path_len :0];
+ }
+
+ fn rel(sd: *const SynthDir) []const u8 {
+ return sd.rel_buf[0..sd.rel_len];
+ }
+
+ fn nodeid(sd: *const SynthDir) u64 {
+ return mountNode(synth_index, sd.slot + 1);
+ }
+};
+
/// One dialed server: its 9P session, its bridge state and its worker
/// thread, plus the queue the dispatcher feeds requests through.
const Mount = struct {
@@ -998,6 +1032,8 @@ const Mntgen = struct {
next_index: u32 = 1,
/// Open directory handles of the synthetic root (dispatcher-owned).
root_dirs: [max_root_dirs]?*DirList = @splat(null),
+ /// Synthetic registry subdirectories (dispatcher-owned), by slot.
+ synths: [max_synth_dirs]?*SynthDir = @splat(null),
fn deinit(mg: *Mntgen) void {
// Wake every worker, then join: at this point the child is gone (or
@@ -1031,6 +1067,12 @@ const Mntgen = struct {
slot.* = null;
}
}
+ for (&mg.synths) |*slot| {
+ if (slot.*) |sd| {
+ mg.gpa.destroy(sd);
+ slot.* = null;
+ }
+ }
if (mg.req_buf.len != 0) mg.gpa.free(mg.req_buf);
if (mg.data_buf.len != 0) mg.gpa.free(mg.data_buf);
if (mg.stage.len != 0) mg.gpa.free(mg.stage);
@@ -1079,12 +1121,23 @@ const Mntgen = struct {
mg.routeInterrupt(in.unique);
return true;
},
+ // A BATCH_FORGET carries entries for many owners at once and
+ // cannot be routed by its header nodeid (the kernel sends 0);
+ // see `distributeForgets`.
+ .batch_forget => {
+ mg.distributeForgets(req);
+ return true;
+ },
else => {},
}
if (h.nodeid == fuse.root_id) {
try mg.handleRoot(req);
return true;
}
+ if (mountIndex(h.nodeid) == synth_index) {
+ try mg.handleSynthDir(req);
+ return true;
+ }
const wants_reply = switch (op) {
.forget, .batch_forget => false,
else => true,
@@ -1133,6 +1186,68 @@ const Mntgen = struct {
}
}
+ /// FUSE_BATCH_FORGET carries (nodeid, nlookup) entries for many owners
+ /// at once — synthetic-root, synthetic-subdirectory and per-mount
+ /// nodeids can all appear in one batch, and the kernel puts 0 in the
+ /// header nodeid. Routing such a batch like an ordinary request would
+ /// hand every entry to one wrong owner and drop the rest: a mount's
+ /// bridge would leak the fid behind each dropped entry forever, and
+ /// a dropped synthetic-subdirectory entry would leak its slot until
+ /// every subdirectory lookup answers EIO. So the dispatcher keeps
+ /// what it owns — the root holds nothing, a synth slot is freed here —
+ /// and forwards each mount-owned entry to its worker as a plain
+ /// single FUSE_FORGET.
+ fn distributeForgets(mg: *Mntgen, req: fuse.Request) void {
+ const in = fuse.body(fuse.BatchForgetIn, req) catch return;
+ const rest = req.body[@sizeOf(fuse.BatchForgetIn)..];
+ const count: usize = in.count;
+ if (rest.len < count * @sizeOf(fuse.ForgetOne)) return;
+ for (0..count) |i| {
+ const one = std.mem.bytesToValue(fuse.ForgetOne, rest[i * @sizeOf(fuse.ForgetOne) ..][0..@sizeOf(fuse.ForgetOne)]);
+ mg.forgetOne(one.nodeid, one.nlookup, req.header.unique);
+ }
+ }
+
+ /// Forgets one node by id, whatever owns it: the root holds nothing,
+ /// a synthetic subdirectory's slot goes back to the pool, and a
+ /// mount-owned node is forwarded to its worker (which clunks the fid
+ /// behind it). Unknown or dead mounts drop the entry, like a single
+ /// FORGET routed by `route`.
+ fn forgetOne(mg: *Mntgen, nodeid: u64, nlookup: u64, unique: u64) void {
+ if (nodeid == fuse.root_id) return;
+ const idx = mountIndex(nodeid);
+ if (idx == synth_index) {
+ const local = nodeid & mount_node_mask;
+ if (local == 0 or local > max_synth_dirs) return;
+ const slot: usize = @intCast(local - 1);
+ if (mg.synths[slot]) |sd| {
+ mg.trace(" synthetic directory slot {d} forgotten", .{slot});
+ mg.gpa.destroy(sd);
+ mg.synths[slot] = null;
+ }
+ return;
+ }
+ const m = if (idx < max_mounts) mg.mounts[idx] else null;
+ if (m == null or m.?.dead.load(.seq_cst)) return;
+ // A synthesized single FORGET (no reply is expected for one, so
+ // the unique is only bookkeeping).
+ var buf: [@sizeOf(fuse.InHeader) + @sizeOf(fuse.ForgetIn)]u8 = undefined;
+ const hdr = fuse.InHeader{
+ .len = @sizeOf(fuse.InHeader) + @sizeOf(fuse.ForgetIn),
+ .opcode = @intFromEnum(fuse.Opcode.forget),
+ .unique = unique,
+ .nodeid = nodeid,
+ .uid = 0,
+ .gid = 0,
+ .pid = 0,
+ .total_extlen = 0,
+ .padding = 0,
+ };
+ @memcpy(buf[0..@sizeOf(fuse.InHeader)], std.mem.asBytes(&hdr));
+ @memcpy(buf[@sizeOf(fuse.InHeader)..], std.mem.asBytes(&fuse.ForgetIn{ .nlookup = nlookup }));
+ mg.enqueue(m.?, &buf);
+ }
+
/// A FUSE_INTERRUPT names the request it wants cancelled; FUSE uniques
/// are unique across the whole connection, so the mount whose bridge is
/// currently serving that unique gets the packet and its session turns
@@ -1257,11 +1372,34 @@ const Mntgen = struct {
const out = mg.rootEntryOut(m.root_node, m.root_attr);
return mg.reply(u, &.{std.mem.asBytes(&out)});
}
- const m = mg.dialMount(name) catch |e| switch (e) {
- error.NotPosted => {
- mg.trace(" lookup '{s}': nothing posted under that name", .{name});
- return mg.replyError(u, .NOENT);
+ // What the entry is decides what a walk into it becomes: a socket
+ // dials (the original behavior), a directory is served like the
+ // root itself (its sockets dial on walk, its directories recurse),
+ // anything else answers EIO.
+ var path_buf: [post.sun_path_len]u8 = undefined;
+ const entry_path = post.registryPath(mg.mo.env, name, &path_buf) catch
+ return mg.replyError(u, .NOENT);
+ const st = std.Io.Dir.statFile(.cwd(), mg.mo.io, entry_path, .{}) catch {
+ mg.trace(" lookup '{s}': nothing posted under that name", .{name});
+ return mg.replyError(u, .NOENT);
+ };
+ switch (st.kind) {
+ .directory => {
+ const sd = mg.newSynth(entry_path, name, 1, fuse.root_id) catch |e| {
+ mg.trace(" lookup '{s}': no synthetic slot: {t}", .{ name, e });
+ return mg.replyError(u, .IO);
+ };
+ const node = sd.nodeid();
+ const out = mg.rootEntryOut(node, mg.synthAttr(node));
+ return mg.reply(u, &.{std.mem.asBytes(&out)});
},
+ .unix_domain_socket => {},
+ else => {
+ mg.trace(" lookup '{s}': registry entry is not a socket", .{name});
+ return mg.replyError(u, .IO);
+ },
+ }
+ const m = mg.dialMount(name) catch |e| switch (e) {
error.Stale => {
mg.trace(" lookup '{s}': registry entry is stale (no server behind it)", .{name});
return mg.replyError(u, .IO);
@@ -1297,6 +1435,204 @@ const Mntgen = struct {
return list;
}
+ // -- synthetic registry subdirectories ------------------------------------
+
+ /// A directory entry in the registry (or in one of its
+ /// subdirectories) is served like the root: a synthetic directory
+ /// listing the real one, whose sockets dial on walk and whose
+ /// directories recurse. Served on the dispatcher thread, like the
+ /// root.
+ fn handleSynthDir(mg: *Mntgen, req: fuse.Request) error{FuseIo}!void {
+ const u = req.header.unique;
+ const local = req.header.nodeid & mount_node_mask;
+ if (local == 0 or local > max_synth_dirs) return mg.replyError(u, .IO);
+ const slot: usize = @intCast(local - 1);
+ const sd = mg.synths[slot] orelse return mg.replyError(u, .IO);
+ switch (req.header.op()) {
+ .forget, .batch_forget => {
+ // The kernel dropped the dentry; the slot goes with it.
+ mg.trace(" synthetic directory slot {d} forgotten", .{slot});
+ mg.gpa.destroy(sd);
+ mg.synths[slot] = null;
+ },
+ .getattr => {
+ const out = fuse.AttrOut{ .attr = mg.synthAttr(sd.nodeid()) };
+ try mg.reply(u, &.{std.mem.asBytes(&out)});
+ },
+ .lookup => try mg.synthLookup(sd, req),
+ .opendir => {
+ var fh: ?usize = null;
+ for (&mg.root_dirs, 0..) |*dir_slot, i| {
+ if (dir_slot.* == null) {
+ fh = i;
+ break;
+ }
+ }
+ const dir_slot = fh orelse return mg.replyError(u, .MFILE);
+ const list = mg.synthListing(sd) catch {
+ return mg.replyError(u, .IO);
+ };
+ mg.root_dirs[dir_slot] = list;
+ const out = fuse.OpenOut{ .fh = dir_slot };
+ try mg.reply(u, &.{std.mem.asBytes(&out)});
+ },
+ .readdir => {
+ const in = fuse.body(fuse.ReadIn, req) catch return mg.replyError(u, .BADF);
+ if (in.fh >= max_root_dirs) return mg.replyError(u, .BADF);
+ const list = mg.root_dirs[@intCast(in.fh)] orelse return mg.replyError(u, .BADF);
+ const size: usize = @min(in.size, max_write);
+ const used = packDirents(list.entries.items, in.offset, mg.data_buf[0..size]);
+ try mg.reply(u, &.{mg.data_buf[0..used]});
+ },
+ .release, .releasedir => {
+ const in = fuse.body(fuse.ReleaseIn, req) catch return mg.replyError(u, .BADF);
+ if (in.fh < max_root_dirs) {
+ if (mg.root_dirs[@intCast(in.fh)]) |list| {
+ list.deinit(mg.gpa);
+ mg.gpa.destroy(list);
+ mg.root_dirs[@intCast(in.fh)] = null;
+ }
+ }
+ try mg.reply(u, &.{});
+ },
+ .statfs => {
+ const out = fuse.StatfsOut{ .st = .{ .bsize = 4096, .namelen = 255, .frsize = 4096 } };
+ try mg.reply(u, &.{std.mem.asBytes(&out)});
+ },
+ .flush, .fsync, .fsyncdir => try mg.reply(u, &.{}),
+ // Capability probes read as "not supported", like the root's.
+ .access, .setxattr, .getxattr, .listxattr, .removexattr, .statx => try mg.replyError(u, .NOSYS),
+ else => try mg.replyError(u, .PERM),
+ }
+ }
+
+ fn synthAttr(mg: *const Mntgen, nodeid: u64) fuse.Attr {
+ // Read-only like the root and like /srv.
+ return .{
+ .ino = nodeid,
+ .mode = fuse.S_IFDIR | 0o555,
+ .nlink = 2,
+ .uid = mg.opts.uid,
+ .gid = mg.opts.gid,
+ .blksize = 4096,
+ };
+ }
+
+ fn synthLookup(mg: *Mntgen, sd: *SynthDir, req: fuse.Request) error{FuseIo}!void {
+ const u = req.header.unique;
+ const name = fuse.nameAfter(void, req) catch return mg.replyError(u, .INVAL);
+ if (std.mem.eql(u8, name, ".")) {
+ const out = mg.rootEntryOut(sd.nodeid(), mg.synthAttr(sd.nodeid()));
+ return mg.reply(u, &.{std.mem.asBytes(&out)});
+ }
+ // The kernel resolves ".." from its own dentry tree and a LOOKUP of
+ // it has never been observed, but it must not alias the directory
+ // onto itself either: answer with the parent's node id (the root's
+ // for a top-level subdirectory — its attr is the same shape).
+ if (std.mem.eql(u8, name, "..")) {
+ const out = mg.rootEntryOut(sd.parent, mg.synthAttr(sd.parent));
+ return mg.reply(u, &.{std.mem.asBytes(&out)});
+ }
+ if (!post.legalName(name)) return mg.replyError(u, .NOENT);
+ // The registry-relative key this child dials under (a mount's
+ // name, for findMount).
+ var key_buf: [post.sun_path_len]u8 = undefined;
+ const key = std.fmt.bufPrint(&key_buf, "{s}/{s}", .{ sd.rel(), name }) catch
+ return mg.replyError(u, .NOTNAM);
+ var path_buf: [post.sun_path_len]u8 = undefined;
+ const child = std.fmt.bufPrintSentinel(&path_buf, "{s}/{s}", .{ sd.path(), name }, 0) catch
+ return mg.replyError(u, .NOTNAM);
+ if (findMount(mg.mounts, key)) |m| {
+ mg.trace(" lookup '{s}': mount {d} already live", .{ key, m.index });
+ const out = mg.rootEntryOut(m.root_node, m.root_attr);
+ return mg.reply(u, &.{std.mem.asBytes(&out)});
+ }
+ const st = std.Io.Dir.statFile(.cwd(), mg.mo.io, child, .{}) catch |e| switch (e) {
+ error.FileNotFound => {
+ mg.trace(" lookup '{s}': no entry", .{key});
+ return mg.replyError(u, .NOENT);
+ },
+ else => {
+ mg.trace(" lookup '{s}': stat failed: {t}", .{ key, e });
+ return mg.replyError(u, .IO);
+ },
+ };
+ switch (st.kind) {
+ .directory => {
+ const child_sd = mg.newSynth(child, key, sd.depth + 1, sd.nodeid()) catch |e| {
+ mg.trace(" lookup '{s}': no synthetic slot: {t}", .{ key, e });
+ return mg.replyError(u, .IO);
+ };
+ const node = child_sd.nodeid();
+ const out = mg.rootEntryOut(node, mg.synthAttr(node));
+ return mg.reply(u, &.{std.mem.asBytes(&out)});
+ },
+ .unix_domain_socket => {
+ const m = mg.dialMountAt(child, key) catch |e| {
+ mg.trace(" lookup '{s}': dial failed: {t}", .{ key, e });
+ return mg.replyError(u, .IO);
+ };
+ mg.trace(" lookup '{s}': dialed as mount {d}", .{ key, m.index });
+ const out = mg.rootEntryOut(m.root_node, m.root_attr);
+ try mg.reply(u, &.{std.mem.asBytes(&out)});
+ },
+ // Not a service and not a directory: the entry answers EIO on
+ // walk, like a plain file in the registry itself.
+ else => {
+ mg.trace(" lookup '{s}': entry is not a socket or directory", .{key});
+ return mg.replyError(u, .IO);
+ },
+ }
+ }
+
+ /// One OPENDIR of a synthetic subdirectory: `.` and `..` plus the
+ /// real directory's entries, snapshotted for the life of the handle
+ /// (a fresh OPENDIR sees fresh entries), like the root.
+ fn synthListing(mg: *Mntgen, sd: *SynthDir) !*DirList {
+ const list = try mg.gpa.create(DirList);
+ errdefer mg.gpa.destroy(list);
+ list.* = .{};
+ errdefer list.deinit(mg.gpa);
+ const node = sd.nodeid();
+ try list.entries.append(mg.gpa, .{ .name = try mg.gpa.dupe(u8, "."), .ino = node, .dtype = fuse.DT_DIR });
+ try list.entries.append(mg.gpa, .{ .name = try mg.gpa.dupe(u8, ".."), .ino = sd.parent, .dtype = fuse.DT_DIR });
+ var names = post.postedDir(mg.mo.io, sd.path(), mg.stage) catch |e| {
+ mg.trace(" directory listing failed: {t}", .{e});
+ return error.Registry;
+ };
+ while (names.next()) |name| {
+ if (!validDirentName(name)) continue;
+ try list.entries.append(mg.gpa, .{ .name = try mg.gpa.dupe(u8, name), .ino = nameIno(name), .dtype = fuse.DT_DIR });
+ }
+ return list;
+ }
+
+ /// Allocates a synthetic directory node mirroring `path`, keyed by
+ /// the registry-relative `key`, at `depth` under `parent`'s node id.
+ fn newSynth(mg: *Mntgen, path: [:0]const u8, key: []const u8, depth: u8, parent: u64) !*SynthDir {
+ if (depth > max_synth_depth) return error.TooDeep;
+ var slot: ?usize = null;
+ for (&mg.synths, 0..) |*s, i| {
+ if (s.* == null) {
+ slot = i;
+ break;
+ }
+ }
+ const i = slot orelse return error.TooMany;
+ const sd = try mg.gpa.create(SynthDir);
+ errdefer mg.gpa.destroy(sd);
+ sd.* = .{ .slot = i, .parent = parent, .depth = depth };
+ if (path.len + 1 > sd.path_buf.len) return error.NameTooLong;
+ @memcpy(sd.path_buf[0..path.len], path);
+ sd.path_buf[path.len] = 0;
+ sd.path_len = @intCast(path.len);
+ if (key.len > sd.rel_buf.len) return error.NameTooLong;
+ @memcpy(sd.rel_buf[0..key.len], key);
+ sd.rel_len = @intCast(key.len);
+ mg.synths[i] = sd;
+ return sd;
+ }
+
// -- dialing ---------------------------------------------------------------
/// Dials `name` out of the registry, attaches, stats the server root and
@@ -1304,10 +1640,22 @@ const Mntgen = struct {
/// of the LOOKUP that triggered it (so a hung server delays that walk,
/// like it would delay any 9P client).
fn dialMount(mg: *Mntgen, name: []const u8) !*Mount {
+ const stream = try post.dial(mg.mo.io, mg.mo.env, name);
+ return mg.mountStream(stream, name);
+ }
+
+ /// Dials the socket at `path` (a registry subdirectory entry) and
+ /// mounts it under `key`, the registry-relative path — the
+ /// subdirectory analogue of `dialMount`.
+ fn dialMountAt(mg: *Mntgen, path: [:0]const u8, key: []const u8) !*Mount {
+ const stream = try post.dialPath(mg.mo.io, path);
+ return mg.mountStream(stream, key);
+ }
+
+ /// The shared dial tail: session, attach, stat, bridge and worker.
+ fn mountStream(mg: *Mntgen, stream: std.Io.net.Stream, key: []const u8) !*Mount {
if (mg.next_index >= max_mounts) return error.TooManyMounts;
const index: u32 = mg.next_index;
-
- const stream = try post.dial(mg.mo.io, mg.mo.env, name);
const fd: i32 = @intCast(stream.socket.handle);
// The session below does blocking I/O: make sure a dial that left
// the descriptor nonblocking cannot spin its read loop on EAGAIN,
@@ -1369,7 +1717,7 @@ const Mntgen = struct {
.gpa = mg.gpa,
.io = mg.io,
.debug = mg.opts.debug,
- .name = try mg.gpa.dupe(u8, name),
+ .name = try mg.gpa.dupe(u8, key),
.index = index,
.root_node = root_node,
.root_attr = attrFromStat(st, inoFromPath(st.qid.path) ^ b.ino_xor, mg.opts.uid, mg.opts.gid),
@@ -1848,6 +2196,16 @@ test "mntgen node id layout: index in the top bits, local ids below" {
try testing.expect(mount_node_mask == (1 << 32) - 1);
}
+test "mntgen synthetic subdirectory node ids route apart from mounts" {
+ for (1..max_synth_dirs + 1) |i| {
+ const node = mountNode(synth_index, i);
+ try testing.expectEqual(synth_index, mountIndex(node));
+ try testing.expectEqual(i, node & mount_node_mask);
+ }
+ // The reserved index can never be a mount ordinal.
+ try testing.expect(synth_index >= max_mounts);
+}
+
test "mntgen synthetic-root inos are deterministic and name-derived" {
const a = nameIno("alpha");
try testing.expectEqual(a, nameIno("alpha"));