diff options
| author | Gabriel Schneider <[email protected]> | 2026-09-21 14:23:27 -0300 |
|---|---|---|
| committer | Gabriel Schneider <[email protected]> | 2026-09-21 15:20:27 -0300 |
| commit | 0d7e295efee1fca0935cf4a8bee9629c007dd2b6 (patch) | |
| tree | d371eb028c63da5ab5b506bd25ac6579cf8a93fc /src/post.zig | |
| parent | 3a23f6a29e47ace901bd4d82b9db4055fcc12bb9 (diff) | |
| download | cloud9-0d7e295efee1fca0935cf4a8bee9629c007dd2b6.tar.gz cloud9-0d7e295efee1fca0935cf4a8bee9629c007dd2b6.zip | |
9ns --mntgen: registry subdirectories are mount points too
A registry entry that is a directory is now served the way the root is:
a synthetic directory listing the real one, dialing the sockets inside
it on walk and recursing into further directories, to max_synth_depth
(8) levels across max_synth_dirs (64) synthetic nodes. That is the
plan9port mntgen shape and the layout zmx now posts under, so a live
session reads at /mnt/9p/zmx/<name>. Before this a directory in the
registry was dialed like a socket and answered EIO for good.
post gains the two entry points the traversal needs: postedDir (the
registry scan, against any directory) and dialPath (a dial by composed
path, no name validation).
Hardening, each from an attack that broke the code:
- BATCH_FORGET carries entries for many owners and puts 0 in the header
nodeid, so routing it by the header dropped all of them: 32 of 64
synthetic slots leaked in one close burst and the subdirectories that
held them answered EIO forever. distributeForgets unpacks the body and
hands each entry to its owner.
- probe() and connectBlocking() copied a caller's path into the kernel
address with no bound: a path past sun_path overran the 110-byte stack
sockaddr (a panic in Debug, silent corruption in ReleaseFast). Both
refuse it now, probe as `.live` so a claim never deletes what it could
not inspect.
- That bound then caught 9proc's own listener, which handed probe() the
whole 108-byte sun_path array instead of the path inside it. The probe
reads `.live` for anything it cannot ask about, so every stale socket
became AlreadyListening and no server could ever take a dead
predecessor's name back. It passes the path now.
Suites: 87/87 root (+7 post/serve attack regressions), 48/48 9ns,
51+88 9ns integration (+4 traversal and slot-recycling checks), 213/0
9ns adversarial, 60/60 9proc plus its adversarial suites with a new
stale-socket takeover check, freestanding green.
Diffstat (limited to 'src/post.zig')
| -rw-r--r-- | src/post.zig | 138 |
1 files changed, 137 insertions, 1 deletions
diff --git a/src/post.zig b/src/post.zig index 47a7d3d..641970d 100644 --- a/src/post.zig +++ b/src/post.zig @@ -124,6 +124,14 @@ pub const PostedError = PathError || error{NoSpace} || Io.Dir.OpenError || Io.Di pub fn posted(io: Io, env: Env, out: []u8) PostedError!Names { var dir_buf: [std.fs.max_path_bytes]u8 = undefined; const dir_path = try registryDir(env, &dir_buf); + return postedDir(io, dir_path, out); +} + +/// Lists one directory's entry names into `out` and returns an iterator +/// over them — the registry scan, generalized for the registry +/// subdirectories that 9ns mntgen serves. Nothing is dialed; a missing +/// directory lists as empty. +pub fn postedDir(io: Io, dir_path: [:0]const u8, out: []u8) PostedError!Names { const dir = Io.Dir.openDirAbsolute(io, dir_path, .{ .iterate = true }) catch |err| switch (err) { error.FileNotFound, error.NotDir => return .{ .bytes = out[0..0] }, else => return err, @@ -154,6 +162,10 @@ pub const Probe = enum { none, stale, live }; /// `live`, because uncertainty must be owned by the server, never /// resolved by deleting what may be someone's socket. pub fn probe(path: [:0]const u8) Probe { + // A path that cannot fit a `sun_path` cannot be asked about at all; + // like `transport.isListening`, uncertainty is `.live` (occupied) so + // a claim never deletes a name it cannot inspect. + if (path.len >= sun_path_len) return .live; const fd = socketNonblocking() catch return .live; defer _ = linux.close(fd); var addr: linux.sockaddr.un = .{ .path = @splat(0) }; @@ -217,14 +229,37 @@ pub fn dial(io: Io, env: Env, name: []const u8) DialError!Io.net.Stream { error.AccessDenied => return error.AccessDenied, error.Loop => return error.SymLinkLoop, error.NotDir => return error.NotDir, + error.NameTooLong => return error.NameTooLong, + }; + _ = io; + return .{ .socket = .{ .handle = fd, .address = .{ .ip4 = .loopback(0) } } }; +} + +/// Dials the socket at `path` — any registry path, including a +/// subdirectory entry — and returns its stream. The same blocking, +/// close-on-exec and refused-detection semantics as `dial`, with no +/// name validation: the caller composed the path. +pub fn dialPath(io: Io, path: [:0]const u8) DialError!Io.net.Stream { + const fd = connectBlocking(path) catch |err| switch (err) { + error.Socket => return error.SystemResources, + error.Noent => return error.NotPosted, + error.Refused => return error.Stale, + error.AccessDenied => return error.AccessDenied, + error.Loop => return error.SymLinkLoop, + error.NotDir => return error.NotDir, + error.NameTooLong => return error.NameTooLong, }; _ = io; return .{ .socket = .{ .handle = fd, .address = .{ .ip4 = .loopback(0) } } }; } -const ConnectError = error{ Socket, Noent, Refused, AccessDenied, Loop, NotDir }; +const ConnectError = error{ Socket, Noent, Refused, AccessDenied, Loop, NotDir, NameTooLong }; fn connectBlocking(path: [:0]const u8) ConnectError!i32 { + // The caller composed the path (dialPath runs no name validation), so + // it must still fit a `sun_path`; beyond it the copy into the kernel + // address would overrun the stack struct. + if (path.len >= sun_path_len) return error.NameTooLong; const rc = linux.socket(linux.AF.UNIX, linux.SOCK.STREAM | linux.SOCK.CLOEXEC, 0); if (rawErrno(@bitCast(rc)) != .SUCCESS) return error.Socket; const fd: i32 = @intCast(rc); @@ -1022,3 +1057,104 @@ test "post watch: queue overflow surfaces, a deleted registry is gone" { try testing.expectEqualStrings("back", back.?.name); try Io.Dir.deleteFileAbsolute(io, try std.fmt.bufPrint(&name_buf, "{s}/back", .{reg})); } + +test "post probe/dialPath: a path past the sun_path budget is refused, never read" { + // A path longer than a sockaddr's `sun_path` cannot be handed to the + // kernel; the raw probe/dial copies must not overrun the stack + // address struct. `probe` owns the uncertainty as `.live` (never + // delete what cannot be inspected); `dialPath` names the failure. + var buf: [512]u8 = undefined; + const over = sun_path_len + 8; + @memset(buf[0..over], 'a'); + buf[over] = 0; + const p: [:0]const u8 = buf[0..over :0]; + try testing.expectEqual(Probe.live, probe(p)); + try testing.expectError(error.NameTooLong, dialPath(testing.io, p)); + // Exactly `sun_path_len` bytes leaves no room for a zero and is not a + // name either. + buf[sun_path_len] = 0; + const at: [:0]const u8 = buf[0..sun_path_len :0]; + try testing.expectEqual(Probe.live, probe(at)); + try testing.expectError(error.NameTooLong, dialPath(testing.io, at)); +} + +test "post dialPath: refused, missing, non-socket, loop and a CLOEXEC stream" { + if (builtin.os.tag != .linux) return error.SkipZigTest; + var s: Scratch = .{ .dir = undefined }; + try s.start(); + defer s.end(); + const io = testing.io; + var real_buf: [std.fs.max_path_bytes]u8 = undefined; + const rlen = try s.dir.dir.realPath(io, &real_buf); + const root = real_buf[0..rlen]; + var path_buf: [std.fs.max_path_bytes]u8 = undefined; + + // A live socket: dialPath yields a blocking, close-on-exec stream + // (the mounted PROGRAM must not inherit a dialed descriptor). + const sock_path = try std.fmt.bufPrintZ(&path_buf, "{s}/live.sock", .{root}); + var server = try (try Io.net.UnixAddress.init(sock_path)).listen(io, .{}); + { + var st = try dialPath(io, sock_path); + const fd = st.socket.handle; + try testing.expect(linux.fcntl(fd, linux.F.GETFD, 0) & linux.FD_CLOEXEC != 0); + st.close(io); + } + // The server dies without unlinking: refused, and the caller can tell. + server.deinit(io); + try testing.expectError(error.Stale, dialPath(io, sock_path)); + + // No entry at all. + try testing.expectError(error.NotPosted, dialPath(io, try std.fmt.bufPrintZ(&path_buf, "{s}/missing", .{root}))); + + // An entry that is not a socket answers the same ECONNREFUSED the + // kernel gives a dead socket; the caller tells them apart by stat + // before dialing (post.post does exactly that). + try testing.expectError(error.Stale, dialPath(io, try std.fmt.bufPrintZ(&path_buf, "{s}", .{root}))); + const file_path = try std.fmt.bufPrintZ(&path_buf, "{s}/plain", .{root}); + { + var f = try Io.Dir.createFileAbsolute(io, file_path, .{}); + f.close(io); + } + try testing.expectError(error.Stale, dialPath(io, file_path)); + + // A symlink loop is its own error, not a missing entry. + const loop_path = try std.fmt.bufPrintZ(&path_buf, "{s}/loop", .{root}); + try Io.Dir.symLinkAbsolute(io, loop_path, loop_path, .{}); + try testing.expectError(error.SymLinkLoop, dialPath(io, loop_path)); +} + +test "post postedDir: a 1000-entry directory lists fully or refuses cleanly" { + if (builtin.os.tag != .linux) return error.SkipZigTest; + var s: Scratch = .{ .dir = undefined }; + try s.start(); + defer s.end(); + const io = testing.io; + var real_buf: [std.fs.max_path_bytes]u8 = undefined; + const rlen = try s.dir.dir.realPath(io, &real_buf); + var dir_buf: [std.fs.max_path_bytes]u8 = undefined; + const dir = try std.fmt.bufPrintZ(&dir_buf, "{s}", .{real_buf[0..rlen]}); + for (0..1000) |i| { + var name_buf: [32]u8 = undefined; + var path_buf: [std.fs.max_path_bytes]u8 = undefined; + const name = try std.fmt.bufPrint(&name_buf, "e{d:0>4}", .{i}); + var f = try Io.Dir.createFileAbsolute(io, try std.fmt.bufPrintZ(&path_buf, "{s}/{s}", .{ dir, name }), .{}); + f.close(io); + } + // A staging buffer that cannot hold every record is an explicit + // NoSpace — never a silently truncated listing. + var small: [512]u8 = undefined; + try testing.expectError(error.NoSpace, postedDir(io, dir, &small)); + // Big enough lists every entry, with the first and last intact. + var big: [16 * 1024]u8 = undefined; + var names = try postedDir(io, dir, &big); + var count: usize = 0; + var have_first = false; + var have_last = false; + while (names.next()) |name| { + count += 1; + have_first = have_first or std.mem.eql(u8, name, "e0000"); + have_last = have_last or std.mem.eql(u8, name, "e0999"); + } + try testing.expectEqual(@as(usize, 1000), count); + try testing.expect(have_first and have_last); +} |
