From b05abcba3ea09ea106ad28364c6e40a3ec31b890 Mon Sep 17 00:00:00 2001 From: Gabriel Schneider Date: Sat, 19 Sep 2026 21:26:05 -0300 Subject: Add 9player and introspect as programs beside the library 9player/: FUSE mount CLI that mounts a 9P2000 tree into a fresh user+mount namespace and runs a program in it (no root, no libfuse, no libc). introspect/: the 9P debug/introspection library (freestanding core, value renderers, Linux probe with threads/stacks/memory/breakpoints/panics) and its demo server. Each has its own build fragment; the root build.zig wires them behind -D9player/-Dintrospect with namespaced steps (9player-itest, introspect-check-freestanding, programs-test, ...) and exports the introspect module for dependents. This is the layout for related programs. Co-Authored-By: Claude Fable 5.1 --- 9player/src/ns.zig | 1082 ++++++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 1082 insertions(+) create mode 100644 9player/src/ns.zig (limited to '9player/src/ns.zig') diff --git a/9player/src/ns.zig b/9player/src/ns.zig new file mode 100644 index 0000000..2c6f202 --- /dev/null +++ b/9player/src/ns.zig @@ -0,0 +1,1082 @@ +//! Namespace and process plumbing for 9player. +//! +//! Everything here is raw `std.os.linux` syscalls (no libc). The child side +//! of `spawn` runs between `fork` and `execve`; it does not allocate except +//! inside `ensureMountpoint` (the process is single-threaded by then, so the +//! inherited allocator is safe to use). +//! +//! Exit codes produced by the child before exec: 125 for namespace/mount +//! setup failures, 126 when the program was found but is not executable, +//! 127 when it was not found. + +const std = @import("std"); +const builtin = @import("builtin"); +const linux = std.os.linux; +const Allocator = std.mem.Allocator; +const E = linux.E; + +pub const Spawn = struct { + /// argv[0] is PATH-searched unless it contains '/'. + argv: []const []const u8, + /// Inherited environment; `NINEPLAYER_MOUNT` is added or replaced. + envp: [*:null]const ?[*:0]const u8, + /// Absolute mountpoint (see `resolveMountpoint`). + mountpoint: []const u8, + uid: u32, + gid: u32, + max_read: u32, + /// When false the namespace is set up (including mountpoint shadowing) + /// but `/dev/fuse` is not opened and nothing is mounted; `Child.fuse_fd` + /// is then -1. Only for smoke tests. + mount_fuse: bool = true, +}; + +pub const Child = struct { + pid: i32, + /// The `/dev/fuse` connection backing the mount, opened by the child + /// inside its user namespace (the kernel refuses to mount a fuse fd that + /// was opened from another user namespace) and handed back over the + /// status socket with SCM_RIGHTS. Owned by the caller; CLOEXEC. + fuse_fd: i32, + /// Parent end of the status socket. The child reports an exec failure + /// on it (see `reportExecFailure`); it reads EOF once exec succeeded. + status_fd: i32, +}; + +/// Exit status used by the child for setup failures (matches 9player's own). +pub const setup_failure_status: u8 = 125; +/// Refuse to shadow a directory with more entries than this. +pub const max_shadow_entries: usize = 4096; + +const default_path = "/usr/local/bin:/bin:/usr/bin"; +const path_max = 4096; + +// --------------------------------------------------------------------------- +// Mountpoint resolution +// --------------------------------------------------------------------------- + +/// Absolute path (relative paths resolved against cwd), duplicate slashes +/// collapsed, `.` and `..` components resolved lexically, no trailing slash. +/// `/` itself is rejected. +pub fn resolveMountpoint(gpa: Allocator, path: []const u8) ![:0]u8 { + var cwd_buf: [path_max]u8 = undefined; + var cwd: []const u8 = "/"; + if (path.len == 0 or path[0] != '/') { + const rc = linux.getcwd(&cwd_buf, cwd_buf.len); + switch (linux.errno(rc)) { + .SUCCESS => {}, + else => |e| { + std.debug.print("9player: getcwd: E{t}\n", .{e}); + return error.Cwd; + }, + } + // rc counts the terminating NUL. + cwd = cwd_buf[0 .. rc - 1]; + } + return normalizePath(gpa, cwd, path); +} + +/// Pure part of `resolveMountpoint`: `cwd` is only used when `path` is relative. +fn normalizePath(gpa: Allocator, cwd: []const u8, path: []const u8) ![:0]u8 { + if (path.len == 0) return error.InvalidMountpoint; + var out: std.ArrayList(u8) = .empty; + defer out.deinit(gpa); + if (path[0] != '/') try appendComponents(gpa, &out, cwd); + try appendComponents(gpa, &out, path); + if (out.items.len == 0) return error.InvalidMountpoint; // "/" or equivalent + return out.toOwnedSliceSentinel(gpa, 0); +} + +fn appendComponents(gpa: Allocator, out: *std.ArrayList(u8), path: []const u8) !void { + var it = std.mem.tokenizeScalar(u8, path, '/'); + while (it.next()) |comp| { + if (std.mem.eql(u8, comp, ".")) continue; + if (std.mem.eql(u8, comp, "..")) { + // Pop the last component (lexically; "/.." stays "/"). + const idx = std.mem.lastIndexOfScalar(u8, out.items, '/') orelse 0; + out.shrinkRetainingCapacity(idx); + continue; + } + try out.append(gpa, '/'); + try out.appendSlice(gpa, comp); + } +} + +// --------------------------------------------------------------------------- +// Environment helpers +// --------------------------------------------------------------------------- + +/// Look a variable up in a raw envp block. +pub fn getenv(envp: [*:null]const ?[*:0]const u8, name: []const u8) ?[]const u8 { + var i: usize = 0; + while (envp[i]) |entry| : (i += 1) { + const kv = std.mem.span(entry); + if (kv.len > name.len and kv[name.len] == '=' and std.mem.eql(u8, kv[0..name.len], name)) { + return kv[name.len + 1 ..]; + } + } + return null; +} + +/// Every path `execve` should try for `name`, in order: just `name` if it +/// contains a '/', else `/` for each `$PATH` element (an empty +/// element means the current directory; `$PATH` unset falls back to +/// `/usr/local/bin:/bin:/usr/bin`). +pub fn pathCandidates(gpa: Allocator, envp: [*:null]const ?[*:0]const u8, name: []const u8) ![]const [:0]const u8 { + if (name.len == 0) return error.EmptyProgramName; + var list: std.ArrayList([:0]const u8) = .empty; + errdefer { + for (list.items) |c| gpa.free(c); + list.deinit(gpa); + } + if (std.mem.indexOfScalar(u8, name, '/') != null) { + try list.append(gpa, try gpa.dupeZ(u8, name)); + return list.toOwnedSlice(gpa); + } + const path = getenv(envp, "PATH") orelse default_path; + var it = std.mem.splitScalar(u8, path, ':'); + while (it.next()) |dir| { + const d = if (dir.len == 0) "." else dir; + try list.append(gpa, try std.fmt.allocPrintSentinel(gpa, "{s}/{s}", .{ d, name }, 0)); + } + return list.toOwnedSlice(gpa); +} + +/// First PATH candidate that is an executable regular file, or the name +/// itself when it contains a '/'. Provided for completeness; `spawn` simply +/// tries `execve` on every candidate instead. +pub fn findInPath(gpa: Allocator, envp: [*:null]const ?[*:0]const u8, name: []const u8) ![:0]u8 { + const cands = try pathCandidates(gpa, envp, name); + defer { + for (cands) |c| gpa.free(c); + gpa.free(cands); + } + for (cands) |c| { + var stx: linux.Statx = undefined; + const rc = linux.statx(linux.AT.FDCWD, c.ptr, 0, .{ .TYPE = true, .MODE = true }, &stx); + if (linux.errno(rc) != .SUCCESS) continue; + if (stx.mode & linux.S.IFMT != linux.S.IFREG) continue; + if (stx.mode & 0o111 == 0) continue; + return gpa.dupeZ(u8, c); + } + return error.FileNotFound; +} + +/// New envp block: every entry of `envp` except `NINEPLAYER_MOUNT=...`, +/// followed by `NINEPLAYER_MOUNT=`. +fn buildEnvp(gpa: Allocator, envp: [*:null]const ?[*:0]const u8, mountpoint: []const u8) ![:null]?[*:0]const u8 { + const key = "NINEPLAYER_MOUNT="; + var keep: usize = 0; + var i: usize = 0; + while (envp[i]) |entry| : (i += 1) { + if (!std.mem.startsWith(u8, std.mem.span(entry), key)) keep += 1; + } + const out = try gpa.allocSentinel(?[*:0]const u8, keep + 1, null); + errdefer gpa.free(out); + var j: usize = 0; + i = 0; + while (envp[i]) |entry| : (i += 1) { + if (std.mem.startsWith(u8, std.mem.span(entry), key)) continue; + out[j] = entry; + j += 1; + } + const mount_entry = try std.fmt.allocPrintSentinel(gpa, key ++ "{s}", .{mountpoint}, 0); + out[j] = mount_entry.ptr; + return out; +} + +fn buildArgv(gpa: Allocator, argv: []const []const u8) ![:null]?[*:0]const u8 { + const out = try gpa.allocSentinel(?[*:0]const u8, argv.len, null); + for (argv, 0..) |a, i| out[i] = (try gpa.dupeZ(u8, a)).ptr; + return out; +} + +// --------------------------------------------------------------------------- +// Mountpoint policy +// --------------------------------------------------------------------------- + +/// Make sure `path` is a directory, inside the *current* mount namespace: +/// +/// * already a directory → done; +/// * else `mkdir`; on `EACCES`/`EPERM`/`EROFS` shadow the parent directory +/// with a tmpfs that re-exposes every existing entry (bind mounts for +/// directories and files, recreated symlinks) and `mkdir` inside it; +/// * anything else fails with the errno and a hint. +/// +/// Every failure prints `9player: : E` to stderr before +/// returning. Meant to be called in the child of `spawn` (or from a +/// throwaway namespace: `unshare -Urm`). +pub fn ensureMountpoint(gpa: Allocator, path: [:0]const u8) !void { + if (fileType(linux.AT.FDCWD, path, false)) |ft| { + if (ft == .dir) return; + std.debug.print("9player: mountpoint {s}: exists but is not a directory\n", .{path}); + return error.Mountpoint; + } + if (fileType(linux.AT.FDCWD, path, true) == .symlink) { + std.debug.print("9player: mountpoint {s}: dangling symlink\n", .{path}); + return error.Mountpoint; + } + const mk = linux.errno(linux.mkdirat(linux.AT.FDCWD, path, 0o755)); + switch (mk) { + .SUCCESS => return, + .ACCES, .PERM, .ROFS => {}, + else => |e| { + std.debug.print("9player: mkdir {s}: E{t} (pass --mount an existing directory)\n", .{ path, e }); + return error.Mountpoint; + }, + } + const parent = std.fs.path.dirname(path) orelse "/"; + if (std.mem.eql(u8, parent, "/") or isSameDirectory(parent, "/")) { + std.debug.print("9player: mkdir {s}: E{t}; refusing to shadow / (pass --mount an existing directory)\n", .{ path, mk }); + return error.Mountpoint; + } + // The shadow rebuilds entries from /proc/self/fd//; a tmpfs + // over /proc (or a subtree of it) would take that away from itself. + if (std.mem.eql(u8, parent, "/proc") or std.mem.startsWith(u8, parent, "/proc/")) { + std.debug.print("9player: mkdir {s}: E{t}; refusing to shadow {s} (pass --mount an existing directory)\n", .{ path, mk, parent }); + return error.Mountpoint; + } + const parent_z = try gpa.dupeZ(u8, parent); + defer gpa.free(parent_z); + try shadowDirectory(gpa, parent_z); + switch (linux.errno(linux.mkdirat(linux.AT.FDCWD, path, 0o755))) { + .SUCCESS => {}, + else => |e| { + std.debug.print("9player: mkdir {s} (in shadow tmpfs): E{t}\n", .{ path, e }); + return error.Mountpoint; + }, + } +} + +const FileType = enum { dir, symlink, other }; + +/// True when both paths resolve (following symlinks, including magic ones +/// such as /proc/self/root) to the same inode. +fn isSameDirectory(a: []const u8, b: [*:0]const u8) bool { + var a_buf: [path_max]u8 = undefined; + const a_z = std.fmt.bufPrintZ(&a_buf, "{s}", .{a}) catch return false; + var sa: linux.Statx = undefined; + var sb: linux.Statx = undefined; + if (linux.errno(linux.statx(linux.AT.FDCWD, a_z, 0, .{ .INO = true }, &sa)) != .SUCCESS) return false; + if (linux.errno(linux.statx(linux.AT.FDCWD, b, 0, .{ .INO = true }, &sb)) != .SUCCESS) return false; + return sa.ino == sb.ino and sa.dev_major == sb.dev_major and sa.dev_minor == sb.dev_minor; +} + +fn fileType(dirfd: i32, name: [*:0]const u8, nofollow: bool) ?FileType { + var stx: linux.Statx = undefined; + const flags: u32 = if (nofollow) linux.AT.SYMLINK_NOFOLLOW else 0; + const rc = linux.statx(dirfd, name, flags, .{ .TYPE = true }, &stx); + if (linux.errno(rc) != .SUCCESS) return null; + return switch (stx.mode & linux.S.IFMT) { + linux.S.IFDIR => .dir, + linux.S.IFLNK => .symlink, + else => .other, + }; +} + +const Entry = struct { name: [:0]u8, kind: FileType }; + +/// Read every entry of the directory open at `fd` (excluding `.` and `..`). +fn listDir(gpa: Allocator, fd: i32, dirpath: []const u8) ![]Entry { + var list: std.ArrayList(Entry) = .empty; + errdefer { + for (list.items) |e| gpa.free(e.name); + list.deinit(gpa); + } + var buf: [32 * 1024]u8 align(@alignOf(linux.dirent64)) = undefined; + while (true) { + const rc = linux.getdents64(fd, &buf, buf.len); + switch (linux.errno(rc)) { + .SUCCESS => {}, + else => |e| { + std.debug.print("9player: getdents64 {s}: E{t}\n", .{ dirpath, e }); + return error.Mountpoint; + }, + } + if (rc == 0) break; + var off: usize = 0; + while (off < rc) { + const d: *align(1) const linux.dirent64 = @ptrCast(&buf[off]); + const name_ptr: [*:0]const u8 = @ptrCast(&buf[off + @offsetOf(linux.dirent64, "name")]); + const name = std.mem.span(name_ptr); + const dtype = d.type; + off += d.reclen; + if (std.mem.eql(u8, name, ".") or std.mem.eql(u8, name, "..")) continue; + if (list.items.len >= max_shadow_entries) { + std.debug.print("9player: refusing to shadow {s}: more than {d} entries\n", .{ dirpath, max_shadow_entries }); + return error.TooManyEntries; + } + const kind: FileType = switch (dtype) { + linux.DT.DIR => .dir, + linux.DT.LNK => .symlink, + linux.DT.UNKNOWN => fileType(fd, name_ptr, true) orelse .other, + else => .other, + }; + try list.append(gpa, .{ .name = try gpa.dupeZ(u8, name), .kind = kind }); + } + } + return list.toOwnedSlice(gpa); +} + +fn shadowDirectory(gpa: Allocator, parent: [:0]const u8) !void { + const open_rc = linux.open(parent, .{ .ACCMODE = .RDONLY, .DIRECTORY = true, .CLOEXEC = true }, 0); + switch (linux.errno(open_rc)) { + .SUCCESS => {}, + else => |e| { + std.debug.print("9player: open {s}: E{t}\n", .{ parent, e }); + return error.Mountpoint; + }, + } + const pfd: i32 = @intCast(open_rc); + defer _ = linux.close(pfd); + + const entries = try listDir(gpa, pfd, parent); + defer { + for (entries) |e| gpa.free(e.name); + gpa.free(entries); + } + + const tmpfs_opts: [*:0]const u8 = "mode=755"; + switch (linux.errno(linux.mount("tmpfs", parent, "tmpfs", linux.MS.NOSUID | linux.MS.NODEV, @intFromPtr(tmpfs_opts)))) { + .SUCCESS => {}, + else => |e| { + std.debug.print("9player: mount tmpfs on {s}: E{t}\n", .{ parent, e }); + return error.Mountpoint; + }, + } + + // `pfd` still refers to the original directory underneath the tmpfs, so + // `/proc/self/fd//` reaches the hidden entries. + var src_buf: [path_max]u8 = undefined; + var dst_buf: [path_max]u8 = undefined; + var link_buf: [path_max]u8 = undefined; + for (entries) |e| { + const src = std.fmt.bufPrintZ(&src_buf, "/proc/self/fd/{d}/{s}", .{ pfd, e.name }) catch { + std.debug.print("9player: shadow {s}/{s}: name too long (skipped)\n", .{ parent, e.name }); + continue; + }; + const dst = std.fmt.bufPrintZ(&dst_buf, "{s}/{s}", .{ parent, e.name }) catch { + std.debug.print("9player: shadow {s}/{s}: name too long (skipped)\n", .{ parent, e.name }); + continue; + }; + switch (e.kind) { + .dir => { + if (!check("mkdir", dst, linux.mkdirat(linux.AT.FDCWD, dst, 0o755))) continue; + _ = check("bind", dst, linux.mount(src, dst, null, linux.MS.BIND | linux.MS.REC, 0)); + }, + .symlink => { + const rc = linux.readlinkat(pfd, e.name, &link_buf, link_buf.len - 1); + if (!check("readlink", dst, rc)) continue; + link_buf[rc] = 0; + const target: [*:0]const u8 = @ptrCast(&link_buf); + _ = check("symlink", dst, linux.symlinkat(target, linux.AT.FDCWD, dst)); + }, + .other => { + const rc = linux.openat(linux.AT.FDCWD, dst, .{ .ACCMODE = .WRONLY, .CREAT = true, .CLOEXEC = true }, 0o644); + if (!check("create", dst, rc)) continue; + _ = linux.close(@intCast(rc)); + _ = check("bind", dst, linux.mount(src, dst, null, linux.MS.BIND | linux.MS.REC, 0)); + }, + } + } +} + +/// Report a failed per-entry step as a warning (the entry is skipped; the +/// rest of the shadow is still useful). Returns true on success. +fn check(step: []const u8, path: [*:0]const u8, rc: usize) bool { + switch (linux.errno(rc)) { + .SUCCESS => return true, + else => |e| { + std.debug.print("9player: shadow: {s} {s}: E{t} (skipped)\n", .{ step, std.mem.span(path), e }); + return false; + }, + } +} + +// --------------------------------------------------------------------------- +// spawn +// --------------------------------------------------------------------------- + +const ChildArgs = struct { + gpa: Allocator, + status_sock: i32, + mountpoint: [:0]const u8, + fuse_opts_prefix: [:0]const u8, // everything after "fd=," + mount_fuse: bool, + uid_map: []const u8, + gid_map: []const u8, + argv: [:null]?[*:0]const u8, + envp: [:null]?[*:0]const u8, + candidates: []const [:0]const u8, + name: []const u8, +}; + +/// Status channel protocol (child → parent, over a CLOEXEC socketpair): +/// a 0 byte means "namespace and mount are up" and carries the fuse fd as +/// SCM_RIGHTS; a non-zero byte is an exit status followed by a message. +/// EOF ends the conversation (exec succeeded, or the child died). +const ok_byte: u8 = 0; + +/// fork; the child unshares user+mount namespaces, maps its uid/gid, +/// makes `/` private, ensures the mountpoint, opens `/dev/fuse`, mounts it +/// on the mountpoint and sends the fd back. `spawn` returns at that point +/// (with `error.ChildFailed` and a message on stderr if any step failed). +/// The child then stats the mountpoint, which makes the kernel fetch the +/// root's attributes once the parent serves (the kernel seeds the fuse root +/// with uid 0, unmapped in the new user namespace, so nothing could be +/// created in the root until then), sets `NINEPLAYER_MOUNT` and execs +/// `argv`. An exec failure is reported on `Child.status_fd` and ends the +/// child with 126/127; collect it with `reportExecFailure` after +/// `bridge.serve` returns. +/// +/// If `installSignals` was called, the pid is stored into the registered +/// variable as soon as fork returns so no SIGCHLD can be missed. +pub fn spawn(gpa: Allocator, s: Spawn) !Child { + if (s.argv.len == 0 or s.argv[0].len == 0) { + std.debug.print("9player: empty program name\n", .{}); + return error.EmptyProgramName; + } + + const mountpoint = try gpa.dupeZ(u8, s.mountpoint); + defer gpa.free(mountpoint); + const fuse_opts_prefix = try std.fmt.allocPrintSentinel(gpa, "rootmode=40000,user_id={d},group_id={d},max_read={d}", .{ s.uid, s.gid, s.max_read }, 0); + defer gpa.free(fuse_opts_prefix); + var uid_buf: [64]u8 = undefined; + var gid_buf: [64]u8 = undefined; + const uid_map = try std.fmt.bufPrint(&uid_buf, "{d} {d} 1\n", .{ s.uid, s.uid }); + const gid_map = try std.fmt.bufPrint(&gid_buf, "{d} {d} 1\n", .{ s.gid, s.gid }); + const argv = try buildArgv(gpa, s.argv); + defer { + for (argv) |a| gpa.free(std.mem.span(a.?)); + gpa.free(argv); + } + const envp = try buildEnvp(gpa, s.envp, s.mountpoint); + defer { + gpa.free(std.mem.span(envp[envp.len - 1].?)); // the NINEPLAYER_MOUNT entry we created + gpa.free(envp); + } + const candidates = try pathCandidates(gpa, s.envp, s.argv[0]); + defer { + for (candidates) |c| gpa.free(c); + gpa.free(candidates); + } + + var sv: [2]i32 = undefined; + switch (linux.errno(linux.socketpair(linux.AF.UNIX, linux.SOCK.STREAM | linux.SOCK.CLOEXEC, 0, &sv))) { + .SUCCESS => {}, + else => |e| { + std.debug.print("9player: socketpair: E{t}\n", .{e}); + return error.SystemResources; + }, + } + + const child_args = ChildArgs{ + .gpa = gpa, + .status_sock = sv[1], + .mountpoint = mountpoint, + .fuse_opts_prefix = fuse_opts_prefix, + .mount_fuse = s.mount_fuse, + .uid_map = uid_map, + .gid_map = gid_map, + .argv = argv, + .envp = envp, + .candidates = candidates, + .name = s.argv[0], + }; + + const fork_rc = linux.fork(); + switch (linux.errno(fork_rc)) { + .SUCCESS => {}, + else => |e| { + _ = linux.close(sv[0]); + _ = linux.close(sv[1]); + std.debug.print("9player: fork: E{t}\n", .{e}); + return error.SystemResources; + }, + } + if (fork_rc == 0) childMain(&child_args); + + const pid: i32 = @intCast(fork_rc); + if (child_pid_ptr) |p| @atomicStore(i32, p, pid, .seq_cst); + _ = linux.close(sv[1]); + + // First byte: ok (with the fuse fd attached) or a failure status. + var first: [1]u8 = .{ok_byte}; // defined even if recvmsg stores nothing + var fuse_fd: i32 = -1; + var n: usize = 0; + while (true) { + const rc = recvWithFd(sv[0], &first, &fuse_fd, 0); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .INTR => continue, + else => break, + } + n = rc; + break; + } + if (n == 1 and first[0] == ok_byte and (fuse_fd >= 0 or !s.mount_fuse)) { + return .{ .pid = pid, .fuse_fd = fuse_fd, .status_fd = sv[0] }; + } + + // Failure. A status byte means the child is exiting on its own and a + // message follows. Anything else (EOF: the child died before reporting; + // an ok byte without the fd: the SCM_RIGHTS transfer was truncated, e.g. + // EMFILE) is a protocol violation: the child may be about to exec with a + // dead mount, so kill it before waiting rather than reading the status + // socket until an exec'd program eventually exits. + const reported = n == 1 and first[0] != ok_byte; + if (!reported) _ = linux.kill(pid, .KILL); + if (fuse_fd >= 0) _ = linux.close(fuse_fd); + var msg: [512]u8 = undefined; + var len: usize = 0; + while (reported and len < msg.len) { + const rc = linux.read(sv[0], msg[len..].ptr, msg.len - len); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .INTR => continue, + else => break, + } + if (rc == 0) break; + len += rc; + } + _ = linux.close(sv[0]); + if (reported) { + std.debug.print("9player: {s}\n", .{msg[0..len]}); + } else if (n == 1) { + std.debug.print("9player: child handshake failed: no fuse fd received (out of file descriptors?)\n", .{}); + } else { + std.debug.print("9player: child exited before reporting\n", .{}); + } + _ = waitChild(pid) catch {}; + if (child_pid_ptr) |p| @atomicStore(i32, p, 0, .seq_cst); + const status: u8 = if (reported) first[0] else setup_failure_status; + return switch (status) { + 126 => error.ExecPermission, + 127 => error.ExecNotFound, + else => error.ChildFailed, + }; +} + +/// After the child is gone (or the mount is dead): print the exec failure +/// the child reported on `status_fd`, if any, and close it. Returns the +/// status byte the child announced, or null when exec succeeded / nothing +/// was reported. Never blocks. +pub fn reportExecFailure(child: Child) ?u8 { + defer _ = linux.close(child.status_fd); + var msg: [512]u8 = undefined; + var len: usize = 0; + while (len < msg.len) { + var iov = [_]std.posix.iovec{.{ .base = msg[len..].ptr, .len = msg.len - len }}; + var hdr = linux.msghdr{ + .name = null, + .namelen = 0, + .iov = &iov, + .iovlen = 1, + .control = null, + .controllen = 0, + .flags = 0, + }; + const rc = linux.recvmsg(child.status_fd, &hdr, linux.MSG.DONTWAIT); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .INTR => continue, + else => break, + } + if (rc == 0) break; + len += rc; + } + if (len == 0) return null; + std.debug.print("9player: {s}\n", .{msg[1..len]}); + return msg[0]; +} + +const cmsg_fd_len = @sizeOf(linux.cmsghdr) + @sizeOf(i32); +const cmsg_fd_space = std.mem.alignForward(usize, cmsg_fd_len, @sizeOf(usize)); + +/// sendmsg one data byte, optionally with `fd` attached as SCM_RIGHTS. +fn sendWithFd(sock: i32, byte: u8, fd: ?i32) usize { + const data = [_]u8{byte}; + const iov = [_]std.posix.iovec_const{.{ .base = &data, .len = 1 }}; + var cbuf: [cmsg_fd_space]u8 align(@alignOf(linux.cmsghdr)) = @splat(0); + var msg = linux.msghdr_const{ + .name = null, + .namelen = 0, + .iov = &iov, + .iovlen = 1, + .control = null, + .controllen = 0, + .flags = 0, + }; + if (fd) |f| { + const hdr: *linux.cmsghdr = @ptrCast(&cbuf); + hdr.* = .{ .len = cmsg_fd_len, .level = linux.SOL.SOCKET, .type = linux.SCM.RIGHTS }; + @memcpy(cbuf[@sizeOf(linux.cmsghdr)..][0..@sizeOf(i32)], std.mem.asBytes(&f)); + msg.control = &cbuf; + msg.controllen = cmsg_fd_space; + } + return linux.sendmsg(sock, &msg, linux.MSG.NOSIGNAL); +} + +/// recvmsg into `buf`; an SCM_RIGHTS fd, if any, is stored in `fd_out`. +fn recvWithFd(sock: i32, buf: []u8, fd_out: *i32, flags: u32) usize { + var iov = [_]std.posix.iovec{.{ .base = buf.ptr, .len = buf.len }}; + var cbuf: [cmsg_fd_space]u8 align(@alignOf(linux.cmsghdr)) = @splat(0); + var msg = linux.msghdr{ + .name = null, + .namelen = 0, + .iov = &iov, + .iovlen = 1, + .control = &cbuf, + .controllen = cbuf.len, + .flags = 0, + }; + const rc = linux.recvmsg(sock, &msg, linux.MSG.CMSG_CLOEXEC | flags); + if (linux.errno(rc) != .SUCCESS) return rc; + if (msg.controllen >= cmsg_fd_len) { + const hdr: *const linux.cmsghdr = @ptrCast(&cbuf); + if (hdr.level == linux.SOL.SOCKET and hdr.type == linux.SCM.RIGHTS and hdr.len >= cmsg_fd_len) { + var fd: i32 = undefined; + @memcpy(std.mem.asBytes(&fd), cbuf[@sizeOf(linux.cmsghdr)..][0..@sizeOf(i32)]); + fd_out.* = fd; + } + } + return rc; +} + +/// Child side of `spawn`. Never returns. +fn childMain(c: *const ChildArgs) noreturn { + resetSignals(); + + const rc_unshare = linux.errno(linux.unshare(linux.CLONE.NEWUSER | linux.CLONE.NEWNS)); + if (rc_unshare != .SUCCESS) childFail(c, setup_failure_status, "unshare(CLONE_NEWUSER|CLONE_NEWNS)", rc_unshare, true); + writeProcFile(c, "/proc/self/setgroups", "deny", true); + writeProcFile(c, "/proc/self/uid_map", c.uid_map, false); + writeProcFile(c, "/proc/self/gid_map", c.gid_map, false); + + const root: [*:0]const u8 = "/"; + const rc_priv = linux.mount(null, root, null, linux.MS.REC | linux.MS.PRIVATE, 0); + if (linux.errno(rc_priv) != .SUCCESS) childFail(c, setup_failure_status, "mount(/, MS_REC|MS_PRIVATE)", linux.errno(rc_priv), true); + + ensureMountpoint(c.gpa, c.mountpoint) catch { + childFail(c, setup_failure_status, "mountpoint setup failed (pass --mount an existing directory)", .SUCCESS, false); + }; + + var fuse_fd: ?i32 = null; + if (c.mount_fuse) { + // Must be opened here, after unshare: the kernel only mounts a fuse + // device opened from the mount's own user namespace. + const rc_open = linux.open("/dev/fuse", .{ .ACCMODE = .RDWR, .CLOEXEC = true }, 0); + switch (linux.errno(rc_open)) { + .SUCCESS => {}, + .NOENT => childFail(c, setup_failure_status, "open /dev/fuse: ENOENT (is the fuse module loaded? try: modprobe fuse)", .SUCCESS, false), + else => |e| childFail(c, setup_failure_status, "open /dev/fuse", e, true), + } + const fd: i32 = @intCast(rc_open); + var opts_buf: [256]u8 = undefined; + const opts = std.fmt.bufPrintZ(&opts_buf, "fd={d},{s}", .{ fd, c.fuse_opts_prefix }) catch unreachable; + const rc = linux.mount("9player", c.mountpoint, "fuse", linux.MS.NOSUID | linux.MS.NODEV, @intFromPtr(opts.ptr)); + if (linux.errno(rc) != .SUCCESS) childFail(c, setup_failure_status, "mount fuse", linux.errno(rc), true); + fuse_fd = fd; + } + const sent = sendWithFd(c.status_sock, ok_byte, fuse_fd); + if (linux.errno(sent) != .SUCCESS) linux.exit_group(setup_failure_status); + if (fuse_fd) |fd| { + _ = linux.close(fd); // the parent holds the connection now + // Force one GETATTR of the root (served by the parent, which is + // entering its serve loop now); see `spawn`. Errors don't matter. + var stx: linux.Statx = undefined; + _ = linux.statx(linux.AT.FDCWD, c.mountpoint, 0, .{ .TYPE = true }, &stx); + } + + var last: E = .NOENT; + var saw_acces = false; + for (c.candidates) |cand| { + const rc = linux.execve(cand.ptr, c.argv.ptr, c.envp.ptr); + last = linux.errno(rc); + switch (last) { + .NOENT, .NOTDIR, .LOOP, .NAMETOOLONG => continue, + .ACCES => { + saw_acces = true; + continue; + }, + else => break, + } + } + var buf: [512]u8 = undefined; + // "Not found" covers every candidate that could not even be resolved + // (a PATH element that is a file gives ENOTDIR, a symlink loop ELOOP); + // a candidate that existed but was not executable wins over those. + const not_found = switch (last) { + .NOENT, .NOTDIR, .LOOP, .NAMETOOLONG => true, + else => false, + }; + if (not_found and saw_acces) last = .ACCES; + const status: u8 = if (not_found and !saw_acces) 127 else 126; + const text = std.fmt.bufPrint(&buf, "exec {s}", .{c.name}) catch "exec"; + childFail(c, status, text, last, true); +} + +fn writeProcFile(c: *const ChildArgs, path: [*:0]const u8, data: []const u8, ignore_missing: bool) void { + const rc = linux.open(path, .{ .ACCMODE = .WRONLY, .CLOEXEC = true }, 0); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .NOENT => if (ignore_missing) return else childFail(c, setup_failure_status, std.mem.span(path), .NOENT, true), + else => |e| childFail(c, setup_failure_status, std.mem.span(path), e, true), + } + const fd: i32 = @intCast(rc); + const w = linux.write(fd, data.ptr, data.len); + const we = linux.errno(w); + _ = linux.close(fd); + if (we != .SUCCESS) childFail(c, setup_failure_status, std.mem.span(path), we, true); + if (w != data.len) childFail(c, setup_failure_status, std.mem.span(path), .IO, true); +} + +/// Write `[: E]` to the status socket and exit. +fn childFail(c: *const ChildArgs, status: u8, step: []const u8, e: E, with_errno: bool) noreturn { + var buf: [600]u8 = undefined; + buf[0] = status; + const rest = if (with_errno) + std.fmt.bufPrint(buf[1..], "{s}: E{t}", .{ step, e }) catch buf[1..1] + else + std.fmt.bufPrint(buf[1..], "{s}", .{step}) catch buf[1..1]; + const msg = buf[0 .. 1 + rest.len]; + var off: usize = 0; + while (off < msg.len) { + const rc = linux.write(c.status_sock, msg[off..].ptr, msg.len - off); + if (linux.errno(rc) == .INTR) continue; + if (linux.errno(rc) != .SUCCESS) break; + off += rc; + } + linux.exit_group(status); +} + +// --------------------------------------------------------------------------- +// Signals +// --------------------------------------------------------------------------- + +var child_pid_ptr: ?*i32 = null; +var chld_pipe_w: i32 = -1; +var reaped = std.atomic.Value(bool).init(false); +var reaped_status = std.atomic.Value(u32).init(0); +/// A second child (the `--spawn` server) that the SIGCHLD handler reaps so +/// it does not linger as a zombie when it dies mid-session. Its exit does +/// not stop the serve loop. 0 = none. +var server_pid = std.atomic.Value(i32).init(0); + +/// Register the `--spawn` server for reaping by the SIGCHLD handler. +pub fn watchServer(pid: i32) void { + server_pid.store(pid, .seq_cst); +} + +/// Seconds the serve loop gets to come back after the child died before +/// the watchdog ends the process anyway. +pub const exit_grace_seconds: isize = 3; + +/// The watched child is already dead but the serve loop has not come back +/// (it is stuck in a 9P request the server never answers): a terminal +/// signal, or the watchdog armed by `onChld`, then ends 9player with the +/// child's status instead of hanging. Nothing is lost: the mount is torn +/// down when the process exits. +fn bailIfChildGone() void { + if (!reaped.load(.acquire)) return; + const srv = server_pid.load(.seq_cst); + if (srv > 0) _ = linux.kill(srv, .TERM); + linux.exit_group(decodeStatus(reaped_status.load(.acquire))); +} + +fn armWatchdog() void { + // setitimer takes an itimerval; std declares it with itimerspec, which + // has the same layout on 64-bit targets (the sub-second field is 0). + const t = linux.itimerspec{ + .it_interval = .{ .sec = 0, .nsec = 0 }, + .it_value = .{ .sec = exit_grace_seconds, .nsec = 0 }, + }; + _ = linux.setitimer(@intFromEnum(linux.ITIMER.REAL), &t, null); +} + +fn onAlarm(_: linux.SIG) callconv(.c) void { + bailIfChildGone(); +} + +fn onForward(sig: linux.SIG) callconv(.c) void { + const p = child_pid_ptr orelse return; + const pid = @atomicLoad(i32, p, .seq_cst); + if (pid > 0) _ = linux.kill(pid, sig); + bailIfChildGone(); +} + +/// SIGINT/SIGQUIT: the child owns the tty and gets them itself; we only +/// react when the child is already gone (see `bailIfChildGone`). +fn onTerminal(_: linux.SIG) callconv(.c) void { + bailIfChildGone(); +} + +/// Only the watched child counts: reap it here (WNOHANG), remember its +/// status, forget its pid (so a later SIGTERM cannot hit a recycled pid) +/// and poke the self-pipe. The `--spawn` server is reaped too but does not +/// interrupt `bridge.serve`; SIGCHLD from anything else is ignored. +fn onChld(_: linux.SIG) callconv(.c) void { + const srv = server_pid.load(.seq_cst); + if (srv > 0) { + var sst: u32 = 0; + const src = linux.waitpid(srv, &sst, linux.W.NOHANG); + if (linux.errno(src) == .SUCCESS and src != 0) server_pid.store(0, .seq_cst); + } + const p = child_pid_ptr orelse return; + const pid = @atomicLoad(i32, p, .seq_cst); + if (pid <= 0) return; + var st: u32 = 0; + const rc = linux.waitpid(pid, &st, linux.W.NOHANG); + if (linux.errno(rc) != .SUCCESS or rc == 0) return; + reaped_status.store(st, .release); + reaped.store(true, .release); + @atomicStore(i32, p, 0, .seq_cst); + const b = [_]u8{'c'}; + _ = linux.write(chld_pipe_w, &b, 1); + armWatchdog(); +} + +/// SIGPIPE ignored; SIGINT/SIGQUIT effectively ignored (the child owns the +/// tty) unless the child is already dead; SIGTERM/SIGHUP forwarded to +/// `*child_pid`; SIGCHLD for `*child_pid` reaps it, writes a byte to a +/// nonblocking self-pipe whose read end is returned (use it as `stop_fd`) +/// and arms a watchdog (`exit_grace_seconds`, SIGALRM) that ends the +/// process with the child's status should the serve loop stay blocked. +/// `*child_pid` is filled in by `spawn`. +pub fn installSignals(child_pid: *i32) !i32 { + child_pid_ptr = child_pid; + var fds: [2]i32 = undefined; + switch (linux.errno(linux.pipe2(&fds, .{ .CLOEXEC = true, .NONBLOCK = true }))) { + .SUCCESS => {}, + else => |e| { + std.debug.print("9player: pipe2: E{t}\n", .{e}); + return error.SystemResources; + }, + } + chld_pipe_w = fds[1]; + + const ign = linux.Sigaction{ .handler = .{ .handler = linux.SIG.IGN }, .mask = linux.sigemptyset(), .flags = 0 }; + const term = linux.Sigaction{ .handler = .{ .handler = &onTerminal }, .mask = linux.sigemptyset(), .flags = linux.SA.RESTART }; + const fwd = linux.Sigaction{ .handler = .{ .handler = &onForward }, .mask = linux.sigemptyset(), .flags = linux.SA.RESTART }; + const chld = linux.Sigaction{ .handler = .{ .handler = &onChld }, .mask = linux.sigemptyset(), .flags = linux.SA.RESTART | linux.SA.NOCLDSTOP }; + const alrm = linux.Sigaction{ .handler = .{ .handler = &onAlarm }, .mask = linux.sigemptyset(), .flags = linux.SA.RESTART }; + std.posix.sigaction(.INT, &term, null); + std.posix.sigaction(.QUIT, &term, null); + std.posix.sigaction(.ALRM, &alrm, null); + std.posix.sigaction(.PIPE, &ign, null); + std.posix.sigaction(.TERM, &fwd, null); + std.posix.sigaction(.HUP, &fwd, null); + std.posix.sigaction(.CHLD, &chld, null); + return fds[0]; +} + +/// Restore default dispositions in the child before exec (ignored signals +/// would otherwise survive execve). +fn resetSignals() void { + const dfl = linux.Sigaction{ .handler = .{ .handler = linux.SIG.DFL }, .mask = linux.sigemptyset(), .flags = 0 }; + inline for (.{ linux.SIG.INT, linux.SIG.QUIT, linux.SIG.PIPE, linux.SIG.TERM, linux.SIG.HUP, linux.SIG.CHLD, linux.SIG.ALRM }) |sig| { + _ = linux.sigaction(sig, &dfl, null); + } +} + +// --------------------------------------------------------------------------- +// Waiting +// --------------------------------------------------------------------------- + +fn takeReaped() ?u32 { + if (!reaped.load(.acquire)) return null; + return reaped_status.load(.acquire); +} + +/// waitpid status → exit code (`128+sig` when killed by a signal). +pub fn decodeStatus(st: u32) u8 { + if (linux.W.IFEXITED(st)) return linux.W.EXITSTATUS(st); + if (linux.W.IFSIGNALED(st)) return 128 +% @as(u8, @truncate(@intFromEnum(linux.W.TERMSIG(st)))); + return 1; +} + +/// Block until `pid` exits (the SIGCHLD handler may have reaped it already). +pub fn waitChild(pid: i32) !u8 { + while (true) { + if (takeReaped()) |st| return decodeStatus(st); + var st: u32 = 0; + const rc = linux.waitpid(pid, &st, 0); + switch (linux.errno(rc)) { + .SUCCESS => return decodeStatus(st), + .INTR => continue, + .CHILD => { + if (takeReaped()) |s| return decodeStatus(s); + return error.NoChild; + }, + else => |e| { + std.debug.print("9player: waitpid: E{t}\n", .{e}); + return error.Wait; + }, + } + } +} + +/// Non-blocking: the exit status of `pid` if it has exited, else null. +pub fn reapIfExited(pid: i32) ?u8 { + if (takeReaped()) |st| return decodeStatus(st); + var st: u32 = 0; + const rc = linux.waitpid(pid, &st, linux.W.NOHANG); + switch (linux.errno(rc)) { + .SUCCESS => return if (rc == 0) null else decodeStatus(st), + .CHILD => return if (takeReaped()) |s| decodeStatus(s) else null, + else => return null, + } +} + +/// Reap any child (used for the `--spawn` server at exit). Non-blocking. +pub fn reapAny(pid: i32) void { + var st: u32 = 0; + _ = linux.waitpid(pid, &st, linux.W.NOHANG); +} + +// --------------------------------------------------------------------------- +// Tests (no namespaces needed; `ensureMountpoint` is exercised by +// test/integration.sh through the 9player binary) +// --------------------------------------------------------------------------- + +const testing = std.testing; + +test "normalizePath: absolute paths" { + const gpa = testing.allocator; + const cases = [_]struct { in: []const u8, out: []const u8 }{ + .{ .in = "/mnt/9p", .out = "/mnt/9p" }, + .{ .in = "/mnt/9p/", .out = "/mnt/9p" }, + .{ .in = "//mnt///9p//", .out = "/mnt/9p" }, + .{ .in = "/mnt/./9p/.", .out = "/mnt/9p" }, + .{ .in = "/mnt/x/../9p", .out = "/mnt/9p" }, + .{ .in = "/../mnt/9p", .out = "/mnt/9p" }, + .{ .in = "/a/b/c/../..", .out = "/a" }, + }; + for (cases) |c| { + const got = try normalizePath(gpa, "/cwd", c.in); + defer gpa.free(got); + try testing.expectEqualStrings(c.out, got); + try testing.expectEqual(@as(u8, 0), got[got.len]); + } +} + +test "normalizePath: relative paths use cwd" { + const gpa = testing.allocator; + const cases = [_]struct { cwd: []const u8, in: []const u8, out: []const u8 }{ + .{ .cwd = "/home/me", .in = "mnt", .out = "/home/me/mnt" }, + .{ .cwd = "/home/me", .in = "./mnt/", .out = "/home/me/mnt" }, + .{ .cwd = "/home/me", .in = "../mnt", .out = "/home/mnt" }, + .{ .cwd = "/home/me/", .in = ".", .out = "/home/me" }, + .{ .cwd = "/", .in = "x", .out = "/x" }, + }; + for (cases) |c| { + const got = try normalizePath(gpa, c.cwd, c.in); + defer gpa.free(got); + try testing.expectEqualStrings(c.out, got); + } +} + +test "normalizePath: rejects root and empty" { + const gpa = testing.allocator; + try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/cwd", "/")); + try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/cwd", "///")); + try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/cwd", "/mnt/..")); + try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/cwd", "")); + try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/", "..")); +} + +test "resolveMountpoint: relative resolves against the real cwd" { + const gpa = testing.allocator; + const got = try resolveMountpoint(gpa, "sub/dir"); + defer gpa.free(got); + try testing.expect(got[0] == '/'); + try testing.expect(std.mem.endsWith(u8, got, "/sub/dir")); +} + +test "getenv" { + const env = [_:null]?[*:0]const u8{ "PATH=/a:/b", "X=", "PATHX=no", "NINEPLAYER_MOUNT=/m" }; + const envp: [*:null]const ?[*:0]const u8 = &env; + try testing.expectEqualStrings("/a:/b", getenv(envp, "PATH").?); + try testing.expectEqualStrings("", getenv(envp, "X").?); + try testing.expectEqualStrings("/m", getenv(envp, "NINEPLAYER_MOUNT").?); + try testing.expect(getenv(envp, "NOPE") == null); + try testing.expect(getenv(envp, "PAT") == null); +} + +test "pathCandidates: PATH search" { + const gpa = testing.allocator; + const env = [_:null]?[*:0]const u8{ "PATH=/usr/local/bin::/usr/bin", "HOME=/h" }; + const cands = try pathCandidates(gpa, &env, "fish"); + defer { + for (cands) |c| gpa.free(c); + gpa.free(cands); + } + try testing.expectEqual(@as(usize, 3), cands.len); + try testing.expectEqualStrings("/usr/local/bin/fish", cands[0]); + try testing.expectEqualStrings("./fish", cands[1]); + try testing.expectEqualStrings("/usr/bin/fish", cands[2]); +} + +test "pathCandidates: slash means no search; default PATH" { + const gpa = testing.allocator; + const env = [_:null]?[*:0]const u8{"HOME=/h"}; + { + const cands = try pathCandidates(gpa, &env, "./bin/x"); + defer { + for (cands) |c| gpa.free(c); + gpa.free(cands); + } + try testing.expectEqual(@as(usize, 1), cands.len); + try testing.expectEqualStrings("./bin/x", cands[0]); + } + { + const cands = try pathCandidates(gpa, &env, "sh"); + defer { + for (cands) |c| gpa.free(c); + gpa.free(cands); + } + try testing.expectEqual(@as(usize, 3), cands.len); + try testing.expectEqualStrings("/usr/local/bin/sh", cands[0]); + try testing.expectEqualStrings("/bin/sh", cands[1]); + } + try testing.expectError(error.EmptyProgramName, pathCandidates(gpa, &env, "")); +} + +test "findInPath finds sh" { + const gpa = testing.allocator; + const env = [_:null]?[*:0]const u8{"PATH=/nonexistent:/bin:/usr/bin"}; + const p = try findInPath(gpa, &env, "sh"); + defer gpa.free(p); + try testing.expect(std.mem.endsWith(u8, p, "/sh")); + try testing.expectError(error.FileNotFound, findInPath(gpa, &env, "definitely-not-a-program-9player")); +} + +test "buildEnvp replaces NINEPLAYER_MOUNT" { + const gpa = testing.allocator; + const env = [_:null]?[*:0]const u8{ "A=1", "NINEPLAYER_MOUNT=/old", "B=2" }; + const out = try buildEnvp(gpa, &env, "/mnt/9p"); + defer { + gpa.free(std.mem.span(out[out.len - 1].?)); + gpa.free(out); + } + try testing.expectEqual(@as(usize, 3), out.len); + try testing.expectEqualStrings("A=1", std.mem.span(out[0].?)); + try testing.expectEqualStrings("B=2", std.mem.span(out[1].?)); + try testing.expectEqualStrings("NINEPLAYER_MOUNT=/mnt/9p", std.mem.span(out[2].?)); + try testing.expect(out[3] == null); + try testing.expectEqualStrings("/mnt/9p", getenv(out.ptr, "NINEPLAYER_MOUNT").?); +} + +test "decodeStatus" { + try testing.expectEqual(@as(u8, 0), decodeStatus(0)); + try testing.expectEqual(@as(u8, 7), decodeStatus(7 << 8)); + try testing.expectEqual(@as(u8, 255), decodeStatus(255 << 8)); + try testing.expectEqual(@as(u8, 128 + 9), decodeStatus(9)); // SIGKILL + try testing.expectEqual(@as(u8, 128 + 15), decodeStatus(15)); // SIGTERM +} + +test "ensureMountpoint: existing directory is accepted, plain file rejected" { + const gpa = testing.allocator; + try ensureMountpoint(gpa, "/tmp"); + try testing.expectError(error.Mountpoint, ensureMountpoint(gpa, "/proc/self/status")); +} -- cgit v1.3