//! Namespace and process plumbing for 9ns. //! //! Everything here is raw `std.os.linux` syscalls (no libc). The child side //! of `spawn` runs between `fork` and `execve`; it does not allocate except //! inside `ensureMountpoint` (the process is single-threaded by then, so the //! inherited allocator is safe to use). //! //! Exit codes produced by the child before exec: 125 for namespace/mount //! setup failures, 126 when the program was found but is not executable, //! 127 when it was not found. const std = @import("std"); const builtin = @import("builtin"); const linux = std.os.linux; const Allocator = std.mem.Allocator; const E = linux.E; pub const Spawn = struct { /// argv[0] is PATH-searched unless it contains '/'. argv: []const []const u8, /// Inherited environment; `NINE_MOUNT` is added or replaced. envp: [*:null]const ?[*:0]const u8, /// Absolute mountpoint (see `resolveMountpoint`). mountpoint: []const u8, uid: u32, gid: u32, max_read: u32, /// When false the namespace is set up (including mountpoint shadowing) /// but `/dev/fuse` is not opened and nothing is mounted; `Child.fuse_fd` /// is then -1. Only for smoke tests. mount_fuse: bool = true, }; pub const Child = struct { pid: i32, /// The `/dev/fuse` connection backing the mount, opened by the child /// inside its user namespace (the kernel refuses to mount a fuse fd that /// was opened from another user namespace) and handed back over the /// status socket with SCM_RIGHTS. Owned by the caller; CLOEXEC. fuse_fd: i32, /// Parent end of the status socket. The child reports an exec failure /// on it (see `reportExecFailure`); it reads EOF once exec succeeded. status_fd: i32, }; /// Exit status used by the child for setup failures (matches 9ns's own). pub const setup_failure_status: u8 = 125; /// Refuse to shadow a directory with more entries than this. pub const max_shadow_entries: usize = 4096; const default_path = "/usr/local/bin:/bin:/usr/bin"; const path_max = 4096; // --------------------------------------------------------------------------- // Mountpoint resolution // --------------------------------------------------------------------------- /// Absolute path (relative paths resolved against cwd), duplicate slashes /// collapsed, `.` and `..` components resolved lexically, no trailing slash. /// `/` itself is rejected. pub fn resolveMountpoint(gpa: Allocator, path: []const u8) ![:0]u8 { var cwd_buf: [path_max]u8 = undefined; var cwd: []const u8 = "/"; if (path.len == 0 or path[0] != '/') { const rc = linux.getcwd(&cwd_buf, cwd_buf.len); switch (linux.errno(rc)) { .SUCCESS => {}, else => |e| { std.debug.print("9ns: getcwd: E{t}\n", .{e}); return error.Cwd; }, } // rc counts the terminating NUL. cwd = cwd_buf[0 .. rc - 1]; } return normalizePath(gpa, cwd, path); } /// Pure part of `resolveMountpoint`: `cwd` is only used when `path` is relative. fn normalizePath(gpa: Allocator, cwd: []const u8, path: []const u8) ![:0]u8 { if (path.len == 0) return error.InvalidMountpoint; var out: std.ArrayList(u8) = .empty; defer out.deinit(gpa); if (path[0] != '/') try appendComponents(gpa, &out, cwd); try appendComponents(gpa, &out, path); if (out.items.len == 0) return error.InvalidMountpoint; // "/" or equivalent return out.toOwnedSliceSentinel(gpa, 0); } fn appendComponents(gpa: Allocator, out: *std.ArrayList(u8), path: []const u8) !void { var it = std.mem.tokenizeScalar(u8, path, '/'); while (it.next()) |comp| { if (std.mem.eql(u8, comp, ".")) continue; if (std.mem.eql(u8, comp, "..")) { // Pop the last component (lexically; "/.." stays "/"). const idx = std.mem.lastIndexOfScalar(u8, out.items, '/') orelse 0; out.shrinkRetainingCapacity(idx); continue; } try out.append(gpa, '/'); try out.appendSlice(gpa, comp); } } // --------------------------------------------------------------------------- // Environment helpers // --------------------------------------------------------------------------- /// Look a variable up in a raw envp block. pub fn getenv(envp: [*:null]const ?[*:0]const u8, name: []const u8) ?[]const u8 { var i: usize = 0; while (envp[i]) |entry| : (i += 1) { const kv = std.mem.span(entry); if (kv.len > name.len and kv[name.len] == '=' and std.mem.eql(u8, kv[0..name.len], name)) { return kv[name.len + 1 ..]; } } return null; } /// Every path `execve` should try for `name`, in order: just `name` if it /// contains a '/', else `/` for each `$PATH` element (an empty /// element means the current directory; `$PATH` unset falls back to /// `/usr/local/bin:/bin:/usr/bin`). pub fn pathCandidates(gpa: Allocator, envp: [*:null]const ?[*:0]const u8, name: []const u8) ![]const [:0]const u8 { if (name.len == 0) return error.EmptyProgramName; var list: std.ArrayList([:0]const u8) = .empty; errdefer { for (list.items) |c| gpa.free(c); list.deinit(gpa); } if (std.mem.indexOfScalar(u8, name, '/') != null) { try list.append(gpa, try gpa.dupeZ(u8, name)); return list.toOwnedSlice(gpa); } const path = getenv(envp, "PATH") orelse default_path; var it = std.mem.splitScalar(u8, path, ':'); while (it.next()) |dir| { const d = if (dir.len == 0) "." else dir; try list.append(gpa, try std.fmt.allocPrintSentinel(gpa, "{s}/{s}", .{ d, name }, 0)); } return list.toOwnedSlice(gpa); } /// First PATH candidate that is an executable regular file, or the name /// itself when it contains a '/'. Provided for completeness; `spawn` simply /// tries `execve` on every candidate instead. pub fn findInPath(gpa: Allocator, envp: [*:null]const ?[*:0]const u8, name: []const u8) ![:0]u8 { const cands = try pathCandidates(gpa, envp, name); defer { for (cands) |c| gpa.free(c); gpa.free(cands); } for (cands) |c| { var stx: linux.Statx = undefined; const rc = linux.statx(linux.AT.FDCWD, c.ptr, 0, .{ .TYPE = true, .MODE = true }, &stx); if (linux.errno(rc) != .SUCCESS) continue; if (stx.mode & linux.S.IFMT != linux.S.IFREG) continue; if (stx.mode & 0o111 == 0) continue; return gpa.dupeZ(u8, c); } return error.FileNotFound; } /// New envp block: every entry of `envp` except `NINE_MOUNT=...`, /// followed by `NINE_MOUNT=`. fn buildEnvp(gpa: Allocator, envp: [*:null]const ?[*:0]const u8, mountpoint: []const u8) ![:null]?[*:0]const u8 { const key = "NINE_MOUNT="; var keep: usize = 0; var i: usize = 0; while (envp[i]) |entry| : (i += 1) { if (!std.mem.startsWith(u8, std.mem.span(entry), key)) keep += 1; } const out = try gpa.allocSentinel(?[*:0]const u8, keep + 1, null); errdefer gpa.free(out); var j: usize = 0; i = 0; while (envp[i]) |entry| : (i += 1) { if (std.mem.startsWith(u8, std.mem.span(entry), key)) continue; out[j] = entry; j += 1; } const mount_entry = try std.fmt.allocPrintSentinel(gpa, key ++ "{s}", .{mountpoint}, 0); out[j] = mount_entry.ptr; return out; } fn buildArgv(gpa: Allocator, argv: []const []const u8) ![:null]?[*:0]const u8 { const out = try gpa.allocSentinel(?[*:0]const u8, argv.len, null); for (argv, 0..) |a, i| out[i] = (try gpa.dupeZ(u8, a)).ptr; return out; } // --------------------------------------------------------------------------- // Mountpoint policy // --------------------------------------------------------------------------- /// Make sure `path` is a directory, inside the *current* mount namespace: /// /// * already a directory → done; /// * else find the deepest existing ancestor and `mkdir` the missing /// components under it one by one (`mkdir -p`); the first of them failing /// with `EACCES`/`EPERM`/`EROFS` (the normal case for `/mnt/9p/` as /// a plain user) means **shadow that ancestor**: mount a `tmpfs` over it /// that re-exposes every existing entry (bind mounts for directories and /// files, recreated symlinks), then create the missing components inside; /// * anything else fails with the errno and a hint. /// /// So `/mnt/9p/x` on a host without `/mnt/9p` shadows `/mnt` and creates /// `9p/x`; with a root-owned `/mnt/9p` it shadows `/mnt/9p`; inside a 9ns /// namespace, where `/mnt/9p` is ours, it just creates `x`. `/` and `/proc` /// are never shadowed, nor a directory with more than `max_shadow_entries`. /// /// Every failure prints `9ns: : E` to stderr before /// returning. Meant to be called in the child of `spawn` (or from a /// throwaway namespace: `unshare -Urm`). pub fn ensureMountpoint(gpa: Allocator, path: [:0]const u8) !void { if (try existingKind(path)) |ft| { if (ft == .dir) return; std.debug.print("9ns: mountpoint {s}: exists but is not a directory\n", .{path}); return error.Mountpoint; } // Deepest existing ancestor: walk up until something is there. var base: []const u8 = path; while (true) { base = std.fs.path.dirname(base) orelse "/"; var base_buf: [path_max]u8 = undefined; const base_z = std.fmt.bufPrintZ(&base_buf, "{s}", .{base}) catch { std.debug.print("9ns: mountpoint {s}: path too long\n", .{path}); return error.Mountpoint; }; const kind = try existingKind(base_z) orelse continue; if (kind != .dir) { std.debug.print("9ns: mountpoint {s}: {s} is not a directory\n", .{ path, base }); return error.Mountpoint; } break; } // Missing components, deepest ancestor first. const missing = path[base.len..]; const mk = mkdirComponents(path, base.len, missing); switch (mk) { .SUCCESS => return, .ACCES, .PERM, .ROFS => {}, else => |e| { std.debug.print("9ns: mkdir {s}: E{t} (pass --mount an existing directory)\n", .{ path, e }); return error.Mountpoint; }, } if (std.mem.eql(u8, base, "/") or isSameDirectory(base, "/")) { std.debug.print("9ns: mkdir {s}: E{t}; refusing to shadow / (pass --mount an existing directory)\n", .{ path, mk }); return error.Mountpoint; } // The shadow rebuilds entries from /proc/self/fd//; a tmpfs // over /proc (or a subtree of it) would take that away from itself. if (std.mem.eql(u8, base, "/proc") or std.mem.startsWith(u8, base, "/proc/")) { std.debug.print("9ns: mkdir {s}: E{t}; refusing to shadow {s} (pass --mount an existing directory)\n", .{ path, mk, base }); return error.Mountpoint; } const base_z = try gpa.dupeZ(u8, base); defer gpa.free(base_z); try shadowDirectory(gpa, base_z); switch (mkdirComponents(path, base.len, missing)) { .SUCCESS => {}, else => |e| { std.debug.print("9ns: mkdir {s} (in shadow tmpfs): E{t}\n", .{ path, e }); return error.Mountpoint; }, } } /// What `path` is, following symlinks: null when nothing is there; an /// error (reported) for a dangling symlink. fn existingKind(path: [*:0]const u8) !?FileType { if (fileType(linux.AT.FDCWD, path, false)) |ft| return ft; if (fileType(linux.AT.FDCWD, path, true) == .symlink) { std.debug.print("9ns: mountpoint {s}: dangling symlink\n", .{std.mem.span(path)}); return error.Mountpoint; } return null; } /// `mkdir` each component of `missing` (which is `path[base_len..]`, so /// it starts with '/') under the existing prefix `path[0..base_len]`, in /// order. Returns the errno of the first failure (`.SUCCESS` when all were /// created); `EEXIST` on a component is fine (another process, or a retry). fn mkdirComponents(path: []const u8, base_len: usize, missing: []const u8) E { var buf: [path_max]u8 = undefined; var end: usize = base_len; var it = std.mem.tokenizeScalar(u8, missing, '/'); while (it.next()) |comp| { end += 1 + comp.len; const prefix = std.fmt.bufPrintZ(&buf, "{s}", .{path[0..end]}) catch return .NAMETOOLONG; switch (linux.errno(linux.mkdirat(linux.AT.FDCWD, prefix, 0o755))) { .SUCCESS, .EXIST => {}, else => |e| return e, } } return .SUCCESS; } const FileType = enum { dir, symlink, other }; /// True when both paths resolve (following symlinks, including magic ones /// such as /proc/self/root) to the same inode. fn isSameDirectory(a: []const u8, b: [*:0]const u8) bool { var a_buf: [path_max]u8 = undefined; const a_z = std.fmt.bufPrintZ(&a_buf, "{s}", .{a}) catch return false; var sa: linux.Statx = undefined; var sb: linux.Statx = undefined; if (linux.errno(linux.statx(linux.AT.FDCWD, a_z, 0, .{ .INO = true }, &sa)) != .SUCCESS) return false; if (linux.errno(linux.statx(linux.AT.FDCWD, b, 0, .{ .INO = true }, &sb)) != .SUCCESS) return false; return sa.ino == sb.ino and sa.dev_major == sb.dev_major and sa.dev_minor == sb.dev_minor; } fn fileType(dirfd: i32, name: [*:0]const u8, nofollow: bool) ?FileType { var stx: linux.Statx = undefined; const flags: u32 = if (nofollow) linux.AT.SYMLINK_NOFOLLOW else 0; const rc = linux.statx(dirfd, name, flags, .{ .TYPE = true }, &stx); if (linux.errno(rc) != .SUCCESS) return null; return switch (stx.mode & linux.S.IFMT) { linux.S.IFDIR => .dir, linux.S.IFLNK => .symlink, else => .other, }; } const Entry = struct { name: [:0]u8, kind: FileType }; /// Read every entry of the directory open at `fd` (excluding `.` and `..`). fn listDir(gpa: Allocator, fd: i32, dirpath: []const u8) ![]Entry { var list: std.ArrayList(Entry) = .empty; errdefer { for (list.items) |e| gpa.free(e.name); list.deinit(gpa); } var buf: [32 * 1024]u8 align(@alignOf(linux.dirent64)) = undefined; while (true) { const rc = linux.getdents64(fd, &buf, buf.len); switch (linux.errno(rc)) { .SUCCESS => {}, else => |e| { std.debug.print("9ns: getdents64 {s}: E{t}\n", .{ dirpath, e }); return error.Mountpoint; }, } if (rc == 0) break; var off: usize = 0; while (off < rc) { const d: *align(1) const linux.dirent64 = @ptrCast(&buf[off]); const name_ptr: [*:0]const u8 = @ptrCast(&buf[off + @offsetOf(linux.dirent64, "name")]); const name = std.mem.span(name_ptr); const dtype = d.type; off += d.reclen; if (std.mem.eql(u8, name, ".") or std.mem.eql(u8, name, "..")) continue; if (list.items.len >= max_shadow_entries) { std.debug.print("9ns: refusing to shadow {s}: more than {d} entries\n", .{ dirpath, max_shadow_entries }); return error.TooManyEntries; } const kind: FileType = switch (dtype) { linux.DT.DIR => .dir, linux.DT.LNK => .symlink, linux.DT.UNKNOWN => fileType(fd, name_ptr, true) orelse .other, else => .other, }; try list.append(gpa, .{ .name = try gpa.dupeZ(u8, name), .kind = kind }); } } return list.toOwnedSlice(gpa); } fn shadowDirectory(gpa: Allocator, parent: [:0]const u8) !void { const open_rc = linux.open(parent, .{ .ACCMODE = .RDONLY, .DIRECTORY = true, .CLOEXEC = true }, 0); switch (linux.errno(open_rc)) { .SUCCESS => {}, else => |e| { std.debug.print("9ns: open {s}: E{t}\n", .{ parent, e }); return error.Mountpoint; }, } const pfd: i32 = @intCast(open_rc); defer _ = linux.close(pfd); const entries = try listDir(gpa, pfd, parent); defer { for (entries) |e| gpa.free(e.name); gpa.free(entries); } const tmpfs_opts: [*:0]const u8 = "mode=755"; switch (linux.errno(linux.mount("tmpfs", parent, "tmpfs", linux.MS.NOSUID | linux.MS.NODEV, @intFromPtr(tmpfs_opts)))) { .SUCCESS => {}, else => |e| { std.debug.print("9ns: mount tmpfs on {s}: E{t}\n", .{ parent, e }); return error.Mountpoint; }, } // `pfd` still refers to the original directory underneath the tmpfs, so // `/proc/self/fd//` reaches the hidden entries. var src_buf: [path_max]u8 = undefined; var dst_buf: [path_max]u8 = undefined; var link_buf: [path_max]u8 = undefined; for (entries) |e| { const src = std.fmt.bufPrintZ(&src_buf, "/proc/self/fd/{d}/{s}", .{ pfd, e.name }) catch { std.debug.print("9ns: shadow {s}/{s}: name too long (skipped)\n", .{ parent, e.name }); continue; }; const dst = std.fmt.bufPrintZ(&dst_buf, "{s}/{s}", .{ parent, e.name }) catch { std.debug.print("9ns: shadow {s}/{s}: name too long (skipped)\n", .{ parent, e.name }); continue; }; switch (e.kind) { .dir => { if (!check("mkdir", dst, linux.mkdirat(linux.AT.FDCWD, dst, 0o755))) continue; _ = check("bind", dst, linux.mount(src, dst, null, linux.MS.BIND | linux.MS.REC, 0)); }, .symlink => { const rc = linux.readlinkat(pfd, e.name, &link_buf, link_buf.len - 1); if (!check("readlink", dst, rc)) continue; link_buf[rc] = 0; const target: [*:0]const u8 = @ptrCast(&link_buf); _ = check("symlink", dst, linux.symlinkat(target, linux.AT.FDCWD, dst)); }, .other => { const rc = linux.openat(linux.AT.FDCWD, dst, .{ .ACCMODE = .WRONLY, .CREAT = true, .CLOEXEC = true }, 0o644); if (!check("create", dst, rc)) continue; _ = linux.close(@intCast(rc)); _ = check("bind", dst, linux.mount(src, dst, null, linux.MS.BIND | linux.MS.REC, 0)); }, } } } /// Report a failed per-entry step as a warning (the entry is skipped; the /// rest of the shadow is still useful). Returns true on success. fn check(step: []const u8, path: [*:0]const u8, rc: usize) bool { switch (linux.errno(rc)) { .SUCCESS => return true, else => |e| { std.debug.print("9ns: shadow: {s} {s}: E{t} (skipped)\n", .{ step, std.mem.span(path), e }); return false; }, } } // --------------------------------------------------------------------------- // spawn // --------------------------------------------------------------------------- const ChildArgs = struct { gpa: Allocator, status_sock: i32, mountpoint: [:0]const u8, fuse_opts_prefix: [:0]const u8, // everything after "fd=," mount_fuse: bool, uid_map: []const u8, gid_map: []const u8, argv: [:null]?[*:0]const u8, envp: [:null]?[*:0]const u8, candidates: []const [:0]const u8, name: []const u8, }; /// Status channel protocol (child → parent, over a CLOEXEC socketpair): /// a 0 byte means "namespace and mount are up" and carries the fuse fd as /// SCM_RIGHTS; a non-zero byte is an exit status followed by a message. /// EOF ends the conversation (exec succeeded, or the child died). const ok_byte: u8 = 0; /// fork; the child unshares user+mount namespaces, maps its uid/gid, /// makes `/` private, ensures the mountpoint, opens `/dev/fuse`, mounts it /// on the mountpoint and sends the fd back. `spawn` returns at that point /// (with `error.ChildFailed` and a message on stderr if any step failed). /// The child then stats the mountpoint, which makes the kernel fetch the /// root's attributes once the parent serves (the kernel seeds the fuse root /// with uid 0, unmapped in the new user namespace, so nothing could be /// created in the root until then), sets `NINE_MOUNT` and execs /// `argv`. An exec failure is reported on `Child.status_fd` and ends the /// child with 126/127; collect it with `reportExecFailure` after /// `bridge.serve` returns. /// /// If `installSignals` was called, the pid is stored into the registered /// variable as soon as fork returns so no SIGCHLD can be missed. pub fn spawn(gpa: Allocator, s: Spawn) !Child { if (s.argv.len == 0 or s.argv[0].len == 0) { std.debug.print("9ns: empty program name\n", .{}); return error.EmptyProgramName; } const mountpoint = try gpa.dupeZ(u8, s.mountpoint); defer gpa.free(mountpoint); const fuse_opts_prefix = try std.fmt.allocPrintSentinel(gpa, "rootmode=40000,user_id={d},group_id={d},max_read={d}", .{ s.uid, s.gid, s.max_read }, 0); defer gpa.free(fuse_opts_prefix); var uid_buf: [64]u8 = undefined; var gid_buf: [64]u8 = undefined; const uid_map = try std.fmt.bufPrint(&uid_buf, "{d} {d} 1\n", .{ s.uid, s.uid }); const gid_map = try std.fmt.bufPrint(&gid_buf, "{d} {d} 1\n", .{ s.gid, s.gid }); const argv = try buildArgv(gpa, s.argv); defer { for (argv) |a| gpa.free(std.mem.span(a.?)); gpa.free(argv); } const envp = try buildEnvp(gpa, s.envp, s.mountpoint); defer { gpa.free(std.mem.span(envp[envp.len - 1].?)); // the NINE_MOUNT entry we created gpa.free(envp); } const candidates = try pathCandidates(gpa, s.envp, s.argv[0]); defer { for (candidates) |c| gpa.free(c); gpa.free(candidates); } var sv: [2]i32 = undefined; switch (linux.errno(linux.socketpair(linux.AF.UNIX, linux.SOCK.STREAM | linux.SOCK.CLOEXEC, 0, &sv))) { .SUCCESS => {}, else => |e| { std.debug.print("9ns: socketpair: E{t}\n", .{e}); return error.SystemResources; }, } const child_args = ChildArgs{ .gpa = gpa, .status_sock = sv[1], .mountpoint = mountpoint, .fuse_opts_prefix = fuse_opts_prefix, .mount_fuse = s.mount_fuse, .uid_map = uid_map, .gid_map = gid_map, .argv = argv, .envp = envp, .candidates = candidates, .name = s.argv[0], }; const fork_rc = linux.fork(); switch (linux.errno(fork_rc)) { .SUCCESS => {}, else => |e| { _ = linux.close(sv[0]); _ = linux.close(sv[1]); std.debug.print("9ns: fork: E{t}\n", .{e}); return error.SystemResources; }, } if (fork_rc == 0) childMain(&child_args); const pid: i32 = @intCast(fork_rc); if (child_pid_ptr) |p| @atomicStore(i32, p, pid, .seq_cst); _ = linux.close(sv[1]); // First byte: ok (with the fuse fd attached) or a failure status. var first: [1]u8 = .{ok_byte}; // defined even if recvmsg stores nothing var fuse_fd: i32 = -1; var n: usize = 0; while (true) { const rc = recvWithFd(sv[0], &first, &fuse_fd, 0); switch (linux.errno(rc)) { .SUCCESS => {}, .INTR => continue, else => break, } n = rc; break; } if (n == 1 and first[0] == ok_byte and (fuse_fd >= 0 or !s.mount_fuse)) { return .{ .pid = pid, .fuse_fd = fuse_fd, .status_fd = sv[0] }; } // Failure. A status byte means the child is exiting on its own and a // message follows. Anything else (EOF: the child died before reporting; // an ok byte without the fd: the SCM_RIGHTS transfer was truncated, e.g. // EMFILE) is a protocol violation: the child may be about to exec with a // dead mount, so kill it before waiting rather than reading the status // socket until an exec'd program eventually exits. const reported = n == 1 and first[0] != ok_byte; if (!reported) _ = linux.kill(pid, .KILL); if (fuse_fd >= 0) _ = linux.close(fuse_fd); var msg: [512]u8 = undefined; var len: usize = 0; while (reported and len < msg.len) { const rc = linux.read(sv[0], msg[len..].ptr, msg.len - len); switch (linux.errno(rc)) { .SUCCESS => {}, .INTR => continue, else => break, } if (rc == 0) break; len += rc; } _ = linux.close(sv[0]); if (reported) { std.debug.print("9ns: {s}\n", .{msg[0..len]}); } else if (n == 1) { std.debug.print("9ns: child handshake failed: no fuse fd received (out of file descriptors?)\n", .{}); } else { std.debug.print("9ns: child exited before reporting\n", .{}); } _ = waitChild(pid) catch {}; if (child_pid_ptr) |p| @atomicStore(i32, p, 0, .seq_cst); const status: u8 = if (reported) first[0] else setup_failure_status; return switch (status) { 126 => error.ExecPermission, 127 => error.ExecNotFound, else => error.ChildFailed, }; } /// After the child is gone (or the mount is dead): print the exec failure /// the child reported on `status_fd`, if any, and close it. Returns the /// status byte the child announced, or null when exec succeeded / nothing /// was reported. Never blocks. pub fn reportExecFailure(child: Child) ?u8 { defer _ = linux.close(child.status_fd); var msg: [512]u8 = undefined; var len: usize = 0; while (len < msg.len) { var iov = [_]std.posix.iovec{.{ .base = msg[len..].ptr, .len = msg.len - len }}; var hdr = linux.msghdr{ .name = null, .namelen = 0, .iov = &iov, .iovlen = 1, .control = null, .controllen = 0, .flags = 0, }; const rc = linux.recvmsg(child.status_fd, &hdr, linux.MSG.DONTWAIT); switch (linux.errno(rc)) { .SUCCESS => {}, .INTR => continue, else => break, } if (rc == 0) break; len += rc; } if (len == 0) return null; std.debug.print("9ns: {s}\n", .{msg[1..len]}); return msg[0]; } const cmsg_fd_len = @sizeOf(linux.cmsghdr) + @sizeOf(i32); const cmsg_fd_space = std.mem.alignForward(usize, cmsg_fd_len, @sizeOf(usize)); /// sendmsg one data byte, optionally with `fd` attached as SCM_RIGHTS. fn sendWithFd(sock: i32, byte: u8, fd: ?i32) usize { const data = [_]u8{byte}; const iov = [_]std.posix.iovec_const{.{ .base = &data, .len = 1 }}; var cbuf: [cmsg_fd_space]u8 align(@alignOf(linux.cmsghdr)) = @splat(0); var msg = linux.msghdr_const{ .name = null, .namelen = 0, .iov = &iov, .iovlen = 1, .control = null, .controllen = 0, .flags = 0, }; if (fd) |f| { const hdr: *linux.cmsghdr = @ptrCast(&cbuf); hdr.* = .{ .len = cmsg_fd_len, .level = linux.SOL.SOCKET, .type = linux.SCM.RIGHTS }; @memcpy(cbuf[@sizeOf(linux.cmsghdr)..][0..@sizeOf(i32)], std.mem.asBytes(&f)); msg.control = &cbuf; msg.controllen = cmsg_fd_space; } return linux.sendmsg(sock, &msg, linux.MSG.NOSIGNAL); } /// recvmsg into `buf`; an SCM_RIGHTS fd, if any, is stored in `fd_out`. fn recvWithFd(sock: i32, buf: []u8, fd_out: *i32, flags: u32) usize { var iov = [_]std.posix.iovec{.{ .base = buf.ptr, .len = buf.len }}; var cbuf: [cmsg_fd_space]u8 align(@alignOf(linux.cmsghdr)) = @splat(0); var msg = linux.msghdr{ .name = null, .namelen = 0, .iov = &iov, .iovlen = 1, .control = &cbuf, .controllen = cbuf.len, .flags = 0, }; const rc = linux.recvmsg(sock, &msg, linux.MSG.CMSG_CLOEXEC | flags); if (linux.errno(rc) != .SUCCESS) return rc; if (msg.controllen >= cmsg_fd_len) { const hdr: *const linux.cmsghdr = @ptrCast(&cbuf); if (hdr.level == linux.SOL.SOCKET and hdr.type == linux.SCM.RIGHTS and hdr.len >= cmsg_fd_len) { var fd: i32 = undefined; @memcpy(std.mem.asBytes(&fd), cbuf[@sizeOf(linux.cmsghdr)..][0..@sizeOf(i32)]); fd_out.* = fd; } } return rc; } /// Child side of `spawn`. Never returns. fn childMain(c: *const ChildArgs) noreturn { resetSignals(); const rc_unshare = linux.errno(linux.unshare(linux.CLONE.NEWUSER | linux.CLONE.NEWNS)); if (rc_unshare != .SUCCESS) childFail(c, setup_failure_status, "unshare(CLONE_NEWUSER|CLONE_NEWNS)", rc_unshare, true); writeProcFile(c, "/proc/self/setgroups", "deny", true); writeProcFile(c, "/proc/self/uid_map", c.uid_map, false); writeProcFile(c, "/proc/self/gid_map", c.gid_map, false); const root: [*:0]const u8 = "/"; const rc_priv = linux.mount(null, root, null, linux.MS.REC | linux.MS.PRIVATE, 0); if (linux.errno(rc_priv) != .SUCCESS) childFail(c, setup_failure_status, "mount(/, MS_REC|MS_PRIVATE)", linux.errno(rc_priv), true); ensureMountpoint(c.gpa, c.mountpoint) catch { childFail(c, setup_failure_status, "mountpoint setup failed (pass --mount an existing directory)", .SUCCESS, false); }; var fuse_fd: ?i32 = null; if (c.mount_fuse) { // Must be opened here, after unshare: the kernel only mounts a fuse // device opened from the mount's own user namespace. const rc_open = linux.open("/dev/fuse", .{ .ACCMODE = .RDWR, .CLOEXEC = true }, 0); switch (linux.errno(rc_open)) { .SUCCESS => {}, .NOENT => childFail(c, setup_failure_status, "open /dev/fuse: ENOENT (is the fuse module loaded? try: modprobe fuse)", .SUCCESS, false), else => |e| childFail(c, setup_failure_status, "open /dev/fuse", e, true), } const fd: i32 = @intCast(rc_open); var opts_buf: [256]u8 = undefined; const opts = std.fmt.bufPrintZ(&opts_buf, "fd={d},{s}", .{ fd, c.fuse_opts_prefix }) catch unreachable; const rc = linux.mount("9ns", c.mountpoint, "fuse", linux.MS.NOSUID | linux.MS.NODEV, @intFromPtr(opts.ptr)); if (linux.errno(rc) != .SUCCESS) childFail(c, setup_failure_status, "mount fuse", linux.errno(rc), true); fuse_fd = fd; } const sent = sendWithFd(c.status_sock, ok_byte, fuse_fd); if (linux.errno(sent) != .SUCCESS) linux.exit_group(setup_failure_status); if (fuse_fd) |fd| { _ = linux.close(fd); // the parent holds the connection now // Force one GETATTR of the root (served by the parent, which is // entering its serve loop now); see `spawn`. Errors don't matter. var stx: linux.Statx = undefined; _ = linux.statx(linux.AT.FDCWD, c.mountpoint, 0, .{ .TYPE = true }, &stx); } var last: E = .NOENT; var saw_acces = false; for (c.candidates) |cand| { const rc = linux.execve(cand.ptr, c.argv.ptr, c.envp.ptr); last = linux.errno(rc); switch (last) { .NOENT, .NOTDIR, .LOOP, .NAMETOOLONG => continue, .ACCES => { saw_acces = true; continue; }, else => break, } } var buf: [512]u8 = undefined; // "Not found" covers every candidate that could not even be resolved // (a PATH element that is a file gives ENOTDIR, a symlink loop ELOOP); // a candidate that existed but was not executable wins over those. const not_found = switch (last) { .NOENT, .NOTDIR, .LOOP, .NAMETOOLONG => true, else => false, }; if (not_found and saw_acces) last = .ACCES; const status: u8 = if (not_found and !saw_acces) 127 else 126; const text = std.fmt.bufPrint(&buf, "exec {s}", .{c.name}) catch "exec"; childFail(c, status, text, last, true); } fn writeProcFile(c: *const ChildArgs, path: [*:0]const u8, data: []const u8, ignore_missing: bool) void { const rc = linux.open(path, .{ .ACCMODE = .WRONLY, .CLOEXEC = true }, 0); switch (linux.errno(rc)) { .SUCCESS => {}, .NOENT => if (ignore_missing) return else childFail(c, setup_failure_status, std.mem.span(path), .NOENT, true), else => |e| childFail(c, setup_failure_status, std.mem.span(path), e, true), } const fd: i32 = @intCast(rc); const w = linux.write(fd, data.ptr, data.len); const we = linux.errno(w); _ = linux.close(fd); if (we != .SUCCESS) childFail(c, setup_failure_status, std.mem.span(path), we, true); if (w != data.len) childFail(c, setup_failure_status, std.mem.span(path), .IO, true); } /// Write `[: E]` to the status socket and exit. fn childFail(c: *const ChildArgs, status: u8, step: []const u8, e: E, with_errno: bool) noreturn { var buf: [600]u8 = undefined; buf[0] = status; const rest = if (with_errno) std.fmt.bufPrint(buf[1..], "{s}: E{t}", .{ step, e }) catch buf[1..1] else std.fmt.bufPrint(buf[1..], "{s}", .{step}) catch buf[1..1]; const msg = buf[0 .. 1 + rest.len]; var off: usize = 0; while (off < msg.len) { const rc = linux.write(c.status_sock, msg[off..].ptr, msg.len - off); if (linux.errno(rc) == .INTR) continue; if (linux.errno(rc) != .SUCCESS) break; off += rc; } linux.exit_group(status); } // --------------------------------------------------------------------------- // Signals // --------------------------------------------------------------------------- var child_pid_ptr: ?*i32 = null; var chld_pipe_w: i32 = -1; var reaped = std.atomic.Value(bool).init(false); var reaped_status = std.atomic.Value(u32).init(0); /// A second child (the `--spawn` server) that the SIGCHLD handler reaps so /// it does not linger as a zombie when it dies mid-session. Its exit does /// not stop the serve loop. 0 = none. var server_pid = std.atomic.Value(i32).init(0); /// Register the `--spawn` server for reaping by the SIGCHLD handler. pub fn watchServer(pid: i32) void { server_pid.store(pid, .seq_cst); } /// Seconds the serve loop gets to come back after the child died before /// the watchdog ends the process anyway. pub const exit_grace_seconds: isize = 3; /// The watched child is already dead but the serve loop has not come back /// (it is stuck in a 9P request the server never answers): a terminal /// signal, or the watchdog armed by `onChld`, then ends 9ns with the /// child's status instead of hanging. Nothing is lost: the mount is torn /// down when the process exits. fn bailIfChildGone() void { if (!reaped.load(.acquire)) return; const srv = server_pid.load(.seq_cst); if (srv > 0) _ = linux.kill(srv, .TERM); linux.exit_group(decodeStatus(reaped_status.load(.acquire))); } fn armWatchdog() void { // setitimer takes an itimerval; std declares it with itimerspec, which // has the same layout on 64-bit targets (the sub-second field is 0). const t = linux.itimerspec{ .it_interval = .{ .sec = 0, .nsec = 0 }, .it_value = .{ .sec = exit_grace_seconds, .nsec = 0 }, }; _ = linux.setitimer(@intFromEnum(linux.ITIMER.REAL), &t, null); } fn onAlarm(_: linux.SIG) callconv(.c) void { bailIfChildGone(); } fn onForward(sig: linux.SIG) callconv(.c) void { const p = child_pid_ptr orelse return; const pid = @atomicLoad(i32, p, .seq_cst); if (pid > 0) _ = linux.kill(pid, sig); bailIfChildGone(); } /// SIGINT/SIGQUIT: the child owns the tty and gets them itself; we only /// react when the child is already gone (see `bailIfChildGone`). fn onTerminal(_: linux.SIG) callconv(.c) void { bailIfChildGone(); } /// Only the watched child counts: reap it here (WNOHANG), remember its /// status, forget its pid (so a later SIGTERM cannot hit a recycled pid) /// and poke the self-pipe. The `--spawn` server is reaped too but does not /// interrupt `bridge.serve`; SIGCHLD from anything else is ignored. fn onChld(_: linux.SIG) callconv(.c) void { const srv = server_pid.load(.seq_cst); if (srv > 0) { var sst: u32 = 0; const src = linux.waitpid(srv, &sst, linux.W.NOHANG); if (linux.errno(src) == .SUCCESS and src != 0) server_pid.store(0, .seq_cst); } const p = child_pid_ptr orelse return; const pid = @atomicLoad(i32, p, .seq_cst); if (pid <= 0) return; var st: u32 = 0; const rc = linux.waitpid(pid, &st, linux.W.NOHANG); if (linux.errno(rc) != .SUCCESS or rc == 0) return; reaped_status.store(st, .release); reaped.store(true, .release); @atomicStore(i32, p, 0, .seq_cst); const b = [_]u8{'c'}; _ = linux.write(chld_pipe_w, &b, 1); armWatchdog(); } /// SIGPIPE ignored; SIGINT/SIGQUIT effectively ignored (the child owns the /// tty) unless the child is already dead; SIGTERM/SIGHUP forwarded to /// `*child_pid`; SIGCHLD for `*child_pid` reaps it, writes a byte to a /// nonblocking self-pipe whose read end is returned (use it as `stop_fd`) /// and arms a watchdog (`exit_grace_seconds`, SIGALRM) that ends the /// process with the child's status should the serve loop stay blocked. /// `*child_pid` is filled in by `spawn`. pub fn installSignals(child_pid: *i32) !i32 { child_pid_ptr = child_pid; var fds: [2]i32 = undefined; switch (linux.errno(linux.pipe2(&fds, .{ .CLOEXEC = true, .NONBLOCK = true }))) { .SUCCESS => {}, else => |e| { std.debug.print("9ns: pipe2: E{t}\n", .{e}); return error.SystemResources; }, } chld_pipe_w = fds[1]; const ign = linux.Sigaction{ .handler = .{ .handler = linux.SIG.IGN }, .mask = linux.sigemptyset(), .flags = 0 }; const term = linux.Sigaction{ .handler = .{ .handler = &onTerminal }, .mask = linux.sigemptyset(), .flags = linux.SA.RESTART }; const fwd = linux.Sigaction{ .handler = .{ .handler = &onForward }, .mask = linux.sigemptyset(), .flags = linux.SA.RESTART }; const chld = linux.Sigaction{ .handler = .{ .handler = &onChld }, .mask = linux.sigemptyset(), .flags = linux.SA.RESTART | linux.SA.NOCLDSTOP }; const alrm = linux.Sigaction{ .handler = .{ .handler = &onAlarm }, .mask = linux.sigemptyset(), .flags = linux.SA.RESTART }; std.posix.sigaction(.INT, &term, null); std.posix.sigaction(.QUIT, &term, null); std.posix.sigaction(.ALRM, &alrm, null); std.posix.sigaction(.PIPE, &ign, null); std.posix.sigaction(.TERM, &fwd, null); std.posix.sigaction(.HUP, &fwd, null); std.posix.sigaction(.CHLD, &chld, null); return fds[0]; } /// Restore default dispositions in the child before exec (ignored signals /// would otherwise survive execve). fn resetSignals() void { const dfl = linux.Sigaction{ .handler = .{ .handler = linux.SIG.DFL }, .mask = linux.sigemptyset(), .flags = 0 }; inline for (.{ linux.SIG.INT, linux.SIG.QUIT, linux.SIG.PIPE, linux.SIG.TERM, linux.SIG.HUP, linux.SIG.CHLD, linux.SIG.ALRM }) |sig| { _ = linux.sigaction(sig, &dfl, null); } } // --------------------------------------------------------------------------- // Waiting // --------------------------------------------------------------------------- fn takeReaped() ?u32 { if (!reaped.load(.acquire)) return null; return reaped_status.load(.acquire); } /// waitpid status → exit code (`128+sig` when killed by a signal). pub fn decodeStatus(st: u32) u8 { if (linux.W.IFEXITED(st)) return linux.W.EXITSTATUS(st); if (linux.W.IFSIGNALED(st)) return 128 +% @as(u8, @truncate(@intFromEnum(linux.W.TERMSIG(st)))); return 1; } /// Block until `pid` exits (the SIGCHLD handler may have reaped it already). pub fn waitChild(pid: i32) !u8 { while (true) { if (takeReaped()) |st| return decodeStatus(st); var st: u32 = 0; const rc = linux.waitpid(pid, &st, 0); switch (linux.errno(rc)) { .SUCCESS => return decodeStatus(st), .INTR => continue, .CHILD => { if (takeReaped()) |s| return decodeStatus(s); return error.NoChild; }, else => |e| { std.debug.print("9ns: waitpid: E{t}\n", .{e}); return error.Wait; }, } } } /// Non-blocking: the exit status of `pid` if it has exited, else null. pub fn reapIfExited(pid: i32) ?u8 { if (takeReaped()) |st| return decodeStatus(st); var st: u32 = 0; const rc = linux.waitpid(pid, &st, linux.W.NOHANG); switch (linux.errno(rc)) { .SUCCESS => return if (rc == 0) null else decodeStatus(st), .CHILD => return if (takeReaped()) |s| decodeStatus(s) else null, else => return null, } } /// Reap any child (used for the `--spawn` server at exit). Non-blocking. pub fn reapAny(pid: i32) void { var st: u32 = 0; _ = linux.waitpid(pid, &st, linux.W.NOHANG); } // --------------------------------------------------------------------------- // Tests (no namespaces needed; `ensureMountpoint` is exercised by // test/integration.sh through the 9ns binary) // --------------------------------------------------------------------------- const testing = std.testing; test "normalizePath: absolute paths" { const gpa = testing.allocator; const cases = [_]struct { in: []const u8, out: []const u8 }{ .{ .in = "/mnt/9p", .out = "/mnt/9p" }, .{ .in = "/mnt/9p/", .out = "/mnt/9p" }, .{ .in = "//mnt///9p//", .out = "/mnt/9p" }, .{ .in = "/mnt/./9p/.", .out = "/mnt/9p" }, .{ .in = "/mnt/x/../9p", .out = "/mnt/9p" }, .{ .in = "/../mnt/9p", .out = "/mnt/9p" }, .{ .in = "/a/b/c/../..", .out = "/a" }, }; for (cases) |c| { const got = try normalizePath(gpa, "/cwd", c.in); defer gpa.free(got); try testing.expectEqualStrings(c.out, got); try testing.expectEqual(@as(u8, 0), got[got.len]); } } test "normalizePath: relative paths use cwd" { const gpa = testing.allocator; const cases = [_]struct { cwd: []const u8, in: []const u8, out: []const u8 }{ .{ .cwd = "/home/me", .in = "mnt", .out = "/home/me/mnt" }, .{ .cwd = "/home/me", .in = "./mnt/", .out = "/home/me/mnt" }, .{ .cwd = "/home/me", .in = "../mnt", .out = "/home/mnt" }, .{ .cwd = "/home/me/", .in = ".", .out = "/home/me" }, .{ .cwd = "/", .in = "x", .out = "/x" }, }; for (cases) |c| { const got = try normalizePath(gpa, c.cwd, c.in); defer gpa.free(got); try testing.expectEqualStrings(c.out, got); } } test "normalizePath: rejects root and empty" { const gpa = testing.allocator; try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/cwd", "/")); try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/cwd", "///")); try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/cwd", "/mnt/..")); try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/cwd", "")); try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/", "..")); } test "resolveMountpoint: relative resolves against the real cwd" { const gpa = testing.allocator; const got = try resolveMountpoint(gpa, "sub/dir"); defer gpa.free(got); try testing.expect(got[0] == '/'); try testing.expect(std.mem.endsWith(u8, got, "/sub/dir")); } test "getenv" { const env = [_:null]?[*:0]const u8{ "PATH=/a:/b", "X=", "PATHX=no", "NINE_MOUNT=/m" }; const envp: [*:null]const ?[*:0]const u8 = &env; try testing.expectEqualStrings("/a:/b", getenv(envp, "PATH").?); try testing.expectEqualStrings("", getenv(envp, "X").?); try testing.expectEqualStrings("/m", getenv(envp, "NINE_MOUNT").?); try testing.expect(getenv(envp, "NOPE") == null); try testing.expect(getenv(envp, "PAT") == null); } test "pathCandidates: PATH search" { const gpa = testing.allocator; const env = [_:null]?[*:0]const u8{ "PATH=/usr/local/bin::/usr/bin", "HOME=/h" }; const cands = try pathCandidates(gpa, &env, "fish"); defer { for (cands) |c| gpa.free(c); gpa.free(cands); } try testing.expectEqual(@as(usize, 3), cands.len); try testing.expectEqualStrings("/usr/local/bin/fish", cands[0]); try testing.expectEqualStrings("./fish", cands[1]); try testing.expectEqualStrings("/usr/bin/fish", cands[2]); } test "pathCandidates: slash means no search; default PATH" { const gpa = testing.allocator; const env = [_:null]?[*:0]const u8{"HOME=/h"}; { const cands = try pathCandidates(gpa, &env, "./bin/x"); defer { for (cands) |c| gpa.free(c); gpa.free(cands); } try testing.expectEqual(@as(usize, 1), cands.len); try testing.expectEqualStrings("./bin/x", cands[0]); } { const cands = try pathCandidates(gpa, &env, "sh"); defer { for (cands) |c| gpa.free(c); gpa.free(cands); } try testing.expectEqual(@as(usize, 3), cands.len); try testing.expectEqualStrings("/usr/local/bin/sh", cands[0]); try testing.expectEqualStrings("/bin/sh", cands[1]); } try testing.expectError(error.EmptyProgramName, pathCandidates(gpa, &env, "")); } test "findInPath finds sh" { const gpa = testing.allocator; const env = [_:null]?[*:0]const u8{"PATH=/nonexistent:/bin:/usr/bin"}; const p = try findInPath(gpa, &env, "sh"); defer gpa.free(p); try testing.expect(std.mem.endsWith(u8, p, "/sh")); try testing.expectError(error.FileNotFound, findInPath(gpa, &env, "definitely-not-a-program-9ns")); } test "buildEnvp replaces NINE_MOUNT" { const gpa = testing.allocator; const env = [_:null]?[*:0]const u8{ "A=1", "NINE_MOUNT=/old", "B=2" }; const out = try buildEnvp(gpa, &env, "/mnt/9p"); defer { gpa.free(std.mem.span(out[out.len - 1].?)); gpa.free(out); } try testing.expectEqual(@as(usize, 3), out.len); try testing.expectEqualStrings("A=1", std.mem.span(out[0].?)); try testing.expectEqualStrings("B=2", std.mem.span(out[1].?)); try testing.expectEqualStrings("NINE_MOUNT=/mnt/9p", std.mem.span(out[2].?)); try testing.expect(out[3] == null); try testing.expectEqualStrings("/mnt/9p", getenv(out.ptr, "NINE_MOUNT").?); } test "decodeStatus" { try testing.expectEqual(@as(u8, 0), decodeStatus(0)); try testing.expectEqual(@as(u8, 7), decodeStatus(7 << 8)); try testing.expectEqual(@as(u8, 255), decodeStatus(255 << 8)); try testing.expectEqual(@as(u8, 128 + 9), decodeStatus(9)); // SIGKILL try testing.expectEqual(@as(u8, 128 + 15), decodeStatus(15)); // SIGTERM } test "ensureMountpoint: existing directory is accepted, plain file rejected" { const gpa = testing.allocator; try ensureMountpoint(gpa, "/tmp"); try testing.expectError(error.Mountpoint, ensureMountpoint(gpa, "/proc/self/status")); }