diff options
Diffstat (limited to '9player/src/ns.zig')
| -rw-r--r-- | 9player/src/ns.zig | 1082 |
1 files changed, 0 insertions, 1082 deletions
diff --git a/9player/src/ns.zig b/9player/src/ns.zig deleted file mode 100644 index 2c6f202..0000000 --- a/9player/src/ns.zig +++ /dev/null @@ -1,1082 +0,0 @@ -//! Namespace and process plumbing for 9player. -//! -//! Everything here is raw `std.os.linux` syscalls (no libc). The child side -//! of `spawn` runs between `fork` and `execve`; it does not allocate except -//! inside `ensureMountpoint` (the process is single-threaded by then, so the -//! inherited allocator is safe to use). -//! -//! Exit codes produced by the child before exec: 125 for namespace/mount -//! setup failures, 126 when the program was found but is not executable, -//! 127 when it was not found. - -const std = @import("std"); -const builtin = @import("builtin"); -const linux = std.os.linux; -const Allocator = std.mem.Allocator; -const E = linux.E; - -pub const Spawn = struct { - /// argv[0] is PATH-searched unless it contains '/'. - argv: []const []const u8, - /// Inherited environment; `NINEPLAYER_MOUNT` is added or replaced. - envp: [*:null]const ?[*:0]const u8, - /// Absolute mountpoint (see `resolveMountpoint`). - mountpoint: []const u8, - uid: u32, - gid: u32, - max_read: u32, - /// When false the namespace is set up (including mountpoint shadowing) - /// but `/dev/fuse` is not opened and nothing is mounted; `Child.fuse_fd` - /// is then -1. Only for smoke tests. - mount_fuse: bool = true, -}; - -pub const Child = struct { - pid: i32, - /// The `/dev/fuse` connection backing the mount, opened by the child - /// inside its user namespace (the kernel refuses to mount a fuse fd that - /// was opened from another user namespace) and handed back over the - /// status socket with SCM_RIGHTS. Owned by the caller; CLOEXEC. - fuse_fd: i32, - /// Parent end of the status socket. The child reports an exec failure - /// on it (see `reportExecFailure`); it reads EOF once exec succeeded. - status_fd: i32, -}; - -/// Exit status used by the child for setup failures (matches 9player's own). -pub const setup_failure_status: u8 = 125; -/// Refuse to shadow a directory with more entries than this. -pub const max_shadow_entries: usize = 4096; - -const default_path = "/usr/local/bin:/bin:/usr/bin"; -const path_max = 4096; - -// --------------------------------------------------------------------------- -// Mountpoint resolution -// --------------------------------------------------------------------------- - -/// Absolute path (relative paths resolved against cwd), duplicate slashes -/// collapsed, `.` and `..` components resolved lexically, no trailing slash. -/// `/` itself is rejected. -pub fn resolveMountpoint(gpa: Allocator, path: []const u8) ![:0]u8 { - var cwd_buf: [path_max]u8 = undefined; - var cwd: []const u8 = "/"; - if (path.len == 0 or path[0] != '/') { - const rc = linux.getcwd(&cwd_buf, cwd_buf.len); - switch (linux.errno(rc)) { - .SUCCESS => {}, - else => |e| { - std.debug.print("9player: getcwd: E{t}\n", .{e}); - return error.Cwd; - }, - } - // rc counts the terminating NUL. - cwd = cwd_buf[0 .. rc - 1]; - } - return normalizePath(gpa, cwd, path); -} - -/// Pure part of `resolveMountpoint`: `cwd` is only used when `path` is relative. -fn normalizePath(gpa: Allocator, cwd: []const u8, path: []const u8) ![:0]u8 { - if (path.len == 0) return error.InvalidMountpoint; - var out: std.ArrayList(u8) = .empty; - defer out.deinit(gpa); - if (path[0] != '/') try appendComponents(gpa, &out, cwd); - try appendComponents(gpa, &out, path); - if (out.items.len == 0) return error.InvalidMountpoint; // "/" or equivalent - return out.toOwnedSliceSentinel(gpa, 0); -} - -fn appendComponents(gpa: Allocator, out: *std.ArrayList(u8), path: []const u8) !void { - var it = std.mem.tokenizeScalar(u8, path, '/'); - while (it.next()) |comp| { - if (std.mem.eql(u8, comp, ".")) continue; - if (std.mem.eql(u8, comp, "..")) { - // Pop the last component (lexically; "/.." stays "/"). - const idx = std.mem.lastIndexOfScalar(u8, out.items, '/') orelse 0; - out.shrinkRetainingCapacity(idx); - continue; - } - try out.append(gpa, '/'); - try out.appendSlice(gpa, comp); - } -} - -// --------------------------------------------------------------------------- -// Environment helpers -// --------------------------------------------------------------------------- - -/// Look a variable up in a raw envp block. -pub fn getenv(envp: [*:null]const ?[*:0]const u8, name: []const u8) ?[]const u8 { - var i: usize = 0; - while (envp[i]) |entry| : (i += 1) { - const kv = std.mem.span(entry); - if (kv.len > name.len and kv[name.len] == '=' and std.mem.eql(u8, kv[0..name.len], name)) { - return kv[name.len + 1 ..]; - } - } - return null; -} - -/// Every path `execve` should try for `name`, in order: just `name` if it -/// contains a '/', else `<dir>/<name>` for each `$PATH` element (an empty -/// element means the current directory; `$PATH` unset falls back to -/// `/usr/local/bin:/bin:/usr/bin`). -pub fn pathCandidates(gpa: Allocator, envp: [*:null]const ?[*:0]const u8, name: []const u8) ![]const [:0]const u8 { - if (name.len == 0) return error.EmptyProgramName; - var list: std.ArrayList([:0]const u8) = .empty; - errdefer { - for (list.items) |c| gpa.free(c); - list.deinit(gpa); - } - if (std.mem.indexOfScalar(u8, name, '/') != null) { - try list.append(gpa, try gpa.dupeZ(u8, name)); - return list.toOwnedSlice(gpa); - } - const path = getenv(envp, "PATH") orelse default_path; - var it = std.mem.splitScalar(u8, path, ':'); - while (it.next()) |dir| { - const d = if (dir.len == 0) "." else dir; - try list.append(gpa, try std.fmt.allocPrintSentinel(gpa, "{s}/{s}", .{ d, name }, 0)); - } - return list.toOwnedSlice(gpa); -} - -/// First PATH candidate that is an executable regular file, or the name -/// itself when it contains a '/'. Provided for completeness; `spawn` simply -/// tries `execve` on every candidate instead. -pub fn findInPath(gpa: Allocator, envp: [*:null]const ?[*:0]const u8, name: []const u8) ![:0]u8 { - const cands = try pathCandidates(gpa, envp, name); - defer { - for (cands) |c| gpa.free(c); - gpa.free(cands); - } - for (cands) |c| { - var stx: linux.Statx = undefined; - const rc = linux.statx(linux.AT.FDCWD, c.ptr, 0, .{ .TYPE = true, .MODE = true }, &stx); - if (linux.errno(rc) != .SUCCESS) continue; - if (stx.mode & linux.S.IFMT != linux.S.IFREG) continue; - if (stx.mode & 0o111 == 0) continue; - return gpa.dupeZ(u8, c); - } - return error.FileNotFound; -} - -/// New envp block: every entry of `envp` except `NINEPLAYER_MOUNT=...`, -/// followed by `NINEPLAYER_MOUNT=<mountpoint>`. -fn buildEnvp(gpa: Allocator, envp: [*:null]const ?[*:0]const u8, mountpoint: []const u8) ![:null]?[*:0]const u8 { - const key = "NINEPLAYER_MOUNT="; - var keep: usize = 0; - var i: usize = 0; - while (envp[i]) |entry| : (i += 1) { - if (!std.mem.startsWith(u8, std.mem.span(entry), key)) keep += 1; - } - const out = try gpa.allocSentinel(?[*:0]const u8, keep + 1, null); - errdefer gpa.free(out); - var j: usize = 0; - i = 0; - while (envp[i]) |entry| : (i += 1) { - if (std.mem.startsWith(u8, std.mem.span(entry), key)) continue; - out[j] = entry; - j += 1; - } - const mount_entry = try std.fmt.allocPrintSentinel(gpa, key ++ "{s}", .{mountpoint}, 0); - out[j] = mount_entry.ptr; - return out; -} - -fn buildArgv(gpa: Allocator, argv: []const []const u8) ![:null]?[*:0]const u8 { - const out = try gpa.allocSentinel(?[*:0]const u8, argv.len, null); - for (argv, 0..) |a, i| out[i] = (try gpa.dupeZ(u8, a)).ptr; - return out; -} - -// --------------------------------------------------------------------------- -// Mountpoint policy -// --------------------------------------------------------------------------- - -/// Make sure `path` is a directory, inside the *current* mount namespace: -/// -/// * already a directory → done; -/// * else `mkdir`; on `EACCES`/`EPERM`/`EROFS` shadow the parent directory -/// with a tmpfs that re-exposes every existing entry (bind mounts for -/// directories and files, recreated symlinks) and `mkdir` inside it; -/// * anything else fails with the errno and a hint. -/// -/// Every failure prints `9player: <step> <path>: E<errno>` to stderr before -/// returning. Meant to be called in the child of `spawn` (or from a -/// throwaway namespace: `unshare -Urm`). -pub fn ensureMountpoint(gpa: Allocator, path: [:0]const u8) !void { - if (fileType(linux.AT.FDCWD, path, false)) |ft| { - if (ft == .dir) return; - std.debug.print("9player: mountpoint {s}: exists but is not a directory\n", .{path}); - return error.Mountpoint; - } - if (fileType(linux.AT.FDCWD, path, true) == .symlink) { - std.debug.print("9player: mountpoint {s}: dangling symlink\n", .{path}); - return error.Mountpoint; - } - const mk = linux.errno(linux.mkdirat(linux.AT.FDCWD, path, 0o755)); - switch (mk) { - .SUCCESS => return, - .ACCES, .PERM, .ROFS => {}, - else => |e| { - std.debug.print("9player: mkdir {s}: E{t} (pass --mount an existing directory)\n", .{ path, e }); - return error.Mountpoint; - }, - } - const parent = std.fs.path.dirname(path) orelse "/"; - if (std.mem.eql(u8, parent, "/") or isSameDirectory(parent, "/")) { - std.debug.print("9player: mkdir {s}: E{t}; refusing to shadow / (pass --mount an existing directory)\n", .{ path, mk }); - return error.Mountpoint; - } - // The shadow rebuilds entries from /proc/self/fd/<fd>/<name>; a tmpfs - // over /proc (or a subtree of it) would take that away from itself. - if (std.mem.eql(u8, parent, "/proc") or std.mem.startsWith(u8, parent, "/proc/")) { - std.debug.print("9player: mkdir {s}: E{t}; refusing to shadow {s} (pass --mount an existing directory)\n", .{ path, mk, parent }); - return error.Mountpoint; - } - const parent_z = try gpa.dupeZ(u8, parent); - defer gpa.free(parent_z); - try shadowDirectory(gpa, parent_z); - switch (linux.errno(linux.mkdirat(linux.AT.FDCWD, path, 0o755))) { - .SUCCESS => {}, - else => |e| { - std.debug.print("9player: mkdir {s} (in shadow tmpfs): E{t}\n", .{ path, e }); - return error.Mountpoint; - }, - } -} - -const FileType = enum { dir, symlink, other }; - -/// True when both paths resolve (following symlinks, including magic ones -/// such as /proc/self/root) to the same inode. -fn isSameDirectory(a: []const u8, b: [*:0]const u8) bool { - var a_buf: [path_max]u8 = undefined; - const a_z = std.fmt.bufPrintZ(&a_buf, "{s}", .{a}) catch return false; - var sa: linux.Statx = undefined; - var sb: linux.Statx = undefined; - if (linux.errno(linux.statx(linux.AT.FDCWD, a_z, 0, .{ .INO = true }, &sa)) != .SUCCESS) return false; - if (linux.errno(linux.statx(linux.AT.FDCWD, b, 0, .{ .INO = true }, &sb)) != .SUCCESS) return false; - return sa.ino == sb.ino and sa.dev_major == sb.dev_major and sa.dev_minor == sb.dev_minor; -} - -fn fileType(dirfd: i32, name: [*:0]const u8, nofollow: bool) ?FileType { - var stx: linux.Statx = undefined; - const flags: u32 = if (nofollow) linux.AT.SYMLINK_NOFOLLOW else 0; - const rc = linux.statx(dirfd, name, flags, .{ .TYPE = true }, &stx); - if (linux.errno(rc) != .SUCCESS) return null; - return switch (stx.mode & linux.S.IFMT) { - linux.S.IFDIR => .dir, - linux.S.IFLNK => .symlink, - else => .other, - }; -} - -const Entry = struct { name: [:0]u8, kind: FileType }; - -/// Read every entry of the directory open at `fd` (excluding `.` and `..`). -fn listDir(gpa: Allocator, fd: i32, dirpath: []const u8) ![]Entry { - var list: std.ArrayList(Entry) = .empty; - errdefer { - for (list.items) |e| gpa.free(e.name); - list.deinit(gpa); - } - var buf: [32 * 1024]u8 align(@alignOf(linux.dirent64)) = undefined; - while (true) { - const rc = linux.getdents64(fd, &buf, buf.len); - switch (linux.errno(rc)) { - .SUCCESS => {}, - else => |e| { - std.debug.print("9player: getdents64 {s}: E{t}\n", .{ dirpath, e }); - return error.Mountpoint; - }, - } - if (rc == 0) break; - var off: usize = 0; - while (off < rc) { - const d: *align(1) const linux.dirent64 = @ptrCast(&buf[off]); - const name_ptr: [*:0]const u8 = @ptrCast(&buf[off + @offsetOf(linux.dirent64, "name")]); - const name = std.mem.span(name_ptr); - const dtype = d.type; - off += d.reclen; - if (std.mem.eql(u8, name, ".") or std.mem.eql(u8, name, "..")) continue; - if (list.items.len >= max_shadow_entries) { - std.debug.print("9player: refusing to shadow {s}: more than {d} entries\n", .{ dirpath, max_shadow_entries }); - return error.TooManyEntries; - } - const kind: FileType = switch (dtype) { - linux.DT.DIR => .dir, - linux.DT.LNK => .symlink, - linux.DT.UNKNOWN => fileType(fd, name_ptr, true) orelse .other, - else => .other, - }; - try list.append(gpa, .{ .name = try gpa.dupeZ(u8, name), .kind = kind }); - } - } - return list.toOwnedSlice(gpa); -} - -fn shadowDirectory(gpa: Allocator, parent: [:0]const u8) !void { - const open_rc = linux.open(parent, .{ .ACCMODE = .RDONLY, .DIRECTORY = true, .CLOEXEC = true }, 0); - switch (linux.errno(open_rc)) { - .SUCCESS => {}, - else => |e| { - std.debug.print("9player: open {s}: E{t}\n", .{ parent, e }); - return error.Mountpoint; - }, - } - const pfd: i32 = @intCast(open_rc); - defer _ = linux.close(pfd); - - const entries = try listDir(gpa, pfd, parent); - defer { - for (entries) |e| gpa.free(e.name); - gpa.free(entries); - } - - const tmpfs_opts: [*:0]const u8 = "mode=755"; - switch (linux.errno(linux.mount("tmpfs", parent, "tmpfs", linux.MS.NOSUID | linux.MS.NODEV, @intFromPtr(tmpfs_opts)))) { - .SUCCESS => {}, - else => |e| { - std.debug.print("9player: mount tmpfs on {s}: E{t}\n", .{ parent, e }); - return error.Mountpoint; - }, - } - - // `pfd` still refers to the original directory underneath the tmpfs, so - // `/proc/self/fd/<pfd>/<name>` reaches the hidden entries. - var src_buf: [path_max]u8 = undefined; - var dst_buf: [path_max]u8 = undefined; - var link_buf: [path_max]u8 = undefined; - for (entries) |e| { - const src = std.fmt.bufPrintZ(&src_buf, "/proc/self/fd/{d}/{s}", .{ pfd, e.name }) catch { - std.debug.print("9player: shadow {s}/{s}: name too long (skipped)\n", .{ parent, e.name }); - continue; - }; - const dst = std.fmt.bufPrintZ(&dst_buf, "{s}/{s}", .{ parent, e.name }) catch { - std.debug.print("9player: shadow {s}/{s}: name too long (skipped)\n", .{ parent, e.name }); - continue; - }; - switch (e.kind) { - .dir => { - if (!check("mkdir", dst, linux.mkdirat(linux.AT.FDCWD, dst, 0o755))) continue; - _ = check("bind", dst, linux.mount(src, dst, null, linux.MS.BIND | linux.MS.REC, 0)); - }, - .symlink => { - const rc = linux.readlinkat(pfd, e.name, &link_buf, link_buf.len - 1); - if (!check("readlink", dst, rc)) continue; - link_buf[rc] = 0; - const target: [*:0]const u8 = @ptrCast(&link_buf); - _ = check("symlink", dst, linux.symlinkat(target, linux.AT.FDCWD, dst)); - }, - .other => { - const rc = linux.openat(linux.AT.FDCWD, dst, .{ .ACCMODE = .WRONLY, .CREAT = true, .CLOEXEC = true }, 0o644); - if (!check("create", dst, rc)) continue; - _ = linux.close(@intCast(rc)); - _ = check("bind", dst, linux.mount(src, dst, null, linux.MS.BIND | linux.MS.REC, 0)); - }, - } - } -} - -/// Report a failed per-entry step as a warning (the entry is skipped; the -/// rest of the shadow is still useful). Returns true on success. -fn check(step: []const u8, path: [*:0]const u8, rc: usize) bool { - switch (linux.errno(rc)) { - .SUCCESS => return true, - else => |e| { - std.debug.print("9player: shadow: {s} {s}: E{t} (skipped)\n", .{ step, std.mem.span(path), e }); - return false; - }, - } -} - -// --------------------------------------------------------------------------- -// spawn -// --------------------------------------------------------------------------- - -const ChildArgs = struct { - gpa: Allocator, - status_sock: i32, - mountpoint: [:0]const u8, - fuse_opts_prefix: [:0]const u8, // everything after "fd=<n>," - mount_fuse: bool, - uid_map: []const u8, - gid_map: []const u8, - argv: [:null]?[*:0]const u8, - envp: [:null]?[*:0]const u8, - candidates: []const [:0]const u8, - name: []const u8, -}; - -/// Status channel protocol (child → parent, over a CLOEXEC socketpair): -/// a 0 byte means "namespace and mount are up" and carries the fuse fd as -/// SCM_RIGHTS; a non-zero byte is an exit status followed by a message. -/// EOF ends the conversation (exec succeeded, or the child died). -const ok_byte: u8 = 0; - -/// fork; the child unshares user+mount namespaces, maps its uid/gid, -/// makes `/` private, ensures the mountpoint, opens `/dev/fuse`, mounts it -/// on the mountpoint and sends the fd back. `spawn` returns at that point -/// (with `error.ChildFailed` and a message on stderr if any step failed). -/// The child then stats the mountpoint, which makes the kernel fetch the -/// root's attributes once the parent serves (the kernel seeds the fuse root -/// with uid 0, unmapped in the new user namespace, so nothing could be -/// created in the root until then), sets `NINEPLAYER_MOUNT` and execs -/// `argv`. An exec failure is reported on `Child.status_fd` and ends the -/// child with 126/127; collect it with `reportExecFailure` after -/// `bridge.serve` returns. -/// -/// If `installSignals` was called, the pid is stored into the registered -/// variable as soon as fork returns so no SIGCHLD can be missed. -pub fn spawn(gpa: Allocator, s: Spawn) !Child { - if (s.argv.len == 0 or s.argv[0].len == 0) { - std.debug.print("9player: empty program name\n", .{}); - return error.EmptyProgramName; - } - - const mountpoint = try gpa.dupeZ(u8, s.mountpoint); - defer gpa.free(mountpoint); - const fuse_opts_prefix = try std.fmt.allocPrintSentinel(gpa, "rootmode=40000,user_id={d},group_id={d},max_read={d}", .{ s.uid, s.gid, s.max_read }, 0); - defer gpa.free(fuse_opts_prefix); - var uid_buf: [64]u8 = undefined; - var gid_buf: [64]u8 = undefined; - const uid_map = try std.fmt.bufPrint(&uid_buf, "{d} {d} 1\n", .{ s.uid, s.uid }); - const gid_map = try std.fmt.bufPrint(&gid_buf, "{d} {d} 1\n", .{ s.gid, s.gid }); - const argv = try buildArgv(gpa, s.argv); - defer { - for (argv) |a| gpa.free(std.mem.span(a.?)); - gpa.free(argv); - } - const envp = try buildEnvp(gpa, s.envp, s.mountpoint); - defer { - gpa.free(std.mem.span(envp[envp.len - 1].?)); // the NINEPLAYER_MOUNT entry we created - gpa.free(envp); - } - const candidates = try pathCandidates(gpa, s.envp, s.argv[0]); - defer { - for (candidates) |c| gpa.free(c); - gpa.free(candidates); - } - - var sv: [2]i32 = undefined; - switch (linux.errno(linux.socketpair(linux.AF.UNIX, linux.SOCK.STREAM | linux.SOCK.CLOEXEC, 0, &sv))) { - .SUCCESS => {}, - else => |e| { - std.debug.print("9player: socketpair: E{t}\n", .{e}); - return error.SystemResources; - }, - } - - const child_args = ChildArgs{ - .gpa = gpa, - .status_sock = sv[1], - .mountpoint = mountpoint, - .fuse_opts_prefix = fuse_opts_prefix, - .mount_fuse = s.mount_fuse, - .uid_map = uid_map, - .gid_map = gid_map, - .argv = argv, - .envp = envp, - .candidates = candidates, - .name = s.argv[0], - }; - - const fork_rc = linux.fork(); - switch (linux.errno(fork_rc)) { - .SUCCESS => {}, - else => |e| { - _ = linux.close(sv[0]); - _ = linux.close(sv[1]); - std.debug.print("9player: fork: E{t}\n", .{e}); - return error.SystemResources; - }, - } - if (fork_rc == 0) childMain(&child_args); - - const pid: i32 = @intCast(fork_rc); - if (child_pid_ptr) |p| @atomicStore(i32, p, pid, .seq_cst); - _ = linux.close(sv[1]); - - // First byte: ok (with the fuse fd attached) or a failure status. - var first: [1]u8 = .{ok_byte}; // defined even if recvmsg stores nothing - var fuse_fd: i32 = -1; - var n: usize = 0; - while (true) { - const rc = recvWithFd(sv[0], &first, &fuse_fd, 0); - switch (linux.errno(rc)) { - .SUCCESS => {}, - .INTR => continue, - else => break, - } - n = rc; - break; - } - if (n == 1 and first[0] == ok_byte and (fuse_fd >= 0 or !s.mount_fuse)) { - return .{ .pid = pid, .fuse_fd = fuse_fd, .status_fd = sv[0] }; - } - - // Failure. A status byte means the child is exiting on its own and a - // message follows. Anything else (EOF: the child died before reporting; - // an ok byte without the fd: the SCM_RIGHTS transfer was truncated, e.g. - // EMFILE) is a protocol violation: the child may be about to exec with a - // dead mount, so kill it before waiting rather than reading the status - // socket until an exec'd program eventually exits. - const reported = n == 1 and first[0] != ok_byte; - if (!reported) _ = linux.kill(pid, .KILL); - if (fuse_fd >= 0) _ = linux.close(fuse_fd); - var msg: [512]u8 = undefined; - var len: usize = 0; - while (reported and len < msg.len) { - const rc = linux.read(sv[0], msg[len..].ptr, msg.len - len); - switch (linux.errno(rc)) { - .SUCCESS => {}, - .INTR => continue, - else => break, - } - if (rc == 0) break; - len += rc; - } - _ = linux.close(sv[0]); - if (reported) { - std.debug.print("9player: {s}\n", .{msg[0..len]}); - } else if (n == 1) { - std.debug.print("9player: child handshake failed: no fuse fd received (out of file descriptors?)\n", .{}); - } else { - std.debug.print("9player: child exited before reporting\n", .{}); - } - _ = waitChild(pid) catch {}; - if (child_pid_ptr) |p| @atomicStore(i32, p, 0, .seq_cst); - const status: u8 = if (reported) first[0] else setup_failure_status; - return switch (status) { - 126 => error.ExecPermission, - 127 => error.ExecNotFound, - else => error.ChildFailed, - }; -} - -/// After the child is gone (or the mount is dead): print the exec failure -/// the child reported on `status_fd`, if any, and close it. Returns the -/// status byte the child announced, or null when exec succeeded / nothing -/// was reported. Never blocks. -pub fn reportExecFailure(child: Child) ?u8 { - defer _ = linux.close(child.status_fd); - var msg: [512]u8 = undefined; - var len: usize = 0; - while (len < msg.len) { - var iov = [_]std.posix.iovec{.{ .base = msg[len..].ptr, .len = msg.len - len }}; - var hdr = linux.msghdr{ - .name = null, - .namelen = 0, - .iov = &iov, - .iovlen = 1, - .control = null, - .controllen = 0, - .flags = 0, - }; - const rc = linux.recvmsg(child.status_fd, &hdr, linux.MSG.DONTWAIT); - switch (linux.errno(rc)) { - .SUCCESS => {}, - .INTR => continue, - else => break, - } - if (rc == 0) break; - len += rc; - } - if (len == 0) return null; - std.debug.print("9player: {s}\n", .{msg[1..len]}); - return msg[0]; -} - -const cmsg_fd_len = @sizeOf(linux.cmsghdr) + @sizeOf(i32); -const cmsg_fd_space = std.mem.alignForward(usize, cmsg_fd_len, @sizeOf(usize)); - -/// sendmsg one data byte, optionally with `fd` attached as SCM_RIGHTS. -fn sendWithFd(sock: i32, byte: u8, fd: ?i32) usize { - const data = [_]u8{byte}; - const iov = [_]std.posix.iovec_const{.{ .base = &data, .len = 1 }}; - var cbuf: [cmsg_fd_space]u8 align(@alignOf(linux.cmsghdr)) = @splat(0); - var msg = linux.msghdr_const{ - .name = null, - .namelen = 0, - .iov = &iov, - .iovlen = 1, - .control = null, - .controllen = 0, - .flags = 0, - }; - if (fd) |f| { - const hdr: *linux.cmsghdr = @ptrCast(&cbuf); - hdr.* = .{ .len = cmsg_fd_len, .level = linux.SOL.SOCKET, .type = linux.SCM.RIGHTS }; - @memcpy(cbuf[@sizeOf(linux.cmsghdr)..][0..@sizeOf(i32)], std.mem.asBytes(&f)); - msg.control = &cbuf; - msg.controllen = cmsg_fd_space; - } - return linux.sendmsg(sock, &msg, linux.MSG.NOSIGNAL); -} - -/// recvmsg into `buf`; an SCM_RIGHTS fd, if any, is stored in `fd_out`. -fn recvWithFd(sock: i32, buf: []u8, fd_out: *i32, flags: u32) usize { - var iov = [_]std.posix.iovec{.{ .base = buf.ptr, .len = buf.len }}; - var cbuf: [cmsg_fd_space]u8 align(@alignOf(linux.cmsghdr)) = @splat(0); - var msg = linux.msghdr{ - .name = null, - .namelen = 0, - .iov = &iov, - .iovlen = 1, - .control = &cbuf, - .controllen = cbuf.len, - .flags = 0, - }; - const rc = linux.recvmsg(sock, &msg, linux.MSG.CMSG_CLOEXEC | flags); - if (linux.errno(rc) != .SUCCESS) return rc; - if (msg.controllen >= cmsg_fd_len) { - const hdr: *const linux.cmsghdr = @ptrCast(&cbuf); - if (hdr.level == linux.SOL.SOCKET and hdr.type == linux.SCM.RIGHTS and hdr.len >= cmsg_fd_len) { - var fd: i32 = undefined; - @memcpy(std.mem.asBytes(&fd), cbuf[@sizeOf(linux.cmsghdr)..][0..@sizeOf(i32)]); - fd_out.* = fd; - } - } - return rc; -} - -/// Child side of `spawn`. Never returns. -fn childMain(c: *const ChildArgs) noreturn { - resetSignals(); - - const rc_unshare = linux.errno(linux.unshare(linux.CLONE.NEWUSER | linux.CLONE.NEWNS)); - if (rc_unshare != .SUCCESS) childFail(c, setup_failure_status, "unshare(CLONE_NEWUSER|CLONE_NEWNS)", rc_unshare, true); - writeProcFile(c, "/proc/self/setgroups", "deny", true); - writeProcFile(c, "/proc/self/uid_map", c.uid_map, false); - writeProcFile(c, "/proc/self/gid_map", c.gid_map, false); - - const root: [*:0]const u8 = "/"; - const rc_priv = linux.mount(null, root, null, linux.MS.REC | linux.MS.PRIVATE, 0); - if (linux.errno(rc_priv) != .SUCCESS) childFail(c, setup_failure_status, "mount(/, MS_REC|MS_PRIVATE)", linux.errno(rc_priv), true); - - ensureMountpoint(c.gpa, c.mountpoint) catch { - childFail(c, setup_failure_status, "mountpoint setup failed (pass --mount an existing directory)", .SUCCESS, false); - }; - - var fuse_fd: ?i32 = null; - if (c.mount_fuse) { - // Must be opened here, after unshare: the kernel only mounts a fuse - // device opened from the mount's own user namespace. - const rc_open = linux.open("/dev/fuse", .{ .ACCMODE = .RDWR, .CLOEXEC = true }, 0); - switch (linux.errno(rc_open)) { - .SUCCESS => {}, - .NOENT => childFail(c, setup_failure_status, "open /dev/fuse: ENOENT (is the fuse module loaded? try: modprobe fuse)", .SUCCESS, false), - else => |e| childFail(c, setup_failure_status, "open /dev/fuse", e, true), - } - const fd: i32 = @intCast(rc_open); - var opts_buf: [256]u8 = undefined; - const opts = std.fmt.bufPrintZ(&opts_buf, "fd={d},{s}", .{ fd, c.fuse_opts_prefix }) catch unreachable; - const rc = linux.mount("9player", c.mountpoint, "fuse", linux.MS.NOSUID | linux.MS.NODEV, @intFromPtr(opts.ptr)); - if (linux.errno(rc) != .SUCCESS) childFail(c, setup_failure_status, "mount fuse", linux.errno(rc), true); - fuse_fd = fd; - } - const sent = sendWithFd(c.status_sock, ok_byte, fuse_fd); - if (linux.errno(sent) != .SUCCESS) linux.exit_group(setup_failure_status); - if (fuse_fd) |fd| { - _ = linux.close(fd); // the parent holds the connection now - // Force one GETATTR of the root (served by the parent, which is - // entering its serve loop now); see `spawn`. Errors don't matter. - var stx: linux.Statx = undefined; - _ = linux.statx(linux.AT.FDCWD, c.mountpoint, 0, .{ .TYPE = true }, &stx); - } - - var last: E = .NOENT; - var saw_acces = false; - for (c.candidates) |cand| { - const rc = linux.execve(cand.ptr, c.argv.ptr, c.envp.ptr); - last = linux.errno(rc); - switch (last) { - .NOENT, .NOTDIR, .LOOP, .NAMETOOLONG => continue, - .ACCES => { - saw_acces = true; - continue; - }, - else => break, - } - } - var buf: [512]u8 = undefined; - // "Not found" covers every candidate that could not even be resolved - // (a PATH element that is a file gives ENOTDIR, a symlink loop ELOOP); - // a candidate that existed but was not executable wins over those. - const not_found = switch (last) { - .NOENT, .NOTDIR, .LOOP, .NAMETOOLONG => true, - else => false, - }; - if (not_found and saw_acces) last = .ACCES; - const status: u8 = if (not_found and !saw_acces) 127 else 126; - const text = std.fmt.bufPrint(&buf, "exec {s}", .{c.name}) catch "exec"; - childFail(c, status, text, last, true); -} - -fn writeProcFile(c: *const ChildArgs, path: [*:0]const u8, data: []const u8, ignore_missing: bool) void { - const rc = linux.open(path, .{ .ACCMODE = .WRONLY, .CLOEXEC = true }, 0); - switch (linux.errno(rc)) { - .SUCCESS => {}, - .NOENT => if (ignore_missing) return else childFail(c, setup_failure_status, std.mem.span(path), .NOENT, true), - else => |e| childFail(c, setup_failure_status, std.mem.span(path), e, true), - } - const fd: i32 = @intCast(rc); - const w = linux.write(fd, data.ptr, data.len); - const we = linux.errno(w); - _ = linux.close(fd); - if (we != .SUCCESS) childFail(c, setup_failure_status, std.mem.span(path), we, true); - if (w != data.len) childFail(c, setup_failure_status, std.mem.span(path), .IO, true); -} - -/// Write `<status byte><step>[: E<errno>]` to the status socket and exit. -fn childFail(c: *const ChildArgs, status: u8, step: []const u8, e: E, with_errno: bool) noreturn { - var buf: [600]u8 = undefined; - buf[0] = status; - const rest = if (with_errno) - std.fmt.bufPrint(buf[1..], "{s}: E{t}", .{ step, e }) catch buf[1..1] - else - std.fmt.bufPrint(buf[1..], "{s}", .{step}) catch buf[1..1]; - const msg = buf[0 .. 1 + rest.len]; - var off: usize = 0; - while (off < msg.len) { - const rc = linux.write(c.status_sock, msg[off..].ptr, msg.len - off); - if (linux.errno(rc) == .INTR) continue; - if (linux.errno(rc) != .SUCCESS) break; - off += rc; - } - linux.exit_group(status); -} - -// --------------------------------------------------------------------------- -// Signals -// --------------------------------------------------------------------------- - -var child_pid_ptr: ?*i32 = null; -var chld_pipe_w: i32 = -1; -var reaped = std.atomic.Value(bool).init(false); -var reaped_status = std.atomic.Value(u32).init(0); -/// A second child (the `--spawn` server) that the SIGCHLD handler reaps so -/// it does not linger as a zombie when it dies mid-session. Its exit does -/// not stop the serve loop. 0 = none. -var server_pid = std.atomic.Value(i32).init(0); - -/// Register the `--spawn` server for reaping by the SIGCHLD handler. -pub fn watchServer(pid: i32) void { - server_pid.store(pid, .seq_cst); -} - -/// Seconds the serve loop gets to come back after the child died before -/// the watchdog ends the process anyway. -pub const exit_grace_seconds: isize = 3; - -/// The watched child is already dead but the serve loop has not come back -/// (it is stuck in a 9P request the server never answers): a terminal -/// signal, or the watchdog armed by `onChld`, then ends 9player with the -/// child's status instead of hanging. Nothing is lost: the mount is torn -/// down when the process exits. -fn bailIfChildGone() void { - if (!reaped.load(.acquire)) return; - const srv = server_pid.load(.seq_cst); - if (srv > 0) _ = linux.kill(srv, .TERM); - linux.exit_group(decodeStatus(reaped_status.load(.acquire))); -} - -fn armWatchdog() void { - // setitimer takes an itimerval; std declares it with itimerspec, which - // has the same layout on 64-bit targets (the sub-second field is 0). - const t = linux.itimerspec{ - .it_interval = .{ .sec = 0, .nsec = 0 }, - .it_value = .{ .sec = exit_grace_seconds, .nsec = 0 }, - }; - _ = linux.setitimer(@intFromEnum(linux.ITIMER.REAL), &t, null); -} - -fn onAlarm(_: linux.SIG) callconv(.c) void { - bailIfChildGone(); -} - -fn onForward(sig: linux.SIG) callconv(.c) void { - const p = child_pid_ptr orelse return; - const pid = @atomicLoad(i32, p, .seq_cst); - if (pid > 0) _ = linux.kill(pid, sig); - bailIfChildGone(); -} - -/// SIGINT/SIGQUIT: the child owns the tty and gets them itself; we only -/// react when the child is already gone (see `bailIfChildGone`). -fn onTerminal(_: linux.SIG) callconv(.c) void { - bailIfChildGone(); -} - -/// Only the watched child counts: reap it here (WNOHANG), remember its -/// status, forget its pid (so a later SIGTERM cannot hit a recycled pid) -/// and poke the self-pipe. The `--spawn` server is reaped too but does not -/// interrupt `bridge.serve`; SIGCHLD from anything else is ignored. -fn onChld(_: linux.SIG) callconv(.c) void { - const srv = server_pid.load(.seq_cst); - if (srv > 0) { - var sst: u32 = 0; - const src = linux.waitpid(srv, &sst, linux.W.NOHANG); - if (linux.errno(src) == .SUCCESS and src != 0) server_pid.store(0, .seq_cst); - } - const p = child_pid_ptr orelse return; - const pid = @atomicLoad(i32, p, .seq_cst); - if (pid <= 0) return; - var st: u32 = 0; - const rc = linux.waitpid(pid, &st, linux.W.NOHANG); - if (linux.errno(rc) != .SUCCESS or rc == 0) return; - reaped_status.store(st, .release); - reaped.store(true, .release); - @atomicStore(i32, p, 0, .seq_cst); - const b = [_]u8{'c'}; - _ = linux.write(chld_pipe_w, &b, 1); - armWatchdog(); -} - -/// SIGPIPE ignored; SIGINT/SIGQUIT effectively ignored (the child owns the -/// tty) unless the child is already dead; SIGTERM/SIGHUP forwarded to -/// `*child_pid`; SIGCHLD for `*child_pid` reaps it, writes a byte to a -/// nonblocking self-pipe whose read end is returned (use it as `stop_fd`) -/// and arms a watchdog (`exit_grace_seconds`, SIGALRM) that ends the -/// process with the child's status should the serve loop stay blocked. -/// `*child_pid` is filled in by `spawn`. -pub fn installSignals(child_pid: *i32) !i32 { - child_pid_ptr = child_pid; - var fds: [2]i32 = undefined; - switch (linux.errno(linux.pipe2(&fds, .{ .CLOEXEC = true, .NONBLOCK = true }))) { - .SUCCESS => {}, - else => |e| { - std.debug.print("9player: pipe2: E{t}\n", .{e}); - return error.SystemResources; - }, - } - chld_pipe_w = fds[1]; - - const ign = linux.Sigaction{ .handler = .{ .handler = linux.SIG.IGN }, .mask = linux.sigemptyset(), .flags = 0 }; - const term = linux.Sigaction{ .handler = .{ .handler = &onTerminal }, .mask = linux.sigemptyset(), .flags = linux.SA.RESTART }; - const fwd = linux.Sigaction{ .handler = .{ .handler = &onForward }, .mask = linux.sigemptyset(), .flags = linux.SA.RESTART }; - const chld = linux.Sigaction{ .handler = .{ .handler = &onChld }, .mask = linux.sigemptyset(), .flags = linux.SA.RESTART | linux.SA.NOCLDSTOP }; - const alrm = linux.Sigaction{ .handler = .{ .handler = &onAlarm }, .mask = linux.sigemptyset(), .flags = linux.SA.RESTART }; - std.posix.sigaction(.INT, &term, null); - std.posix.sigaction(.QUIT, &term, null); - std.posix.sigaction(.ALRM, &alrm, null); - std.posix.sigaction(.PIPE, &ign, null); - std.posix.sigaction(.TERM, &fwd, null); - std.posix.sigaction(.HUP, &fwd, null); - std.posix.sigaction(.CHLD, &chld, null); - return fds[0]; -} - -/// Restore default dispositions in the child before exec (ignored signals -/// would otherwise survive execve). -fn resetSignals() void { - const dfl = linux.Sigaction{ .handler = .{ .handler = linux.SIG.DFL }, .mask = linux.sigemptyset(), .flags = 0 }; - inline for (.{ linux.SIG.INT, linux.SIG.QUIT, linux.SIG.PIPE, linux.SIG.TERM, linux.SIG.HUP, linux.SIG.CHLD, linux.SIG.ALRM }) |sig| { - _ = linux.sigaction(sig, &dfl, null); - } -} - -// --------------------------------------------------------------------------- -// Waiting -// --------------------------------------------------------------------------- - -fn takeReaped() ?u32 { - if (!reaped.load(.acquire)) return null; - return reaped_status.load(.acquire); -} - -/// waitpid status → exit code (`128+sig` when killed by a signal). -pub fn decodeStatus(st: u32) u8 { - if (linux.W.IFEXITED(st)) return linux.W.EXITSTATUS(st); - if (linux.W.IFSIGNALED(st)) return 128 +% @as(u8, @truncate(@intFromEnum(linux.W.TERMSIG(st)))); - return 1; -} - -/// Block until `pid` exits (the SIGCHLD handler may have reaped it already). -pub fn waitChild(pid: i32) !u8 { - while (true) { - if (takeReaped()) |st| return decodeStatus(st); - var st: u32 = 0; - const rc = linux.waitpid(pid, &st, 0); - switch (linux.errno(rc)) { - .SUCCESS => return decodeStatus(st), - .INTR => continue, - .CHILD => { - if (takeReaped()) |s| return decodeStatus(s); - return error.NoChild; - }, - else => |e| { - std.debug.print("9player: waitpid: E{t}\n", .{e}); - return error.Wait; - }, - } - } -} - -/// Non-blocking: the exit status of `pid` if it has exited, else null. -pub fn reapIfExited(pid: i32) ?u8 { - if (takeReaped()) |st| return decodeStatus(st); - var st: u32 = 0; - const rc = linux.waitpid(pid, &st, linux.W.NOHANG); - switch (linux.errno(rc)) { - .SUCCESS => return if (rc == 0) null else decodeStatus(st), - .CHILD => return if (takeReaped()) |s| decodeStatus(s) else null, - else => return null, - } -} - -/// Reap any child (used for the `--spawn` server at exit). Non-blocking. -pub fn reapAny(pid: i32) void { - var st: u32 = 0; - _ = linux.waitpid(pid, &st, linux.W.NOHANG); -} - -// --------------------------------------------------------------------------- -// Tests (no namespaces needed; `ensureMountpoint` is exercised by -// test/integration.sh through the 9player binary) -// --------------------------------------------------------------------------- - -const testing = std.testing; - -test "normalizePath: absolute paths" { - const gpa = testing.allocator; - const cases = [_]struct { in: []const u8, out: []const u8 }{ - .{ .in = "/mnt/9p", .out = "/mnt/9p" }, - .{ .in = "/mnt/9p/", .out = "/mnt/9p" }, - .{ .in = "//mnt///9p//", .out = "/mnt/9p" }, - .{ .in = "/mnt/./9p/.", .out = "/mnt/9p" }, - .{ .in = "/mnt/x/../9p", .out = "/mnt/9p" }, - .{ .in = "/../mnt/9p", .out = "/mnt/9p" }, - .{ .in = "/a/b/c/../..", .out = "/a" }, - }; - for (cases) |c| { - const got = try normalizePath(gpa, "/cwd", c.in); - defer gpa.free(got); - try testing.expectEqualStrings(c.out, got); - try testing.expectEqual(@as(u8, 0), got[got.len]); - } -} - -test "normalizePath: relative paths use cwd" { - const gpa = testing.allocator; - const cases = [_]struct { cwd: []const u8, in: []const u8, out: []const u8 }{ - .{ .cwd = "/home/me", .in = "mnt", .out = "/home/me/mnt" }, - .{ .cwd = "/home/me", .in = "./mnt/", .out = "/home/me/mnt" }, - .{ .cwd = "/home/me", .in = "../mnt", .out = "/home/mnt" }, - .{ .cwd = "/home/me/", .in = ".", .out = "/home/me" }, - .{ .cwd = "/", .in = "x", .out = "/x" }, - }; - for (cases) |c| { - const got = try normalizePath(gpa, c.cwd, c.in); - defer gpa.free(got); - try testing.expectEqualStrings(c.out, got); - } -} - -test "normalizePath: rejects root and empty" { - const gpa = testing.allocator; - try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/cwd", "/")); - try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/cwd", "///")); - try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/cwd", "/mnt/..")); - try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/cwd", "")); - try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/", "..")); -} - -test "resolveMountpoint: relative resolves against the real cwd" { - const gpa = testing.allocator; - const got = try resolveMountpoint(gpa, "sub/dir"); - defer gpa.free(got); - try testing.expect(got[0] == '/'); - try testing.expect(std.mem.endsWith(u8, got, "/sub/dir")); -} - -test "getenv" { - const env = [_:null]?[*:0]const u8{ "PATH=/a:/b", "X=", "PATHX=no", "NINEPLAYER_MOUNT=/m" }; - const envp: [*:null]const ?[*:0]const u8 = &env; - try testing.expectEqualStrings("/a:/b", getenv(envp, "PATH").?); - try testing.expectEqualStrings("", getenv(envp, "X").?); - try testing.expectEqualStrings("/m", getenv(envp, "NINEPLAYER_MOUNT").?); - try testing.expect(getenv(envp, "NOPE") == null); - try testing.expect(getenv(envp, "PAT") == null); -} - -test "pathCandidates: PATH search" { - const gpa = testing.allocator; - const env = [_:null]?[*:0]const u8{ "PATH=/usr/local/bin::/usr/bin", "HOME=/h" }; - const cands = try pathCandidates(gpa, &env, "fish"); - defer { - for (cands) |c| gpa.free(c); - gpa.free(cands); - } - try testing.expectEqual(@as(usize, 3), cands.len); - try testing.expectEqualStrings("/usr/local/bin/fish", cands[0]); - try testing.expectEqualStrings("./fish", cands[1]); - try testing.expectEqualStrings("/usr/bin/fish", cands[2]); -} - -test "pathCandidates: slash means no search; default PATH" { - const gpa = testing.allocator; - const env = [_:null]?[*:0]const u8{"HOME=/h"}; - { - const cands = try pathCandidates(gpa, &env, "./bin/x"); - defer { - for (cands) |c| gpa.free(c); - gpa.free(cands); - } - try testing.expectEqual(@as(usize, 1), cands.len); - try testing.expectEqualStrings("./bin/x", cands[0]); - } - { - const cands = try pathCandidates(gpa, &env, "sh"); - defer { - for (cands) |c| gpa.free(c); - gpa.free(cands); - } - try testing.expectEqual(@as(usize, 3), cands.len); - try testing.expectEqualStrings("/usr/local/bin/sh", cands[0]); - try testing.expectEqualStrings("/bin/sh", cands[1]); - } - try testing.expectError(error.EmptyProgramName, pathCandidates(gpa, &env, "")); -} - -test "findInPath finds sh" { - const gpa = testing.allocator; - const env = [_:null]?[*:0]const u8{"PATH=/nonexistent:/bin:/usr/bin"}; - const p = try findInPath(gpa, &env, "sh"); - defer gpa.free(p); - try testing.expect(std.mem.endsWith(u8, p, "/sh")); - try testing.expectError(error.FileNotFound, findInPath(gpa, &env, "definitely-not-a-program-9player")); -} - -test "buildEnvp replaces NINEPLAYER_MOUNT" { - const gpa = testing.allocator; - const env = [_:null]?[*:0]const u8{ "A=1", "NINEPLAYER_MOUNT=/old", "B=2" }; - const out = try buildEnvp(gpa, &env, "/mnt/9p"); - defer { - gpa.free(std.mem.span(out[out.len - 1].?)); - gpa.free(out); - } - try testing.expectEqual(@as(usize, 3), out.len); - try testing.expectEqualStrings("A=1", std.mem.span(out[0].?)); - try testing.expectEqualStrings("B=2", std.mem.span(out[1].?)); - try testing.expectEqualStrings("NINEPLAYER_MOUNT=/mnt/9p", std.mem.span(out[2].?)); - try testing.expect(out[3] == null); - try testing.expectEqualStrings("/mnt/9p", getenv(out.ptr, "NINEPLAYER_MOUNT").?); -} - -test "decodeStatus" { - try testing.expectEqual(@as(u8, 0), decodeStatus(0)); - try testing.expectEqual(@as(u8, 7), decodeStatus(7 << 8)); - try testing.expectEqual(@as(u8, 255), decodeStatus(255 << 8)); - try testing.expectEqual(@as(u8, 128 + 9), decodeStatus(9)); // SIGKILL - try testing.expectEqual(@as(u8, 128 + 15), decodeStatus(15)); // SIGTERM -} - -test "ensureMountpoint: existing directory is accepted, plain file rejected" { - const gpa = testing.allocator; - try ensureMountpoint(gpa, "/tmp"); - try testing.expectError(error.Mountpoint, ensureMountpoint(gpa, "/proc/self/status")); -} |
