From ba996acfcad1698adbf4a1834fe50e73b1c6cab9 Mon Sep 17 00:00:00 2001 From: Gabriel Schneider Date: Sat, 19 Sep 2026 23:28:22 -0300 Subject: Rename programs: 9player -> 9ns, introspect -> 9proc, app -> web (9web) Directories, binaries, build options (-D9ns, -D9proc), step names, module name (9proc), thread and fs names, env var NINEPLAYER_MOUNT -> NINE_MOUNT, docs and test scripts. Browser assets move to web/static. Co-Authored-By: Claude Fable 5.1 --- 9ns/README.md | 156 ++++++ 9ns/build.zig | 71 +++ 9ns/docs/DESIGN.md | 405 ++++++++++++++ 9ns/src/bridge.zig | 971 ++++++++++++++++++++++++++++++++++ 9ns/src/fuse.zig | 653 +++++++++++++++++++++++ 9ns/src/main.zig | 444 ++++++++++++++++ 9ns/src/nine.zig | 756 ++++++++++++++++++++++++++ 9ns/src/ns.zig | 1082 ++++++++++++++++++++++++++++++++++++++ 9ns/test/adv_bridge_hostile.py | 487 +++++++++++++++++ 9ns/test/adv_bridge_hostile.sh | 181 +++++++ 9ns/test/adv_bridge_semantics.sh | 205 ++++++++ 9ns/test/adv_bridge_stress.sh | 64 +++ 9ns/test/adv_ns_process.sh | 202 +++++++ 9ns/test/adversarial.sh | 15 + 9ns/test/integration.sh | 160 ++++++ 15 files changed, 5852 insertions(+) create mode 100644 9ns/README.md create mode 100644 9ns/build.zig create mode 100644 9ns/docs/DESIGN.md create mode 100644 9ns/src/bridge.zig create mode 100644 9ns/src/fuse.zig create mode 100644 9ns/src/main.zig create mode 100644 9ns/src/nine.zig create mode 100644 9ns/src/ns.zig create mode 100755 9ns/test/adv_bridge_hostile.py create mode 100755 9ns/test/adv_bridge_hostile.sh create mode 100755 9ns/test/adv_bridge_semantics.sh create mode 100755 9ns/test/adv_bridge_stress.sh create mode 100755 9ns/test/adv_ns_process.sh create mode 100755 9ns/test/adversarial.sh create mode 100755 9ns/test/integration.sh (limited to '9ns') diff --git a/9ns/README.md b/9ns/README.md new file mode 100644 index 0000000..04edffb --- /dev/null +++ b/9ns/README.md @@ -0,0 +1,156 @@ +# 9ns + +Mount a 9P2000 file tree into a fresh mount namespace and run a program in it, +as a plain user, without touching the host's mount table. + +```sh +9ns --unix /run/user/1000/acme -- fish # a shell that sees the tree at /mnt/9p +9ns --tcp 127.0.0.1:564 -- claude # an agent that sees it too +9ns --spawn '9proc-demo --stdio' -- bash # start the server yourself, talk over a socketpair +``` + +Inside, the tree is ordinary files: `ls`, `cat`, `echo x > ctl`, editors, +`find`, `rsync`, whatever. `$NINE_MOUNT` tells programs where it is +(default `/mnt/9p`). When the program exits, 9ns exits with its status +and the namespace, mount and connection disappear. + +## How it works + +The kernel's own `9p` filesystem cannot be mounted inside an unprivileged user +namespace, so 9ns is a small FUSE server that speaks 9P2000 to the real +server. There is no libfuse and no libc: `src/fuse.zig` implements the subset +of the kernel FUSE protocol needed, straight from `linux/fuse.h`. + +``` + program (fish/bash/claude) 9ns (parent) 9P server + in a new user+mount namespace │ + /mnt/9p ── FUSE ──▶ kernel ────▶│ fuse.zig ─▶ bridge.zig ─▶ nine.zig ──▶ unix / tcp / socketpair + │ (framing) (translation) (cloud9 Client) +``` + +1. The parent connects to the 9P server (version + attach) so failures are + reported before anything is forked. +2. The child does `unshare(CLONE_NEWUSER|CLONE_NEWNS)`, maps its own uid/gid, + makes every mount private, opens `/dev/fuse` (it must be opened inside the + new user namespace), mounts it on the mountpoint and passes the descriptor + back to the parent over `SCM_RIGHTS`, then execs the program. +3. The parent serves FUSE requests by translating them into 9P transactions + (`Twalk`, `Topen`, `Tread`, `Twrite`, `Tcreate`, `Tremove`, `Tstat`, + `Twstat`) until the child exits or the namespace disappears. + +Files are opened with `FOPEN_DIRECT_IO`, so synthetic files that report length +0 (the 9P convention for control files) still read correctly, and `O_TRUNC` +travels inside the 9P open mode (`OTRUNC`) rather than as a separate +truncate. Repeated lookups of the same qid map to the same inode. + +If `/mnt/9p` does not exist and cannot be created (the normal case), 9ns +mounts a tmpfs over `/mnt` *inside the namespace only* and bind-mounts every +existing entry of `/mnt` back into it, so nothing is hidden. Pass `--mount DIR` +to use any other directory. + +## Building and testing + +9ns lives in the [cloud9](../) repository as `cloud9/9ns/`, beside +the 9P2000 protocol library it is built on, and is wired into cloud9's +`build.zig` through the fragment `9ns/build.zig`. Everything is run from +the cloud9 root with Zig 0.16: + +```sh +zig build # zig-out/bin/{9ns,9proc-demo,9web,cloud9-probe} +zig build 9ns # build and install only zig-out/bin/9ns +zig build 9ns-test # unit tests (protocol structs, session, bridge, namespace helpers) +zig build 9ns-itest # integration tests: real namespaces, real FUSE, + # 9proc-demo over unix/tcp/socketpair, and plan9port's + # ramfs when /usr/lib/plan9/bin/ramfs is installed +zig build 9ns-adv # adversarial suites: hostile 9P servers, FUSE semantics, + # namespace/signal edge cases, stress (several minutes) +zig build -Doptimize=ReleaseSafe +zig build -D9ns=false # leave 9ns out (the default on non-Linux targets) +``` + +The integration suites mount the `9proc-demo` server (`../9proc`), +so they need `-D9proc=true` (the default on Linux), unprivileged user +namespaces (`kernel.unprivileged_userns_clone=1` on distributions that have +the knob), `/dev/fuse` and Python 3; they skip themselves otherwise. +`zig build programs-test` and `programs-itest` run the unit and integration +steps of every program in the repository. + +## Usage + +``` +9ns [options] -- PROGRAM [ARGS...] + +Transport (exactly one): + --unix PATH Unix stream socket + --tcp IP:PORT TCP (IPv4/IPv6 literal) + --fd N an already-connected inherited descriptor + --spawn CMD run CMD via /bin/sh -c with a socketpair on its stdin/stdout + +Options: + --mount PATH mountpoint inside the new namespace (default /mnt/9p) + --uname NAME 9P user name (default $USER) + --aname NAME 9P tree to attach (default "") + --msize BYTES maximum 9P message size to request (default 131072, max 16 MiB) + --cache SECONDS attr/entry cache validity, fractional allowed (default 1) + --no-direct-io let the kernel cache file pages (trusts stat length) + --debug trace FUSE and 9P operations on stderr + --help, --version +``` + +PROGRAM defaults to `$SHELL`. Exit status is the program's (`128+signal` if it +was killed); 125 means 9ns itself failed (usage, connect, namespace, +mount); 126/127 are exec failures as usual. + +## 9proc-demo: a demo 9P server + +`9proc-demo` is a single-binary 9P2000 server whose file tree is the binary +itself: build-time facts, `comptime` reflection and live runtime state. It is +the demo of the [9proc library](../9proc) (`../9proc/demo/main.zig`), +built and installed by `zig build 9proc`. + +``` +/README +/build/{zig_version,target,optimize,time,change} captured by build.zig (jj change id, UTC time) +/comptime/types//{name,size,align,fields} @sizeOf/@alignOf/@typeInfo, generated at comptime +/comptime/decls pub declarations of the server module +/runtime/{pid,ppid,uptime,argv,cwd,env,clients} +/runtime/fn/ reading calls a Zig function (hostname, now, random, uname, fib30); + the directory is generated from @typeInfo of the Fns struct +/runtime/ctl write "fib N" | "add A B" | "echo TEXT" | "sleep-ms N", read the result +/scratch/ in-memory read/write tree +``` + +`/runtime/env` exposes the server's whole environment, so serve 9proc-demo on +a Unix socket or loopback only. + +```sh +zig-out/bin/9proc-demo --unix /tmp/intro.sock & +zig-out/bin/9ns --unix /tmp/intro.sock -- sh -c ' + cat $NINE_MOUNT/build/zig_version; echo + cat $NINE_MOUNT/comptime/types/Qid/fields + echo "fib 20" > $NINE_MOUNT/runtime/ctl; cat $NINE_MOUNT/runtime/ctl' +``` + +The same command with `claude -p "explore /mnt/9p ..."` as the program gives an +agent a live, file-shaped view into a running process; that is the intended +use. + +## Limitations + +* One 9P request is in flight at a time; a server read that blocks (event + files) stalls the mount until it returns. +* Base 9P2000 only: no symlinks, ownership, xattrs or locks. Every file is + reported as owned by the invoking user. Cross-directory rename is `EXDEV`. +* No PID namespace and no `/proc` remount. `--tcp` takes IP literals only + (no libc, no resolver). +* Linux only. + +## Relation to cloud9 + +9ns consumes cloud9 as the module `cloud9` and keeps all mounting, +namespace and process policy on its side, which is what cloud9's design asks +of applications. It ships from the cloud9 repository as the sibling directory +`9ns/` (sources in `src/`, suites in `test/`, this README and +`docs/DESIGN.md`) with a build fragment that the root `build.zig` enables with +`-D9ns` on Linux targets; nothing in the code depends on that layout. +`docs/DESIGN.md` has the full module contracts. diff --git a/9ns/build.zig b/9ns/build.zig new file mode 100644 index 0000000..7f2155f --- /dev/null +++ b/9ns/build.zig @@ -0,0 +1,71 @@ +//! Build fragment for 9ns: mount a 9P2000 tree into a fresh mount +//! namespace via FUSE and run a program in it (Linux only, no libc). It is +//! `@import`ed by the root build.zig and called with the root builder, so +//! every `b.path(...)` here is relative to the cloud9 root (hence the +//! `9ns/` prefix), every option is defined by the root (no +//! `standardTargetOptions` here) and every step it registers lands in the +//! root's step list under the `9ns` prefix. +//! +//! Steps: 9ns, 9ns-test, 9ns-itest, 9ns-adv. +const std = @import("std"); + +/// What the root passes in. The root owns target/optimize resolution and the +/// cloud9 module; the integration suites also need a 9P server to mount, +/// which is the 9proc demo built by the sibling fragment. +pub const Context = struct { + target: std.Build.ResolvedTarget, + optimize: std.builtin.OptimizeMode, + cloud9: *std.Build.Module, + /// The `9proc-demo` server; null when 9proc is disabled, in + /// which case the end-to-end steps exist but fail with a notice. + proc_demo: ?*std.Build.Step.Compile, +}; + +pub const Artifacts = struct { + /// The 9ns executable (also installed by the plain `zig build`). + exe: *std.Build.Step.Compile, + /// `9ns-test`, `9ns-itest`, `9ns-adv`. + test_step: *std.Build.Step, + itest_step: *std.Build.Step, + adv_step: *std.Build.Step, +}; + +pub fn add(b: *std.Build, ctx: Context) Artifacts { + const target = ctx.target; + const optimize = ctx.optimize; + std.debug.assert(target.result.os.tag == .linux); // the root only enables 9ns on Linux + + const ns_mod = b.createModule(.{ + .root_source_file = b.path("9ns/src/main.zig"), + .target = target, + .optimize = optimize, + .imports = &.{.{ .name = "cloud9", .module = ctx.cloud9 }}, + }); + const ns = b.addExecutable(.{ .name = "9ns", .root_module = ns_mod }); + const install = b.addInstallArtifact(ns, .{}); + b.getInstallStep().dependOn(&install.step); + b.step("9ns", "Build and install only the 9ns binary").dependOn(&install.step); + + // Unit tests: protocol structs, session, bridge, namespace helpers (main.zig + // reaches every module). + const test_step = b.step("9ns-test", "Run 9ns's unit tests"); + test_step.dependOn(&b.addRunArtifact(b.addTest(.{ .root_module = ns_mod })).step); + + // End-to-end suites: real namespaces, real FUSE, a real 9P server. + const itest = b.step("9ns-itest", "Run 9ns/test/integration.sh (needs unprivileged user namespaces and /dev/fuse)"); + const adv = b.step("9ns-adv", "Run 9ns/test/adversarial.sh (hostile servers, namespaces, stress; several minutes)"); + const demo = ctx.proc_demo orelse { + const fail = b.addFail("9ns-itest and 9ns-adv need the 9proc-demo server (build with -D9proc=true)"); + itest.dependOn(&fail.step); + adv.dependOn(&fail.step); + return .{ .exe = ns, .test_step = test_step, .itest_step = itest, .adv_step = adv }; + }; + inline for (.{ .{ itest, "integration" }, .{ adv, "adversarial" } }) |pair| { + const run = b.addSystemCommand(&.{"bash"}); + run.addFileArg(b.path("9ns/test/" ++ pair[1] ++ ".sh")); + run.addArtifactArg(ns); + run.addArtifactArg(demo); + pair[0].dependOn(&run.step); + } + return .{ .exe = ns, .test_step = test_step, .itest_step = itest, .adv_step = adv }; +} diff --git a/9ns/docs/DESIGN.md b/9ns/docs/DESIGN.md new file mode 100644 index 0000000..7f943a8 --- /dev/null +++ b/9ns/docs/DESIGN.md @@ -0,0 +1,405 @@ +# 9ns design + +`9ns` mounts a 9P2000 file tree served over a Unix or TCP stream socket +into a **fresh mount namespace** and runs a program inside it. The program +(fish, bash, `claude`, anything) sees the 9P tree as ordinary files, without +root and without touching the host's mount table. + +## Why FUSE + +The kernel's own `9p` filesystem is not mountable inside an unprivileged user +namespace (it lacks `FS_USERNS_MOUNT`) and loading it needs root. FUSE has been +user-namespace mountable since Linux 4.18, and `/dev/fuse` is world read/write. +So 9ns is a tiny FUSE server that speaks 9P2000 to the real server: + +``` + program (fish/bash/claude) 9ns (parent) 9P server + in new user+mount namespace │ (9proc-demo, + /mnt/9p ─── FUSE ───▶ kernel ──▶│ fuse.zig ──▶ bridge.zig ──▶ nine.zig ──▶ ramfs, ...) + │ (framing) (translation) (cloud9 Client) +``` + +No libfuse: `src/fuse.zig` implements the small subset of the kernel FUSE +protocol we need directly against `/usr/include/linux/fuse.h`. + +## Toolchain facts (Zig 0.16) + +* Zig 0.16.0 at `/usr/bin/zig`, std at `/usr/lib/zig/std`. **Grep the std + tree before assuming an API exists**; 0.16 moved a lot of process/fs code + behind `std.Io`. Raw Linux syscalls in `std.os.linux` (`fork`, `execve`, + `mount`, `unshare`, `waitpid`, `pipe2`, `socketpair`, `poll`, `read`, `write`, + `open`, `openat`, `getdents64`, `sigaction`, `kill`, `readlinkat`, `mkdirat`, + `symlinkat`) are the intended low-level path. They return `usize`; decode + with `std.os.linux.errno(rc)` (an `E` enum, `.SUCCESS` when ok). +* `std.posix.poll`, `std.posix.sigaction`, `std.posix.read`, `std.posix.kill` + exist. `std.posix.fork/execve/waitpid/pipe2/socketpair` do **not**. +* `pub fn main() !void` and `pub fn main(init: std.process.Init) !void` are + both supported. Prefer `main(init: std.process.Init)`; `init.gpa` is a + general purpose allocator, `init.arena` an arena, `init.minimal.args` the + argv (`toSlice(allocator)`), `init.minimal.environ.block` the envp block. +* No libc is linked. Do not use `std.c.*`. Hostname lookups are therefore out + of scope: `--tcp` takes IP literals only. +* 9ns lives in the cloud9 repository as `cloud9/9ns/` and is built by + the root `build.zig` through the fragment `9ns/build.zig` (steps + `9ns`, `9ns-test`, `9ns-itest`, `9ns-adv`; toggle + `-D9ns`). cloud9 itself is imported as module `cloud9` + (`@import("cloud9")`). Read `../src/client.zig`, `Server.zig`, `wire.zig` + and `../docs/design.md`. Its core is allocation-free and caller-driven: you + push bytes in, take results out. The demo 9P server the tests mount is the + sibling program `../9proc` (`zig build 9proc`). +* Standalone module tests while other files are missing (from the cloud9 + root): + `zig test --dep cloud9 -Mroot=9ns/src/.zig -Mcloud9=src/root.zig`. +* Format everything with `zig fmt`. + +## Process model + +``` +9ns [options] -- PROGRAM [ARGS...] +``` + +1. Parent parses args, probes that `/dev/fuse` exists, connects to the 9P + server, negotiates `version` and `attach`es (fid 0 = root). Connection + failures are reported before anything is forked. +2. Parent forks with a `socketpair` status channel. **Child**: + 1. `unshare(CLONE_NEWUSER | CLONE_NEWNS)`. + 2. Writes `/proc/self/setgroups` = `deny`, `/proc/self/uid_map` = + `" 1"`, `/proc/self/gid_map` = `" 1"` (same ids + inside as outside; the child creating the namespace holds full + capabilities in it until exec). + 3. `mount(NULL, "/", NULL, MS_REC|MS_PRIVATE, NULL)` so nothing propagates. + 4. Ensures the mountpoint exists (see below). + 5. Opens `/dev/fuse` (`O_RDWR|O_CLOEXEC`). The kernel refuses to mount a + fuse descriptor opened from a different user namespace than the mount + ("wrong user namespace for fuse device"), so this must happen here, not + in the parent. + 6. `mount("9ns", mountpoint, "fuse", MS_NOSUID|MS_NODEV, + "fd=,rootmode=40000,user_id=,group_id=,max_read=")`. + 7. Sends the fuse fd to the parent over the status socket (`SCM_RIGHTS`). + 8. `statx` of the mountpoint: this forces one GETATTR, which the parent + serves. Without it the kernel keeps the root inode's initial uid 0 + (unmapped in the namespace) and every create in the root gets `EACCES`. + 9. Sets `NINE_MOUNT=` in the environment. + 10. `execve` of PROGRAM with PATH search (implemented by hand; no libc). + Exec failures are reported through the `CLOEXEC` status socket + (errno + message); the parent prints them after the serve loop ends. +3. **Parent** receives the fuse fd, then runs the FUSE loop (`bridge.serve`) + until either the child exits (SIGCHLD via self-pipe) or the FUSE fd reports + `ENODEV` (last process in the namespace gone, mount destroyed). It then + closes the fuse fd and exits with the child's status (`128+sig` if + signalled). The self-pipe is also watched by the 9P session while a reply + is outstanding (`Session.stop_fd` → `error.Stopped`), so a server that + never answers cannot keep 9ns alive after the child is gone; a 3 s + watchdog armed from the SIGCHLD handler is the last resort. +4. Signals in the parent: `SIGINT`/`SIGQUIT` ignored (the child owns the tty + and gets them itself); `SIGTERM`/`SIGHUP` forwarded to the child; + `SIGPIPE` ignored; `SIGCHLD` → self-pipe. + +The FUSE fd is shared with the child only until exec (CLOEXEC); the parent's +copy keeps the connection alive. + +### Mountpoint policy + +Default mountpoint: `/mnt/9p`. A relative `--mount` is resolved against cwd. + +* If the path is a directory: use it. +* Else try `mkdir`. If that fails with `EACCES`/`EPERM`/`EROFS` (the normal + case for `/mnt/9p` as a plain user), **shadow the parent directory**: + open an fd to the parent, mount a `tmpfs` over it, then recreate every + existing entry inside the tmpfs: directories → `mkdir` + bind mount from + `/proc/self/fd//`; symlinks → `readlinkat` + `symlink`; anything + else → empty regular file + bind mount. Then `mkdir` the target inside. + Refuse (with a clear message) if the parent has more than 4096 entries or + is `/`. This only affects the new namespace. +* Else fail with the errno and a hint to pass `--mount` an existing dir. + +## Module contracts + +### `src/fuse.zig` — kernel FUSE protocol (no policy) + +Extern structs mirroring `linux/fuse.h`, with `comptime` size asserts: +`InHeader` (40), `OutHeader` (16), `Attr` (88), `EntryOut` (128), +`AttrOut` (104), `GetattrIn` (16), `SetattrIn` (88), `OpenIn` (8), +`OpenOut` (16), `ReleaseIn` (24), `FlushIn` (24), `ReadIn` (40), +`WriteIn` (40), `WriteOut` (8), `CreateIn` (16), `MkdirIn` (8), +`RenameIn` (8), `Rename2In` (16), `ForgetIn` (8), `BatchForgetIn` (8), +`ForgetOne` (16), `FsyncIn` (16), `AccessIn` (8), `InterruptIn` (8), +`Kstatfs` (80), `StatfsOut` (80), `InitIn` (64), `InitOut` (64), +`Dirent` (24 header, name padded to 8), `LseekIn` (24). + +`pub const Opcode = enum(u32) { lookup = 1, forget = 2, getattr = 3, setattr = 4, +readlink = 5, symlink = 6, mknod = 8, mkdir = 9, unlink = 10, rmdir = 11, +rename = 12, link = 13, open = 14, read = 15, write = 16, statfs = 17, +release = 18, fsync = 20, setxattr = 21, getxattr = 22, listxattr = 23, +removexattr = 24, flush = 25, init = 26, opendir = 27, readdir = 28, +releasedir = 29, fsyncdir = 30, getlk = 31, setlk = 32, setlkw = 33, +access = 34, create = 35, interrupt = 36, bmap = 37, destroy = 38, +ioctl = 39, poll = 40, notify_reply = 41, batch_forget = 42, fallocate = 43, +readdirplus = 44, rename2 = 45, lseek = 46, copy_file_range = 47, +setupmapping = 48, removemapping = 49, syncfs = 50, tmpfile = 51, statx = 52, _ }` + +Constants: `kernel_version = 7`, `kernel_minor = 31` (what we answer; the +kernel adapts to the lower minor), `FOPEN_DIRECT_IO = 1`, `FOPEN_KEEP_CACHE = 2`, +`FOPEN_NONSEEKABLE = 4`, `FUSE_ASYNC_READ = 1`, `FUSE_MAX_PAGES = 1<<22`, +`FATTR_MODE=1, FATTR_UID=2, FATTR_GID=4, FATTR_SIZE=8, FATTR_ATIME=16, +FATTR_MTIME=32, FATTR_FH=64, FATTR_ATIME_NOW=128, FATTR_MTIME_NOW=256, +FATTR_LOCKOWNER=512, FATTR_CTIME=1024`. `root_id = 1`. + +I/O helpers (blocking fd, no allocation beyond the caller's buffer): + +```zig +pub const Request = struct { header: InHeader, body: []const u8 }; +/// One kernel request. Returns null on ENODEV (unmounted). Retries EINTR/EAGAIN/ENOENT. +pub fn readRequest(fd: i32, buf: []u8) !?Request; +/// Success reply: header + concatenated payload slices, single writev. +pub fn reply(fd: i32, unique: u64, payloads: []const []const u8) !void; +/// Error reply: negative errno. +pub fn replyError(fd: i32, unique: u64, err: std.os.linux.E) !void; +/// Append a fuse_dirent (8-byte padded) to `buf`; returns false if it doesn't fit. +pub fn addDirent(buf: []u8, used: *usize, ino: u64, off: u64, dtype: u32, name: []const u8) bool; +pub fn body(comptime T: type, req: Request) !*const T; // aligned copy-free view, checks size +pub fn nameAfter(comptime T: type, req: Request) ![]const u8; // NUL-terminated name after a struct +``` + +The request buffer must be ≥ `max_write + 4096`; 9ns uses 1 MiB + 4 KiB. +Requests with an unknown/unsupported opcode get `ENOSYS`. + +### `src/nine.zig` — synchronous 9P2000 session on a blocking fd + +Thin, synchronous RPC layer over `cloud9.Client` (which is push/take, +non-blocking-agnostic). One outstanding request at a time (the FUSE loop is +single-threaded). Fids are allocated from a free list. + +```zig +pub const Address = union(enum) { unix: []const u8, tcp: struct { host: []const u8, port: u16 }, fd: i32 }; +pub const Session = struct { + pub const Error = error{ Nine, Protocol, Io, Closed, TooLarge, OutOfMemory }; + /// After error.Nine, `ename` holds the server's Rerror text (copied, bounded). + ename: [256]u8, ename_len: usize, + msize: u32, + + pub fn connect(gpa: std.mem.Allocator, address: Address, msize: u32) !Session; // socket+connect, version + pub fn deinit(s: *Session) void; + pub fn attach(s: *Session, fid: u32, uname: []const u8, aname: []const u8) Error!cloud9.Qid; + pub fn allocFid(s: *Session) u32; + pub fn freeFid(s: *Session, fid: u32) void; + /// Generic RPC. Result slices borrow the input buffer until the next call. + pub fn rpc(s: *Session, req: cloud9.Client.Request) Error!cloud9.Client.Result; + // Conveniences (all built on rpc): + pub fn walk(s, fid: u32, newfid: u32, names: []const []const u8) Error!Walk; // Walk = { nwqid, wqid[16] }; partial walk → error.Nine with ename "not found"-ish + pub fn clone(s, fid: u32) Error!u32; // allocFid + walk with no names + pub fn open(s, fid: u32, mode: u8) Error!Open; // Open = { qid, iounit } + pub fn create(s, fid: u32, name: []const u8, perm: u32, mode: u8) Error!Open; + pub fn read(s, fid: u32, offset: u64, buf: []u8) Error!usize; // chunks by maxRead/iounit; stops at short read + pub fn write(s, fid: u32, offset: u64, data: []const u8) Error!usize; // chunks; stops at short write + pub fn stat(s, fid: u32) Error!cloud9.Stat; // strings borrow the input buffer + pub fn wstat(s, fid: u32, st: cloud9.Stat) Error!void; + pub fn clunk(s, fid: u32) Error!void; // frees the fid even on error + pub fn remove(s, fid: u32) Error!void; // frees the fid even on error + pub fn errno(s: *const Session) std.os.linux.E; // map ename → errno (see below) +}; +pub const dontcare = cloud9.Stat{ .type = 0xFFFF, .dev = 0xFFFF_FFFF, .qid = .{ .type = 0xFF, .version = 0xFFFF_FFFF, .path = 0xFFFF_FFFF_FFFF_FFFF }, .mode = 0xFFFF_FFFF, .atime = 0xFFFF_FFFF, .mtime = 0xFFFF_FFFF, .length = 0xFFFF_FFFF_FFFF_FFFF, .name = "", .uid = "", .gid = "", .muid = "" }; +``` + +`connect`: for `.unix` and `.tcp` create a blocking `SOCK_STREAM|SOCK_CLOEXEC` +socket and connect (`TCP_NODELAY` on TCP); for `.fd` adopt it. Buffers of +`msize` bytes for in/out are heap allocated. Then submit `.version`, drain +output to the socket, read until `take()` yields the version result. The +negotiated msize is `result.version.msize`; if the server answered +`"unknown"`, fail with `error.Protocol`. + +`rpc`: submit, write all of `client.output()` (calling `wrote`), then loop: +`take()`; if null, `read` from the fd into a temp buffer and `push` (push +returns how much fit; the frame is at most msize so it always fits after a +`take`). If the fd returns 0 → `error.Closed`. If the client dies → +`error.Protocol`. A `.fail` result copies the ename and returns `error.Nine`. + +Rerror text → errno mapping (case-insensitive substring, in this order): +`"not exist"`, `"not found"`, `"no such"` → `ENOENT`; `"exists"` → `EEXIST`; +`"not empty"` → `ENOTEMPTY`; `"not a dir"` → `ENOTDIR`; +`"is a dir"` → `EISDIR`; `"permission"`, `"denied"` → `EACCES`; +`"read-only"`, `"read only"`, `"readonly"` → `EROFS`; `"no space"` → `ENOSPC`; +`"not allowed"`, `"not permitted"`, `"cannot"` → `EPERM`; +`"fid"` → `EBADF`; `"bad offset"`, `"invalid"`, `"bad "` → `EINVAL`; +`"busy"`, `"in use"` → `EBUSY`; `"too long"` → `ENAMETOOLONG`; +`"not supported"`, `"unsupported"` → `ENOTSUP`; otherwise `EIO`. + +### `src/bridge.zig` — FUSE ↔ 9P translation + +```zig +pub const Options = struct { + uid: u32, gid: u32, // reported owner of every file + attr_timeout_ns: u64 = 1e9, // attr/entry cache validity (0 = none) + direct_io: bool = true, // FOPEN_DIRECT_IO on every regular file + debug: bool = false, // trace to stderr +}; +/// Runs until the FUSE fd reports ENODEV or `stop_fd` becomes readable. +pub fn serve(gpa: std.mem.Allocator, fuse_fd: i32, nine: *nine.Session, root_fid: u32, stop_fd: i32, opts: Options) !void; +``` + +State: + +* `inodes: AutoHashMap(u64 /*nodeid*/, Inode{ fid: u32, qid: Qid, nlookup: u64 })`. + Node 1 is the root (`root_fid`, never forgotten). +* `by_qid: AutoHashMap(u64 /*qid.path*/, u64 /*nodeid*/)` so that repeated + lookups of the same file map to the same inode (the old fid is clunked and + the fresh one kept). Dedupe only merges when the qid type (dir bit) also + matches, so a server reusing a path across a file and a directory cannot + poison an inode. `ino` in attrs is `qid.path` (root, or anything carrying + the root's path: 1). +* `handles: AutoHashMap(u64 /*fh*/, Handle{ fid: u32, dir: ?DirList })`. + `DirList` is the entire directory read at first `READDIR` offset 0: + `[]Entry{ name: []u8, ino: u64, dtype: u32 }` with synthetic `.` and `..` + first. `READDIR` offsets are indices into that list; a `READDIR` at offset 0 + re-reads the directory (rewinddir). + +Op mapping (9P2000 has no symlinks, links, xattrs, locks, mknod): + +| FUSE | 9P | +|---|---| +| INIT | reply `InitOut{ major=7, minor=31, max_readahead=in.max_readahead, flags = FUSE_ASYNC_READ \| FUSE_ATOMIC_O_TRUNC \| FUSE_AUTO_INVAL_DATA \| FUSE_BIG_WRITES (plus FUSE_MAX_PAGES with max_pages=256 if offered), max_background=16, congestion_threshold=12, max_write=1 MiB, time_gran=1 }`. Atomic O_TRUNC matters: without it the kernel truncates via a separate SETATTR(size=0) that synthetic control files reject; with it `O_TRUNC` becomes 9P `OTRUNC` inside the open | +| LOOKUP(parent,name) | `walk(parent.fid → newfid, [name])`; `stat(newfid)`; dedupe by qid; `EntryOut` | +| FORGET / BATCH_FORGET | `nlookup -= n`; at 0 `clunk` and drop (no reply) | +| GETATTR | `stat(inode.fid)` → `AttrOut` | +| SETATTR | `stat` then `wstat` with a *dontcare* Stat: `FATTR_SIZE`→length; `FATTR_MODE`→`(old.mode & ~0o777) \| (mode & 0o777)`; `FATTR_MTIME`→mtime (`FATTR_MTIME_NOW` → now); `FATTR_ATIME` ignored; `FATTR_UID/GID` → `EPERM` unless unchanged; then `stat` again for the reply | +| OPEN | `clone(inode.fid)` then `open(newfid, mode)`; mode from `O_ACCMODE` (`oread/owrite/ordwr`), `O_TRUNC` → `otrunc`; reply `OpenOut{ fh, open_flags = FOPEN_DIRECT_IO }`; on failure clunk | +| OPENDIR | same with `oread`; `fh` with `dir = null` | +| READ | `read(fh.fid, offset, buf[0..min(size, 1 MiB)])`; reply data | +| WRITE | `write(fh.fid, offset, data)`; `WriteOut{ size = n }` | +| READDIR | fill `Dirent`s from the `DirList` starting at `offset`, up to `size` bytes | +| RELEASE / RELEASEDIR | `clunk(fh.fid)`; free DirList | +| FLUSH / FSYNC / FSYNCDIR | ok (no-op) | +| CREATE(parent,name,flags,mode) | `clone(parent)`; `create(fid, name, mode & 0o777, openmode)` → this fid is the **open** file; then `walk(parent → fid2, [name])` + `stat(fid2)` for the inode; reply `EntryOut ++ OpenOut` | +| MKDIR | `clone(parent)`; `create(fid, name, DMDIR \| (mode & 0o777), oread)`; `clunk`; then lookup as above | +| UNLINK / RMDIR | `walk(parent → tmp, [name])`; `remove(tmp)` | +| RENAME / RENAME2 | if `newdir != parent` → `EXDEV`; else `walk(parent → tmp, [oldname])`, `wstat(tmp, dontcare with .name = newname)`, `clunk`. 9P rename never replaces, POSIX does: when the target exists (and `RENAME_NOREPLACE` is not set) a directory target is removed first; a file target is parked under a temporary name, the rename retried, and the parked file removed only after success (restored on failure) | +| STATFS | constant `Kstatfs{ bsize = 4096, namelen = 255, frsize = 4096 }` | +| ACCESS | `ENOSYS` (kernel stops asking; the server enforces permissions on open) | +| READLINK, SYMLINK, LINK, MKNOD, *XATTR, *LK, IOCTL, POLL, BMAP, FALLOCATE, LSEEK, COPY_FILE_RANGE, TMPFILE, STATX | `ENOSYS` | +| INTERRUPT | ignored (reply nothing) | +| DESTROY | return from `serve` | + +Attr mapping from `cloud9.Stat`: `mode = (S_IFDIR if DMDIR else S_IFREG) | +(st.mode & 0o777)`; `nlink = 1`; `size = length`; `blocks = (length+511)/512`; +`blksize = 4096`; `atime/mtime/ctime = st.atime/st.mtime/st.mtime`; +`uid/gid = opts.uid/gid`. `Dirent.type` = `DT_DIR` (4) / `DT_REG` (8). + +Errors: `nine.Session.Error.Nine` → `nine.errno()`; `Closed`/`Protocol`/`Io` +→ `EIO` and, since the session is dead, `serve` returns `error.Closed` after +replying so 9ns can report "9P server went away". + +With `direct_io` the kernel never trusts `length` for reads: synthetic files +that report length 0 (very common in 9P) still `cat` correctly, and reads run +until the server returns a short read. With `--no-direct-io` the bridge forces +`attr_timeout_ns = 0`, because a cached stale size truncates reads (observed +data loss on a 4 MiB copy otherwise). + +Hostile-server rules: directory listings are capped at 64 MiB (a server that +ignores read offsets otherwise loops forever); directory records with names +containing `/`, NUL, empty, `.`/`..` or longer than `FUSE_NAME_MAX` are dropped +rather than poisoning the whole READDIR reply; `length` near 2^64 is clamped +to `i64` max; the errno of a failing 9P call is latched before any cleanup +clunk overwrites the session's ename. + +### `src/ns.zig` — namespace and process plumbing + +```zig +pub const Spawn = struct { + argv: []const []const u8, // argv[0] is PATH-searched unless it contains '/' + envp: [*:null]const ?[*:0]const u8, // inherited environment + mountpoint: []const u8, // absolute + fuse_fd: i32, + uid: u32, gid: u32, + max_read: u32, +}; +pub const Child = struct { pid: i32 }; +/// fork; the child sets up the namespace, mounts, and execs. Returns once exec succeeded +/// (status pipe closed) or fails with the child's error (message on stderr). +pub fn spawn(gpa: std.mem.Allocator, s: Spawn) !Child; +pub fn ensureMountpoint(path: [:0]const u8) !void; // the shadowing logic, testable alone +pub fn resolveMountpoint(gpa, path: []const u8) ![:0]u8; // absolute, no trailing slash +pub fn findInPath(gpa, envp, name) ![:0]u8; +``` + +Also exports the signal plumbing used by `main.zig`: +`installSignals(child_pid_ptr: *i32) !i32` returning the SIGCHLD self-pipe +read end (used as `stop_fd` for `bridge.serve`), and +`waitChild(pid) !u8` → exit status (`128+sig` on signal death). + +### `src/main.zig` — CLI + +``` +Usage: 9ns [options] -- PROGRAM [ARGS...] +Transport (exactly one): + --unix PATH Unix stream socket + --tcp IP:PORT TCP (IPv4/IPv6 literal) + --fd N already-connected inherited descriptor + --spawn CMD run CMD (via /bin/sh -c) with a socketpair on its stdin/stdout +Options: + --mount PATH mountpoint inside the new namespace (default /mnt/9p) + --uname NAME 9P user name (default $USER, else "none") + --aname NAME 9P tree to attach (default "") + --msize BYTES maximum 9P message size to request (default 131072, max 16 MiB) + --cache SECONDS attr/entry cache validity, may be fractional (default 1) + --no-direct-io let the kernel cache file pages (trusts stat length) + --debug trace FUSE and 9P operations on stderr + --help, --version +PROGRAM defaults to $SHELL (else /bin/sh). The mountpoint is exported as $NINE_MOUNT. +``` + +Exit codes: child's status; 125 for 9ns's own failures (usage, connect, +mount); 126/127 as usual for exec failures. + +### `../9proc/demo/main.zig` — demo 9P2000 server (binary `9proc-demo`) + +The demo server is a separate program in this repository, built on the +9proc library; see `../9proc/docs/LIBRARY.md` for the library +contract (freestanding core, value renderers, Linux debug probe). The tree it serves keeps the paths the integration tests read +(`/build/*`, `/comptime/types//*`, `/comptime/decls`, `/runtime/fn/*`, +`/runtime/ctl`, `/runtime/{pid,ppid,uptime,argv,cwd,env,clients}`, +`/scratch/`) and adds `/vars`, `/threads`, `/addr`, `/mem`, `/hex`, +`/breakpoints` and `/panic`. + +## Integration test plan (`test/integration.sh`) + +Run by `zig build 9ns-itest`; args: path to `9ns`, path to `9proc-demo`. +Everything under a temp dir. Skips (exit 0 with a notice) when +`unshare -Urm true` fails or `/dev/fuse` is missing. + +1. 9proc-demo on a Unix socket; `9ns --unix … -- sh -c` scripts: + `cat /mnt/9p/build/zig_version` == `zig version`; `ls` listings; `stat` + sizes; `/runtime/fn/now` is numeric; `ctl` round trip; `/scratch`: create, + append (`>>`), overwrite, truncate, `mkdir -p a/b/c`, rename within dir, + `mv` across dirs fails with `EXDEV`-ish message, `rm`, `rmdir`, 1 MiB + random file round trip compared with `sha256sum`, `dd` with odd block + sizes, many small files, `find`, exit-status propagation (`exit 7` → 7), + `$NINE_MOUNT` set, nested `9ns` inside `9ns`. +2. `--spawn "<9proc-demo> --stdio"` variant. +3. `--tcp 127.0.0.1:` variant. +4. If `/usr/lib/plan9/bin/ramfs` exists: `NAMESPACE=$tmp ramfs -s ramfs` + creates `$tmp/ramfs`; run the scratch battery against it. +5. `--mount` with an existing dir, with a relative path, and the default + `/mnt/9p` (exercises parent shadowing; verify `/mnt`'s other entries are + still visible inside). +6. Kill tests: 9ns exits when the child exits; server death during use + yields `EIO`, not a hang. + +## Verification + +`zig build 9ns-test` (unit), `zig build 9ns-itest` (74 end-to-end +checks against 9proc-demo over unix/tcp/socketpair and against plan9port's +`ramfs`) and `zig build 9ns-adv` (adversarial suites: a scriptable +hostile 9P server with ~30 misbehaviour modes, FUSE semantics through the +bridge, process/namespace/signal edge cases with 51 checks, and stress). The +suites that attack the 9proc server itself (a hostile raw-9P client with +181 checks, the core, the Linux layer) moved with it to +`../9proc/test` (`zig build 9proc-adv`). All pass in Debug and +ReleaseSafe. + +## Out of scope for v1 (documented, not hidden) + +* One 9P request in flight at a time: a 9P read that blocks (event files) + stalls the whole mount until it returns (but not past the child's exit). +* No 9P2000.u/.L: no symlinks, ownership, or extended attributes. +* No PID namespace, no `/proc` remount. `--tcp` needs an IP literal. +* Cross-directory rename returns `EXDEV` (9P2000 cannot move files). diff --git a/9ns/src/bridge.zig b/9ns/src/bridge.zig new file mode 100644 index 0000000..3a072ec --- /dev/null +++ b/9ns/src/bridge.zig @@ -0,0 +1,971 @@ +//! FUSE ↔ 9P2000 translation: the request loop that turns kernel FUSE requests +//! into synchronous 9P calls on a `nine.Session` and sends the replies back. +//! +//! Everything here is single-threaded and one request at a time. State is three +//! tables: inodes (nodeid → fid/qid, deduplicated by qid.path), open handles +//! (fh → fid plus a cached directory listing), and the reverse qid map. +const std = @import("std"); +const cloud9 = @import("cloud9"); +const fuse = @import("fuse.zig"); +const nine = @import("nine.zig"); +const linux = std.os.linux; + +pub const Options = struct { + /// Reported owner of every file. + uid: u32, + gid: u32, + /// attr/entry cache validity (0 = none). + attr_timeout_ns: u64 = 1_000_000_000, + /// FOPEN_DIRECT_IO on every regular file. + direct_io: bool = true, + /// Trace every request, reply and 9P call to stderr. + debug: bool = false, +}; + +/// Largest single READ/WRITE payload we accept from the kernel. +pub const max_write: u32 = 1 << 20; +/// Upper bound on the raw bytes of one directory listing (about a million entries); +/// past it the listing fails with EIO instead of eating memory. +pub const max_dir_bytes: u64 = 64 << 20; +/// The kernel refuses dirents longer than this (FUSE_NAME_MAX) with EIO. +pub const max_name_len: usize = 1024; +/// Request buffer: `max_write` plus room for the header and the largest in-struct. +pub const request_buf_len: usize = max_write + 4096; + +const Inode = struct { + fid: u32, + qid: cloud9.Qid, + nlookup: u64, + /// nodeid of the directory this inode was looked up in (root: itself). Used for "..". + parent: u64, +}; + +pub const Entry = struct { name: []u8, ino: u64, dtype: u32 }; + +pub const DirList = struct { + entries: std.ArrayList(Entry) = .empty, + + pub fn deinit(d: *DirList, gpa: std.mem.Allocator) void { + for (d.entries.items) |e| gpa.free(e.name); + d.entries.deinit(gpa); + } +}; + +const Handle = struct { + fid: u32, + nodeid: u64, + dir: ?DirList, +}; + +/// Errors a request handler may surface. Policy failures are ordinary errors +/// that the dispatcher maps to an errno; `FuseIo` means the kernel side is broken. +const HandlerError = nine.Session.Error || error{ + BadRequest, + NoEntry, + BadHandle, + Exdev, + Perm, + NotSup, + /// A directory listing the server sent could not be parsed (EIO, not fatal). + BadDir, + FuseIo, +}; + +/// Runs until the FUSE fd reports ENODEV, a DESTROY arrives, or `stop_fd` +/// becomes readable (also while a 9P reply is outstanding). Returns +/// `error.Closed` if the 9P server went away. +pub fn serve(gpa: std.mem.Allocator, fuse_fd: i32, session: *nine.Session, root_fid: u32, stop_fd: i32, opts: Options) !void { + var effective = opts; + // With page caching on, a nonzero attr cache lets the kernel trust a stale + // (often zero) size and truncate reads: 9P sizes are authoritative and change + // under us. direct_io ignores the cached size, so the cache is safe only there. + if (!effective.direct_io) effective.attr_timeout_ns = 0; + var b: Bridge = .{ + .gpa = gpa, + .fuse_fd = fuse_fd, + .nine = session, + .opts = effective, + }; + defer b.deinit(); + + b.req_buf = try gpa.alignedAlloc(u8, .@"8", request_buf_len); + b.data_buf = try gpa.alloc(u8, max_write); + + // Abandon any pending 9P reply once the child is gone (stop_fd readable), + // including the initial root stat below: a silent server must not pin us. + session.stop_fd = stop_fd; + defer session.stop_fd = -1; + + // Node 1 is the root; its qid comes from a stat so lookups resolving back to + // it (e.g. via a walk) dedupe onto node 1. + var root_qid: cloud9.Qid = .{ .type = cloud9.qtdir, .version = 0, .path = 0 }; + if (b.stat(root_fid)) |st| { + root_qid = st.qid; + b.root_path = st.qid.path; + try b.by_qid.put(gpa, st.qid.path, fuse.root_id); + } else |e| switch (e) { + error.Nine => {}, + error.Stopped => return, + else => return error.Closed, + } + try b.inodes.put(gpa, fuse.root_id, .{ .fid = root_fid, .qid = root_qid, .nlookup = 1, .parent = fuse.root_id }); + + var pfds = [_]linux.pollfd{ + .{ .fd = fuse_fd, .events = linux.POLL.IN, .revents = 0 }, + .{ .fd = stop_fd, .events = linux.POLL.IN, .revents = 0 }, + }; + while (true) { + pfds[0].revents = 0; + pfds[1].revents = 0; + const rc = linux.poll(&pfds, pfds.len, -1); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .INTR, .AGAIN => continue, + else => return error.Io, + } + if (pfds[1].revents != 0) { + b.trace("stop_fd readable; leaving serve loop", .{}); + return; + } + if (pfds[0].revents == 0) continue; + const req = (fuse.readRequest(fuse_fd, b.req_buf) catch |e| switch (e) { + error.Protocol => return error.FuseProtocol, + else => return error.FuseIo, + }) orelse { + b.trace("fuse fd reports ENODEV; unmounted", .{}); + return; + }; + if (!try b.dispatch(req)) return; + } +} + +const Bridge = struct { + gpa: std.mem.Allocator, + fuse_fd: i32, + nine: *nine.Session, + opts: Options, + req_buf: []align(8) u8 = &.{}, + data_buf: []u8 = &.{}, + inodes: std.AutoHashMapUnmanaged(u64, Inode) = .empty, + by_qid: std.AutoHashMapUnmanaged(u64, u64) = .empty, + handles: std.AutoHashMapUnmanaged(u64, Handle) = .empty, + next_node: u64 = 2, + next_fh: u64 = 1, + /// qid.path of the root, reported as ino 1 wherever it shows up. + root_path: u64 = 0, + /// errno of the most recent Rerror that a handler did not swallow. Kept here + /// because `Session.rpc` clears its ename on every call, and error paths + /// clunk (an rpc) before the dispatcher maps the failure to an errno. + last_err: linux.E = .IO, + + fn deinit(b: *Bridge) void { + var it = b.handles.valueIterator(); + while (it.next()) |h| if (h.dir) |*d| d.deinit(b.gpa); + b.handles.deinit(b.gpa); + b.inodes.deinit(b.gpa); + b.by_qid.deinit(b.gpa); + if (b.req_buf.len != 0) b.gpa.free(b.req_buf); + if (b.data_buf.len != 0) b.gpa.free(b.data_buf); + } + + fn trace(b: *const Bridge, comptime fmt: []const u8, args: anytype) void { + if (b.opts.debug) std.debug.print("9ns: " ++ fmt ++ "\n", args); + } + + // -- dispatch -------------------------------------------------------------------- + + /// Handles one request. Returns false when the loop should stop (DESTROY). + /// Fatal errors (dead 9P session, broken FUSE fd) propagate. + fn dispatch(b: *Bridge, req: fuse.Request) !bool { + const h = req.header; + const op = h.op(); + b.trace("<- {s} unique={d} nodeid={d} len={d} (fids={d} inodes={d} handles={d})", .{ opName(op), h.unique, h.nodeid, h.len, b.nine.fidsInUse(), b.inodes.count(), b.handles.count() }); + const wants_reply = switch (op) { + .forget, .batch_forget, .interrupt => false, + else => true, + }; + if (op == .destroy) { + b.reply(h.unique, &.{}) catch {}; + return false; + } + b.handle(req) catch |e| { + const code: linux.E = switch (e) { + error.Nine => b.last_err, + error.BadRequest => .INVAL, + error.NoEntry => .NOENT, + error.BadHandle => .BADF, + error.Exdev => .XDEV, + error.Perm => .PERM, + error.NotSup => .NOSYS, + error.OutOfMemory => .NOMEM, + error.TooLarge => .NAMETOOLONG, + error.BadDir => .IO, + error.Closed, error.Protocol, error.Io, error.Stopped => .IO, + error.FuseIo => return error.FuseIo, + }; + if (wants_reply) try b.replyError(h.unique, code); + switch (e) { + error.Closed, error.Protocol, error.Io => return error.Closed, + error.Stopped => return false, // the child is gone; the mount is being torn down + else => {}, + } + }; + return true; + } + + fn handle(b: *Bridge, req: fuse.Request) HandlerError!void { + const u = req.header.unique; + switch (req.header.op()) { + .init => { + const in = try body(fuse.InitIn, req); + const out = fuse.initReply(in, max_write); + try b.reply(u, &.{std.mem.asBytes(&out)}); + }, + .lookup => { + const name = try nameAfter(void, req); + const entry = try b.lookupEntry(req.header.nodeid, name); + try b.reply(u, &.{std.mem.asBytes(&entry)}); + }, + .forget => { + const in = try body(fuse.ForgetIn, req); + try b.forget(req.header.nodeid, in.nlookup); + }, + .batch_forget => { + const in = try body(fuse.BatchForgetIn, req); + const rest = req.body[@sizeOf(fuse.BatchForgetIn)..]; + const count: usize = in.count; + if (rest.len < count * @sizeOf(fuse.ForgetOne)) return error.BadRequest; + for (0..count) |i| { + const one = std.mem.bytesToValue(fuse.ForgetOne, rest[i * @sizeOf(fuse.ForgetOne) ..][0..@sizeOf(fuse.ForgetOne)]); + try b.forget(one.nodeid, one.nlookup); + } + }, + .getattr => { + const ino = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; + const st = try b.stat(ino.fid); + const out = b.attrOut(st, b.inoOf(req.header.nodeid, ino.qid)); + try b.reply(u, &.{std.mem.asBytes(&out)}); + }, + .setattr => try b.setattr(req), + .open => try b.openFile(req, false), + .opendir => try b.openFile(req, true), + .read => { + const in = try body(fuse.ReadIn, req); + const h = b.handles.get(in.fh) orelse return error.BadHandle; + const want: usize = @min(in.size, max_write); + const n = try b.read(h.fid, in.offset, b.data_buf[0..want]); + try b.reply(u, &.{b.data_buf[0..n]}); + }, + .write => { + const in = try body(fuse.WriteIn, req); + const h = b.handles.get(in.fh) orelse return error.BadHandle; + const rest = req.body[@sizeOf(fuse.WriteIn)..]; + if (rest.len < in.size) return error.BadRequest; + const n = try b.write(h.fid, in.offset, rest[0..in.size]); + const out = fuse.WriteOut{ .size = @intCast(n) }; + try b.reply(u, &.{std.mem.asBytes(&out)}); + }, + .readdir => try b.readdir(req), + .release, .releasedir => { + const in = try body(fuse.ReleaseIn, req); + const kv = b.handles.fetchRemove(in.fh) orelse return error.BadHandle; + var h = kv.value; + if (h.dir) |*d| d.deinit(b.gpa); + try b.clunk(h.fid); + try b.reply(u, &.{}); + }, + .flush, .fsync, .fsyncdir => try b.reply(u, &.{}), + .create => try b.create(req), + .mkdir => { + const in = try body(fuse.MkdirIn, req); + const name = try nameAfter(fuse.MkdirIn, req); + const parent = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; + const fid = try b.clone(parent.fid); + _ = b.create9(fid, name, cloud9.dmdir | (in.mode & 0o777), cloud9.oread) catch |e| { + b.clunkQuiet(fid); + return e; + }; + try b.clunk(fid); + const entry = try b.lookupEntry(req.header.nodeid, name); + try b.reply(u, &.{std.mem.asBytes(&entry)}); + }, + .unlink, .rmdir => { + const name = try nameAfter(void, req); + const parent = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; + const tmp = try b.walkName(parent.fid, name); + try b.remove(tmp); + try b.reply(u, &.{}); + }, + .rename => { + const in = try body(fuse.RenameIn, req); + const old = try nameAfter(fuse.RenameIn, req); + const new = try secondName(req, old, @sizeOf(fuse.RenameIn)); + try b.rename(req.header.nodeid, in.newdir, old, new, 0); + try b.reply(u, &.{}); + }, + .rename2 => { + const in = try body(fuse.Rename2In, req); + const old = try nameAfter(fuse.Rename2In, req); + const new = try secondName(req, old, @sizeOf(fuse.Rename2In)); + try b.rename(req.header.nodeid, in.newdir, old, new, in.flags); + try b.reply(u, &.{}); + }, + .statfs => { + const out = fuse.StatfsOut{ .st = .{ .bsize = 4096, .namelen = 255, .frsize = 4096 } }; + try b.reply(u, &.{std.mem.asBytes(&out)}); + }, + .interrupt => {}, + .destroy => unreachable, // handled in dispatch + .access => return error.NotSup, + else => return error.NotSup, + } + } + + // -- handlers ---------------------------------------------------------------------- + + /// walk(parent → new fid, [name]) + stat, deduplicated by qid.path. Bumps nlookup. + fn lookupEntry(b: *Bridge, parent_id: u64, name: []const u8) HandlerError!fuse.EntryOut { + const parent = b.inodes.get(parent_id) orelse return error.NoEntry; + const newfid = try b.walkName(parent.fid, name); + const st = b.stat(newfid) catch |e| { + b.clunkQuiet(newfid); + return e; + }; + const qid = st.qid; + var nodeid: u64 = undefined; + if (b.by_qid.get(qid.path)) |existing| { + // A directory and a file sharing a qid.path (a server bug) must not + // share a node: the kernel would mark the inode bad, and for the + // root that is fatal for the whole mount. + const merge = if (b.inodes.getPtr(existing)) |ino| (ino.qid.type & cloud9.qtdir) == (qid.type & cloud9.qtdir) else false; + if (merge) { + const ino = b.inodes.getPtr(existing).?; + ino.nlookup += 1; + ino.qid = qid; + nodeid = existing; + if (existing == fuse.root_id) { + b.clunkQuiet(newfid); + } else { + // Keep the fresh fid (it is bound to the current file at this + // name) and retire the older one. + const stale = ino.fid; + ino.fid = newfid; + b.clunkQuiet(stale); + } + } else { + // Stale reverse entry, or a type clash: bind a fresh node to it. + nodeid = try b.newInode(newfid, qid, parent_id); + } + } else { + nodeid = try b.newInode(newfid, qid, parent_id); + } + var out = fuse.EntryOut{ + .nodeid = nodeid, + .generation = 0, + .attr = b.attrFrom(st, b.inoOf(nodeid, qid)), + }; + out.entry_valid = b.opts.attr_timeout_ns / 1_000_000_000; + out.entry_valid_nsec = @intCast(b.opts.attr_timeout_ns % 1_000_000_000); + out.attr_valid = out.entry_valid; + out.attr_valid_nsec = out.entry_valid_nsec; + return out; + } + + fn newInode(b: *Bridge, fid: u32, qid: cloud9.Qid, parent: u64) HandlerError!u64 { + const nodeid = b.next_node; + b.inodes.put(b.gpa, nodeid, .{ .fid = fid, .qid = qid, .nlookup = 1, .parent = parent }) catch |e| { + b.clunkQuiet(fid); + return e; + }; + b.by_qid.put(b.gpa, qid.path, nodeid) catch |e| { + _ = b.inodes.remove(nodeid); + b.clunkQuiet(fid); + return e; + }; + b.next_node += 1; + return nodeid; + } + + fn forget(b: *Bridge, nodeid: u64, n: u64) HandlerError!void { + if (nodeid == fuse.root_id) return; + const ino = b.inodes.getPtr(nodeid) orelse return; + if (ino.nlookup > n) { + ino.nlookup -= n; + return; + } + const fid = ino.fid; + const path = ino.qid.path; + _ = b.inodes.remove(nodeid); + if (b.by_qid.get(path)) |mapped| { + if (mapped == nodeid) _ = b.by_qid.remove(path); + } + b.clunk(fid) catch |e| switch (e) { + error.Nine => {}, + else => return e, + }; + } + + fn setattr(b: *Bridge, req: fuse.Request) HandlerError!void { + const in = try body(fuse.SetattrIn, req); + const ino = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; + const old = try b.stat(ino.fid); + const old_mode = old.mode; + + var st = nine.dontcare; + var changed = false; + if (in.valid & fuse.FATTR_UID != 0 and in.uid != b.opts.uid) return error.Perm; + if (in.valid & fuse.FATTR_GID != 0 and in.gid != b.opts.gid) return error.Perm; + if (in.valid & fuse.FATTR_SIZE != 0) { + st.length = in.size; + changed = true; + } + if (in.valid & fuse.FATTR_MODE != 0) { + st.mode = (old_mode & ~@as(u32, 0o777)) | (in.mode & 0o777); + changed = true; + } + if (in.valid & fuse.FATTR_MTIME_NOW != 0) { + st.mtime = nowSeconds(); + changed = true; + } else if (in.valid & fuse.FATTR_MTIME != 0) { + st.mtime = @truncate(in.mtime); + changed = true; + } + if (changed) try b.wstat(ino.fid, st); + const fresh = try b.stat(ino.fid); + const out = b.attrOut(fresh, b.inoOf(req.header.nodeid, ino.qid)); + try b.reply(req.header.unique, &.{std.mem.asBytes(&out)}); + } + + fn openFile(b: *Bridge, req: fuse.Request, is_dir: bool) HandlerError!void { + const in = try body(fuse.OpenIn, req); + const ino = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; + const mode: u8 = if (is_dir) cloud9.oread else openMode(in.flags); + const fid = try b.clone(ino.fid); + _ = b.open9(fid, mode) catch |e| { + b.clunkQuiet(fid); + return e; + }; + const fh = try b.newHandle(fid, req.header.nodeid); + const out = fuse.OpenOut{ + .fh = fh, + .open_flags = if (!is_dir and b.opts.direct_io) fuse.FOPEN_DIRECT_IO else 0, + }; + try b.reply(req.header.unique, &.{std.mem.asBytes(&out)}); + } + + fn newHandle(b: *Bridge, fid: u32, nodeid: u64) HandlerError!u64 { + const fh = b.next_fh; + b.handles.put(b.gpa, fh, .{ .fid = fid, .nodeid = nodeid, .dir = null }) catch |e| { + b.clunkQuiet(fid); + return e; + }; + b.next_fh += 1; + return fh; + } + + fn create(b: *Bridge, req: fuse.Request) HandlerError!void { + const in = try body(fuse.CreateIn, req); + const name = try nameAfter(fuse.CreateIn, req); + const parent = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; + // The created fid becomes the open file. + const fid = try b.clone(parent.fid); + _ = b.create9(fid, name, in.mode & 0o777, openMode(in.flags)) catch |e| { + b.clunkQuiet(fid); + return e; + }; + const entry = b.lookupEntry(req.header.nodeid, name) catch |e| { + b.clunkQuiet(fid); + return e; + }; + const fh = try b.newHandle(fid, entry.nodeid); + const oo = fuse.OpenOut{ + .fh = fh, + .open_flags = if (b.opts.direct_io) fuse.FOPEN_DIRECT_IO else 0, + }; + try b.reply(req.header.unique, &.{ std.mem.asBytes(&entry), std.mem.asBytes(&oo) }); + } + + fn rename(b: *Bridge, parent_id: u64, newdir: u64, old: []const u8, new: []const u8, flags: u32) HandlerError!void { + if (newdir != parent_id) return error.Exdev; + const rf: linux.RENAME = @bitCast(flags); + if (rf.EXCHANGE or rf.WHITEOUT) return error.BadRequest; + const parent = b.inodes.get(parent_id) orelse return error.NoEntry; + const tmp = try b.walkName(parent.fid, old); + defer b.clunkQuiet(tmp); + var st = nine.dontcare; + st.name = new; + b.wstat(tmp, st) catch |e| { + // 9P2000 rename never replaces an existing name; POSIX rename does. + if (e != error.Nine or rf.NOREPLACE or b.nine.errno() != .EXIST) return e; + try b.renameOver(parent.fid, tmp, new); + }; + } + + /// Replace `new` with the file behind `src`. An (empty) directory target is + /// removed first: it holds no data and the VFS already ruled out mismatched + /// types. A file target is parked under a temporary name so that a failing + /// second rename can put it back instead of having destroyed it. + fn renameOver(b: *Bridge, parent_fid: u32, src: u32, new: []const u8) HandlerError!void { + const victim = try b.walkName(parent_fid, new); + const vst = b.stat(victim) catch |e| { + b.clunkQuiet(victim); + return e; + }; + var st = nine.dontcare; + st.name = new; + if (vst.mode & cloud9.dmdir != 0) { + b.trace(" rename target is a directory; removing it and retrying", .{}); + try b.remove(victim); + return b.wstat(src, st); + } + var park_buf: [48]u8 = undefined; + const park = std.fmt.bufPrint(&park_buf, ".9ns-rename-{x}", .{randomU64()}) catch unreachable; + b.trace(" rename target exists; parking it as {s} and retrying", .{park}); + var pst = nine.dontcare; + pst.name = park; + b.wstat(victim, pst) catch |e| { + b.clunkQuiet(victim); + return e; + }; + b.wstat(src, st) catch |e| { + b.trace(" rename still failed; restoring the target", .{}); + const saved = b.last_err; + b.wstat(victim, st) catch {}; + b.last_err = saved; + b.clunkQuiet(victim); + return e; + }; + b.remove(victim) catch b.trace(" could not remove the parked target {s}", .{park}); + } + + fn readdir(b: *Bridge, req: fuse.Request) HandlerError!void { + const in = try body(fuse.ReadIn, req); + const h = b.handles.getPtr(in.fh) orelse return error.BadHandle; + if (in.offset == 0 or h.dir == null) { + if (h.dir) |*d| d.deinit(b.gpa); + h.dir = null; + h.dir = try b.loadDir(h.fid, h.nodeid); + } + const dir = &h.dir.?; + const size: usize = @min(in.size, max_write); + const used = packDirents(dir.entries.items, in.offset, b.data_buf[0..size]); + try b.reply(req.header.unique, &.{b.data_buf[0..used]}); + } + + /// Reads the whole directory and builds its listing, "." and ".." first. + fn loadDir(b: *Bridge, fid: u32, nodeid: u64) HandlerError!DirList { + var list: DirList = .{}; + errdefer list.deinit(b.gpa); + const self_ino = b.inoOfNode(nodeid); + const parent_ino = if (b.inodes.get(nodeid)) |ino| b.inoOfNode(ino.parent) else self_ino; + try list.entries.append(b.gpa, .{ .name = try b.gpa.dupe(u8, "."), .ino = self_ino, .dtype = fuse.DT_DIR }); + try list.entries.append(b.gpa, .{ .name = try b.gpa.dupe(u8, ".."), .ino = parent_ino, .dtype = fuse.DT_DIR }); + + var offset: u64 = 0; + while (true) { + // A server that ignores the offset would otherwise feed us forever. + if (offset >= max_dir_bytes) return error.BadDir; + const n = try b.read(fid, offset, b.data_buf); + if (n == 0) break; + try parseDirRecords(b.gpa, b.data_buf[0..n], &list); + offset += n; + } + // Entries carrying the root's own qid.path get the root's ino (1), as GETATTR would report it. + for (list.entries.items[2..]) |*e| if (e.ino == b.root_path) { + e.ino = fuse.root_id; + }; + return list; + } + + // -- 9P wrappers (tracing) ----------------------------------------------------------- + + fn stat(b: *Bridge, fid: u32) nine.Session.Error!cloud9.Stat { + const st = b.nine.stat(fid) catch |e| return b.nineErr("stat", fid, e); + b.trace(" 9p stat fid={d} -> name={s} mode={o} len={d} qid={x}", .{ fid, st.name, st.mode, st.length, st.qid.path }); + return st; + } + + fn walkName(b: *Bridge, fid: u32, name: []const u8) nine.Session.Error!u32 { + const newfid = b.nine.allocFid(); + _ = b.nine.walk(fid, newfid, &.{name}) catch |e| { + b.nine.freeFid(newfid); + return b.nineErr("walk", fid, e); + }; + b.trace(" 9p walk fid={d} newfid={d} name={s} -> ok", .{ fid, newfid, name }); + return newfid; + } + + fn clone(b: *Bridge, fid: u32) nine.Session.Error!u32 { + const newfid = b.nine.clone(fid) catch |e| return b.nineErr("clone", fid, e); + b.trace(" 9p walk fid={d} newfid={d} (clone) -> ok", .{ fid, newfid }); + return newfid; + } + + fn open9(b: *Bridge, fid: u32, mode: u8) nine.Session.Error!nine.Session.Open { + const o = b.nine.open(fid, mode) catch |e| return b.nineErr("open", fid, e); + b.trace(" 9p open fid={d} mode={d} -> iounit={d}", .{ fid, mode, o.iounit }); + return o; + } + + fn create9(b: *Bridge, fid: u32, name: []const u8, perm: u32, mode: u8) nine.Session.Error!nine.Session.Open { + const o = b.nine.create(fid, name, perm, mode) catch |e| return b.nineErr("create", fid, e); + b.trace(" 9p create fid={d} name={s} perm={o} mode={d} -> iounit={d}", .{ fid, name, perm, mode, o.iounit }); + return o; + } + + fn read(b: *Bridge, fid: u32, offset: u64, buf: []u8) nine.Session.Error!usize { + const n = b.nine.read(fid, offset, buf) catch |e| return b.nineErr("read", fid, e); + b.trace(" 9p read fid={d} offset={d} count={d} -> {d}", .{ fid, offset, buf.len, n }); + return n; + } + + fn write(b: *Bridge, fid: u32, offset: u64, data: []const u8) nine.Session.Error!usize { + const n = b.nine.write(fid, offset, data) catch |e| return b.nineErr("write", fid, e); + b.trace(" 9p write fid={d} offset={d} count={d} -> {d}", .{ fid, offset, data.len, n }); + return n; + } + + fn wstat(b: *Bridge, fid: u32, st: cloud9.Stat) nine.Session.Error!void { + b.nine.wstat(fid, st) catch |e| return b.nineErr("wstat", fid, e); + b.trace(" 9p wstat fid={d} name={s} mode={x} len={x} mtime={x} -> ok", .{ fid, st.name, st.mode, st.length, st.mtime }); + } + + fn clunk(b: *Bridge, fid: u32) nine.Session.Error!void { + b.nine.clunk(fid) catch |e| return b.nineErr("clunk", fid, e); + b.trace(" 9p clunk fid={d} -> ok", .{fid}); + } + + /// Best-effort clunk during error unwinding; a dead session surfaces on the + /// next call. Does not disturb the errno of the failure being unwound. + fn clunkQuiet(b: *Bridge, fid: u32) void { + const saved = b.last_err; + defer b.last_err = saved; + b.clunk(fid) catch {}; + } + + fn remove(b: *Bridge, fid: u32) nine.Session.Error!void { + b.nine.remove(fid) catch |e| return b.nineErr("remove", fid, e); + b.trace(" 9p remove fid={d} -> ok", .{fid}); + } + + fn nineErr(b: *Bridge, what: []const u8, fid: u32, e: nine.Session.Error) nine.Session.Error { + if (e == error.Nine) { + b.last_err = b.nine.errno(); + b.trace(" 9p {s} fid={d} -> Rerror \"{s}\" ({s})", .{ what, fid, b.nine.ename[0..b.nine.ename_len], @tagName(b.nine.errno()) }); + } else { + b.trace(" 9p {s} fid={d} -> {s}", .{ what, fid, @errorName(e) }); + } + return e; + } + + // -- FUSE wrappers (tracing) -------------------------------------------------------- + + fn reply(b: *Bridge, unique: u64, payloads: []const []const u8) error{FuseIo}!void { + var total: usize = 0; + for (payloads) |p| total += p.len; + b.trace("-> unique={d} ok ({d} bytes)", .{ unique, total }); + fuse.reply(b.fuse_fd, unique, payloads) catch return error.FuseIo; + } + + fn replyError(b: *Bridge, unique: u64, code: linux.E) error{FuseIo}!void { + b.trace("-> unique={d} error E{s}", .{ unique, @tagName(code) }); + fuse.replyError(b.fuse_fd, unique, code) catch return error.FuseIo; + } + + // -- attrs --------------------------------------------------------------------------- + + fn inoOf(b: *const Bridge, nodeid: u64, qid: cloud9.Qid) u64 { + return if (nodeid == fuse.root_id or qid.path == b.root_path) fuse.root_id else qid.path; + } + + fn inoOfNode(b: *const Bridge, nodeid: u64) u64 { + if (nodeid == fuse.root_id) return fuse.root_id; + const ino = b.inodes.get(nodeid) orelse return nodeid; + return b.inoOf(nodeid, ino.qid); + } + + fn attrFrom(b: *const Bridge, st: cloud9.Stat, ino: u64) fuse.Attr { + return attrFromStat(st, ino, b.opts.uid, b.opts.gid); + } + + fn attrOut(b: *const Bridge, st: cloud9.Stat, ino: u64) fuse.AttrOut { + return .{ + .attr_valid = b.opts.attr_timeout_ns / 1_000_000_000, + .attr_valid_nsec = @intCast(b.opts.attr_timeout_ns % 1_000_000_000), + .attr = b.attrFrom(st, ino), + }; + } +}; + +// -- pure helpers (unit-tested) ------------------------------------------------------------ + +/// Attr from a 9P Stat: DMDIR → S_IFDIR else S_IFREG, low 9 permission bits kept. +pub fn attrFromStat(st: cloud9.Stat, ino: u64, uid: u32, gid: u32) fuse.Attr { + const ftype: u32 = if (st.mode & cloud9.dmdir != 0) fuse.S_IFDIR else fuse.S_IFREG; + return .{ + .ino = ino, + // The kernel marks an inode bad when size > LLONG_MAX; clamp hostile lengths. + .size = @min(st.length, std.math.maxInt(i64)), + // Saturating: a hostile length of 2^64-1 must not overflow. + .blocks = st.length / 512 + @intFromBool(st.length % 512 != 0), + .atime = st.atime, + .mtime = st.mtime, + .ctime = st.mtime, + .mode = ftype | (st.mode & 0o777), + .nlink = 1, + .uid = uid, + .gid = gid, + .blksize = 4096, + }; +} + +/// Kernel open(2) flags → 9P open mode. O_APPEND has no 9P equivalent and is ignored. +pub fn openMode(flags: u32) u8 { + const o: linux.O = @bitCast(flags); + var mode: u8 = switch (o.ACCMODE) { + .RDONLY => cloud9.oread, + .WRONLY => cloud9.owrite, + .RDWR => cloud9.ordwr, + }; + if (o.TRUNC) mode |= cloud9.otrunc; + return mode; +} + +/// Parses consecutive 9P directory records (2-byte size + Stat) and appends entries. +pub fn parseDirRecords(gpa: std.mem.Allocator, bytes: []const u8, list: *DirList) error{ OutOfMemory, BadDir }!void { + var pos: usize = 0; + while (pos < bytes.len) { + if (bytes.len - pos < 2) return error.BadDir; + const size: usize = std.mem.readInt(u16, bytes[pos..][0..2], .little); + if (bytes.len - pos < 2 + size) return error.BadDir; + const st = cloud9.Stat.decode(bytes[pos..][0 .. 2 + size]) catch return error.BadDir; + pos += 2 + size; + // The kernel rejects a whole READDIR reply (EIO) over one bad name, and + // "." and ".." are synthesised by loadDir: drop such records instead. + if (!validDirentName(st.name)) continue; + const name = try gpa.dupe(u8, st.name); + errdefer gpa.free(name); + try list.entries.append(gpa, .{ + .name = name, + .ino = st.qid.path, + .dtype = if (st.mode & cloud9.dmdir != 0) fuse.DT_DIR else fuse.DT_REG, + }); + } +} + +/// A name the kernel will accept in a dirent and that does not duplicate the synthetic "." / "..". +pub fn validDirentName(name: []const u8) bool { + if (name.len == 0 or name.len > max_name_len) return false; + if (std.mem.indexOfAny(u8, name, "/\x00") != null) return false; + if (std.mem.eql(u8, name, ".") or std.mem.eql(u8, name, "..")) return false; + return true; +} + +/// Packs dirents from `entries[offset..]` into `buf`; each record's `off` is its index + 1. +/// Returns the number of bytes used. +pub fn packDirents(entries: []const Entry, offset: u64, buf: []u8) usize { + var used: usize = 0; + var i: usize = @intCast(@min(offset, entries.len)); + while (i < entries.len) : (i += 1) { + const e = entries[i]; + if (!fuse.addDirent(buf, &used, e.ino, @as(u64, i) + 1, e.dtype, e.name)) break; + } + return used; +} + +fn randomU64() u64 { + var bytes: [8]u8 = undefined; + if (linux.errno(linux.getrandom(&bytes, bytes.len, 0)) == .SUCCESS) return std.mem.readInt(u64, &bytes, .little); + var ts: linux.timespec = undefined; + _ = linux.clock_gettime(.MONOTONIC, &ts); + return @as(u64, @bitCast(ts.nsec)) ^ (@as(u64, @bitCast(ts.sec)) << 32); +} + +fn nowSeconds() u32 { + var ts: linux.timespec = undefined; + if (linux.errno(linux.clock_gettime(.REALTIME, &ts)) != .SUCCESS) return 0; + return @intCast(@as(u64, @intCast(ts.sec)) & 0xFFFF_FFFF); +} + +fn opName(op: fuse.Opcode) []const u8 { + return switch (op) { + _ => "unknown", + else => @tagName(op), + }; +} + +// Thin adapters so fuse.zig's parse errors become HandlerError.BadRequest. +fn body(comptime T: type, req: fuse.Request) error{BadRequest}!*const T { + return fuse.body(T, req) catch error.BadRequest; +} + +fn nameAfter(comptime T: type, req: fuse.Request) error{BadRequest}![]const u8 { + return fuse.nameAfter(T, req) catch error.BadRequest; +} + +fn secondName(req: fuse.Request, first: []const u8, offset: usize) error{BadRequest}![]const u8 { + return fuse.secondName(req, first, offset) catch error.BadRequest; +} + +// -- tests ------------------------------------------------------------------------------ + +const testing = std.testing; + +test { + // Force semantic analysis of `serve` and the whole dispatch path, which no + // unit test can exercise without a FUSE mount. + testing.refAllDecls(@This()); +} + +fn testStat(name: []const u8, mode: u32, length: u64, path: u64) cloud9.Stat { + return .{ + .type = 0, + .dev = 0, + .qid = .{ .type = if (mode & cloud9.dmdir != 0) cloud9.qtdir else 0, .version = 0, .path = path }, + .mode = mode, + .atime = 100, + .mtime = 200, + .length = length, + .name = name, + .uid = "u", + .gid = "g", + .muid = "u", + }; +} + +test "attr mapping: DMDIR → S_IFDIR|perm, length → size/blocks" { + const d = attrFromStat(testStat("d", cloud9.dmdir | 0o755, 0, 9), 9, 1000, 1001); + try testing.expectEqual(fuse.S_IFDIR | 0o755, d.mode); + try testing.expectEqual(@as(u64, 9), d.ino); + try testing.expectEqual(@as(u64, 0), d.size); + try testing.expectEqual(@as(u64, 0), d.blocks); + try testing.expectEqual(@as(u32, 1000), d.uid); + try testing.expectEqual(@as(u32, 1001), d.gid); + try testing.expectEqual(@as(u32, 1), d.nlink); + + const f = attrFromStat(testStat("f", 0o640 | cloud9.dmappend, 1025, 4), 4, 0, 0); + try testing.expectEqual(fuse.S_IFREG | 0o640, f.mode); // dmappend bit not leaked + try testing.expectEqual(@as(u64, 1025), f.size); + try testing.expectEqual(@as(u64, 3), f.blocks); + try testing.expectEqual(@as(u32, 4096), f.blksize); + try testing.expectEqual(@as(u64, 100), f.atime); + try testing.expectEqual(@as(u64, 200), f.mtime); + try testing.expectEqual(@as(u64, 200), f.ctime); + + try testing.expectEqual(@as(u64, 1), attrFromStat(testStat("f", 0o600, 512, 4), 4, 0, 0).blocks); + try testing.expectEqual(@as(u64, 2), attrFromStat(testStat("f", 0o600, 513, 4), 4, 0, 0).blocks); +} + +test "open flag → 9P mode mapping" { + const rdonly: u32 = @bitCast(linux.O{ .ACCMODE = .RDONLY }); + const wronly: u32 = @bitCast(linux.O{ .ACCMODE = .WRONLY }); + const rdwr: u32 = @bitCast(linux.O{ .ACCMODE = .RDWR }); + const trunc: u32 = @bitCast(linux.O{ .TRUNC = true }); + const append: u32 = @bitCast(linux.O{ .APPEND = true }); + const creat: u32 = @bitCast(linux.O{ .CREAT = true }); + try testing.expectEqual(cloud9.oread, openMode(rdonly)); + try testing.expectEqual(cloud9.owrite, openMode(wronly)); + try testing.expectEqual(cloud9.ordwr, openMode(rdwr)); + try testing.expectEqual(cloud9.owrite | cloud9.otrunc, openMode(wronly | trunc)); + try testing.expectEqual(cloud9.ordwr | cloud9.otrunc, openMode(rdwr | trunc | creat)); + try testing.expectEqual(cloud9.owrite, openMode(wronly | append)); // O_APPEND ignored +} + +test "dirlist parsing from two hand-encoded Stat records" { + var buf: [512]u8 = undefined; + const a = try cloud9.Stat.encode(testStat("alpha", 0o644, 10, 0x11), &buf); + const bb = try cloud9.Stat.encode(testStat("beta", cloud9.dmdir | 0o755, 0, 0x22), buf[a.len..]); + const bytes = buf[0 .. a.len + bb.len]; + // Sanity: the record is prefixed by its own 2-byte size. + try testing.expectEqual(a.len - 2, std.mem.readInt(u16, bytes[0..2], .little)); + + var list: DirList = .{}; + defer list.deinit(testing.allocator); + try parseDirRecords(testing.allocator, bytes, &list); + try testing.expectEqual(@as(usize, 2), list.entries.items.len); + try testing.expectEqualStrings("alpha", list.entries.items[0].name); + try testing.expectEqual(@as(u64, 0x11), list.entries.items[0].ino); + try testing.expectEqual(fuse.DT_REG, list.entries.items[0].dtype); + try testing.expectEqualStrings("beta", list.entries.items[1].name); + try testing.expectEqual(@as(u64, 0x22), list.entries.items[1].ino); + try testing.expectEqual(fuse.DT_DIR, list.entries.items[1].dtype); + + // Truncated input is a protocol error and leaves earlier entries intact. + try testing.expectError(error.BadDir, parseDirRecords(testing.allocator, bytes[0 .. bytes.len - 1], &list)); + try testing.expectEqual(@as(usize, 3), list.entries.items.len); +} + +test "readdir packing and offset resumption" { + const names = [_][]const u8{ ".", "..", "one", "two", "three" }; + var entries: [names.len]Entry = undefined; + for (&entries, names, 0..) |*e, n, i| e.* = .{ .name = @constCast(n), .ino = 100 + i, .dtype = if (i < 2) fuse.DT_DIR else fuse.DT_REG }; + + // Everything fits: five records, off = index + 1. + var big: [1024]u8 = undefined; + const used = packDirents(&entries, 0, &big); + var pos: usize = 0; + var idx: usize = 0; + while (pos < used) : (idx += 1) { + const d = std.mem.bytesToValue(fuse.Dirent, big[pos..][0..@sizeOf(fuse.Dirent)]); + try testing.expectEqual(@as(u64, 100 + idx), d.ino); + try testing.expectEqual(@as(u64, idx + 1), d.off); + try testing.expectEqualStrings(names[idx], big[pos + @sizeOf(fuse.Dirent) ..][0..d.namelen]); + pos += (@sizeOf(fuse.Dirent) + d.namelen + 7) & ~@as(usize, 7); + } + try testing.expectEqual(names.len, idx); + + // A buffer that fits exactly two records ("." = 32, ".." = 32) stops there… + var small: [64]u8 = undefined; + const first_used = packDirents(&entries, 0, &small); + try testing.expectEqual(@as(usize, 64), first_used); + const last = std.mem.bytesToValue(fuse.Dirent, small[32..][0..@sizeOf(fuse.Dirent)]); + try testing.expectEqual(@as(u64, 2), last.off); + // …and resuming at the last `off` yields "one" next. + const second_used = packDirents(&entries, last.off, &small); + const next = std.mem.bytesToValue(fuse.Dirent, small[0..@sizeOf(fuse.Dirent)]); + try testing.expectEqualStrings("one", small[@sizeOf(fuse.Dirent)..][0..next.namelen]); + try testing.expectEqual(@as(u64, 3), next.off); + try testing.expect(second_used > 0); + + // Past the end: nothing (EOF for the kernel). + try testing.expectEqual(@as(usize, 0), packDirents(&entries, names.len, &big)); + try testing.expectEqual(@as(usize, 0), packDirents(&entries, 1000, &big)); +} + +test "attr mapping saturates hostile lengths instead of overflowing" { + const a = attrFromStat(testStat("f", 0o600, std.math.maxInt(u64), 4), 4, 0, 0); + try testing.expectEqual(@as(u64, std.math.maxInt(i64)), a.size); + try testing.expectEqual(@as(u64, std.math.maxInt(u64) / 512 + 1), a.blocks); + const b = attrFromStat(testStat("f", 0o600, 1024, 4), 4, 0, 0); + try testing.expectEqual(@as(u64, 2), b.blocks); + try testing.expectEqual(@as(u64, 1024), b.size); +} + +test "dirent names the kernel would reject are dropped from listings" { + try testing.expect(validDirentName("a")); + try testing.expect(validDirentName("x" ** 1024)); + try testing.expect(!validDirentName("")); + try testing.expect(!validDirentName("a/b")); + try testing.expect(!validDirentName("a\x00b")); + try testing.expect(!validDirentName(".")); + try testing.expect(!validDirentName("..")); + try testing.expect(!validDirentName("x" ** 1025)); + + var buf: [4096]u8 = undefined; + var n: usize = 0; + for ([_][]const u8{ ".", "..", "", "a/b", "keep", "x" ** 1025, "also" }) |name| { + n += (try cloud9.Stat.encode(testStat(name, 0o644, 1, 0x30), buf[n..])).len; + } + var list: DirList = .{}; + defer list.deinit(testing.allocator); + try parseDirRecords(testing.allocator, buf[0..n], &list); + try testing.expectEqual(@as(usize, 2), list.entries.items.len); + try testing.expectEqualStrings("keep", list.entries.items[0].name); + try testing.expectEqualStrings("also", list.entries.items[1].name); +} + +test "DirList frees its names" { + var list: DirList = .{}; + try list.entries.append(testing.allocator, .{ .name = try testing.allocator.dupe(u8, "x"), .ino = 1, .dtype = fuse.DT_REG }); + list.deinit(testing.allocator); +} diff --git a/9ns/src/fuse.zig b/9ns/src/fuse.zig new file mode 100644 index 0000000..682b3d5 --- /dev/null +++ b/9ns/src/fuse.zig @@ -0,0 +1,653 @@ +//! Kernel FUSE protocol subset (no libfuse, no libc, no policy). +//! +//! Extern structs mirror `/usr/include/linux/fuse.h` (kernel header 7.45); +//! every layout is checked against the header's size at comptime. Only the +//! opcodes and structs 9ns needs are here. The I/O helpers are blocking +//! and allocation-free: the caller owns a single request buffer. +//! +//! Wire rules worth remembering: +//! * The kernel delivers exactly one request per `read(2)` on `/dev/fuse`, +//! and a reply must be exactly one `write(2)`/`writev(2)`. +//! * Request bodies start right after the 40-byte `InHeader`; since every +//! in-struct is 8-byte aligned in the header, `body()` requires the caller's +//! buffer to be 8-byte aligned (`std.heap` page allocations and +//! `align(8)` arrays both qualify). +//! * A write that fails with `ENOENT` means the request was interrupted and +//! the kernel already forgot it: the reply is silently dropped. +//! * `ENODEV` on read means the filesystem was unmounted. + +const std = @import("std"); +const linux = std.os.linux; + +pub const kernel_version: u32 = 7; +/// The minor we answer; the kernel adapts to the lower of the two. +pub const kernel_minor: u32 = 31; +pub const root_id: u64 = 1; + +pub const FOPEN_DIRECT_IO: u32 = 1 << 0; +pub const FOPEN_KEEP_CACHE: u32 = 1 << 1; +pub const FOPEN_NONSEEKABLE: u32 = 1 << 2; + +pub const FUSE_ASYNC_READ: u32 = 1 << 0; +/// The kernel passes O_TRUNC in OPEN instead of a separate SETATTR(size=0); 9P has OTRUNC for exactly this. +pub const FUSE_ATOMIC_O_TRUNC: u32 = 1 << 3; +/// Without this the kernel's cached-write path (`--no-direct-io`) sends one 4 KiB WRITE per page. +pub const FUSE_BIG_WRITES: u32 = 1 << 5; +/// Re-fetch a cached inode's size/mtime and drop stale pages when they change. +/// Required for `--no-direct-io` correctness: 9P sizes change under us, and +/// without this the kernel trusts a stale cached size and truncates reads. +pub const FUSE_AUTO_INVAL_DATA: u32 = 1 << 12; +pub const FUSE_MAX_PAGES: u32 = 1 << 22; + +pub const FATTR_MODE: u32 = 1 << 0; +pub const FATTR_UID: u32 = 1 << 1; +pub const FATTR_GID: u32 = 1 << 2; +pub const FATTR_SIZE: u32 = 1 << 3; +pub const FATTR_ATIME: u32 = 1 << 4; +pub const FATTR_MTIME: u32 = 1 << 5; +pub const FATTR_FH: u32 = 1 << 6; +pub const FATTR_ATIME_NOW: u32 = 1 << 7; +pub const FATTR_MTIME_NOW: u32 = 1 << 8; +pub const FATTR_LOCKOWNER: u32 = 1 << 9; +pub const FATTR_CTIME: u32 = 1 << 10; + +/// File type bits for `Attr.mode` and `Dirent.type` (used by the bridge). +pub const S_IFDIR: u32 = linux.S.IFDIR; +pub const S_IFREG: u32 = linux.S.IFREG; +pub const DT_DIR: u32 = linux.DT.DIR; +pub const DT_REG: u32 = linux.DT.REG; + +pub const Opcode = enum(u32) { + lookup = 1, + forget = 2, + getattr = 3, + setattr = 4, + readlink = 5, + symlink = 6, + mknod = 8, + mkdir = 9, + unlink = 10, + rmdir = 11, + rename = 12, + link = 13, + open = 14, + read = 15, + write = 16, + statfs = 17, + release = 18, + fsync = 20, + setxattr = 21, + getxattr = 22, + listxattr = 23, + removexattr = 24, + flush = 25, + init = 26, + opendir = 27, + readdir = 28, + releasedir = 29, + fsyncdir = 30, + getlk = 31, + setlk = 32, + setlkw = 33, + access = 34, + create = 35, + interrupt = 36, + bmap = 37, + destroy = 38, + ioctl = 39, + poll = 40, + notify_reply = 41, + batch_forget = 42, + fallocate = 43, + readdirplus = 44, + rename2 = 45, + lseek = 46, + copy_file_range = 47, + setupmapping = 48, + removemapping = 49, + syncfs = 50, + tmpfile = 51, + statx = 52, + _, +}; + +// --------------------------------------------------------------------------- +// Structs (field order and widths follow linux/fuse.h exactly) +// --------------------------------------------------------------------------- + +pub const InHeader = extern struct { + len: u32, + opcode: u32, + unique: u64, + nodeid: u64, + uid: u32, + gid: u32, + pid: u32, + total_extlen: u16, + padding: u16, + + pub fn op(h: InHeader) Opcode { + return @enumFromInt(h.opcode); + } +}; + +pub const OutHeader = extern struct { + len: u32, + @"error": i32, + unique: u64, +}; + +pub const Attr = extern struct { + ino: u64 = 0, + size: u64 = 0, + blocks: u64 = 0, + atime: u64 = 0, + mtime: u64 = 0, + ctime: u64 = 0, + atimensec: u32 = 0, + mtimensec: u32 = 0, + ctimensec: u32 = 0, + mode: u32 = 0, + nlink: u32 = 0, + uid: u32 = 0, + gid: u32 = 0, + rdev: u32 = 0, + blksize: u32 = 0, + flags: u32 = 0, +}; + +pub const EntryOut = extern struct { + nodeid: u64 = 0, + generation: u64 = 0, + entry_valid: u64 = 0, + attr_valid: u64 = 0, + entry_valid_nsec: u32 = 0, + attr_valid_nsec: u32 = 0, + attr: Attr = .{}, +}; + +pub const AttrOut = extern struct { + attr_valid: u64 = 0, + attr_valid_nsec: u32 = 0, + dummy: u32 = 0, + attr: Attr = .{}, +}; + +pub const GetattrIn = extern struct { getattr_flags: u32, dummy: u32, fh: u64 }; + +pub const SetattrIn = extern struct { + valid: u32, + padding: u32, + fh: u64, + size: u64, + lock_owner: u64, + atime: u64, + mtime: u64, + ctime: u64, + atimensec: u32, + mtimensec: u32, + ctimensec: u32, + mode: u32, + unused4: u32, + uid: u32, + gid: u32, + unused5: u32, +}; + +pub const OpenIn = extern struct { flags: u32, open_flags: u32 }; +pub const OpenOut = extern struct { fh: u64 = 0, open_flags: u32 = 0, backing_id: i32 = 0 }; +pub const ReleaseIn = extern struct { fh: u64, flags: u32, release_flags: u32, lock_owner: u64 }; +pub const FlushIn = extern struct { fh: u64, unused: u32, padding: u32, lock_owner: u64 }; + +pub const ReadIn = extern struct { + fh: u64, + offset: u64, + size: u32, + read_flags: u32, + lock_owner: u64, + flags: u32, + padding: u32, +}; + +pub const WriteIn = extern struct { + fh: u64, + offset: u64, + size: u32, + write_flags: u32, + lock_owner: u64, + flags: u32, + padding: u32, +}; + +pub const WriteOut = extern struct { size: u32, padding: u32 = 0 }; +pub const CreateIn = extern struct { flags: u32, mode: u32, umask: u32, open_flags: u32 }; +pub const MkdirIn = extern struct { mode: u32, umask: u32 }; +pub const RenameIn = extern struct { newdir: u64 }; +pub const Rename2In = extern struct { newdir: u64, flags: u32, padding: u32 }; +pub const ForgetIn = extern struct { nlookup: u64 }; +pub const BatchForgetIn = extern struct { count: u32, dummy: u32 }; +pub const ForgetOne = extern struct { nodeid: u64, nlookup: u64 }; +pub const FsyncIn = extern struct { fh: u64, fsync_flags: u32, padding: u32 }; +pub const AccessIn = extern struct { mask: u32, padding: u32 }; +pub const InterruptIn = extern struct { unique: u64 }; +pub const LseekIn = extern struct { fh: u64, offset: u64, whence: u32, padding: u32 }; + +pub const Kstatfs = extern struct { + blocks: u64 = 0, + bfree: u64 = 0, + bavail: u64 = 0, + files: u64 = 0, + ffree: u64 = 0, + bsize: u32 = 0, + namelen: u32 = 0, + frsize: u32 = 0, + padding: u32 = 0, + spare: [6]u32 = [_]u32{0} ** 6, +}; + +pub const StatfsOut = extern struct { st: Kstatfs = .{} }; + +pub const InitIn = extern struct { + major: u32, + minor: u32, + max_readahead: u32, + flags: u32, + flags2: u32, + unused: [11]u32, +}; + +/// 64 bytes; the kernel accepts this size whenever the answered minor >= 23. +pub const InitOut = extern struct { + major: u32 = kernel_version, + minor: u32 = kernel_minor, + max_readahead: u32 = 0, + flags: u32 = 0, + max_background: u16 = 0, + congestion_threshold: u16 = 0, + max_write: u32 = 0, + time_gran: u32 = 0, + max_pages: u16 = 0, + map_alignment: u16 = 0, + flags2: u32 = 0, + max_stack_depth: u32 = 0, + request_timeout: u16 = 0, + unused: [11]u16 = [_]u16{0} ** 11, +}; + +/// Fixed 24-byte head of `fuse_dirent`; the name follows, padded to 8 bytes. +pub const Dirent = extern struct { ino: u64, off: u64, namelen: u32, type: u32 }; + +comptime { + std.debug.assert(@sizeOf(InHeader) == 40); + std.debug.assert(@sizeOf(OutHeader) == 16); + std.debug.assert(@sizeOf(Attr) == 88); + std.debug.assert(@sizeOf(EntryOut) == 128); + std.debug.assert(@sizeOf(AttrOut) == 104); + std.debug.assert(@sizeOf(GetattrIn) == 16); + std.debug.assert(@sizeOf(SetattrIn) == 88); + std.debug.assert(@sizeOf(OpenIn) == 8); + std.debug.assert(@sizeOf(OpenOut) == 16); + std.debug.assert(@sizeOf(ReleaseIn) == 24); + std.debug.assert(@sizeOf(FlushIn) == 24); + std.debug.assert(@sizeOf(ReadIn) == 40); + std.debug.assert(@sizeOf(WriteIn) == 40); + std.debug.assert(@sizeOf(WriteOut) == 8); + std.debug.assert(@sizeOf(CreateIn) == 16); + std.debug.assert(@sizeOf(MkdirIn) == 8); + std.debug.assert(@sizeOf(RenameIn) == 8); + std.debug.assert(@sizeOf(Rename2In) == 16); + std.debug.assert(@sizeOf(ForgetIn) == 8); + std.debug.assert(@sizeOf(BatchForgetIn) == 8); + std.debug.assert(@sizeOf(ForgetOne) == 16); + std.debug.assert(@sizeOf(FsyncIn) == 16); + std.debug.assert(@sizeOf(AccessIn) == 8); + std.debug.assert(@sizeOf(InterruptIn) == 8); + std.debug.assert(@sizeOf(Kstatfs) == 80); + std.debug.assert(@sizeOf(StatfsOut) == 80); + std.debug.assert(@sizeOf(InitIn) == 64); + std.debug.assert(@sizeOf(InitOut) == 64); + std.debug.assert(@sizeOf(Dirent) == 24); + std.debug.assert(@sizeOf(LseekIn) == 24); +} + +// --------------------------------------------------------------------------- +// Request / reply helpers +// --------------------------------------------------------------------------- + +pub const Error = error{ Protocol, Io, TooManyPayloads }; + +pub const Request = struct { + header: InHeader, + /// Bytes after the header; a slice into the caller's buffer. + body: []const u8, +}; + +/// Reads one kernel request with a single `read(2)`. Returns null on ENODEV +/// (unmounted). Retries EINTR/EAGAIN/ENOENT. `buf` should be at least +/// `max_write + 4096` bytes and 8-byte aligned so `body()` can view it. +pub fn readRequest(fd: i32, buf: []u8) Error!?Request { + while (true) { + const rc = linux.read(fd, buf.ptr, buf.len); + switch (linux.errno(rc)) { + .SUCCESS => { + const n: usize = rc; + if (n < @sizeOf(InHeader)) return error.Protocol; + const header = std.mem.bytesToValue(InHeader, buf[0..@sizeOf(InHeader)]); + if (header.len != n) return error.Protocol; + return .{ .header = header, .body = buf[@sizeOf(InHeader)..n] }; + }, + .INTR, .AGAIN, .NOENT => continue, + .NODEV => return null, + else => return error.Io, + } + } +} + +/// Maximum number of payload slices a single `reply` can carry. +pub const max_payloads = 7; + +/// Success reply: `OutHeader` followed by the concatenated `payloads`, sent in +/// one `writev(2)`. An ENOENT from the kernel means the request was +/// interrupted; the reply is dropped and this returns normally. +pub fn reply(fd: i32, unique: u64, payloads: []const []const u8) Error!void { + if (payloads.len > max_payloads) return error.TooManyPayloads; + var total: usize = @sizeOf(OutHeader); + for (payloads) |p| total += p.len; + if (total > std.math.maxInt(u32)) return error.Protocol; + const header = OutHeader{ .len = @intCast(total), .@"error" = 0, .unique = unique }; + var iov: [max_payloads + 1]std.posix.iovec_const = undefined; + iov[0] = .{ .base = @ptrCast(&header), .len = @sizeOf(OutHeader) }; + for (payloads, 1..) |p, i| iov[i] = .{ .base = p.ptr, .len = p.len }; + return writeAll(fd, &iov, payloads.len + 1, total); +} + +/// Error reply: an `OutHeader` carrying `-errno` and no payload. +pub fn replyError(fd: i32, unique: u64, err: linux.E) Error!void { + const code: i32 = @intCast(@intFromEnum(err)); + const header = OutHeader{ .len = @sizeOf(OutHeader), .@"error" = -code, .unique = unique }; + var iov = [_]std.posix.iovec_const{.{ .base = @ptrCast(&header), .len = @sizeOf(OutHeader) }}; + return writeAll(fd, &iov, 1, @sizeOf(OutHeader)); +} + +fn writeAll(fd: i32, iov: [*]const std.posix.iovec_const, count: usize, total: usize) Error!void { + while (true) { + const rc = linux.writev(fd, iov, count); + switch (linux.errno(rc)) { + .SUCCESS => return if (rc == total) {} else error.Protocol, + .INTR => continue, + .NOENT => return, // request was interrupted; reply dropped + else => return error.Io, + } + } +} + +/// Appends a `fuse_dirent` (head + name, padded to a multiple of 8) at +/// `buf[used.*..]`. Returns false and leaves `buf`/`used` unchanged if the +/// record does not fit. +pub fn addDirent(buf: []u8, used: *usize, ino: u64, off: u64, dtype: u32, name: []const u8) bool { + const raw = @sizeOf(Dirent) + name.len; + const rec = (raw + 7) & ~@as(usize, 7); + if (used.* > buf.len or buf.len - used.* < rec) return false; + const dst = buf[used.*..][0..rec]; + const head = Dirent{ .ino = ino, .off = off, .namelen = @intCast(name.len), .type = dtype }; + @memcpy(dst[0..@sizeOf(Dirent)], std.mem.asBytes(&head)); + @memcpy(dst[@sizeOf(Dirent)..raw], name); + @memset(dst[raw..rec], 0); + used.* += rec; + return true; +} + +/// Views the first `@sizeOf(T)` bytes of `req.body` as `T` (copy-free). +/// Fails with `error.Protocol` if the body is too short or misaligned. +pub fn body(comptime T: type, req: Request) Error!*const T { + if (req.body.len < @sizeOf(T)) return error.Protocol; + if (@intFromPtr(req.body.ptr) % @alignOf(T) != 0) return error.Protocol; + return @ptrCast(@alignCast(req.body.ptr)); +} + +/// The NUL-terminated string at `req.body[offset..]`, without the NUL. +pub fn nameAt(req: Request, offset: usize) Error![]const u8 { + if (offset > req.body.len) return error.Protocol; + const rest = req.body[offset..]; + const end = std.mem.indexOfScalar(u8, rest, 0) orelse return error.Protocol; + return rest[0..end]; +} + +/// The NUL-terminated string following a `T` body (or at offset 0 when +/// `T == void`), e.g. LOOKUP's name (`void`) or MKDIR's name (`MkdirIn`). +pub fn nameAfter(comptime T: type, req: Request) Error![]const u8 { + const offset = if (T == void) 0 else @sizeOf(T); + return nameAt(req, offset); +} + +/// The string that follows `first` (obtained via `nameAt(req, offset)`), +/// for "old\0new\0" pairs such as RENAME's. +pub fn secondName(req: Request, first: []const u8, offset: usize) Error![]const u8 { + return nameAt(req, offset + first.len + 1); +} + +/// Builds the INIT reply per docs/DESIGN.md. +pub fn initReply(in: *const InitIn, max_write: u32) InitOut { + var out = InitOut{ + .major = kernel_version, + .minor = @min(kernel_minor, in.minor), + .max_readahead = in.max_readahead, + .flags = FUSE_ASYNC_READ | FUSE_ATOMIC_O_TRUNC | FUSE_BIG_WRITES | FUSE_AUTO_INVAL_DATA, + .max_background = 16, + .congestion_threshold = 12, + .max_write = max_write, + .time_gran = 1, + }; + if (in.flags & FUSE_MAX_PAGES != 0) { + out.flags |= FUSE_MAX_PAGES; + out.max_pages = 256; + } + return out; +} + +// --------------------------------------------------------------------------- +// Tests +// --------------------------------------------------------------------------- + +const testing = std.testing; + +test "struct sizes match linux/fuse.h" { + // The comptime block above is the real check; this makes it run under + // `zig test` even if the module is otherwise unreferenced. + try testing.expectEqual(@as(usize, 40), @sizeOf(InHeader)); + try testing.expectEqual(@as(usize, 64), @sizeOf(InitOut)); + try testing.expectEqual(@as(usize, 24), @sizeOf(Dirent)); + try testing.expectEqual(@as(u32, 26), @intFromEnum(Opcode.init)); + try testing.expectEqual(Opcode.statx, @as(Opcode, @enumFromInt(52))); +} + +test "addDirent pads records to 8 bytes and refuses when full" { + var buf: [1024]u8 = undefined; + var used: usize = 0; + const name = "abcdefghijklmnopq"; // 17 chars + var expect_total: usize = 0; + var n: usize = 1; + while (n <= 17) : (n += 1) { + const before = used; + try testing.expect(addDirent(&buf, &used, n, n, DT_REG, name[0..n])); + const rec = used - before; + try testing.expectEqual(@as(usize, 0), rec % 8); + try testing.expectEqual((24 + n + 7) & ~@as(usize, 7), rec); + // check head fields and NUL padding + const head = std.mem.bytesToValue(Dirent, buf[before..][0..24]); + try testing.expectEqual(n, head.ino); + try testing.expectEqual(@as(u32, @intCast(n)), head.namelen); + try testing.expectEqualStrings(name[0..n], buf[before + 24 ..][0..n]); + for (buf[before + 24 + n .. used]) |b| try testing.expectEqual(@as(u8, 0), b); + expect_total += rec; + } + try testing.expectEqual(expect_total, used); + + // A record that does not fit leaves everything untouched. + var small: [40]u8 = undefined; + var used2: usize = 0; + try testing.expect(addDirent(&small, &used2, 1, 1, DT_DIR, "0123456789abcdef")); // 24+16 = 40 + try testing.expectEqual(@as(usize, 40), used2); + try testing.expect(!addDirent(&small, &used2, 2, 2, DT_DIR, "x")); + try testing.expectEqual(@as(usize, 40), used2); + var tight: [31]u8 = undefined; + var used3: usize = 0; + try testing.expect(!addDirent(&tight, &used3, 1, 1, DT_REG, "a")); // needs 32 + try testing.expectEqual(@as(usize, 0), used3); +} + +test "body/nameAfter/secondName on hand-built requests" { + var buf: [128]u8 align(8) = undefined; + // LOOKUP(parent=1, "hello") + const name = "hello"; + const hdr = InHeader{ + .len = @intCast(@sizeOf(InHeader) + name.len + 1), + .opcode = @intFromEnum(Opcode.lookup), + .unique = 7, + .nodeid = root_id, + .uid = 1000, + .gid = 1000, + .pid = 42, + .total_extlen = 0, + .padding = 0, + }; + @memcpy(buf[0..40], std.mem.asBytes(&hdr)); + @memcpy(buf[40..45], name); + buf[45] = 0; + const req = Request{ .header = hdr, .body = buf[40..hdr.len] }; + try testing.expectEqual(Opcode.lookup, req.header.op()); + try testing.expectEqualStrings("hello", try nameAfter(void, req)); + try testing.expectError(error.Protocol, body(MkdirIn, Request{ .header = hdr, .body = buf[40..44] })); + + // MKDIR(mode=0o755) + "dir" + const mk = MkdirIn{ .mode = 0o755, .umask = 0o22 }; + @memcpy(buf[40..48], std.mem.asBytes(&mk)); + @memcpy(buf[48..51], "dir"); + buf[51] = 0; + const mreq = Request{ .header = hdr, .body = buf[40..52] }; + const got = try body(MkdirIn, mreq); + try testing.expectEqual(@as(u32, 0o755), got.mode); + try testing.expectEqualStrings("dir", try nameAfter(MkdirIn, mreq)); + + // RENAME(newdir) + "old\0new\0" + const rn = RenameIn{ .newdir = 9 }; + @memcpy(buf[40..48], std.mem.asBytes(&rn)); + @memcpy(buf[48..56], "old\x00new\x00"); + const rreq = Request{ .header = hdr, .body = buf[40..56] }; + try testing.expectEqual(@as(u64, 9), (try body(RenameIn, rreq)).newdir); + const old = try nameAfter(RenameIn, rreq); + try testing.expectEqualStrings("old", old); + try testing.expectEqualStrings("new", try secondName(rreq, old, @sizeOf(RenameIn))); + try testing.expectError(error.Protocol, secondName(rreq, "new", @sizeOf(RenameIn) + 4)); + + // Missing NUL and misalignment are protocol errors. + try testing.expectError(error.Protocol, nameAt(Request{ .header = hdr, .body = buf[48..51] }, 0)); + try testing.expectError(error.Protocol, body(MkdirIn, Request{ .header = hdr, .body = buf[41..57] })); +} + +test "initReply fields" { + var in = InitIn{ .major = 7, .minor = 45, .max_readahead = 131072, .flags = 0, .flags2 = 0, .unused = [_]u32{0} ** 11 }; + const a = initReply(&in, 1 << 20); + try testing.expectEqual(@as(u32, 7), a.major); + try testing.expectEqual(@as(u32, 31), a.minor); + try testing.expectEqual(@as(u32, 131072), a.max_readahead); + try testing.expectEqual(FUSE_ASYNC_READ | FUSE_ATOMIC_O_TRUNC | FUSE_BIG_WRITES | FUSE_AUTO_INVAL_DATA, a.flags); + try testing.expectEqual(@as(u16, 0), a.max_pages); + try testing.expectEqual(@as(u16, 16), a.max_background); + try testing.expectEqual(@as(u16, 12), a.congestion_threshold); + try testing.expectEqual(@as(u32, 1 << 20), a.max_write); + try testing.expectEqual(@as(u32, 1), a.time_gran); + + in.flags = FUSE_MAX_PAGES | FUSE_ASYNC_READ; + in.minor = 27; + const b = initReply(&in, 4096); + try testing.expectEqual(@as(u32, 27), b.minor); + try testing.expectEqual(FUSE_ASYNC_READ | FUSE_ATOMIC_O_TRUNC | FUSE_BIG_WRITES | FUSE_AUTO_INVAL_DATA | FUSE_MAX_PAGES, b.flags); + try testing.expectEqual(@as(u16, 256), b.max_pages); + try testing.expectEqual(@as(u32, 4096), b.max_write); +} + +fn makePipe() ![2]i32 { + var fds: [2]i32 = undefined; + if (linux.errno(linux.pipe2(&fds, .{ .CLOEXEC = true })) != .SUCCESS) return error.Io; + return fds; +} + +fn readExact(fd: i32, out: []u8) !void { + var got: usize = 0; + while (got < out.len) { + const rc = linux.read(fd, out[got..].ptr, out.len - got); + if (linux.errno(rc) != .SUCCESS or rc == 0) return error.Io; + got += rc; + } +} + +test "reply writes header + payloads through a pipe" { + const fds = try makePipe(); + defer _ = linux.close(fds[0]); + defer _ = linux.close(fds[1]); + + const oo = OpenOut{ .fh = 0x1234, .open_flags = FOPEN_DIRECT_IO }; + try reply(fds[1], 99, &.{ std.mem.asBytes(&oo), "tail" }); + + var out: [16 + 16 + 4]u8 = undefined; + try readExact(fds[0], &out); + const h = std.mem.bytesToValue(OutHeader, out[0..16]); + try testing.expectEqual(@as(u32, 36), h.len); + try testing.expectEqual(@as(i32, 0), h.@"error"); + try testing.expectEqual(@as(u64, 99), h.unique); + try testing.expectEqualSlices(u8, std.mem.asBytes(&oo), out[16..32]); + try testing.expectEqualStrings("tail", out[32..36]); + + // Empty payload list: header only. + try reply(fds[1], 5, &.{}); + var only: [16]u8 = undefined; + try readExact(fds[0], &only); + try testing.expectEqual(@as(u32, 16), std.mem.bytesToValue(OutHeader, &only).len); + + var too_many: [max_payloads + 1][]const u8 = undefined; + for (&too_many) |*p| p.* = "x"; + try testing.expectError(error.TooManyPayloads, reply(fds[1], 1, &too_many)); +} + +test "replyError writes a negative errno" { + const fds = try makePipe(); + defer _ = linux.close(fds[0]); + defer _ = linux.close(fds[1]); + + try replyError(fds[1], 0xdead_beef, .NOENT); + var out: [16]u8 = undefined; + try readExact(fds[0], &out); + const h = std.mem.bytesToValue(OutHeader, &out); + try testing.expectEqual(@as(u32, 16), h.len); + try testing.expectEqual(@as(i32, -2), h.@"error"); + try testing.expectEqual(@as(u64, 0xdead_beef), h.unique); + + try replyError(fds[1], 1, .NOSYS); + try readExact(fds[0], &out); + try testing.expectEqual(-@as(i32, @intCast(@intFromEnum(linux.E.NOSYS))), std.mem.bytesToValue(OutHeader, &out).@"error"); +} + +test "readRequest parses one request from a pipe and rejects bad lengths" { + const fds = try makePipe(); + defer _ = linux.close(fds[0]); + defer _ = linux.close(fds[1]); + + var wire: [48]u8 align(8) = undefined; + const hdr = InHeader{ .len = 48, .opcode = @intFromEnum(Opcode.forget), .unique = 3, .nodeid = 2, .uid = 0, .gid = 0, .pid = 0, .total_extlen = 0, .padding = 0 }; + @memcpy(wire[0..40], std.mem.asBytes(&hdr)); + @memcpy(wire[40..48], std.mem.asBytes(&ForgetIn{ .nlookup = 11 })); + try testing.expectEqual(@as(usize, 48), linux.write(fds[1], &wire, wire.len)); + + var buf: [4096]u8 align(8) = undefined; + const req = (try readRequest(fds[0], &buf)) orelse return error.Io; + try testing.expectEqual(Opcode.forget, req.header.op()); + try testing.expectEqual(@as(u64, 2), req.header.nodeid); + try testing.expectEqual(@as(u64, 11), (try body(ForgetIn, req)).nlookup); + + // Header length disagreeing with what was read is a protocol error. + var bad = wire; + std.mem.bytesAsValue(InHeader, bad[0..40]).len = 40; + try testing.expectEqual(@as(usize, 48), linux.write(fds[1], &bad, bad.len)); + try testing.expectError(error.Protocol, readRequest(fds[0], &buf)); +} diff --git a/9ns/src/main.zig b/9ns/src/main.zig new file mode 100644 index 0000000..26bd699 --- /dev/null +++ b/9ns/src/main.zig @@ -0,0 +1,444 @@ +//! 9ns: mount a 9P2000 tree into a fresh user+mount namespace via FUSE +//! and run a program inside it. +//! +//! Exit codes: the child's status (128+sig if signalled); 125 for 9ns's +//! own failures (usage, connect, attach, namespace/mount); 126/127 for exec +//! failures. + +const std = @import("std"); +const linux = std.os.linux; +const ns = @import("ns.zig"); +const nine = @import("nine.zig"); +const bridge = @import("bridge.zig"); + +const version_string = "9ns 0.1.0"; + +const usage_text = + \\Usage: 9ns [options] -- PROGRAM [ARGS...] + \\Transport (exactly one): + \\ --unix PATH Unix stream socket + \\ --tcp IP:PORT TCP (IPv4/IPv6 literal) + \\ --fd N already-connected inherited descriptor + \\ --spawn CMD run CMD (via /bin/sh -c) with a socketpair on its stdin/stdout + \\Options: + \\ --mount PATH mountpoint inside the new namespace (default /mnt/9p) + \\ --uname NAME 9P user name (default $USER, else "none") + \\ --aname NAME 9P tree to attach (default "") + \\ --msize BYTES maximum 9P message size to request (default 131072) + \\ --cache SECONDS attr/entry cache validity, may be fractional (default 1) + \\ --no-direct-io let the kernel cache file pages (trusts stat length) + \\ --debug trace FUSE and 9P operations on stderr + \\ --help, --version + \\PROGRAM defaults to $SHELL (else /bin/sh). The mountpoint is exported as $NINE_MOUNT. + \\ +; + +const own_failure: u8 = 125; +/// Largest 9P message size we agree to request: the session allocates two +/// buffers of this size up front, before the server negotiates it down. +const max_msize: u32 = 16 * 1024 * 1024; + +/// Write `text` to stdout (informational output such as --help); errors are +/// ignored, there is nowhere better to report them. +fn printStdout(text: []const u8) void { + var off: usize = 0; + while (off < text.len) { + const rc = linux.write(1, text[off..].ptr, text.len - off); + switch (linux.errno(rc)) { + .SUCCESS => off += rc, + .INTR => continue, + else => return, + } + } +} + +const Config = struct { + address: ?nine.Address = null, + spawn_cmd: ?[]const u8 = null, + mount: []const u8 = "/mnt/9p", + uname: ?[]const u8 = null, + aname: []const u8 = "", + msize: u32 = 131072, + cache_ns: u64 = 1_000_000_000, + direct_io: bool = true, + debug: bool = false, + /// Empty means "default program". + program: []const []const u8 = &.{}, +}; + +const ParseResult = union(enum) { + run: Config, + /// Usage error, already reported on stderr; exit with this status. + exit: u8, + /// --help/--version: text for stdout, then exit 0. Printing is left to + /// `main` so that no test path writes to fd 1 (under `zig build test` + /// that is the test runner's protocol pipe). + info: []const u8, +}; + +fn usageError(comptime fmt: []const u8, args: anytype) ParseResult { + std.debug.print("9ns: " ++ fmt ++ "\n(try 9ns --help)\n", args); + return .{ .exit = own_failure }; +} + +fn parseArgs(arena: std.mem.Allocator, args: []const [:0]const u8) !ParseResult { + var cfg = Config{}; + var transports: usize = 0; + var i: usize = 1; + var program_start: ?usize = null; + while (i < args.len) : (i += 1) { + const arg: []const u8 = args[i]; + if (std.mem.eql(u8, arg, "--")) { + program_start = i + 1; + break; + } + if (!std.mem.startsWith(u8, arg, "--")) { + // A single-dash word is a typo for an option, not a program. + if (arg.len > 1 and arg[0] == '-') return usageError("unknown option {s} (options start with --)", .{arg}); + // A bare word starts PROGRAM, as if "--" were given. + program_start = i; + break; + } + // Split "--opt=value". + var name = arg; + var inline_value: ?[]const u8 = null; + if (std.mem.indexOfScalar(u8, arg, '=')) |eq| { + name = arg[0..eq]; + inline_value = arg[eq + 1 ..]; + } + const Opt = enum { unix, tcp, fd, spawn, mount, uname, aname, msize, cache, @"no-direct-io", debug, help, version, unknown }; + const opt = std.meta.stringToEnum(Opt, name[2..]) orelse .unknown; + switch (opt) { + .@"no-direct-io", .debug, .help, .version => if (inline_value != null) return usageError("{s} takes no value", .{name}), + .unknown => return usageError("unknown option {s}", .{name}), + else => {}, + } + const value: []const u8 = switch (opt) { + .@"no-direct-io", .debug, .help, .version, .unknown => "", + else => inline_value orelse blk: { + i += 1; + if (i >= args.len) return usageError("{s} needs a value", .{name}); + break :blk args[i]; + }, + }; + switch (opt) { + .unix => { + if (value.len == 0) return usageError("--unix wants a socket path", .{}); + cfg.address = .{ .unix = value }; + transports += 1; + }, + .tcp => { + cfg.address = parseTcp(value) orelse return usageError("--tcp wants IP:PORT (IPv6 as [ADDR]:PORT), got '{s}'", .{value}); + transports += 1; + }, + .fd => { + const n = std.fmt.parseInt(i32, value, 10) catch return usageError("--fd wants a number, got '{s}'", .{value}); + if (n < 0) return usageError("--fd wants a non-negative number", .{}); + cfg.address = .{ .fd = n }; + transports += 1; + }, + .spawn => { + if (value.len == 0) return usageError("--spawn wants a command", .{}); + cfg.spawn_cmd = value; + transports += 1; + }, + .mount => { + if (value.len == 0) return usageError("--mount wants a path", .{}); + cfg.mount = value; + }, + .uname => cfg.uname = value, + .aname => cfg.aname = value, + .msize => { + cfg.msize = std.fmt.parseInt(u32, value, 10) catch return usageError("--msize wants a number, got '{s}'", .{value}); + if (cfg.msize < 4096 or cfg.msize > max_msize) return usageError("--msize must be between 4096 and {d}", .{max_msize}); + }, + .cache => { + const secs = std.fmt.parseFloat(f64, value) catch return usageError("--cache wants seconds, got '{s}'", .{value}); + if (!(secs >= 0) or secs > 1e9) return usageError("--cache out of range", .{}); + cfg.cache_ns = @intFromFloat(secs * 1e9); + }, + .@"no-direct-io" => cfg.direct_io = false, + .debug => cfg.debug = true, + .help => return .{ .info = usage_text }, + .version => return .{ .info = version_string ++ "\n" }, + .unknown => unreachable, + } + } + if (transports == 0) return usageError("one transport is required (--unix, --tcp, --fd or --spawn)", .{}); + if (transports > 1) return usageError("exactly one transport is allowed", .{}); + if (program_start) |start| { + const prog = try arena.alloc([]const u8, args.len - start); + for (args[start..], 0..) |a, j| prog[j] = a; + cfg.program = prog; + } + return .{ .run = cfg }; +} + +fn parseTcp(spec: []const u8) ?nine.Address { + const colon = std.mem.lastIndexOfScalar(u8, spec, ':') orelse return null; + var host = spec[0..colon]; + if (host.len >= 2 and host[0] == '[' and host[host.len - 1] == ']') host = host[1 .. host.len - 1]; + if (host.len == 0) return null; + const port = std.fmt.parseInt(u16, spec[colon + 1 ..], 10) catch return null; + return .{ .tcp = .{ .host = host, .port = port } }; +} + +/// `--spawn`: run CMD under /bin/sh with one end of a socketpair as its +/// stdin/stdout; the other end is the 9P transport. +const Server = struct { pid: i32, fd: i32 }; + +fn spawnServer(cmd: [:0]const u8, envp: [*:null]const ?[*:0]const u8) !Server { + var sv: [2]i32 = undefined; + switch (linux.errno(linux.socketpair(linux.AF.UNIX, linux.SOCK.STREAM | linux.SOCK.CLOEXEC, 0, &sv))) { + .SUCCESS => {}, + else => |e| { + std.debug.print("9ns: socketpair: E{t}\n", .{e}); + return error.SystemResources; + }, + } + const rc = linux.fork(); + switch (linux.errno(rc)) { + .SUCCESS => {}, + else => |e| { + _ = linux.close(sv[0]); + _ = linux.close(sv[1]); + std.debug.print("9ns: fork: E{t}\n", .{e}); + return error.SystemResources; + }, + } + if (rc == 0) { + // Child: dup2 clears CLOEXEC on 0 and 1; everything else is CLOEXEC. + if (linux.errno(linux.dup2(sv[1], 0)) != .SUCCESS or linux.errno(linux.dup2(sv[1], 1)) != .SUCCESS) linux.exit_group(125); + // The server shares our process group, so a Ctrl-C meant for the + // program would kill it and take the mount down with it: ignore the + // tty signals (inherited across exec). SIGPIPE goes back to its + // default, we only ignore it for ourselves. + ignoreSignal(.INT); + ignoreSignal(.QUIT); + defaultSignal(.PIPE); + const argv = [_:null]?[*:0]const u8{ "sh", "-c", cmd.ptr }; + const e = linux.errno(linux.execve("/bin/sh", &argv, envp)); + std.debug.print("9ns: --spawn: exec /bin/sh: E{t}\n", .{e}); + linux.exit_group(127); + } + _ = linux.close(sv[1]); + return .{ .pid = @intCast(rc), .fd = sv[0] }; +} + +fn stopServer(server: ?Server) void { + const s = server orelse return; + _ = linux.kill(s.pid, .TERM); + ns.reapAny(s.pid); +} + +/// Fail early (before spawning servers or forking) if /dev/fuse is unusable. +fn probeFuseDevice() bool { + const rc = linux.open("/dev/fuse", .{ .ACCMODE = .RDWR, .CLOEXEC = true }, 0); + switch (linux.errno(rc)) { + .SUCCESS => { + _ = linux.close(@intCast(rc)); + return true; + }, + .NOENT => std.debug.print("9ns: /dev/fuse: ENOENT (is the fuse module loaded? try: modprobe fuse)\n", .{}), + else => |e| std.debug.print("9ns: open /dev/fuse: E{t}\n", .{e}), + } + return false; +} + +fn ignoreSignal(sig: linux.SIG) void { + const ign = linux.Sigaction{ .handler = .{ .handler = linux.SIG.IGN }, .mask = linux.sigemptyset(), .flags = 0 }; + std.posix.sigaction(sig, &ign, null); +} + +fn defaultSignal(sig: linux.SIG) void { + const dfl = linux.Sigaction{ .handler = .{ .handler = linux.SIG.DFL }, .mask = linux.sigemptyset(), .flags = 0 }; + std.posix.sigaction(sig, &dfl, null); +} + +/// `--fd N`: the descriptor is ours from now on; it must not leak into the +/// program (which could otherwise read 9P replies meant for us). Fails on a +/// bad descriptor, which is the earliest place to report it. +fn adoptFd(fd: i32) bool { + switch (linux.errno(linux.fcntl(fd, linux.F.SETFD, linux.FD_CLOEXEC))) { + .SUCCESS => return true, + else => |e| { + std.debug.print("9ns: --fd {d}: E{t}\n", .{ fd, e }); + return false; + }, + } +} + +fn describeAddress(a: nine.Address, buf: []u8) []const u8 { + return switch (a) { + .unix => |p| std.fmt.bufPrint(buf, "unix socket {s}", .{p}) catch "unix socket", + .tcp => |t| std.fmt.bufPrint(buf, "tcp {s}:{d}", .{ t.host, t.port }) catch "tcp", + .fd => |fd| std.fmt.bufPrint(buf, "fd {d}", .{fd}) catch "fd", + }; +} + +pub fn main(init: std.process.Init) !u8 { + const gpa = init.gpa; + const arena = init.arena.allocator(); + const args = try init.minimal.args.toSlice(arena); + const envp: [*:null]const ?[*:0]const u8 = init.minimal.environ.block.slice.ptr; + + var cfg = switch (try parseArgs(arena, args)) { + .exit => |code| return code, + .info => |text| { + printStdout(text); + return 0; + }, + .run => |c| c, + }; + + // Defaults that come from the environment. + if (cfg.program.len == 0) { + const env_shell = ns.getenv(envp, "SHELL") orelse ""; + const shell = if (env_shell.len == 0) "/bin/sh" else env_shell; + cfg.program = try arena.dupe([]const u8, &.{shell}); + } + const uname = cfg.uname orelse ns.getenv(envp, "USER") orelse "none"; + const mountpoint = ns.resolveMountpoint(gpa, cfg.mount) catch |err| { + std.debug.print("9ns: --mount {s}: {t}\n", .{ cfg.mount, err }); + return own_failure; + }; + defer gpa.free(mountpoint); + + if (!probeFuseDevice()) return own_failure; + + // Writes to a dead server socket must not kill us. + ignoreSignal(.PIPE); + + var server: ?Server = null; + var address: nine.Address = undefined; + if (cfg.spawn_cmd) |cmd| { + const cmd_z = try arena.dupeZ(u8, cmd); + server = spawnServer(cmd_z, envp) catch return own_failure; + ns.watchServer(server.?.pid); + address = .{ .fd = server.?.fd }; + } else { + address = cfg.address.?; + if (address == .fd and !adoptFd(address.fd)) return own_failure; + } + + var addr_buf: [256]u8 = undefined; + var session = nine.Session.connect(gpa, address, cfg.msize) catch |err| { + std.debug.print("9ns: connect to {s}: {t}\n", .{ describeAddress(address, &addr_buf), err }); + stopServer(server); + return own_failure; + }; + defer session.deinit(); + defer stopServer(server); + + _ = session.attach(0, uname, cfg.aname) catch |err| { + switch (err) { + error.Nine => std.debug.print("9ns: attach (uname={s}, aname='{s}'): {s}\n", .{ uname, cfg.aname, session.ename[0..session.ename_len] }), + else => std.debug.print("9ns: attach: {t}\n", .{err}), + } + return own_failure; + }; + if (cfg.debug) std.debug.print("9ns: attached to {s} (msize {d}), mounting on {s}\n", .{ describeAddress(address, &addr_buf), session.msize, mountpoint }); + + var child_pid: i32 = 0; + const stop_fd = ns.installSignals(&child_pid) catch return own_failure; + + const uid = linux.getuid(); + const gid = linux.getgid(); + const child = ns.spawn(gpa, .{ + .argv = cfg.program, + .envp = envp, + .mountpoint = mountpoint, + .uid = uid, + .gid = gid, + .max_read = bridge.max_write, + }) catch return own_failure; + + bridge.serve(gpa, child.fuse_fd, &session, 0, stop_fd, .{ + .uid = uid, + .gid = gid, + .attr_timeout_ns = cfg.cache_ns, + .direct_io = cfg.direct_io, + .debug = cfg.debug, + }) catch |err| switch (err) { + error.Closed => std.debug.print("9ns: 9P server connection closed\n", .{}), + else => std.debug.print("9ns: fuse: {t}\n", .{err}), + }; + + // Closing the device aborts the FUSE connection: anything still using + // the mount gets ENOTCONN instead of hanging on an unserved request. + _ = linux.close(child.fuse_fd); + + const status = ns.reapIfExited(child.pid) orelse ns.waitChild(child.pid) catch own_failure; + // An exec failure (126/127) is already in `status`; this prints its message. + _ = ns.reportExecFailure(child); + return status; +} + +test "parseTcp" { + const a = parseTcp("127.0.0.1:564").?; + try std.testing.expectEqualStrings("127.0.0.1", a.tcp.host); + try std.testing.expectEqual(@as(u16, 564), a.tcp.port); + const b = parseTcp("[::1]:9999").?; + try std.testing.expectEqualStrings("::1", b.tcp.host); + try std.testing.expectEqual(@as(u16, 9999), b.tcp.port); + try std.testing.expect(parseTcp("nohost") == null); + try std.testing.expect(parseTcp(":564") == null); + try std.testing.expect(parseTcp("1.2.3.4:") == null); + try std.testing.expect(parseTcp("1.2.3.4:70000") == null); +} + +test "parseArgs" { + const arena = std.testing.allocator; + { + const args = [_][:0]const u8{ "9ns", "--unix", "/s", "--cache", "0.5", "--msize=8192", "--no-direct-io", "--", "sh", "-c", "x" }; + const r = try parseArgs(arena, &args); + defer arena.free(r.run.program); + try std.testing.expectEqualStrings("/s", r.run.address.?.unix); + try std.testing.expectEqual(@as(u64, 500_000_000), r.run.cache_ns); + try std.testing.expectEqual(@as(u32, 8192), r.run.msize); + try std.testing.expect(!r.run.direct_io); + try std.testing.expectEqual(@as(usize, 3), r.run.program.len); + try std.testing.expectEqualStrings("x", r.run.program[2]); + } + { + const args = [_][:0]const u8{ "9ns", "--fd", "3" }; + const r = try parseArgs(arena, &args); + try std.testing.expectEqual(@as(i32, 3), r.run.address.?.fd); + try std.testing.expectEqual(@as(usize, 0), r.run.program.len); + try std.testing.expectEqualStrings("/mnt/9p", r.run.mount); + } + { + // Two transports, no transport, unknown option, missing value: all 125. + const two = [_][:0]const u8{ "9ns", "--fd", "3", "--unix", "/s" }; + try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &two)).exit); + const none = [_][:0]const u8{ "9ns", "--", "sh" }; + try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &none)).exit); + const unknown = [_][:0]const u8{ "9ns", "--bogus" }; + try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &unknown)).exit); + const missing = [_][:0]const u8{ "9ns", "--unix" }; + try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &missing)).exit); + const badcache = [_][:0]const u8{ "9ns", "--fd", "3", "--cache", "abc" }; + try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &badcache)).exit); + // Empty values, a single-dash typo, and an msize that would allocate gigabytes. + const emptyunix = [_][:0]const u8{ "9ns", "--unix=", "--", "sh" }; + try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &emptyunix)).exit); + const emptymount = [_][:0]const u8{ "9ns", "--fd", "3", "--mount", "" }; + try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &emptymount)).exit); + const singledash = [_][:0]const u8{ "9ns", "--fd", "3", "-mount", "/x" }; + try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &singledash)).exit); + const hugemsize = [_][:0]const u8{ "9ns", "--fd", "3", "--msize", "4294967295" }; + try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &hugemsize)).exit); + const okmsize = [_][:0]const u8{ "9ns", "--fd", "3", "--msize", "16777216" }; + try std.testing.expectEqual(@as(u32, 16777216), (try parseArgs(arena, &okmsize)).run.msize); + } + { + const ver = [_][:0]const u8{ "9ns", "--version" }; + try std.testing.expectEqualStrings(version_string ++ "\n", (try parseArgs(arena, &ver)).info); + const help = [_][:0]const u8{ "9ns", "--help" }; + try std.testing.expect(std.mem.startsWith(u8, (try parseArgs(arena, &help)).info, "Usage: 9ns")); + } +} + +test { + _ = ns; +} diff --git a/9ns/src/nine.zig b/9ns/src/nine.zig new file mode 100644 index 0000000..70633e6 --- /dev/null +++ b/9ns/src/nine.zig @@ -0,0 +1,756 @@ +//! Synchronous 9P2000 session over a blocking file descriptor. +//! +//! A thin RPC layer over `cloud9.Client` (push/take, allocation-free). One request +//! is outstanding at a time: the FUSE loop that drives this is single-threaded, so +//! every call here blocks until its reply (or the connection's death) arrives. +//! Fids are handed out from a free list; fid 0 is reserved for the root. +const std = @import("std"); +const cloud9 = @import("cloud9"); +const linux = std.os.linux; + +pub const Address = union(enum) { + unix: []const u8, + tcp: struct { host: []const u8, port: u16 }, + fd: i32, +}; + +/// A Stat whose every field means "leave unchanged" in a Twstat. +pub const dontcare = cloud9.Stat{ + .type = 0xFFFF, + .dev = 0xFFFF_FFFF, + .qid = .{ .type = 0xFF, .version = 0xFFFF_FFFF, .path = 0xFFFF_FFFF_FFFF_FFFF }, + .mode = 0xFFFF_FFFF, + .atime = 0xFFFF_FFFF, + .mtime = 0xFFFF_FFFF, + .length = 0xFFFF_FFFF_FFFF_FFFF, + .name = "", + .uid = "", + .gid = "", + .muid = "", +}; + +pub const Session = struct { + pub const Error = error{ Nine, Protocol, Io, Closed, Stopped, TooLarge, OutOfMemory }; + + pub const Walk = struct { nwqid: u16, wqid: [cloud9.max_welem]cloud9.Qid }; + pub const Open = struct { qid: cloud9.Qid, iounit: u32 }; + + gpa: std.mem.Allocator, + fd: i32, + client: cloud9.Client, + in_buf: []u8, + out_buf: []u8, + /// After `error.Nine`, the server's Rerror text (copied, bounded). + ename: [256]u8 = undefined, + ename_len: usize = 0, + /// Negotiated maximum message size. + msize: u32, + next_fid: u32 = 1, + free_fids: std.ArrayList(u32) = .empty, + /// Per-fid iounit learned from open/create (0 = none); used to chunk read/write. + iounits: std.AutoHashMapUnmanaged(u32, u32) = .empty, + /// Optional descriptor watched while waiting for a reply: when it becomes + /// readable (the bridge's "child exited" pipe) the pending rpc fails with + /// `error.Stopped` instead of blocking on a server that never answers. + stop_fd: i32 = -1, + + /// Connect to `address`, then negotiate the protocol version. + /// `msize` is the maximum message size to ask for (0 = the buffers' size). + pub fn connect(gpa: std.mem.Allocator, address: Address, msize: u32) !Session { + const want: u32 = if (msize == 0) 8192 else @max(msize, 24); + const fd = try openTransport(address); + errdefer if (address != .fd) { + _ = linux.close(fd); + }; + + const in_buf = try gpa.alloc(u8, want); + errdefer gpa.free(in_buf); + const out_buf = try gpa.alloc(u8, want); + errdefer gpa.free(out_buf); + + var s: Session = .{ + .gpa = gpa, + .fd = fd, + .client = .init(.{ .in = in_buf, .out = out_buf }), + .in_buf = in_buf, + .out_buf = out_buf, + .msize = want, + }; + const r = try s.rpc(.{ .version = .{ .msize = want } }); + if (!std.mem.eql(u8, r.version.version, "9P2000")) return error.Protocol; + s.msize = r.version.msize; + return s; + } + + /// Closes the descriptor and frees the buffers. Fids are not clunked. + pub fn deinit(s: *Session) void { + _ = linux.close(s.fd); + s.free_fids.deinit(s.gpa); + s.iounits.deinit(s.gpa); + s.gpa.free(s.in_buf); + s.gpa.free(s.out_buf); + s.* = undefined; + } + + pub fn attach(s: *Session, fid: u32, uname: []const u8, aname: []const u8) Error!cloud9.Qid { + const r = try s.rpc(.{ .attach = .{ .fid = fid, .uname = uname, .aname = aname } }); + return r.attach; + } + + /// Fid 0 is never handed out: it belongs to the root attach. + pub fn allocFid(s: *Session) u32 { + if (s.free_fids.pop()) |fid| return fid; + const fid = s.next_fid; + s.next_fid += 1; + return fid; + } + + /// Fids currently bound (excluding fid 0); a debugging aid for leak hunting. + pub fn fidsInUse(s: *const Session) usize { + return (s.next_fid - 1) - s.free_fids.items.len; + } + + pub fn freeFid(s: *Session, fid: u32) void { + _ = s.iounits.remove(fid); + // If the free list cannot grow the fid is simply leaked; the counter keeps going. + s.free_fids.append(s.gpa, fid) catch {}; + } + + /// Generic RPC. Result slices borrow the input buffer until the next call. + pub fn rpc(s: *Session, req: cloud9.Client.Request) Error!cloud9.Client.Result { + s.ename_len = 0; + _ = s.client.submit(req) catch |e| switch (e) { + error.NoTags, error.Handshake, error.Dead => return error.Protocol, + error.NoSpace, error.TooLarge => return error.TooLarge, + error.BadRequest => { + s.setEname("bad request"); + return error.Nine; + }, + }; + try s.flush(); + var tmp: [64 * 1024]u8 = undefined; + while (true) { + if (s.client.take()) |done| { + switch (done.result) { + .fail => |ename| { + s.setEname(ename); + return error.Nine; + }, + else => return done.result, + } + } + if (s.client.dead) return error.Protocol; + // After take() returned null the previous frame is gone, so the free + // space is at least what the pending frame still needs. + const room = s.client.in.len - s.client.in_len; + if (room == 0) return error.Protocol; + const n = try readSome(s.fd, s.stop_fd, tmp[0..@min(room, tmp.len)]); + if (n == 0) return error.Closed; + const pushed = s.client.push(tmp[0..n]); + if (pushed != n) return error.Protocol; + } + } + + /// Walk `names` from `fid` to `newfid`. A partial walk leaves `newfid` unbound + /// (9P semantics) and reports `error.Nine` with ename "file does not exist". + pub fn walk(s: *Session, fid: u32, newfid: u32, names: []const []const u8) Error!Walk { + const r = try s.rpc(.{ .walk = .{ .fid = fid, .newfid = newfid, .names = names } }); + if (r.walk.nwqid < names.len) { + s.setEname("file does not exist"); + return error.Nine; + } + return .{ .nwqid = r.walk.nwqid, .wqid = r.walk.wqid }; + } + + /// allocFid + zero-element walk. The fid is released again on failure. + pub fn clone(s: *Session, fid: u32) Error!u32 { + const newfid = s.allocFid(); + errdefer s.freeFid(newfid); + _ = try s.walk(fid, newfid, &.{}); + return newfid; + } + + pub fn open(s: *Session, fid: u32, mode: u8) Error!Open { + const r = try s.rpc(.{ .open = .{ .fid = fid, .mode = mode } }); + s.noteIounit(fid, r.open.iounit); + return .{ .qid = r.open.qid, .iounit = r.open.iounit }; + } + + pub fn create(s: *Session, fid: u32, name: []const u8, perm: u32, mode: u8) Error!Open { + const r = try s.rpc(.{ .create = .{ .fid = fid, .name = name, .perm = perm, .mode = mode } }); + s.noteIounit(fid, r.create.iounit); + return .{ .qid = r.create.qid, .iounit = r.create.iounit }; + } + + /// Reads into `buf`, chunking by min(maxRead, iounit) and stopping at the first + /// short read. Returns the number of bytes read (0 at end of file). + pub fn read(s: *Session, fid: u32, offset: u64, buf: []u8) Error!usize { + return readWith(s, rpc, fid, offset, buf, s.chunk(fid)); + } + + /// Writes `data`, chunking like `read` and stopping at the first short write. + pub fn write(s: *Session, fid: u32, offset: u64, data: []const u8) Error!usize { + return writeWith(s, rpc, fid, offset, data, s.chunkWrite(fid)); + } + + /// The returned Stat's strings (name/uid/gid/muid) borrow the session's input + /// buffer: they are valid only until the next rpc. Copy what must outlive it. + pub fn stat(s: *Session, fid: u32) Error!cloud9.Stat { + const r = try s.rpc(.{ .stat = .{ .fid = fid } }); + return r.stat; + } + + pub fn wstat(s: *Session, fid: u32, st: cloud9.Stat) Error!void { + _ = try s.rpc(.{ .wstat = .{ .fid = fid, .stat = st } }); + } + + /// Frees the fid locally even when the server reports an error. + pub fn clunk(s: *Session, fid: u32) Error!void { + defer s.freeFid(fid); + _ = try s.rpc(.{ .clunk = .{ .fid = fid } }); + } + + /// Frees the fid locally even when the server reports an error. + pub fn remove(s: *Session, fid: u32) Error!void { + defer s.freeFid(fid); + _ = try s.rpc(.{ .remove = .{ .fid = fid } }); + } + + /// Maps the last Rerror text to an errno (case-insensitive substring match). + pub fn errno(s: *const Session) linux.E { + return enameToErrno(s.ename[0..s.ename_len]); + } + + // -- internals -------------------------------------------------------------- + + fn setEname(s: *Session, text: []const u8) void { + const n = @min(text.len, 255); + @memcpy(s.ename[0..n], text[0..n]); + s.ename_len = n; + } + + fn noteIounit(s: *Session, fid: u32, iounit: u32) void { + if (iounit == 0) { + _ = s.iounits.remove(fid); + } else { + s.iounits.put(s.gpa, fid, iounit) catch {}; + } + } + + fn chunk(s: *Session, fid: u32) u32 { + return chunkSize(s.client.maxRead(), s.iounits.get(fid) orelse 0); + } + + fn chunkWrite(s: *Session, fid: u32) u32 { + return chunkSize(s.client.maxWrite(), s.iounits.get(fid) orelse 0); + } + + /// Writes everything in the client's output buffer to the socket. + fn flush(s: *Session) Error!void { + while (s.client.output().len != 0) { + const out = s.client.output(); + const rc = linux.write(s.fd, out.ptr, out.len); + switch (linux.errno(rc)) { + .SUCCESS => { + if (rc == 0) return error.Closed; + s.client.wrote(rc); + }, + .INTR, .AGAIN => continue, + .PIPE, .CONNRESET => return error.Closed, + else => return error.Io, + } + } + } +}; + +fn chunkSize(max: u32, iounit: u32) u32 { + if (iounit != 0 and iounit < max) return iounit; + return max; +} + +/// Chunked read over any rpc-shaped function (injected so the loop is testable). +fn readWith( + s: anytype, + comptime rpcFn: anytype, + fid: u32, + offset: u64, + buf: []u8, + max_chunk: u32, +) Session.Error!usize { + if (max_chunk == 0) return error.Protocol; + var done: usize = 0; + while (done < buf.len) { + const want: u32 = @intCast(@min(buf.len - done, max_chunk)); + const r = try rpcFn(s, .{ .read = .{ .fid = fid, .offset = offset + done, .count = want } }); + const data = r.read; + @memcpy(buf[done..][0..data.len], data); + done += data.len; + if (data.len < want) break; + } + return done; +} + +/// Chunked write over any rpc-shaped function. +fn writeWith( + s: anytype, + comptime rpcFn: anytype, + fid: u32, + offset: u64, + data: []const u8, + max_chunk: u32, +) Session.Error!usize { + if (max_chunk == 0) return error.Protocol; + var done: usize = 0; + while (done < data.len) { + const want: usize = @min(data.len - done, max_chunk); + const r = try rpcFn(s, .{ .write = .{ .fid = fid, .offset = offset + done, .data = data[done..][0..want] } }); + done += r.write; + if (r.write < want) break; + } + return done; +} + +fn readSome(fd: i32, stop_fd: i32, buf: []u8) Session.Error!usize { + while (true) { + if (stop_fd >= 0) { + var pfds = [_]linux.pollfd{ + .{ .fd = fd, .events = linux.POLL.IN, .revents = 0 }, + .{ .fd = stop_fd, .events = linux.POLL.IN, .revents = 0 }, + }; + const prc = linux.poll(&pfds, pfds.len, -1); + switch (linux.errno(prc)) { + .SUCCESS => {}, + .INTR, .AGAIN => continue, + else => return error.Io, + } + if (pfds[1].revents != 0 and pfds[0].revents == 0) return error.Stopped; + } + const rc = linux.read(fd, buf.ptr, buf.len); + switch (linux.errno(rc)) { + .SUCCESS => return rc, + .INTR, .AGAIN => continue, + .CONNRESET => return error.Closed, + else => return error.Io, + } + } +} + +/// Rerror text → errno, per docs/DESIGN.md (first match wins). +pub fn enameToErrno(ename: []const u8) linux.E { + const Rule = struct { needle: []const u8, err: linux.E }; + const rules = [_]Rule{ + .{ .needle = "not exist", .err = .NOENT }, + .{ .needle = "not found", .err = .NOENT }, + .{ .needle = "no such", .err = .NOENT }, + .{ .needle = "exists", .err = .EXIST }, + .{ .needle = "not empty", .err = .NOTEMPTY }, + .{ .needle = "not a dir", .err = .NOTDIR }, + .{ .needle = "is a dir", .err = .ISDIR }, + .{ .needle = "permission", .err = .ACCES }, + .{ .needle = "denied", .err = .ACCES }, + .{ .needle = "read-only", .err = .ROFS }, + .{ .needle = "read only", .err = .ROFS }, + .{ .needle = "readonly", .err = .ROFS }, + .{ .needle = "no space", .err = .NOSPC }, + .{ .needle = "not allowed", .err = .PERM }, + .{ .needle = "not permitted", .err = .PERM }, + .{ .needle = "cannot", .err = .PERM }, + .{ .needle = "fid", .err = .BADF }, + .{ .needle = "bad offset", .err = .INVAL }, + .{ .needle = "invalid", .err = .INVAL }, + .{ .needle = "bad ", .err = .INVAL }, + .{ .needle = "busy", .err = .BUSY }, + .{ .needle = "in use", .err = .BUSY }, + .{ .needle = "too long", .err = .NAMETOOLONG }, + .{ .needle = "not supported", .err = .OPNOTSUPP }, + .{ .needle = "unsupported", .err = .OPNOTSUPP }, + }; + for (rules) |rule| { + if (std.ascii.findIgnoreCase(ename, rule.needle) != null) return rule.err; + } + return .IO; +} + +// -- transport ------------------------------------------------------------------ + +fn openTransport(address: Address) !i32 { + switch (address) { + .fd => |fd| return fd, + .unix => |path| { + if (path.len == 0 or path.len >= 108) return error.NameTooLong; + var sa: linux.sockaddr.un = .{ .path = @splat(0) }; + @memcpy(sa.path[0..path.len], path); + const fd = try newSocket(linux.AF.UNIX, 0); + errdefer _ = linux.close(fd); + try doConnect(fd, @ptrCast(&sa), @sizeOf(linux.sockaddr.un)); + return fd; + }, + .tcp => |t| { + const ip = std.Io.net.IpAddress.parse(t.host, t.port) catch return error.InvalidAddress; + switch (ip) { + .ip4 => |a| { + const sa: linux.sockaddr.in = .{ + .port = std.mem.nativeToBig(u16, t.port), + .addr = @bitCast(a.bytes), + }; + const fd = try newSocket(linux.AF.INET, linux.IPPROTO.TCP); + errdefer _ = linux.close(fd); + setNodelay(fd); + try doConnect(fd, @ptrCast(&sa), @sizeOf(linux.sockaddr.in)); + return fd; + }, + .ip6 => |a| { + const sa: linux.sockaddr.in6 = .{ + .port = std.mem.nativeToBig(u16, t.port), + .flowinfo = 0, + .addr = a.bytes, + .scope_id = 0, + }; + const fd = try newSocket(linux.AF.INET6, linux.IPPROTO.TCP); + errdefer _ = linux.close(fd); + setNodelay(fd); + try doConnect(fd, @ptrCast(&sa), @sizeOf(linux.sockaddr.in6)); + return fd; + }, + } + }, + } +} + +fn newSocket(domain: u32, protocol: u32) !i32 { + const rc = linux.socket(domain, linux.SOCK.STREAM | linux.SOCK.CLOEXEC, protocol); + switch (linux.errno(rc)) { + .SUCCESS => return @intCast(rc), + .MFILE, .NFILE => return error.ProcessFdQuotaExceeded, + .AFNOSUPPORT, .PROTONOSUPPORT => return error.AddressFamilyNotSupported, + .ACCES => return error.AccessDenied, + .NOMEM, .NOBUFS => return error.SystemResources, + else => return error.Unexpected, + } +} + +fn setNodelay(fd: i32) void { + const one: u32 = 1; + _ = linux.setsockopt(fd, linux.IPPROTO.TCP, linux.TCP.NODELAY, @ptrCast(&one), @sizeOf(u32)); +} + +fn doConnect(fd: i32, addr: *const linux.sockaddr, len: linux.socklen_t) !void { + while (true) { + const rc = linux.connect(fd, addr, len); + switch (linux.errno(rc)) { + .SUCCESS => return, + .INTR => continue, + .CONNREFUSED => return error.ConnectionRefused, + .NOENT, .NOTDIR => return error.FileNotFound, + .ACCES, .PERM => return error.AccessDenied, + .TIMEDOUT => return error.ConnectionTimedOut, + .NETUNREACH, .HOSTUNREACH => return error.NetworkUnreachable, + .ADDRNOTAVAIL => return error.AddressNotAvailable, + .AGAIN, .INPROGRESS => return error.WouldBlock, + else => return error.Unexpected, + } + } +} + +// -- tests ---------------------------------------------------------------------- + +const testing = std.testing; + +test { + testing.refAllDecls(@This()); +} + +test "ename → errno mapping" { + try testing.expectEqual(linux.E.NOENT, enameToErrno("file does not exist")); + try testing.expectEqual(linux.E.NOENT, enameToErrno("No Such File")); + try testing.expectEqual(linux.E.NOENT, enameToErrno("directory entry not found")); + try testing.expectEqual(linux.E.EXIST, enameToErrno("file already exists")); + try testing.expectEqual(linux.E.NOTEMPTY, enameToErrno("directory not empty")); + try testing.expectEqual(linux.E.NOTDIR, enameToErrno("not a directory")); + try testing.expectEqual(linux.E.ISDIR, enameToErrno("is a directory")); + try testing.expectEqual(linux.E.ACCES, enameToErrno("permission denied")); + try testing.expectEqual(linux.E.ACCES, enameToErrno("access denied")); + try testing.expectEqual(linux.E.ROFS, enameToErrno("read-only file system")); + try testing.expectEqual(linux.E.NOSPC, enameToErrno("no space left")); + try testing.expectEqual(linux.E.PERM, enameToErrno("operation not permitted")); + try testing.expectEqual(linux.E.PERM, enameToErrno("cannot remove root")); + try testing.expectEqual(linux.E.BADF, enameToErrno("unknown fid")); + try testing.expectEqual(linux.E.BADF, enameToErrno("fid in use")); // "fid" precedes "in use" + try testing.expectEqual(linux.E.INVAL, enameToErrno("bad offset")); + try testing.expectEqual(linux.E.INVAL, enameToErrno("invalid argument")); + try testing.expectEqual(linux.E.INVAL, enameToErrno("bad request")); + try testing.expectEqual(linux.E.BUSY, enameToErrno("device busy")); + try testing.expectEqual(linux.E.NAMETOOLONG, enameToErrno("name too long")); + try testing.expectEqual(linux.E.OPNOTSUPP, enameToErrno("operation not supported")); + try testing.expectEqual(linux.E.IO, enameToErrno("something odd happened")); + try testing.expectEqual(linux.E.IO, enameToErrno("")); +} + +test "fid allocator recycles and never hands out 0" { + var s: Session = undefined; + s.gpa = testing.allocator; + s.next_fid = 1; + s.free_fids = .empty; + s.iounits = .empty; + defer s.free_fids.deinit(s.gpa); + defer s.iounits.deinit(s.gpa); + + const a = s.allocFid(); + const b = s.allocFid(); + const c = s.allocFid(); + try testing.expectEqual(@as(u32, 1), a); + try testing.expectEqual(@as(u32, 2), b); + try testing.expectEqual(@as(u32, 3), c); + s.freeFid(b); + try testing.expectEqual(b, s.allocFid()); + s.freeFid(a); + s.freeFid(c); + const x = s.allocFid(); + const y = s.allocFid(); + try testing.expect((x == a and y == c) or (x == c and y == a)); + try testing.expectEqual(@as(u32, 4), s.allocFid()); + try testing.expect(a != 0 and b != 0 and c != 0); +} + +test "chunkSize honours iounit only when smaller" { + try testing.expectEqual(@as(u32, 100), chunkSize(100, 0)); + try testing.expectEqual(@as(u32, 40), chunkSize(100, 40)); + try testing.expectEqual(@as(u32, 100), chunkSize(100, 400)); +} + +/// Fake rpc for the chunked read/write loops: a file of `len` bytes where byte i == i & 0xff. +const FakeFile = struct { + len: usize, + calls: usize = 0, + max_count: u32 = 0, + short_write_at: ?usize = null, + scratch: [4096]u8 = undefined, + + fn rpc(f: *FakeFile, req: cloud9.Client.Request) Session.Error!cloud9.Client.Result { + f.calls += 1; + switch (req) { + .read => |r| { + f.max_count = @max(f.max_count, r.count); + if (r.offset >= f.len) return .{ .read = "" }; + const n: usize = @min(@as(usize, r.count), f.len - @as(usize, @intCast(r.offset))); + for (f.scratch[0..n], 0..) |*b, i| b.* = @truncate(r.offset + i); + return .{ .read = f.scratch[0..n] }; + }, + .write => |w| { + f.max_count = @max(f.max_count, @as(u32, @intCast(w.data.len))); + if (f.short_write_at) |at| { + if (w.offset + w.data.len > at) { + const n: usize = if (w.offset >= at) 0 else @intCast(at - w.offset); + return .{ .write = @intCast(n) }; + } + } + return .{ .write = @intCast(w.data.len) }; + }, + else => unreachable, + } + } +}; + +test "read chunks by max_chunk and stops at a short read" { + var f: FakeFile = .{ .len = 2500 }; + var buf: [4000]u8 = undefined; + const n = try readWith(&f, FakeFile.rpc, 7, 0, &buf, 1000); + try testing.expectEqual(@as(usize, 2500), n); + try testing.expectEqual(@as(usize, 3), f.calls); // 1000, 1000, 500 (short → stop) + try testing.expectEqual(@as(u32, 1000), f.max_count); + for (buf[0..n], 0..) |b, i| try testing.expectEqual(@as(u8, @truncate(i)), b); + + // Reading exactly up to a chunk boundary uses one call per chunk and no more. + f = .{ .len = 2000 }; + try testing.expectEqual(@as(usize, 2000), try readWith(&f, FakeFile.rpc, 7, 0, buf[0..2000], 1000)); + try testing.expectEqual(@as(usize, 2), f.calls); + + // Offset past EOF → 0. + f = .{ .len = 10 }; + try testing.expectEqual(@as(usize, 0), try readWith(&f, FakeFile.rpc, 7, 50, &buf, 1000)); +} + +test "write chunks and stops at a short write" { + var f: FakeFile = .{ .len = 0 }; + var data: [2500]u8 = undefined; + for (&data, 0..) |*b, i| b.* = @truncate(i); + try testing.expectEqual(@as(usize, 2500), try writeWith(&f, FakeFile.rpc, 7, 0, &data, 1000)); + try testing.expectEqual(@as(usize, 3), f.calls); + try testing.expectEqual(@as(u32, 1000), f.max_count); + + f = .{ .len = 0, .short_write_at = 1500 }; + try testing.expectEqual(@as(usize, 1500), try writeWith(&f, FakeFile.rpc, 7, 0, &data, 1000)); + try testing.expectEqual(@as(usize, 2), f.calls); +} + +// -- in-process server test --------------------------------------------------------- + +/// A tiny 9P2000 backend on a cloud9.Server: answers version/attach/walk/stat/open/ +/// read/clunk/remove with canned data. Runs in its own thread over a socketpair. +const FakeServer = struct { + fd: i32, + msize: u32, + max_read_count: u32 = 0, + file_len: usize, + + const file_qid: cloud9.Qid = .{ .type = 0, .version = 3, .path = 0x1234 }; + const dir_qid: cloud9.Qid = .{ .type = cloud9.qtdir, .version = 1, .path = 0x1 }; + + fn run(fs: *FakeServer) void { + fs.loop() catch |e| std.debug.print("fake server: {s}\n", .{@errorName(e)}); + _ = linux.close(fs.fd); + } + + fn loop(fs: *FakeServer) !void { + const gpa = testing.allocator; + const in = try gpa.alloc(u8, fs.msize); + defer gpa.free(in); + const out = try gpa.alloc(u8, fs.msize * 2); + defer gpa.free(out); + var srv: cloud9.Server = .init(.{ .in = in, .out = out }); + var tmp: [4096]u8 = undefined; + var data: [8192]u8 = undefined; + while (true) { + while (try srv.receive()) |req| { + const tag = req.tag; + switch (req.msg) { + .tversion => |m| try srv.negotiate(m.msize, m.version), + .tattach => try srv.reply(tag, .{ .rattach = .{ .qid = dir_qid } }), + .twalk => |m| { + var wq: [cloud9.max_welem]cloud9.Qid = @splat(dir_qid); + var n: u16 = 0; + for (m.wname[0..m.nwname]) |name| { + if (std.mem.eql(u8, name, "file")) { + wq[n] = file_qid; + } else if (std.mem.eql(u8, name, "dir")) { + wq[n] = dir_qid; + } else break; + n += 1; + } + if (n == 0 and m.nwname != 0) { + try srv.reply(tag, .{ .rerror = .{ .ename = "file does not exist" } }); + } else { + try srv.reply(tag, .{ .rwalk = .{ .nwqid = n, .wqid = wq } }); + } + }, + .tstat => try srv.reply(tag, .{ .rstat = .{ .stat = .{ + .type = 0, + .dev = 0, + .qid = file_qid, + .mode = 0o644, + .atime = 1, + .mtime = 2, + .length = fs.file_len, + .name = "file", + .uid = "u", + .gid = "g", + .muid = "u", + } } }), + .topen => |m| try srv.reply(tag, .{ .ropen = .{ .qid = file_qid, .iounit = if (m.mode == cloud9.owrite) 700 else 0 } }), + .tread => |m| { + fs.max_read_count = @max(fs.max_read_count, m.count); + var n: usize = 0; + if (m.offset < fs.file_len) n = @min(@as(usize, m.count), fs.file_len - @as(usize, @intCast(m.offset))); + n = @min(n, data.len); + for (data[0..n], 0..) |*b, i| b.* = @truncate(m.offset + i); + try srv.reply(tag, .{ .rread = .{ .data = data[0..n] } }); + }, + .twrite => |m| try srv.reply(tag, .{ .rwrite = .{ .count = @intCast(m.data.len) } }), + .tclunk => try srv.reply(tag, .rclunk), + .tremove => try srv.reply(tag, .{ .rerror = .{ .ename = "permission denied" } }), + .twstat => try srv.reply(tag, .rwstat), + // A flush is the test's "hang up now" signal. + .tflush => return, + else => try srv.reply(tag, .{ .rerror = .{ .ename = "not supported" } }), + } + srv.release(); + } + while (srv.output().len != 0) { + const o = srv.output(); + const rc = linux.write(fs.fd, o.ptr, o.len); + if (linux.errno(rc) != .SUCCESS) return error.Write; + srv.wrote(rc); + } + const rc = linux.read(fs.fd, &tmp, tmp.len); + if (linux.errno(rc) != .SUCCESS) return error.Read; + if (rc == 0) return; + if (srv.push(tmp[0..rc]) != rc) return error.Overflow; + } + } +}; + +test "session against an in-process cloud9.Server" { + var fds: [2]i32 = undefined; + try testing.expectEqual(linux.E.SUCCESS, linux.errno(linux.socketpair(linux.AF.UNIX, linux.SOCK.STREAM | linux.SOCK.CLOEXEC, 0, &fds))); + + var fs: FakeServer = .{ .fd = fds[1], .msize = 8192, .file_len = 20_000 }; + const th = try std.Thread.spawn(.{}, FakeServer.run, .{&fs}); + + var s = try Session.connect(testing.allocator, .{ .fd = fds[0] }, 8192); + defer { + s.deinit(); + th.join(); + } + try testing.expectEqual(@as(u32, 8192), s.msize); + + const root = try s.attach(0, "me", ""); + try testing.expectEqual(FakeServer.dir_qid.path, root.path); + + // Plain rpc + stat borrowing the input buffer. + const fid = s.allocFid(); + const w = try s.walk(0, fid, &.{"file"}); + try testing.expectEqual(@as(u16, 1), w.nwqid); + try testing.expectEqual(FakeServer.file_qid.path, w.wqid[0].path); + const st = try s.stat(fid); + try testing.expectEqualStrings("file", st.name); + try testing.expectEqual(@as(u64, 20_000), st.length); + + // Chunked read: 20000 bytes at maxRead = msize - 11 = 8181 per chunk. + _ = try s.open(fid, cloud9.oread); + const buf = try testing.allocator.alloc(u8, 30_000); + defer testing.allocator.free(buf); + const n = try s.read(fid, 0, buf); + try testing.expectEqual(@as(usize, 20_000), n); + for (buf[0..n], 0..) |b, i| try testing.expectEqual(@as(u8, @truncate(i)), b); + try testing.expectEqual(@as(u32, 8181), fs.max_read_count); + try testing.expectEqual(@as(usize, 0), try s.read(fid, 20_000, buf)); + + // iounit from open bounds the chunk. + const wfid = try s.clone(fid); + _ = try s.open(wfid, cloud9.owrite); + fs.max_read_count = 0; + _ = try s.read(wfid, 0, buf[0..3000]); + try testing.expectEqual(@as(u32, 700), fs.max_read_count); + try testing.expectEqual(@as(usize, 3000), try s.write(wfid, 0, buf[0..3000])); + + // Partial walk → error.Nine with a "not exist" ename → ENOENT. + const pfid = s.allocFid(); + try testing.expectError(error.Nine, s.walk(0, pfid, &.{ "dir", "nope" })); + try testing.expectEqual(linux.E.NOENT, s.errno()); + try testing.expectEqualStrings("file does not exist", s.ename[0..s.ename_len]); + s.freeFid(pfid); + + // Server Rerror → error.Nine, ename copied, fid freed by remove even on error. + try testing.expectError(error.Nine, s.remove(wfid)); + try testing.expectEqual(linux.E.ACCES, s.errno()); + try testing.expectEqual(wfid, s.allocFid()); // recycled + s.freeFid(wfid); + + // Unsupported op → "not supported" → ENOTSUP; a plain wstat succeeds. + try testing.expectError(error.Nine, s.rpc(.{ .auth = .{ .afid = 5, .uname = "me" } })); + try testing.expectEqual(linux.E.OPNOTSUPP, s.errno()); + try s.wstat(fid, dontcare); + try s.clunk(fid); + try testing.expectEqual(fid, s.allocFid()); + s.freeFid(fid); + + // A clone bound to a fid that then fails to walk must release the fid. + const before = s.next_fid; + const cfid = s.allocFid(); + s.freeFid(cfid); + try testing.expectError(error.Nine, s.walk(0, cfid, &.{"nope"})); + try testing.expectEqual(before, s.next_fid); + + // The server hanging up makes the pending rpc fail with error.Closed. + try testing.expectError(error.Closed, s.rpc(.{ .flush = .{ .oldtag = 0 } })); +} diff --git a/9ns/src/ns.zig b/9ns/src/ns.zig new file mode 100644 index 0000000..6da2c7f --- /dev/null +++ b/9ns/src/ns.zig @@ -0,0 +1,1082 @@ +//! Namespace and process plumbing for 9ns. +//! +//! Everything here is raw `std.os.linux` syscalls (no libc). The child side +//! of `spawn` runs between `fork` and `execve`; it does not allocate except +//! inside `ensureMountpoint` (the process is single-threaded by then, so the +//! inherited allocator is safe to use). +//! +//! Exit codes produced by the child before exec: 125 for namespace/mount +//! setup failures, 126 when the program was found but is not executable, +//! 127 when it was not found. + +const std = @import("std"); +const builtin = @import("builtin"); +const linux = std.os.linux; +const Allocator = std.mem.Allocator; +const E = linux.E; + +pub const Spawn = struct { + /// argv[0] is PATH-searched unless it contains '/'. + argv: []const []const u8, + /// Inherited environment; `NINE_MOUNT` is added or replaced. + envp: [*:null]const ?[*:0]const u8, + /// Absolute mountpoint (see `resolveMountpoint`). + mountpoint: []const u8, + uid: u32, + gid: u32, + max_read: u32, + /// When false the namespace is set up (including mountpoint shadowing) + /// but `/dev/fuse` is not opened and nothing is mounted; `Child.fuse_fd` + /// is then -1. Only for smoke tests. + mount_fuse: bool = true, +}; + +pub const Child = struct { + pid: i32, + /// The `/dev/fuse` connection backing the mount, opened by the child + /// inside its user namespace (the kernel refuses to mount a fuse fd that + /// was opened from another user namespace) and handed back over the + /// status socket with SCM_RIGHTS. Owned by the caller; CLOEXEC. + fuse_fd: i32, + /// Parent end of the status socket. The child reports an exec failure + /// on it (see `reportExecFailure`); it reads EOF once exec succeeded. + status_fd: i32, +}; + +/// Exit status used by the child for setup failures (matches 9ns's own). +pub const setup_failure_status: u8 = 125; +/// Refuse to shadow a directory with more entries than this. +pub const max_shadow_entries: usize = 4096; + +const default_path = "/usr/local/bin:/bin:/usr/bin"; +const path_max = 4096; + +// --------------------------------------------------------------------------- +// Mountpoint resolution +// --------------------------------------------------------------------------- + +/// Absolute path (relative paths resolved against cwd), duplicate slashes +/// collapsed, `.` and `..` components resolved lexically, no trailing slash. +/// `/` itself is rejected. +pub fn resolveMountpoint(gpa: Allocator, path: []const u8) ![:0]u8 { + var cwd_buf: [path_max]u8 = undefined; + var cwd: []const u8 = "/"; + if (path.len == 0 or path[0] != '/') { + const rc = linux.getcwd(&cwd_buf, cwd_buf.len); + switch (linux.errno(rc)) { + .SUCCESS => {}, + else => |e| { + std.debug.print("9ns: getcwd: E{t}\n", .{e}); + return error.Cwd; + }, + } + // rc counts the terminating NUL. + cwd = cwd_buf[0 .. rc - 1]; + } + return normalizePath(gpa, cwd, path); +} + +/// Pure part of `resolveMountpoint`: `cwd` is only used when `path` is relative. +fn normalizePath(gpa: Allocator, cwd: []const u8, path: []const u8) ![:0]u8 { + if (path.len == 0) return error.InvalidMountpoint; + var out: std.ArrayList(u8) = .empty; + defer out.deinit(gpa); + if (path[0] != '/') try appendComponents(gpa, &out, cwd); + try appendComponents(gpa, &out, path); + if (out.items.len == 0) return error.InvalidMountpoint; // "/" or equivalent + return out.toOwnedSliceSentinel(gpa, 0); +} + +fn appendComponents(gpa: Allocator, out: *std.ArrayList(u8), path: []const u8) !void { + var it = std.mem.tokenizeScalar(u8, path, '/'); + while (it.next()) |comp| { + if (std.mem.eql(u8, comp, ".")) continue; + if (std.mem.eql(u8, comp, "..")) { + // Pop the last component (lexically; "/.." stays "/"). + const idx = std.mem.lastIndexOfScalar(u8, out.items, '/') orelse 0; + out.shrinkRetainingCapacity(idx); + continue; + } + try out.append(gpa, '/'); + try out.appendSlice(gpa, comp); + } +} + +// --------------------------------------------------------------------------- +// Environment helpers +// --------------------------------------------------------------------------- + +/// Look a variable up in a raw envp block. +pub fn getenv(envp: [*:null]const ?[*:0]const u8, name: []const u8) ?[]const u8 { + var i: usize = 0; + while (envp[i]) |entry| : (i += 1) { + const kv = std.mem.span(entry); + if (kv.len > name.len and kv[name.len] == '=' and std.mem.eql(u8, kv[0..name.len], name)) { + return kv[name.len + 1 ..]; + } + } + return null; +} + +/// Every path `execve` should try for `name`, in order: just `name` if it +/// contains a '/', else `/` for each `$PATH` element (an empty +/// element means the current directory; `$PATH` unset falls back to +/// `/usr/local/bin:/bin:/usr/bin`). +pub fn pathCandidates(gpa: Allocator, envp: [*:null]const ?[*:0]const u8, name: []const u8) ![]const [:0]const u8 { + if (name.len == 0) return error.EmptyProgramName; + var list: std.ArrayList([:0]const u8) = .empty; + errdefer { + for (list.items) |c| gpa.free(c); + list.deinit(gpa); + } + if (std.mem.indexOfScalar(u8, name, '/') != null) { + try list.append(gpa, try gpa.dupeZ(u8, name)); + return list.toOwnedSlice(gpa); + } + const path = getenv(envp, "PATH") orelse default_path; + var it = std.mem.splitScalar(u8, path, ':'); + while (it.next()) |dir| { + const d = if (dir.len == 0) "." else dir; + try list.append(gpa, try std.fmt.allocPrintSentinel(gpa, "{s}/{s}", .{ d, name }, 0)); + } + return list.toOwnedSlice(gpa); +} + +/// First PATH candidate that is an executable regular file, or the name +/// itself when it contains a '/'. Provided for completeness; `spawn` simply +/// tries `execve` on every candidate instead. +pub fn findInPath(gpa: Allocator, envp: [*:null]const ?[*:0]const u8, name: []const u8) ![:0]u8 { + const cands = try pathCandidates(gpa, envp, name); + defer { + for (cands) |c| gpa.free(c); + gpa.free(cands); + } + for (cands) |c| { + var stx: linux.Statx = undefined; + const rc = linux.statx(linux.AT.FDCWD, c.ptr, 0, .{ .TYPE = true, .MODE = true }, &stx); + if (linux.errno(rc) != .SUCCESS) continue; + if (stx.mode & linux.S.IFMT != linux.S.IFREG) continue; + if (stx.mode & 0o111 == 0) continue; + return gpa.dupeZ(u8, c); + } + return error.FileNotFound; +} + +/// New envp block: every entry of `envp` except `NINE_MOUNT=...`, +/// followed by `NINE_MOUNT=`. +fn buildEnvp(gpa: Allocator, envp: [*:null]const ?[*:0]const u8, mountpoint: []const u8) ![:null]?[*:0]const u8 { + const key = "NINE_MOUNT="; + var keep: usize = 0; + var i: usize = 0; + while (envp[i]) |entry| : (i += 1) { + if (!std.mem.startsWith(u8, std.mem.span(entry), key)) keep += 1; + } + const out = try gpa.allocSentinel(?[*:0]const u8, keep + 1, null); + errdefer gpa.free(out); + var j: usize = 0; + i = 0; + while (envp[i]) |entry| : (i += 1) { + if (std.mem.startsWith(u8, std.mem.span(entry), key)) continue; + out[j] = entry; + j += 1; + } + const mount_entry = try std.fmt.allocPrintSentinel(gpa, key ++ "{s}", .{mountpoint}, 0); + out[j] = mount_entry.ptr; + return out; +} + +fn buildArgv(gpa: Allocator, argv: []const []const u8) ![:null]?[*:0]const u8 { + const out = try gpa.allocSentinel(?[*:0]const u8, argv.len, null); + for (argv, 0..) |a, i| out[i] = (try gpa.dupeZ(u8, a)).ptr; + return out; +} + +// --------------------------------------------------------------------------- +// Mountpoint policy +// --------------------------------------------------------------------------- + +/// Make sure `path` is a directory, inside the *current* mount namespace: +/// +/// * already a directory → done; +/// * else `mkdir`; on `EACCES`/`EPERM`/`EROFS` shadow the parent directory +/// with a tmpfs that re-exposes every existing entry (bind mounts for +/// directories and files, recreated symlinks) and `mkdir` inside it; +/// * anything else fails with the errno and a hint. +/// +/// Every failure prints `9ns: : E` to stderr before +/// returning. Meant to be called in the child of `spawn` (or from a +/// throwaway namespace: `unshare -Urm`). +pub fn ensureMountpoint(gpa: Allocator, path: [:0]const u8) !void { + if (fileType(linux.AT.FDCWD, path, false)) |ft| { + if (ft == .dir) return; + std.debug.print("9ns: mountpoint {s}: exists but is not a directory\n", .{path}); + return error.Mountpoint; + } + if (fileType(linux.AT.FDCWD, path, true) == .symlink) { + std.debug.print("9ns: mountpoint {s}: dangling symlink\n", .{path}); + return error.Mountpoint; + } + const mk = linux.errno(linux.mkdirat(linux.AT.FDCWD, path, 0o755)); + switch (mk) { + .SUCCESS => return, + .ACCES, .PERM, .ROFS => {}, + else => |e| { + std.debug.print("9ns: mkdir {s}: E{t} (pass --mount an existing directory)\n", .{ path, e }); + return error.Mountpoint; + }, + } + const parent = std.fs.path.dirname(path) orelse "/"; + if (std.mem.eql(u8, parent, "/") or isSameDirectory(parent, "/")) { + std.debug.print("9ns: mkdir {s}: E{t}; refusing to shadow / (pass --mount an existing directory)\n", .{ path, mk }); + return error.Mountpoint; + } + // The shadow rebuilds entries from /proc/self/fd//; a tmpfs + // over /proc (or a subtree of it) would take that away from itself. + if (std.mem.eql(u8, parent, "/proc") or std.mem.startsWith(u8, parent, "/proc/")) { + std.debug.print("9ns: mkdir {s}: E{t}; refusing to shadow {s} (pass --mount an existing directory)\n", .{ path, mk, parent }); + return error.Mountpoint; + } + const parent_z = try gpa.dupeZ(u8, parent); + defer gpa.free(parent_z); + try shadowDirectory(gpa, parent_z); + switch (linux.errno(linux.mkdirat(linux.AT.FDCWD, path, 0o755))) { + .SUCCESS => {}, + else => |e| { + std.debug.print("9ns: mkdir {s} (in shadow tmpfs): E{t}\n", .{ path, e }); + return error.Mountpoint; + }, + } +} + +const FileType = enum { dir, symlink, other }; + +/// True when both paths resolve (following symlinks, including magic ones +/// such as /proc/self/root) to the same inode. +fn isSameDirectory(a: []const u8, b: [*:0]const u8) bool { + var a_buf: [path_max]u8 = undefined; + const a_z = std.fmt.bufPrintZ(&a_buf, "{s}", .{a}) catch return false; + var sa: linux.Statx = undefined; + var sb: linux.Statx = undefined; + if (linux.errno(linux.statx(linux.AT.FDCWD, a_z, 0, .{ .INO = true }, &sa)) != .SUCCESS) return false; + if (linux.errno(linux.statx(linux.AT.FDCWD, b, 0, .{ .INO = true }, &sb)) != .SUCCESS) return false; + return sa.ino == sb.ino and sa.dev_major == sb.dev_major and sa.dev_minor == sb.dev_minor; +} + +fn fileType(dirfd: i32, name: [*:0]const u8, nofollow: bool) ?FileType { + var stx: linux.Statx = undefined; + const flags: u32 = if (nofollow) linux.AT.SYMLINK_NOFOLLOW else 0; + const rc = linux.statx(dirfd, name, flags, .{ .TYPE = true }, &stx); + if (linux.errno(rc) != .SUCCESS) return null; + return switch (stx.mode & linux.S.IFMT) { + linux.S.IFDIR => .dir, + linux.S.IFLNK => .symlink, + else => .other, + }; +} + +const Entry = struct { name: [:0]u8, kind: FileType }; + +/// Read every entry of the directory open at `fd` (excluding `.` and `..`). +fn listDir(gpa: Allocator, fd: i32, dirpath: []const u8) ![]Entry { + var list: std.ArrayList(Entry) = .empty; + errdefer { + for (list.items) |e| gpa.free(e.name); + list.deinit(gpa); + } + var buf: [32 * 1024]u8 align(@alignOf(linux.dirent64)) = undefined; + while (true) { + const rc = linux.getdents64(fd, &buf, buf.len); + switch (linux.errno(rc)) { + .SUCCESS => {}, + else => |e| { + std.debug.print("9ns: getdents64 {s}: E{t}\n", .{ dirpath, e }); + return error.Mountpoint; + }, + } + if (rc == 0) break; + var off: usize = 0; + while (off < rc) { + const d: *align(1) const linux.dirent64 = @ptrCast(&buf[off]); + const name_ptr: [*:0]const u8 = @ptrCast(&buf[off + @offsetOf(linux.dirent64, "name")]); + const name = std.mem.span(name_ptr); + const dtype = d.type; + off += d.reclen; + if (std.mem.eql(u8, name, ".") or std.mem.eql(u8, name, "..")) continue; + if (list.items.len >= max_shadow_entries) { + std.debug.print("9ns: refusing to shadow {s}: more than {d} entries\n", .{ dirpath, max_shadow_entries }); + return error.TooManyEntries; + } + const kind: FileType = switch (dtype) { + linux.DT.DIR => .dir, + linux.DT.LNK => .symlink, + linux.DT.UNKNOWN => fileType(fd, name_ptr, true) orelse .other, + else => .other, + }; + try list.append(gpa, .{ .name = try gpa.dupeZ(u8, name), .kind = kind }); + } + } + return list.toOwnedSlice(gpa); +} + +fn shadowDirectory(gpa: Allocator, parent: [:0]const u8) !void { + const open_rc = linux.open(parent, .{ .ACCMODE = .RDONLY, .DIRECTORY = true, .CLOEXEC = true }, 0); + switch (linux.errno(open_rc)) { + .SUCCESS => {}, + else => |e| { + std.debug.print("9ns: open {s}: E{t}\n", .{ parent, e }); + return error.Mountpoint; + }, + } + const pfd: i32 = @intCast(open_rc); + defer _ = linux.close(pfd); + + const entries = try listDir(gpa, pfd, parent); + defer { + for (entries) |e| gpa.free(e.name); + gpa.free(entries); + } + + const tmpfs_opts: [*:0]const u8 = "mode=755"; + switch (linux.errno(linux.mount("tmpfs", parent, "tmpfs", linux.MS.NOSUID | linux.MS.NODEV, @intFromPtr(tmpfs_opts)))) { + .SUCCESS => {}, + else => |e| { + std.debug.print("9ns: mount tmpfs on {s}: E{t}\n", .{ parent, e }); + return error.Mountpoint; + }, + } + + // `pfd` still refers to the original directory underneath the tmpfs, so + // `/proc/self/fd//` reaches the hidden entries. + var src_buf: [path_max]u8 = undefined; + var dst_buf: [path_max]u8 = undefined; + var link_buf: [path_max]u8 = undefined; + for (entries) |e| { + const src = std.fmt.bufPrintZ(&src_buf, "/proc/self/fd/{d}/{s}", .{ pfd, e.name }) catch { + std.debug.print("9ns: shadow {s}/{s}: name too long (skipped)\n", .{ parent, e.name }); + continue; + }; + const dst = std.fmt.bufPrintZ(&dst_buf, "{s}/{s}", .{ parent, e.name }) catch { + std.debug.print("9ns: shadow {s}/{s}: name too long (skipped)\n", .{ parent, e.name }); + continue; + }; + switch (e.kind) { + .dir => { + if (!check("mkdir", dst, linux.mkdirat(linux.AT.FDCWD, dst, 0o755))) continue; + _ = check("bind", dst, linux.mount(src, dst, null, linux.MS.BIND | linux.MS.REC, 0)); + }, + .symlink => { + const rc = linux.readlinkat(pfd, e.name, &link_buf, link_buf.len - 1); + if (!check("readlink", dst, rc)) continue; + link_buf[rc] = 0; + const target: [*:0]const u8 = @ptrCast(&link_buf); + _ = check("symlink", dst, linux.symlinkat(target, linux.AT.FDCWD, dst)); + }, + .other => { + const rc = linux.openat(linux.AT.FDCWD, dst, .{ .ACCMODE = .WRONLY, .CREAT = true, .CLOEXEC = true }, 0o644); + if (!check("create", dst, rc)) continue; + _ = linux.close(@intCast(rc)); + _ = check("bind", dst, linux.mount(src, dst, null, linux.MS.BIND | linux.MS.REC, 0)); + }, + } + } +} + +/// Report a failed per-entry step as a warning (the entry is skipped; the +/// rest of the shadow is still useful). Returns true on success. +fn check(step: []const u8, path: [*:0]const u8, rc: usize) bool { + switch (linux.errno(rc)) { + .SUCCESS => return true, + else => |e| { + std.debug.print("9ns: shadow: {s} {s}: E{t} (skipped)\n", .{ step, std.mem.span(path), e }); + return false; + }, + } +} + +// --------------------------------------------------------------------------- +// spawn +// --------------------------------------------------------------------------- + +const ChildArgs = struct { + gpa: Allocator, + status_sock: i32, + mountpoint: [:0]const u8, + fuse_opts_prefix: [:0]const u8, // everything after "fd=," + mount_fuse: bool, + uid_map: []const u8, + gid_map: []const u8, + argv: [:null]?[*:0]const u8, + envp: [:null]?[*:0]const u8, + candidates: []const [:0]const u8, + name: []const u8, +}; + +/// Status channel protocol (child → parent, over a CLOEXEC socketpair): +/// a 0 byte means "namespace and mount are up" and carries the fuse fd as +/// SCM_RIGHTS; a non-zero byte is an exit status followed by a message. +/// EOF ends the conversation (exec succeeded, or the child died). +const ok_byte: u8 = 0; + +/// fork; the child unshares user+mount namespaces, maps its uid/gid, +/// makes `/` private, ensures the mountpoint, opens `/dev/fuse`, mounts it +/// on the mountpoint and sends the fd back. `spawn` returns at that point +/// (with `error.ChildFailed` and a message on stderr if any step failed). +/// The child then stats the mountpoint, which makes the kernel fetch the +/// root's attributes once the parent serves (the kernel seeds the fuse root +/// with uid 0, unmapped in the new user namespace, so nothing could be +/// created in the root until then), sets `NINE_MOUNT` and execs +/// `argv`. An exec failure is reported on `Child.status_fd` and ends the +/// child with 126/127; collect it with `reportExecFailure` after +/// `bridge.serve` returns. +/// +/// If `installSignals` was called, the pid is stored into the registered +/// variable as soon as fork returns so no SIGCHLD can be missed. +pub fn spawn(gpa: Allocator, s: Spawn) !Child { + if (s.argv.len == 0 or s.argv[0].len == 0) { + std.debug.print("9ns: empty program name\n", .{}); + return error.EmptyProgramName; + } + + const mountpoint = try gpa.dupeZ(u8, s.mountpoint); + defer gpa.free(mountpoint); + const fuse_opts_prefix = try std.fmt.allocPrintSentinel(gpa, "rootmode=40000,user_id={d},group_id={d},max_read={d}", .{ s.uid, s.gid, s.max_read }, 0); + defer gpa.free(fuse_opts_prefix); + var uid_buf: [64]u8 = undefined; + var gid_buf: [64]u8 = undefined; + const uid_map = try std.fmt.bufPrint(&uid_buf, "{d} {d} 1\n", .{ s.uid, s.uid }); + const gid_map = try std.fmt.bufPrint(&gid_buf, "{d} {d} 1\n", .{ s.gid, s.gid }); + const argv = try buildArgv(gpa, s.argv); + defer { + for (argv) |a| gpa.free(std.mem.span(a.?)); + gpa.free(argv); + } + const envp = try buildEnvp(gpa, s.envp, s.mountpoint); + defer { + gpa.free(std.mem.span(envp[envp.len - 1].?)); // the NINE_MOUNT entry we created + gpa.free(envp); + } + const candidates = try pathCandidates(gpa, s.envp, s.argv[0]); + defer { + for (candidates) |c| gpa.free(c); + gpa.free(candidates); + } + + var sv: [2]i32 = undefined; + switch (linux.errno(linux.socketpair(linux.AF.UNIX, linux.SOCK.STREAM | linux.SOCK.CLOEXEC, 0, &sv))) { + .SUCCESS => {}, + else => |e| { + std.debug.print("9ns: socketpair: E{t}\n", .{e}); + return error.SystemResources; + }, + } + + const child_args = ChildArgs{ + .gpa = gpa, + .status_sock = sv[1], + .mountpoint = mountpoint, + .fuse_opts_prefix = fuse_opts_prefix, + .mount_fuse = s.mount_fuse, + .uid_map = uid_map, + .gid_map = gid_map, + .argv = argv, + .envp = envp, + .candidates = candidates, + .name = s.argv[0], + }; + + const fork_rc = linux.fork(); + switch (linux.errno(fork_rc)) { + .SUCCESS => {}, + else => |e| { + _ = linux.close(sv[0]); + _ = linux.close(sv[1]); + std.debug.print("9ns: fork: E{t}\n", .{e}); + return error.SystemResources; + }, + } + if (fork_rc == 0) childMain(&child_args); + + const pid: i32 = @intCast(fork_rc); + if (child_pid_ptr) |p| @atomicStore(i32, p, pid, .seq_cst); + _ = linux.close(sv[1]); + + // First byte: ok (with the fuse fd attached) or a failure status. + var first: [1]u8 = .{ok_byte}; // defined even if recvmsg stores nothing + var fuse_fd: i32 = -1; + var n: usize = 0; + while (true) { + const rc = recvWithFd(sv[0], &first, &fuse_fd, 0); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .INTR => continue, + else => break, + } + n = rc; + break; + } + if (n == 1 and first[0] == ok_byte and (fuse_fd >= 0 or !s.mount_fuse)) { + return .{ .pid = pid, .fuse_fd = fuse_fd, .status_fd = sv[0] }; + } + + // Failure. A status byte means the child is exiting on its own and a + // message follows. Anything else (EOF: the child died before reporting; + // an ok byte without the fd: the SCM_RIGHTS transfer was truncated, e.g. + // EMFILE) is a protocol violation: the child may be about to exec with a + // dead mount, so kill it before waiting rather than reading the status + // socket until an exec'd program eventually exits. + const reported = n == 1 and first[0] != ok_byte; + if (!reported) _ = linux.kill(pid, .KILL); + if (fuse_fd >= 0) _ = linux.close(fuse_fd); + var msg: [512]u8 = undefined; + var len: usize = 0; + while (reported and len < msg.len) { + const rc = linux.read(sv[0], msg[len..].ptr, msg.len - len); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .INTR => continue, + else => break, + } + if (rc == 0) break; + len += rc; + } + _ = linux.close(sv[0]); + if (reported) { + std.debug.print("9ns: {s}\n", .{msg[0..len]}); + } else if (n == 1) { + std.debug.print("9ns: child handshake failed: no fuse fd received (out of file descriptors?)\n", .{}); + } else { + std.debug.print("9ns: child exited before reporting\n", .{}); + } + _ = waitChild(pid) catch {}; + if (child_pid_ptr) |p| @atomicStore(i32, p, 0, .seq_cst); + const status: u8 = if (reported) first[0] else setup_failure_status; + return switch (status) { + 126 => error.ExecPermission, + 127 => error.ExecNotFound, + else => error.ChildFailed, + }; +} + +/// After the child is gone (or the mount is dead): print the exec failure +/// the child reported on `status_fd`, if any, and close it. Returns the +/// status byte the child announced, or null when exec succeeded / nothing +/// was reported. Never blocks. +pub fn reportExecFailure(child: Child) ?u8 { + defer _ = linux.close(child.status_fd); + var msg: [512]u8 = undefined; + var len: usize = 0; + while (len < msg.len) { + var iov = [_]std.posix.iovec{.{ .base = msg[len..].ptr, .len = msg.len - len }}; + var hdr = linux.msghdr{ + .name = null, + .namelen = 0, + .iov = &iov, + .iovlen = 1, + .control = null, + .controllen = 0, + .flags = 0, + }; + const rc = linux.recvmsg(child.status_fd, &hdr, linux.MSG.DONTWAIT); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .INTR => continue, + else => break, + } + if (rc == 0) break; + len += rc; + } + if (len == 0) return null; + std.debug.print("9ns: {s}\n", .{msg[1..len]}); + return msg[0]; +} + +const cmsg_fd_len = @sizeOf(linux.cmsghdr) + @sizeOf(i32); +const cmsg_fd_space = std.mem.alignForward(usize, cmsg_fd_len, @sizeOf(usize)); + +/// sendmsg one data byte, optionally with `fd` attached as SCM_RIGHTS. +fn sendWithFd(sock: i32, byte: u8, fd: ?i32) usize { + const data = [_]u8{byte}; + const iov = [_]std.posix.iovec_const{.{ .base = &data, .len = 1 }}; + var cbuf: [cmsg_fd_space]u8 align(@alignOf(linux.cmsghdr)) = @splat(0); + var msg = linux.msghdr_const{ + .name = null, + .namelen = 0, + .iov = &iov, + .iovlen = 1, + .control = null, + .controllen = 0, + .flags = 0, + }; + if (fd) |f| { + const hdr: *linux.cmsghdr = @ptrCast(&cbuf); + hdr.* = .{ .len = cmsg_fd_len, .level = linux.SOL.SOCKET, .type = linux.SCM.RIGHTS }; + @memcpy(cbuf[@sizeOf(linux.cmsghdr)..][0..@sizeOf(i32)], std.mem.asBytes(&f)); + msg.control = &cbuf; + msg.controllen = cmsg_fd_space; + } + return linux.sendmsg(sock, &msg, linux.MSG.NOSIGNAL); +} + +/// recvmsg into `buf`; an SCM_RIGHTS fd, if any, is stored in `fd_out`. +fn recvWithFd(sock: i32, buf: []u8, fd_out: *i32, flags: u32) usize { + var iov = [_]std.posix.iovec{.{ .base = buf.ptr, .len = buf.len }}; + var cbuf: [cmsg_fd_space]u8 align(@alignOf(linux.cmsghdr)) = @splat(0); + var msg = linux.msghdr{ + .name = null, + .namelen = 0, + .iov = &iov, + .iovlen = 1, + .control = &cbuf, + .controllen = cbuf.len, + .flags = 0, + }; + const rc = linux.recvmsg(sock, &msg, linux.MSG.CMSG_CLOEXEC | flags); + if (linux.errno(rc) != .SUCCESS) return rc; + if (msg.controllen >= cmsg_fd_len) { + const hdr: *const linux.cmsghdr = @ptrCast(&cbuf); + if (hdr.level == linux.SOL.SOCKET and hdr.type == linux.SCM.RIGHTS and hdr.len >= cmsg_fd_len) { + var fd: i32 = undefined; + @memcpy(std.mem.asBytes(&fd), cbuf[@sizeOf(linux.cmsghdr)..][0..@sizeOf(i32)]); + fd_out.* = fd; + } + } + return rc; +} + +/// Child side of `spawn`. Never returns. +fn childMain(c: *const ChildArgs) noreturn { + resetSignals(); + + const rc_unshare = linux.errno(linux.unshare(linux.CLONE.NEWUSER | linux.CLONE.NEWNS)); + if (rc_unshare != .SUCCESS) childFail(c, setup_failure_status, "unshare(CLONE_NEWUSER|CLONE_NEWNS)", rc_unshare, true); + writeProcFile(c, "/proc/self/setgroups", "deny", true); + writeProcFile(c, "/proc/self/uid_map", c.uid_map, false); + writeProcFile(c, "/proc/self/gid_map", c.gid_map, false); + + const root: [*:0]const u8 = "/"; + const rc_priv = linux.mount(null, root, null, linux.MS.REC | linux.MS.PRIVATE, 0); + if (linux.errno(rc_priv) != .SUCCESS) childFail(c, setup_failure_status, "mount(/, MS_REC|MS_PRIVATE)", linux.errno(rc_priv), true); + + ensureMountpoint(c.gpa, c.mountpoint) catch { + childFail(c, setup_failure_status, "mountpoint setup failed (pass --mount an existing directory)", .SUCCESS, false); + }; + + var fuse_fd: ?i32 = null; + if (c.mount_fuse) { + // Must be opened here, after unshare: the kernel only mounts a fuse + // device opened from the mount's own user namespace. + const rc_open = linux.open("/dev/fuse", .{ .ACCMODE = .RDWR, .CLOEXEC = true }, 0); + switch (linux.errno(rc_open)) { + .SUCCESS => {}, + .NOENT => childFail(c, setup_failure_status, "open /dev/fuse: ENOENT (is the fuse module loaded? try: modprobe fuse)", .SUCCESS, false), + else => |e| childFail(c, setup_failure_status, "open /dev/fuse", e, true), + } + const fd: i32 = @intCast(rc_open); + var opts_buf: [256]u8 = undefined; + const opts = std.fmt.bufPrintZ(&opts_buf, "fd={d},{s}", .{ fd, c.fuse_opts_prefix }) catch unreachable; + const rc = linux.mount("9ns", c.mountpoint, "fuse", linux.MS.NOSUID | linux.MS.NODEV, @intFromPtr(opts.ptr)); + if (linux.errno(rc) != .SUCCESS) childFail(c, setup_failure_status, "mount fuse", linux.errno(rc), true); + fuse_fd = fd; + } + const sent = sendWithFd(c.status_sock, ok_byte, fuse_fd); + if (linux.errno(sent) != .SUCCESS) linux.exit_group(setup_failure_status); + if (fuse_fd) |fd| { + _ = linux.close(fd); // the parent holds the connection now + // Force one GETATTR of the root (served by the parent, which is + // entering its serve loop now); see `spawn`. Errors don't matter. + var stx: linux.Statx = undefined; + _ = linux.statx(linux.AT.FDCWD, c.mountpoint, 0, .{ .TYPE = true }, &stx); + } + + var last: E = .NOENT; + var saw_acces = false; + for (c.candidates) |cand| { + const rc = linux.execve(cand.ptr, c.argv.ptr, c.envp.ptr); + last = linux.errno(rc); + switch (last) { + .NOENT, .NOTDIR, .LOOP, .NAMETOOLONG => continue, + .ACCES => { + saw_acces = true; + continue; + }, + else => break, + } + } + var buf: [512]u8 = undefined; + // "Not found" covers every candidate that could not even be resolved + // (a PATH element that is a file gives ENOTDIR, a symlink loop ELOOP); + // a candidate that existed but was not executable wins over those. + const not_found = switch (last) { + .NOENT, .NOTDIR, .LOOP, .NAMETOOLONG => true, + else => false, + }; + if (not_found and saw_acces) last = .ACCES; + const status: u8 = if (not_found and !saw_acces) 127 else 126; + const text = std.fmt.bufPrint(&buf, "exec {s}", .{c.name}) catch "exec"; + childFail(c, status, text, last, true); +} + +fn writeProcFile(c: *const ChildArgs, path: [*:0]const u8, data: []const u8, ignore_missing: bool) void { + const rc = linux.open(path, .{ .ACCMODE = .WRONLY, .CLOEXEC = true }, 0); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .NOENT => if (ignore_missing) return else childFail(c, setup_failure_status, std.mem.span(path), .NOENT, true), + else => |e| childFail(c, setup_failure_status, std.mem.span(path), e, true), + } + const fd: i32 = @intCast(rc); + const w = linux.write(fd, data.ptr, data.len); + const we = linux.errno(w); + _ = linux.close(fd); + if (we != .SUCCESS) childFail(c, setup_failure_status, std.mem.span(path), we, true); + if (w != data.len) childFail(c, setup_failure_status, std.mem.span(path), .IO, true); +} + +/// Write `[: E]` to the status socket and exit. +fn childFail(c: *const ChildArgs, status: u8, step: []const u8, e: E, with_errno: bool) noreturn { + var buf: [600]u8 = undefined; + buf[0] = status; + const rest = if (with_errno) + std.fmt.bufPrint(buf[1..], "{s}: E{t}", .{ step, e }) catch buf[1..1] + else + std.fmt.bufPrint(buf[1..], "{s}", .{step}) catch buf[1..1]; + const msg = buf[0 .. 1 + rest.len]; + var off: usize = 0; + while (off < msg.len) { + const rc = linux.write(c.status_sock, msg[off..].ptr, msg.len - off); + if (linux.errno(rc) == .INTR) continue; + if (linux.errno(rc) != .SUCCESS) break; + off += rc; + } + linux.exit_group(status); +} + +// --------------------------------------------------------------------------- +// Signals +// --------------------------------------------------------------------------- + +var child_pid_ptr: ?*i32 = null; +var chld_pipe_w: i32 = -1; +var reaped = std.atomic.Value(bool).init(false); +var reaped_status = std.atomic.Value(u32).init(0); +/// A second child (the `--spawn` server) that the SIGCHLD handler reaps so +/// it does not linger as a zombie when it dies mid-session. Its exit does +/// not stop the serve loop. 0 = none. +var server_pid = std.atomic.Value(i32).init(0); + +/// Register the `--spawn` server for reaping by the SIGCHLD handler. +pub fn watchServer(pid: i32) void { + server_pid.store(pid, .seq_cst); +} + +/// Seconds the serve loop gets to come back after the child died before +/// the watchdog ends the process anyway. +pub const exit_grace_seconds: isize = 3; + +/// The watched child is already dead but the serve loop has not come back +/// (it is stuck in a 9P request the server never answers): a terminal +/// signal, or the watchdog armed by `onChld`, then ends 9ns with the +/// child's status instead of hanging. Nothing is lost: the mount is torn +/// down when the process exits. +fn bailIfChildGone() void { + if (!reaped.load(.acquire)) return; + const srv = server_pid.load(.seq_cst); + if (srv > 0) _ = linux.kill(srv, .TERM); + linux.exit_group(decodeStatus(reaped_status.load(.acquire))); +} + +fn armWatchdog() void { + // setitimer takes an itimerval; std declares it with itimerspec, which + // has the same layout on 64-bit targets (the sub-second field is 0). + const t = linux.itimerspec{ + .it_interval = .{ .sec = 0, .nsec = 0 }, + .it_value = .{ .sec = exit_grace_seconds, .nsec = 0 }, + }; + _ = linux.setitimer(@intFromEnum(linux.ITIMER.REAL), &t, null); +} + +fn onAlarm(_: linux.SIG) callconv(.c) void { + bailIfChildGone(); +} + +fn onForward(sig: linux.SIG) callconv(.c) void { + const p = child_pid_ptr orelse return; + const pid = @atomicLoad(i32, p, .seq_cst); + if (pid > 0) _ = linux.kill(pid, sig); + bailIfChildGone(); +} + +/// SIGINT/SIGQUIT: the child owns the tty and gets them itself; we only +/// react when the child is already gone (see `bailIfChildGone`). +fn onTerminal(_: linux.SIG) callconv(.c) void { + bailIfChildGone(); +} + +/// Only the watched child counts: reap it here (WNOHANG), remember its +/// status, forget its pid (so a later SIGTERM cannot hit a recycled pid) +/// and poke the self-pipe. The `--spawn` server is reaped too but does not +/// interrupt `bridge.serve`; SIGCHLD from anything else is ignored. +fn onChld(_: linux.SIG) callconv(.c) void { + const srv = server_pid.load(.seq_cst); + if (srv > 0) { + var sst: u32 = 0; + const src = linux.waitpid(srv, &sst, linux.W.NOHANG); + if (linux.errno(src) == .SUCCESS and src != 0) server_pid.store(0, .seq_cst); + } + const p = child_pid_ptr orelse return; + const pid = @atomicLoad(i32, p, .seq_cst); + if (pid <= 0) return; + var st: u32 = 0; + const rc = linux.waitpid(pid, &st, linux.W.NOHANG); + if (linux.errno(rc) != .SUCCESS or rc == 0) return; + reaped_status.store(st, .release); + reaped.store(true, .release); + @atomicStore(i32, p, 0, .seq_cst); + const b = [_]u8{'c'}; + _ = linux.write(chld_pipe_w, &b, 1); + armWatchdog(); +} + +/// SIGPIPE ignored; SIGINT/SIGQUIT effectively ignored (the child owns the +/// tty) unless the child is already dead; SIGTERM/SIGHUP forwarded to +/// `*child_pid`; SIGCHLD for `*child_pid` reaps it, writes a byte to a +/// nonblocking self-pipe whose read end is returned (use it as `stop_fd`) +/// and arms a watchdog (`exit_grace_seconds`, SIGALRM) that ends the +/// process with the child's status should the serve loop stay blocked. +/// `*child_pid` is filled in by `spawn`. +pub fn installSignals(child_pid: *i32) !i32 { + child_pid_ptr = child_pid; + var fds: [2]i32 = undefined; + switch (linux.errno(linux.pipe2(&fds, .{ .CLOEXEC = true, .NONBLOCK = true }))) { + .SUCCESS => {}, + else => |e| { + std.debug.print("9ns: pipe2: E{t}\n", .{e}); + return error.SystemResources; + }, + } + chld_pipe_w = fds[1]; + + const ign = linux.Sigaction{ .handler = .{ .handler = linux.SIG.IGN }, .mask = linux.sigemptyset(), .flags = 0 }; + const term = linux.Sigaction{ .handler = .{ .handler = &onTerminal }, .mask = linux.sigemptyset(), .flags = linux.SA.RESTART }; + const fwd = linux.Sigaction{ .handler = .{ .handler = &onForward }, .mask = linux.sigemptyset(), .flags = linux.SA.RESTART }; + const chld = linux.Sigaction{ .handler = .{ .handler = &onChld }, .mask = linux.sigemptyset(), .flags = linux.SA.RESTART | linux.SA.NOCLDSTOP }; + const alrm = linux.Sigaction{ .handler = .{ .handler = &onAlarm }, .mask = linux.sigemptyset(), .flags = linux.SA.RESTART }; + std.posix.sigaction(.INT, &term, null); + std.posix.sigaction(.QUIT, &term, null); + std.posix.sigaction(.ALRM, &alrm, null); + std.posix.sigaction(.PIPE, &ign, null); + std.posix.sigaction(.TERM, &fwd, null); + std.posix.sigaction(.HUP, &fwd, null); + std.posix.sigaction(.CHLD, &chld, null); + return fds[0]; +} + +/// Restore default dispositions in the child before exec (ignored signals +/// would otherwise survive execve). +fn resetSignals() void { + const dfl = linux.Sigaction{ .handler = .{ .handler = linux.SIG.DFL }, .mask = linux.sigemptyset(), .flags = 0 }; + inline for (.{ linux.SIG.INT, linux.SIG.QUIT, linux.SIG.PIPE, linux.SIG.TERM, linux.SIG.HUP, linux.SIG.CHLD, linux.SIG.ALRM }) |sig| { + _ = linux.sigaction(sig, &dfl, null); + } +} + +// --------------------------------------------------------------------------- +// Waiting +// --------------------------------------------------------------------------- + +fn takeReaped() ?u32 { + if (!reaped.load(.acquire)) return null; + return reaped_status.load(.acquire); +} + +/// waitpid status → exit code (`128+sig` when killed by a signal). +pub fn decodeStatus(st: u32) u8 { + if (linux.W.IFEXITED(st)) return linux.W.EXITSTATUS(st); + if (linux.W.IFSIGNALED(st)) return 128 +% @as(u8, @truncate(@intFromEnum(linux.W.TERMSIG(st)))); + return 1; +} + +/// Block until `pid` exits (the SIGCHLD handler may have reaped it already). +pub fn waitChild(pid: i32) !u8 { + while (true) { + if (takeReaped()) |st| return decodeStatus(st); + var st: u32 = 0; + const rc = linux.waitpid(pid, &st, 0); + switch (linux.errno(rc)) { + .SUCCESS => return decodeStatus(st), + .INTR => continue, + .CHILD => { + if (takeReaped()) |s| return decodeStatus(s); + return error.NoChild; + }, + else => |e| { + std.debug.print("9ns: waitpid: E{t}\n", .{e}); + return error.Wait; + }, + } + } +} + +/// Non-blocking: the exit status of `pid` if it has exited, else null. +pub fn reapIfExited(pid: i32) ?u8 { + if (takeReaped()) |st| return decodeStatus(st); + var st: u32 = 0; + const rc = linux.waitpid(pid, &st, linux.W.NOHANG); + switch (linux.errno(rc)) { + .SUCCESS => return if (rc == 0) null else decodeStatus(st), + .CHILD => return if (takeReaped()) |s| decodeStatus(s) else null, + else => return null, + } +} + +/// Reap any child (used for the `--spawn` server at exit). Non-blocking. +pub fn reapAny(pid: i32) void { + var st: u32 = 0; + _ = linux.waitpid(pid, &st, linux.W.NOHANG); +} + +// --------------------------------------------------------------------------- +// Tests (no namespaces needed; `ensureMountpoint` is exercised by +// test/integration.sh through the 9ns binary) +// --------------------------------------------------------------------------- + +const testing = std.testing; + +test "normalizePath: absolute paths" { + const gpa = testing.allocator; + const cases = [_]struct { in: []const u8, out: []const u8 }{ + .{ .in = "/mnt/9p", .out = "/mnt/9p" }, + .{ .in = "/mnt/9p/", .out = "/mnt/9p" }, + .{ .in = "//mnt///9p//", .out = "/mnt/9p" }, + .{ .in = "/mnt/./9p/.", .out = "/mnt/9p" }, + .{ .in = "/mnt/x/../9p", .out = "/mnt/9p" }, + .{ .in = "/../mnt/9p", .out = "/mnt/9p" }, + .{ .in = "/a/b/c/../..", .out = "/a" }, + }; + for (cases) |c| { + const got = try normalizePath(gpa, "/cwd", c.in); + defer gpa.free(got); + try testing.expectEqualStrings(c.out, got); + try testing.expectEqual(@as(u8, 0), got[got.len]); + } +} + +test "normalizePath: relative paths use cwd" { + const gpa = testing.allocator; + const cases = [_]struct { cwd: []const u8, in: []const u8, out: []const u8 }{ + .{ .cwd = "/home/me", .in = "mnt", .out = "/home/me/mnt" }, + .{ .cwd = "/home/me", .in = "./mnt/", .out = "/home/me/mnt" }, + .{ .cwd = "/home/me", .in = "../mnt", .out = "/home/mnt" }, + .{ .cwd = "/home/me/", .in = ".", .out = "/home/me" }, + .{ .cwd = "/", .in = "x", .out = "/x" }, + }; + for (cases) |c| { + const got = try normalizePath(gpa, c.cwd, c.in); + defer gpa.free(got); + try testing.expectEqualStrings(c.out, got); + } +} + +test "normalizePath: rejects root and empty" { + const gpa = testing.allocator; + try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/cwd", "/")); + try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/cwd", "///")); + try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/cwd", "/mnt/..")); + try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/cwd", "")); + try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/", "..")); +} + +test "resolveMountpoint: relative resolves against the real cwd" { + const gpa = testing.allocator; + const got = try resolveMountpoint(gpa, "sub/dir"); + defer gpa.free(got); + try testing.expect(got[0] == '/'); + try testing.expect(std.mem.endsWith(u8, got, "/sub/dir")); +} + +test "getenv" { + const env = [_:null]?[*:0]const u8{ "PATH=/a:/b", "X=", "PATHX=no", "NINE_MOUNT=/m" }; + const envp: [*:null]const ?[*:0]const u8 = &env; + try testing.expectEqualStrings("/a:/b", getenv(envp, "PATH").?); + try testing.expectEqualStrings("", getenv(envp, "X").?); + try testing.expectEqualStrings("/m", getenv(envp, "NINE_MOUNT").?); + try testing.expect(getenv(envp, "NOPE") == null); + try testing.expect(getenv(envp, "PAT") == null); +} + +test "pathCandidates: PATH search" { + const gpa = testing.allocator; + const env = [_:null]?[*:0]const u8{ "PATH=/usr/local/bin::/usr/bin", "HOME=/h" }; + const cands = try pathCandidates(gpa, &env, "fish"); + defer { + for (cands) |c| gpa.free(c); + gpa.free(cands); + } + try testing.expectEqual(@as(usize, 3), cands.len); + try testing.expectEqualStrings("/usr/local/bin/fish", cands[0]); + try testing.expectEqualStrings("./fish", cands[1]); + try testing.expectEqualStrings("/usr/bin/fish", cands[2]); +} + +test "pathCandidates: slash means no search; default PATH" { + const gpa = testing.allocator; + const env = [_:null]?[*:0]const u8{"HOME=/h"}; + { + const cands = try pathCandidates(gpa, &env, "./bin/x"); + defer { + for (cands) |c| gpa.free(c); + gpa.free(cands); + } + try testing.expectEqual(@as(usize, 1), cands.len); + try testing.expectEqualStrings("./bin/x", cands[0]); + } + { + const cands = try pathCandidates(gpa, &env, "sh"); + defer { + for (cands) |c| gpa.free(c); + gpa.free(cands); + } + try testing.expectEqual(@as(usize, 3), cands.len); + try testing.expectEqualStrings("/usr/local/bin/sh", cands[0]); + try testing.expectEqualStrings("/bin/sh", cands[1]); + } + try testing.expectError(error.EmptyProgramName, pathCandidates(gpa, &env, "")); +} + +test "findInPath finds sh" { + const gpa = testing.allocator; + const env = [_:null]?[*:0]const u8{"PATH=/nonexistent:/bin:/usr/bin"}; + const p = try findInPath(gpa, &env, "sh"); + defer gpa.free(p); + try testing.expect(std.mem.endsWith(u8, p, "/sh")); + try testing.expectError(error.FileNotFound, findInPath(gpa, &env, "definitely-not-a-program-9ns")); +} + +test "buildEnvp replaces NINE_MOUNT" { + const gpa = testing.allocator; + const env = [_:null]?[*:0]const u8{ "A=1", "NINE_MOUNT=/old", "B=2" }; + const out = try buildEnvp(gpa, &env, "/mnt/9p"); + defer { + gpa.free(std.mem.span(out[out.len - 1].?)); + gpa.free(out); + } + try testing.expectEqual(@as(usize, 3), out.len); + try testing.expectEqualStrings("A=1", std.mem.span(out[0].?)); + try testing.expectEqualStrings("B=2", std.mem.span(out[1].?)); + try testing.expectEqualStrings("NINE_MOUNT=/mnt/9p", std.mem.span(out[2].?)); + try testing.expect(out[3] == null); + try testing.expectEqualStrings("/mnt/9p", getenv(out.ptr, "NINE_MOUNT").?); +} + +test "decodeStatus" { + try testing.expectEqual(@as(u8, 0), decodeStatus(0)); + try testing.expectEqual(@as(u8, 7), decodeStatus(7 << 8)); + try testing.expectEqual(@as(u8, 255), decodeStatus(255 << 8)); + try testing.expectEqual(@as(u8, 128 + 9), decodeStatus(9)); // SIGKILL + try testing.expectEqual(@as(u8, 128 + 15), decodeStatus(15)); // SIGTERM +} + +test "ensureMountpoint: existing directory is accepted, plain file rejected" { + const gpa = testing.allocator; + try ensureMountpoint(gpa, "/tmp"); + try testing.expectError(error.Mountpoint, ensureMountpoint(gpa, "/proc/self/status")); +} diff --git a/9ns/test/adv_bridge_hostile.py b/9ns/test/adv_bridge_hostile.py new file mode 100755 index 0000000..d353541 --- /dev/null +++ b/9ns/test/adv_bridge_hostile.py @@ -0,0 +1,487 @@ +#!/usr/bin/env python3 +"""A scriptable, hostile 9P2000 server on a Unix socket (stdlib only). + +Usage: adv_bridge_hostile.py SOCKET MODE + +Serves a tiny in-memory tree: + /f "hello world\\n" + /d/g "in d\\n" + /fids reading it returns the number of fids currently bound + /big 1 MiB of pseudo-random bytes +plus create/write/remove/wstat so the scratch battery can run in `ok` mode. + +MODE selects one misbehaviour (see MODES below). Everything not covered by +the mode behaves normally, so 9ns gets through version/attach/stat(root). +""" +import os +import random +import socket +import struct +import sys +import time + +NOTAG = 0xFFFF +NOFID = 0xFFFFFFFF +QTDIR = 0x80 +DMDIR = 0x80000000 + +Tversion, Rversion = 100, 101 +Tauth, Rauth = 102, 103 +Tattach, Rattach = 104, 105 +Rerror = 107 +Tflush, Rflush = 108, 109 +Twalk, Rwalk = 110, 111 +Topen, Ropen = 112, 113 +Tcreate, Rcreate = 114, 115 +Tread, Rread = 116, 117 +Twrite, Rwrite = 118, 119 +Tclunk, Rclunk = 120, 121 +Tremove, Rremove = 122, 123 +Tstat, Rstat = 124, 125 +Twstat, Rwstat = 126, 127 + +MODES = """ +ok behave (qid paths are recycled LIFO after remove, like many servers) +trunc Rread on /f: send half the frame, then close +short_frame Rread on /f: frame whose size field is 3 +huge_frame Rread on /f: frame whose size field is msize+1 +wrong_tag Rread on /f: reply carries tag+1 +wrong_type Tstat on /f: answer with an Rwalk +rread_big Rread on /f: count = requested+1 +rwalk_many Twalk to f: nwqid = nwname+1 +rwalk_zero Twalk to nope: Rwalk nwqid=0 instead of Rerror +rstat_garbage Tstat on /f: random bytes as the stat +rstat_overlong Tstat on /f: inner stat size disagrees with outer +dir_split Tread on /: a stat record split across two Rreads +dir_forever Tread on /: ignore offset, always return the same records +qid_collide every file and dir shares qid.path 7 (root keeps its own) +qid_zero every qid.path is 0, including the root +name_slash / has an entry "a/b" +name_empty / has an entry "" +name_huge / has an entry with a 60000-byte name +name_dots / lists "." and ".." too +rerror_big Twalk to nope: Rerror with 65535 bytes of text +extra_reply Rread on /f: an unsolicited Rclunk (tag 9) precedes the real reply +never Tread on /f: never reply (hang) +close_mid Tread on /f: close the socket without replying +renegotiate Tread on /f: an unsolicited Rversion precedes the real reply +length_max Tstat on /f: length = 2**64-1 +iounit_one Ropen: iounit = 1 +rwrite_big Rwrite: count = requested+1 +msize_tiny Rversion msize = 64 +version_unknown Rversion "unknown" +rename_fail Twstat with a new name always fails "file already exists" +slow every reply delayed 20 ms (for interrupt tests) +""" + + +def s8(x): return struct.pack('= 4: + n = struct.unpack('= n: + msg, buf = buf[:n], buf[n:] + out = self.handle(msg) + if out is None: + return # hang up / hang + if self.mode == 'slow': + time.sleep(0.02) + conn.sendall(out) + continue + data = conn.recv(65536) + if not data: + return + buf += data + + def handle(self, msg): + typ = msg[4] + tag = struct.unpack(' len(node.content): + node.content.extend(b'\0' * (off - len(node.content))) + node.content[off:off + len(data)] = data + node.mtime = int(time.time()) + n = len(data) + 1 if self.mode == 'rwrite_big' else len(data) + return self.frame(Rwrite, tag, s32(n)) + if typ == Tclunk: + fid = r.u32() + if fid not in self.fids: + return self.err(tag, 'unknown fid') + del self.fids[fid] + return self.frame(Rclunk, tag, b'') + if typ == Tremove: + fid = r.u32() + if fid not in self.fids: + return self.err(tag, 'unknown fid') + node = self.fids[fid][0] + del self.fids[fid] + if node is self.root: + return self.err(tag, 'cannot remove root') + if node.isdir and node.children: + return self.err(tag, 'directory not empty') + parent = self.find_parent(self.root, node) + if parent is not None: + del parent.children[node.name] + self.free_paths.append(node.path) + node.removed = True + return self.frame(Rremove, tag, b'') + if typ == Tstat: + fid = r.u32() + if fid not in self.fids: + return self.err(tag, 'unknown fid') + node = self.fids[fid][0] + if node is self.root.children.get('f'): + m = self.mode + if m == 'wrong_type': + return self.frame(Rwalk, tag, s16(0)) + if m == 'rstat_garbage': + junk = bytes([0xAB] * 60) + return self.frame(Rstat, tag, s16(len(junk)) + junk) + if m == 'rstat_overlong': + st = node.stat_bytes(self) + inner = st[2:] + return self.frame(Rstat, tag, s16(len(inner) + 5) + inner) + if m == 'length_max': + st = node.stat_bytes(self, length=2 ** 64 - 1) + return self.frame(Rstat, tag, s16(len(st)) + st) + st = node.stat_bytes(self) + return self.frame(Rstat, tag, s16(len(st)) + st) + if typ == Twstat: + fid = r.u32() + r.u16() + st = r.bytes(r.u16()) + if fid not in self.fids: + return self.err(tag, 'unknown fid') + node = self.fids[fid][0] + sr = Reader(st) + sr.u16(); sr.u32(); sr.bytes(13) + mode = sr.u32(); sr.u32(); mtime = sr.u32(); length = sr.u64() + name = sr.str() + if name and name != node.name: + if self.mode == 'rename_fail': + return self.err(tag, 'file already exists') + parent = self.find_parent(self.root, node) + if name in parent.children: + return self.err(tag, 'file already exists') + del parent.children[node.name] + node.name = name + parent.children[name] = node + if mode != 0xFFFFFFFF: + node.mode = mode & 0o777 + if mtime != 0xFFFFFFFF: + node.mtime = mtime + if length != 0xFFFFFFFFFFFFFFFF and not node.isdir: + if length < len(node.content): + del node.content[length:] + else: + node.content.extend(b'\0' * (length - len(node.content))) + return self.frame(Rwstat, tag, b'') + return self.err(tag, 'unsupported message') + + def find_parent(self, cur, node): + for c in cur.children.values(): + if c is node: + return cur + if c.isdir: + p = self.find_parent(c, node) + if p is not None: + return p + return None + + def readdir(self, tag, node, off, count): + recs = [] + if node is self.root: + m = self.mode + if m == 'name_slash': + recs.append(node.stat_bytes(self, name='a/b')) + if m == 'name_empty': + recs.append(node.stat_bytes(self, name='')) + if m == 'name_huge': + recs.append(node.stat_bytes(self, name='h' * 60000)) + if m == 'name_dots': + recs.append(node.stat_bytes(self, name='.')) + recs.append(node.stat_bytes(self, name='..')) + for c in node.children.values(): + recs.append(c.stat_bytes(self)) + blob = b''.join(recs) + if node is self.root and self.mode == 'dir_forever': + return self.frame(Rread, tag, s32(len(blob)) + blob) + if node is self.root and self.mode == 'dir_split': + # first read: up to the middle of the second record; second read: the rest + cut = len(recs[0]) + len(recs[1]) // 2 + if off == 0: + data = blob[:cut] + elif off == cut: + data = blob[cut:] + else: + data = b'' + return self.frame(Rread, tag, s32(len(data)) + data) + # 9P rule: offset 0 or previous offset+count; never split a record. + out = b'' + pos = 0 + for rec in recs: + if pos >= off and len(out) + len(rec) <= count: + out += rec + elif pos >= off: + break + pos += len(rec) + return self.frame(Rread, tag, s32(len(out)) + out) + + +class Reader: + def __init__(self, b): + self.b = b + self.i = 0 + + def bytes(self, n): + v = self.b[self.i:self.i + n] + self.i += n + return v + + def u8(self): return struct.unpack(' <9proc-demo> (9proc unused; part of zig build 9ns-adv) +set -u +NS=$(realpath "${1:?path to 9ns}") +HERE=$(cd "$(dirname "$0")" && pwd) +SRV=$HERE/adv_bridge_hostile.py +TMP=$(mktemp -d "${TMPDIR:-/tmp}/9ns-adv.XXXXXX") +M=/mnt/9p +FAILED=0 +PASSED=0 +SRVPID= + +cleanup() { [ -n "$SRVPID" ] && kill "$SRVPID" 2>/dev/null; pkill -f "adv_bridge_hostile.py $TMP" 2>/dev/null; rm -rf "$TMP"; } +trap cleanup EXIT + +if ! unshare -Urm true 2>/dev/null || [ ! -c /dev/fuse ]; then echo "SKIP: no user namespaces or /dev/fuse"; exit 0; fi + +pass() { PASSED=$((PASSED + 1)); echo "ok - $1"; } +fail() { FAILED=$((FAILED + 1)); echo "FAIL - $1"; shift; [ $# -gt 0 ] && printf ' %s\n' "$@"; } +expect_eq() { if [ "$2" = "$3" ]; then pass "$1"; else fail "$1" "expected: $(printf %q "$2")" "actual: $(printf %q "$3")"; fi; } +expect_contains() { case "$3" in *"$2"*) pass "$1" ;; *) fail "$1" "missing: $(printf %q "$2")" "in: $(printf %q "$3")" ;; esac; } + +start_server() { # mode + [ -n "$SRVPID" ] && { kill "$SRVPID" 2>/dev/null; wait "$SRVPID" 2>/dev/null; } + SOCK=$TMP/$1.sock + rm -f "$SOCK" + python3 "$SRV" "$SOCK" "$1" >"$TMP/$1.srv.out" 2>&1 "$TMP/stderr") + RC=$? + STDERR=$(cat "$TMP/stderr") +} + +# 9ns must not die of a signal or panic. RC 124 = timeout(1) fired. +no_crash() { # name + if [ "$RC" -ge 128 ] || [ "$RC" -eq 124 ]; then fail "$1: 9ns exit $RC" "$STDERR"; return; fi + case "$STDERR" in *panic*|*"Segmentation"*|*"integer overflow"*|*"reached unreachable"*|*"index out of bounds"*) fail "$1: crash text in stderr" "$STDERR";; *) pass "$1: no crash (exit $RC)";; esac +} + +echo "# sanity: the hostile server behaves in 'ok' mode" +run ok "cat $M/f"; expect_eq "ok: cat f" "hello world" "$OUT" +run ok "cat $M/d/g"; expect_eq "ok: nested" "in d" "$OUT" +run ok "cat $M/nope 2>&1 | sed 's/.*: //'"; expect_eq "ok: ENOENT" "No such file or directory" "$OUT" +run ok "head -c 1048576 $M/big | wc -c | grep -q 1048576 && echo yes"; expect_eq "ok: 1 MiB read matches" "yes" "$OUT" + +echo "# unlink + recreate with a recycled qid.path" +run ok "echo 1 > $M/a; rm $M/a; echo 2 > $M/a; cat $M/a; rm $M/a"; expect_eq "qid reuse: new content, not stale" "2" "$OUT" +run ok "echo 1 > $M/x; rm $M/x; mkdir $M/x; stat -c %F $M/x; rmdir $M/x"; expect_eq "qid reuse: file→dir on the same path" "directory" "$OUT" + +echo "# fids do not grow with the number of operations" +loop='i=0; while [ $i -lt N ]; do echo hi > M/t; cat M/t >/dev/null; mkdir M/dd; rmdir M/dd; rm M/t; i=$((i+1)); done; cat M/fids' +run ok "$(echo "$loop" | sed "s|N|20|; s|M/|$M/|g")"; a=$OUT +run ok "$(echo "$loop" | sed "s|N|200|; s|M/|$M/|g")"; b=$OUT +expect_eq "fids after 20 == after 200 iterations ($a)" "$a" "$b" +floop='i=0; while [ $i -lt N ]; do cat M/nope 2>/dev/null; echo x > M/fids 2>/dev/null; mkdir M/f 2>/dev/null; rm M/d 2>/dev/null; mv M/f M/d 2>/dev/null; i=$((i+1)); done; cat M/fids' +run ok "$(echo "$floop" | sed "s|N|20|; s|M/|$M/|g")"; a=$OUT +run ok "$(echo "$floop" | sed "s|N|200|; s|M/|$M/|g")"; b=$OUT +expect_eq "fids after 20 == after 200 failing iterations ($a)" "$a" "$b" + +echo "# rename over an existing file must not lose the target when the rename fails" +run rename_fail "echo A > $M/a; echo B > $M/b; mv $M/a $M/b 2>/dev/null; echo mv=\$?; cat $M/b; cat $M/a" +expect_contains "rename_fail: mv reports failure" "mv=1" "$OUT" +expect_contains "rename_fail: target b still has its content" "B" "$OUT" +expect_contains "rename_fail: source a still has its content" "A" "$OUT" + +echo "# protocol violations on a data read must yield an error, not a crash" +for mode in trunc short_frame huge_frame wrong_tag rread_big extra_reply close_mid renegotiate rwrite_big; do + if [ "$mode" = rwrite_big ]; then script="dd if=/dev/zero of=$M/f bs=10 count=1 2>&1; echo status=\$?"; else script="cat $M/f 2>&1; echo status=\$?"; fi + run "$mode" "$script" + no_crash "$mode" + expect_contains "$mode: child sees an error" "status=1" "$OUT" + case "$OUT" in *"Input/output error"*|*"not connected"*) pass "$mode: EIO/ENOTCONN";; *) fail "$mode: errno text" "$OUT";; esac +done + +echo "# protocol violations on lookup/stat" +for mode in wrong_type rwalk_many rstat_garbage rstat_overlong; do + run "$mode" "stat -c %s $M/f 2>&1; echo status=\$?" + no_crash "$mode" + expect_contains "$mode: child sees an error" "status=1" "$OUT" +done +run rwalk_zero "cat $M/nope 2>&1; echo status=\$?; cat $M/f 2>&1" +no_crash "rwalk_zero" +expect_contains "rwalk_zero: child sees an error" "status=1" "$OUT" +run rerror_big "cat $M/nope 2>&1; echo status=\$?; cat $M/f" +no_crash "rerror_big" +expect_contains "rerror_big: child sees an error" "status=1" "$OUT" +expect_contains "rerror_big: session survives a 64 KiB Rerror" "hello world" "$OUT" + +echo "# hostile stat contents" +run length_max "stat -c '%s %b' $M/f 2>&1; echo status=\$?" +no_crash "length_max" +expect_contains "length_max: stat succeeds with a saturated block count" "status=0" "$OUT" +run iounit_one "cat $M/f; head -c 3000 $M/big | wc -c" +no_crash "iounit_one" +expect_contains "iounit_one: read still complete" "hello world" "$OUT" +expect_contains "iounit_one: 3000 bytes" "3000" "$OUT" + +echo "# hostile directory listings" +run dir_split "ls $M 2>&1; echo status=\$?; cat $M/f" +no_crash "dir_split" +expect_contains "dir_split: readdir fails with EIO" "Input/output error" "$OUT" +expect_contains "dir_split: session survives" "hello world" "$OUT" +run dir_forever "ls $M 2>&1 | tail -c 200; echo status=\$?; cat $M/f; grep VmRSS /proc/\$PPID/status" +no_crash "dir_forever" +expect_contains "dir_forever: infinite directory is cut off with EIO" "Input/output error" "$OUT" +expect_contains "dir_forever: session survives" "hello world" "$OUT" +for mode in name_slash name_empty name_huge name_dots; do + run "$mode" "ls -a $M | tr '\n' ' '; echo; cat $M/f" + no_crash "$mode" + expect_contains "$mode: listing still works" "big d f fids" "$OUT" + expect_contains "$mode: file readable" "hello world" "$OUT" + case "$mode" in + name_slash) expect_eq "$mode: slash entry dropped" "" "$(printf '%s' "$OUT" | grep -o 'a/b')";; + name_dots) expect_eq "$mode: exactly one . and one .." "1 1" "$(printf '%s %s' "$(printf '%s\n' "$OUT" | head -1 | tr ' ' '\n' | grep -c '^\.$')" "$(printf '%s\n' "$OUT" | head -1 | tr ' ' '\n' | grep -c '^\.\.$')")";; + esac +done + +echo "# qid collisions" +run qid_collide "cat $M/f; cat $M/d/g; ls $M/d; stat -c %i $M/f $M/d 2>&1; echo status=\$?" +no_crash "qid_collide" +expect_contains "qid_collide: reads work" "hello world" "$OUT" +run qid_zero "cat $M/f; ls $M | tr '\n' ' '; echo; cat $M/d/g; echo status=\$?" +no_crash "qid_zero" +expect_contains "qid_zero: file with root's qid.path is still a readable file" "hello world" "$OUT" +expect_contains "qid_zero: root still lists" "big d f fids" "$OUT" +expect_contains "qid_zero: nested file readable" "in d" "$OUT" + +echo "# version negotiation" +run version_unknown "echo ran" +expect_eq "version_unknown: 9ns refuses (125)" "125" "$RC" +run msize_tiny "cat $M/f 2>&1; echo status=\$?" +no_crash "msize_tiny" + +echo "# a server that never replies" +start_server never +# SIGTERM is forwarded to the child; once the child is gone 9ns must leave the +# pending 9P reply behind and exit even though the server stays silent. +timeout -s TERM 3 "$NS" --unix "$SOCK" -- sh -c "cat $M/f; echo unreachable" >"$TMP/never.out" 2>"$TMP/never.err" & +TPID=$! +sleep 4 +if kill -0 "$TPID" 2>/dev/null; then + fail "never: SIGTERM did not end 9ns while a reply was outstanding"; kill -9 "$TPID" +else + pass "never: SIGTERM ends 9ns even while the server is silent" +fi +wait "$TPID" 2>/dev/null +# Without a signal the mount hangs (documented v1 limitation) until the server dies. +timeout 30 "$NS" --unix "$SOCK" -- sh -c "cat $M/f; echo unreachable" >"$TMP/never.out" 2>"$TMP/never.err" & +TPID=$! +sleep 1.5 +if kill -0 "$TPID" 2>/dev/null; then + pass "never: mount hangs while the server is silent (documented v1 limitation)" + kill "$SRVPID"; wait "$SRVPID" 2>/dev/null; SRVPID= + for _ in $(seq 1 50); do kill -0 "$TPID" 2>/dev/null || break; sleep 0.1; done + if kill -0 "$TPID" 2>/dev/null; then fail "never: 9ns still alive after its server died"; kill -9 "$TPID"; else pass "never: killing the server unblocks 9ns"; fi +else + wait "$TPID"; fail "never: 9ns exited early ($?)" "$(cat "$TMP/never.err")" +fi + +echo "# interrupting a slow read (INTERRUPT must not confuse reply matching)" +run slow "cat $M/big > /dev/null; cat $M/f; cat $M/fids" --msize 8192 +base=$(printf '%s\n' "$OUT" | tail -1) +run slow "(cat $M/big > /dev/null & sleep 0.3; kill -INT \$!; wait \$!) 2>/dev/null; cat $M/f; cat $M/fids" --msize 8192 +no_crash "slow" +expect_contains "slow: read after interrupted read works" "hello world" "$OUT" +expect_eq "slow: fids after an interrupted read == after a complete one ($base)" "$base" "$(printf '%s\n' "$OUT" | tail -1)" + +echo +echo "passed=$PASSED failed=$FAILED" +[ "$FAILED" -eq 0 ] diff --git a/9ns/test/adv_bridge_semantics.sh b/9ns/test/adv_bridge_semantics.sh new file mode 100755 index 0000000..b38ce9c --- /dev/null +++ b/9ns/test/adv_bridge_semantics.sh @@ -0,0 +1,205 @@ +#!/usr/bin/env bash +# FUSE semantics through the bridge against 9proc-demo's /scratch tree. +# Usage: bash 9ns/test/adv_bridge_semantics.sh <9ns> <9proc-demo> (part of zig build 9ns-adv) +set -u +NS=$(realpath "${1:?path to 9ns}") +PROC=$(realpath "${2:?path to 9proc-demo}") +TMP=$(mktemp -d "${TMPDIR:-/tmp}/9ns-sem.XXXXXX") +M=/mnt/9p +S=$M/scratch +FAILED=0 +PASSED=0 +SRVPID= +cleanup() { [ -n "$SRVPID" ] && kill "$SRVPID" 2>/dev/null; rm -rf "$TMP"; } +trap cleanup EXIT +if ! unshare -Urm true 2>/dev/null || [ ! -c /dev/fuse ]; then echo "SKIP: no user namespaces or /dev/fuse"; exit 0; fi + +pass() { PASSED=$((PASSED + 1)); echo "ok - $1"; } +fail() { FAILED=$((FAILED + 1)); echo "FAIL - $1"; shift; [ $# -gt 0 ] && printf ' %s\n' "$@"; } +expect_eq() { if [ "$2" = "$3" ]; then pass "$1"; else fail "$1" "expected: $(printf %q "$2")" "actual: $(printf %q "$3")"; fi; } +expect_contains() { case "$3" in *"$2"*) pass "$1" ;; *) fail "$1" "missing: $(printf %q "$2")" "in: $(printf %q "$3")" ;; esac; } + +SOCK=$TMP/i.sock +"$PROC" --unix "$SOCK" >"$TMP/srv.out" 2>&1 & +SRVPID=$! +for _ in $(seq 1 100); do [ -S "$SOCK" ] && break; sleep 0.02; done +# Each run is a fresh session; state persists in the server, so tests clean up after themselves. +run() { OUT=$(timeout 120 "$NS" --unix "$SOCK" "${EXTRA[@]}" -- sh -c "$1" 2>"$TMP/stderr"); RC=$?; STDERR=$(cat "$TMP/stderr"); } +EXTRA=() +py() { run "python3 - <<'PYEOF' +$1 +PYEOF"; } + +echo "# open/create flags" +py " +import os, errno +p='$S/excl' +fd=os.open(p, os.O_CREAT|os.O_WRONLY, 0o644); os.write(fd, b'x'); os.close(fd) +try: + os.open(p, os.O_CREAT|os.O_EXCL|os.O_WRONLY, 0o644); print('no error') +except OSError as e: print(errno.errorcode[e.errno]) +os.unlink(p) +fd=os.open('$S/ro', os.O_CREAT|os.O_RDONLY, 0o644); print(os.read(fd, 10)); os.close(fd) +print(os.path.exists('$S/ro')); os.unlink('$S/ro') +" +expect_eq "O_EXCL on an existing file is EEXIST" "EEXIST" "$(printf '%s\n' "$OUT" | sed -n 1p)" +expect_eq "create with O_RDONLY works and reads empty" $'b\'\'\nTrue' "$(printf '%s\n' "$OUT" | sed -n 2,3p)" + +run "mkdir -m 700 $S/m7 && stat -c %a $S/m7; chmod 755 $S/m7 && stat -c %a $S/m7; rmdir $S/m7" +expect_eq "mkdir -m 700 then chmod 755" $'700\n755' "$OUT" +run "echo x > $S/c && chmod 600 $S/c && stat -c %a $S/c; chmod 444 $S/c && stat -c %a $S/c; rm -f $S/c" +expect_eq "chmod on a file" $'600\n444' "$OUT" +run "echo x > $S/t && touch -d @1000000000 $S/t && stat -c %Y $S/t; touch $S/t && [ \$(stat -c %Y $S/t) -gt 1000000000 ] && echo now; rm $S/t" +expect_eq "utimes (explicit) and touch (now)" $'1000000000\nnow' "$OUT" +run "echo abc > $S/tr && truncate -s 10 $S/tr && stat -c %s $S/tr && od -An -c $S/tr | tr -s ' ' | tr -d '\n'; echo; rm $S/tr" +expect_eq "truncate to larger zero-fills" $'10\n a b c \\n \\0 \\0 \\0 \\0 \\0 \\0' "$OUT" +run "echo a > $S/ap && echo b >> $S/ap && echo c >> $S/ap && cat $S/ap | tr '\n' ' '; rm $S/ap" +expect_eq "shell append" "a b c " "$OUT" +py " +import os +p='$S/ap2' +f1=os.open(p, os.O_CREAT|os.O_WRONLY|os.O_APPEND, 0o644) +f2=os.open(p, os.O_WRONLY|os.O_APPEND) +os.write(f1, b'one '); os.write(f2, b'two '); os.write(f1, b'three') +os.close(f1); os.close(f2) +print(open(p).read()); os.unlink(p) +" +expect_eq "O_APPEND from two descriptors interleaves in order" "one two three" "$OUT" +run "echo 0123456789 > $S/tt && (echo X > $S/tt) && cat $S/tt && stat -c %s $S/tt; rm $S/tt" +expect_eq "O_TRUNC (atomic_o_trunc) truncates before write" $'X\n2' "$OUT" + +echo "# reads at the edges" +run "printf hello > $S/e; dd if=$S/e bs=1 skip=100 count=5 2>/dev/null | wc -c; head -c 0 $S/e | wc -c; dd if=/dev/null of=$S/e bs=1 count=0 conv=notrunc 2>/dev/null; cat $S/e; echo; rm $S/e" +expect_eq "read past EOF is 0 bytes; 0-byte read/write are no-ops" $'0\n0\nhello' "$OUT" +head -c 4194304 /dev/urandom >"$TMP/four" +SUM=$(sha256sum <"$TMP/four" | cut -d' ' -f1) +run "dd if=$TMP/four of=$S/four bs=4M status=none && dd if=$S/four bs=4M status=none | sha256sum | cut -d' ' -f1; stat -c %s $S/four; rm $S/four" +expect_eq "4 MiB single-request dd round trip" "$SUM"$'\n4194304' "$OUT" +py " +import os +p='$S/lseek' +open(p,'w').write('0123456789') +f=open(p,'rb'); f.seek(0, 2); print(f.tell()); f.seek(-3, 2); print(f.read()); f.close() +os.unlink(p) +" +expect_eq "lseek SEEK_END on a direct_io file" $'10\nb\'789\'' "$OUT" + +echo "# unlink of an open file" +py " +import os +p='$S/unl' +fd=os.open(p, os.O_CREAT|os.O_RDWR, 0o644) +os.write(fd, b'before') +os.unlink(p) +print(os.path.exists(p)) +os.lseek(fd, 0, 0); print(os.read(fd, 100)) +os.write(fd, b'-after'); os.lseek(fd, 0, 0); print(os.read(fd, 100)) +os.close(fd) +" +expect_eq "read/write through the fd after unlink" $'False\nb\'before\'\nb\'before-after\'' "$OUT" + +echo "# rename" +run "echo A > $S/ra; echo B > $S/rb; mv $S/ra $S/rb && cat $S/rb; ls $S | tr '\n' ' '; echo; rm $S/rb" +expect_eq "rename over an existing file replaces it, no leftovers" $'A\nrb ' "$OUT" +run "mkdir $S/rd1 && echo x > $S/rd1/f && mv $S/rd1 $S/rd2 && cat $S/rd2/f && ls $S/rd2; rm -r $S/rd2; ls $S | wc -l" +expect_eq "rename of a directory" $'x\nf\n0' "$OUT" +run "mkdir $S/e1 $S/e2 && mv -T $S/e1 $S/e2 && ls $S | tr '\n' ' '; rmdir $S/e2" +expect_eq "rename dir over an empty dir" "e2 " "$OUT" +run "mkdir $S/n1 $S/n2 && echo x > $S/n2/f && mv -T $S/n1 $S/n2 2>&1 | sed 's/.*: //'; rm -r $S/n1 $S/n2" +expect_eq "rename dir over a non-empty dir is ENOTEMPTY" "Directory not empty" "$OUT" +py " +import os +p='$S/same'; open(p,'w').write('x'); os.rename(p, p); print(open(p).read()); os.unlink(p) +" +expect_eq "rename onto itself is a no-op" "x" "$OUT" +py " +import os, ctypes, errno +libc = ctypes.CDLL(None, use_errno=True) +a, b = b'$S/nra', b'$S/nrb' +open(a,'w').write('A'); open(b,'w').write('B') +r = libc.renameat2(-100, a, -100, b, 1) # RENAME_NOREPLACE +print('rc', r, errno.errorcode.get(ctypes.get_errno())) +print(open(b).read()) +os.unlink(a); os.unlink(b) +" +expect_eq "RENAME_NOREPLACE keeps the target" $'rc -1 EEXIST\nB' "$OUT" + +echo "# directories" +run "mkdir $S/many && cd $S/many && i=0; while [ \$i -lt 5000 ]; do : > f\$i; i=\$((i+1)); done; ls | wc -l; ls -l | wc -l; grep VmRSS /proc/\$PPID/status | awk '{print \$2}' > $TMP/rss1; rm -f $S/many/*; rmdir $S/many; ls | wc -l; grep VmRSS /proc/\$PPID/status | awk '{print \$2}' > $TMP/rss2" +expect_eq "5000 entries: ls and ls -l" $'5000\n5001\n0' "$OUT" +r1=$(cat "$TMP/rss1"); r2=$(cat "$TMP/rss2") +if [ "$r2" -le $((r1 + 2048)) ]; then pass "RSS after cleanup ($r2 KiB) <= after listing ($r1 KiB)+2 MiB"; else fail "RSS grew after cleanup: $r1 -> $r2 KiB"; fi +py " +import os +d='$S/chg'; os.mkdir(d) +for i in range(50): open(f'{d}/a{i}','w').close() +it = os.scandir(d); first = next(it).name +for i in range(3000): open(f'{d}/b{i}','w').close() +rest = [e.name for e in it] +print(first[0], len(rest) >= 49, len(set(rest)) == len(rest)) +for n in os.listdir(d): os.unlink(f'{d}/{n}') +os.rmdir(d) +" +expect_eq "readdir of a directory that changes mid-iteration" "a True True" "$OUT" +run "ls $M/.. > /dev/null && echo ok; stat -c %i $M $M/. $M/scratch/..; cd $M/scratch && ls .. | grep -c scratch" +expect_eq ".. of the root and of a subdir" $'ok\n1\n1\n1\n1' "$OUT" +run "cd $M && find . -type d | wc -l && find . -type f | head -1 && find $S -type f | wc -l" +expect_contains "find -type works" "./README" "$OUT" +run "stat -f -c '%T %S %l' $M; df -P $M | tail -1 | awk '{print \$1}'; sync -f $M && echo synced; sync && echo synced2" +expect_eq "statfs, df, syncfs, sync" $'fuse 4096 255\n9ns\nsynced\nsynced2' "$OUT" + +echo "# server refusals keep their errno through the error path" +run "echo x > $M/build/zig_version; a=\$?; mkdir $M/build/x 2>/dev/null; b=\$?; rm $M/README 2>/dev/null; c=\$?; rmdir $M/build 2>/dev/null; d=\$?; echo \$a\$b\$c\$d" 2>/dev/null +expect_eq "open-for-write / mkdir / rm / rmdir on read-only nodes all fail (errno preserved through error path)" "1111" "$(printf '%s\n' "$OUT" | tail -1)" + +echo "# unsupported operations fail cleanly" +run "echo x > $S/l1; ln $S/l1 $S/l2 2>&1 | sed 's/.*: //'; ln -s l1 $S/l3 2>&1 | sed 's/.*: //'; mkfifo $S/p 2>&1 | sed 's/.*: //'; ls $S | tr '\n' ' '; echo; rm $S/l1" +# The kernel turns ENOSYS from LINK into EPERM (fuse_link); symlink/mknod keep ENOSYS. +expect_eq "link/symlink/mknod fail cleanly" $'Operation not permitted\nFunction not implemented\nFunction not implemented\nl1 ' "$OUT" +run "echo x > $S/x1; setfattr -n user.a -v 1 $S/x1 2>&1 | sed 's/.*: //'; getfattr -n user.a $S/x1 2>&1 | sed 's/.*: //'; getfattr -d $S/x1 2>&1 | sed 's/.*: //'; rm $S/x1" +expect_eq "xattr ops are EOPNOTSUPP" $'Operation not supported\nOperation not supported\nOperation not supported' "$OUT" +py " +import os, fcntl, mmap, errno +p='$S/mm'; open(p,'w').write('mapme') +fd=os.open(p, os.O_RDWR) +fcntl.flock(fd, fcntl.LOCK_EX); fcntl.flock(fd, fcntl.LOCK_UN); fcntl.lockf(fd, fcntl.LOCK_EX); fcntl.lockf(fd, fcntl.LOCK_UN); print('locks ok') +try: + m = mmap.mmap(fd, 5); print('shared', bytes(m)); m.close() +except OSError as e: print('shared', errno.errorcode[e.errno]) +try: + m = mmap.mmap(fd, 5, flags=mmap.MAP_PRIVATE, prot=mmap.PROT_READ); print('private', bytes(m)); m.close() +except OSError as e: print('private', errno.errorcode[e.errno]) +os.close(fd); os.unlink(p) +" +expect_contains "flock/lockf work (local locks)" "locks ok" "$OUT" +case "$OUT" in *"shared ENODEV"*|*"shared b'mapme'"*) pass "shared mmap: clean result ($(printf '%s\n' "$OUT" | sed -n 2p))";; *) fail "shared mmap" "$OUT";; esac +expect_contains "private mmap reads the file" "private b'mapme'" "$OUT" + +echo "# tools" +mkdir -p "$TMP/tree/sub/deeper"; echo one > "$TMP/tree/a"; echo two > "$TMP/tree/sub/b"; head -c 70000 /dev/urandom > "$TMP/tree/sub/deeper/blob"; chmod 640 "$TMP/tree/a" +run "cp -a $TMP/tree $S/tree 2>&1; diff -r $TMP/tree $S/tree && echo same; stat -c %a $S/tree/a; cp -a $S/tree $TMP/back && diff -r $TMP/tree $TMP/back && echo back; rm -r $S/tree" +expect_eq "cp -a there and back" $'same\n640\nback' "$OUT" +run "cd $TMP && tar cf $S/t.tar tree && cd $S && mkdir tx && tar xf t.tar -C tx && diff -r $TMP/tree tx/tree && echo tar-ok; rm -r $S/tx $S/t.tar" +expect_eq "tar into and out of the mount" "tar-ok" "$OUT" +run "rsync -a $TMP/tree/ $S/rs/ && diff -r $TMP/tree $S/rs && echo rsync-ok; sleep 1.1; echo mod > $TMP/tree/a; rsync -a $TMP/tree/ $S/rs/ && cat $S/rs/a; rm -r $S/rs" +expect_eq "rsync -a twice" $'rsync-ok\nmod' "$OUT" +run "cd $S && mkdir repo && cd repo && git init -q . && git config user.email a@b && git config user.name n && echo hi > f && git add f && git commit -qm init && git log --oneline | wc -l && git status --porcelain | wc -l; cd $S && rm -rf repo; ls $S | wc -l" +expect_eq "git init/add/commit inside the mount" $'1\n0\n0' "$OUT" + +echo "# --no-direct-io" +EXTRA=(--no-direct-io) +run "cp $TMP/four $S/nd && cmp $TMP/four $S/nd && echo same; stat -c %s $S/nd; rm $S/nd" +expect_eq "no-direct-io: 4 MiB round trip" $'same\n4194304' "$OUT" +[ "$RC" -eq 0 ] || echo " stderr: $STDERR" +# Buffered writes are per-page without a writeback cache (kernel behaviour); the +# point here is only that a large buffered write is delivered intact. +EXTRA=(--no-direct-io) +head -c 262144 /dev/urandom > "$TMP/w" +WSUM=$(sha256sum <"$TMP/w" | cut -d' ' -f1) +run "cp $TMP/w $S/w && sha256sum < $S/w | cut -d' ' -f1; stat -c %s $S/w; rm $S/w" +expect_eq "no-direct-io: 256 KiB buffered write is intact" "$WSUM"$'\n262144' "$OUT" +EXTRA=() + +echo +echo "passed=$PASSED failed=$FAILED" +[ "$FAILED" -eq 0 ] diff --git a/9ns/test/adv_bridge_stress.sh b/9ns/test/adv_bridge_stress.sh new file mode 100755 index 0000000..b306b28 --- /dev/null +++ b/9ns/test/adv_bridge_stress.sh @@ -0,0 +1,64 @@ +#!/usr/bin/env bash +# Resource and concurrency stress through the bridge against 9proc-demo. +# Usage: bash 9ns/test/adv_bridge_stress.sh <9ns> <9proc-demo> (~1-2 min; part of zig build 9ns-adv) +set -u +NS=$(realpath "${1:?path to 9ns}") +PROC=$(realpath "${2:?path to 9proc-demo}") +TMP=$(mktemp -d "${TMPDIR:-/tmp}/9ns-stress.XXXXXX") +M=/mnt/9p +S=$M/scratch +FAILED=0 +PASSED=0 +SRVPID= +cleanup() { [ -n "$SRVPID" ] && kill "$SRVPID" 2>/dev/null; rm -rf "$TMP"; } +trap cleanup EXIT +if ! unshare -Urm true 2>/dev/null || [ ! -c /dev/fuse ]; then echo "SKIP: no user namespaces or /dev/fuse"; exit 0; fi +pass() { PASSED=$((PASSED + 1)); echo "ok - $1"; } +fail() { FAILED=$((FAILED + 1)); echo "FAIL - $1"; shift; [ $# -gt 0 ] && printf ' %s\n' "$@"; } +expect_eq() { if [ "$2" = "$3" ]; then pass "$1"; else fail "$1" "expected: $(printf %q "$2")" "actual: $(printf %q "$3")"; fi; } + +SOCK=$TMP/i.sock +"$PROC" --unix "$SOCK" >"$TMP/srv.out" 2>&1 & +SRVPID=$! +for _ in $(seq 1 100); do [ -S "$SOCK" ] && break; sleep 0.02; done +run() { OUT=$(timeout 600 "$NS" --unix "$SOCK" -- sh -c "$1" 2>"$TMP/stderr"); RC=$?; STDERR=$(cat "$TMP/stderr"); } + +echo "# 100k+ 9P operations in one session; RSS must plateau" +# Each iteration: create+write+close, open+read+close, unlink, plus a failing lookup: ~15 RPCs. +run "rss() { grep VmRSS /proc/\$PPID/status | awk '{print \$2}'; } +i=0; while [ \$i -lt 8000 ]; do echo \$i > $S/s; cat $S/s > /dev/null; rm $S/s; cat $S/none 2>/dev/null; i=\$((i+1)); if [ \$i -eq 2000 ]; then rss; fi; done; rss; ls $S | wc -l" +r1=$(printf '%s\n' "$OUT" | sed -n 1p); r2=$(printf '%s\n' "$OUT" | sed -n 2p); left=$(printf '%s\n' "$OUT" | sed -n 3p) +expect_eq "scratch left clean" "0" "$left" +if [ -n "$r1" ] && [ -n "$r2" ] && [ "$r2" -le $((r1 + 4096)) ]; then pass "RSS at 2000 iterations = $r1 KiB, at 8000 = $r2 KiB"; else fail "RSS grows: $r1 -> $r2 KiB" "$STDERR"; fi +case "$STDERR" in *leak*) fail "allocator reported leaks" "$STDERR";; *) pass "no leak report from the debug allocator";; esac + +echo "# eight processes hammering the mount concurrently" +run "mkdir $S/par; for p in 1 2 3 4 5 6 7 8; do ( + d=$S/par/p\$p; mkdir \$d; i=0; bad=0 + while [ \$i -lt 300 ]; do + printf '%s-%s' \$p \$i > \$d/f\$((i % 7)); v=\$(cat \$d/f\$((i % 7))); [ \"\$v\" = \"\$p-\$i\" ] || bad=\$((bad+1)) + mkdir \$d/dd; rmdir \$d/dd; ls \$d > /dev/null; i=\$((i+1)) + done; rm -r \$d; echo \$p:\$bad ) & done; wait; ls $S/par | wc -l; rmdir $S/par" +expect_eq "all workers verified their own data" "1:0 2:0 3:0 4:0 5:0 6:0 7:0 8:0" "$(printf '%s\n' "$OUT" | grep ':' | sort | tr '\n' ' ' | sed 's/ $//')" +expect_eq "parallel tree fully removed" "0" "$(printf '%s\n' "$OUT" | grep -v ':')" + +echo "# a process killed mid-read and mid-write" +run "head -c 8000000 /dev/urandom > $S/kb; (cat $S/kb > /dev/null & sleep 0.05; kill -9 \$!; wait \$!) 2>/dev/null; (cat /dev/zero > $S/kw & sleep 0.05; kill -9 \$!; wait \$!) 2>/dev/null; sha256sum < $S/kb | cut -c1-8 > /dev/null && echo readable; [ -f $S/kw ] && echo written; rm $S/kb $S/kw; ls $S | wc -l" +expect_eq "survives SIGKILL mid-read/mid-write" $'readable\nwritten\n0' "$OUT" + +echo "# many open handles at once (fh counter, fid table)" +run "python3 - <<'EOF' +import os +d='$S/fh'; os.mkdir(d) +fds=[] +for i in range(1500): + fd=os.open(f'{d}/h{i%50}', os.O_CREAT|os.O_RDWR, 0o644); os.write(fd, b'z'); fds.append(fd) +for fd in fds: os.close(fd) +for i in range(50): os.unlink(f'{d}/h{i}') +os.rmdir(d); print('ok') +EOF" +expect_eq "1500 simultaneous handles" "ok" "$OUT" + +echo +echo "passed=$PASSED failed=$FAILED" +[ "$FAILED" -eq 0 ] diff --git a/9ns/test/adv_ns_process.sh b/9ns/test/adv_ns_process.sh new file mode 100755 index 0000000..db759ca --- /dev/null +++ b/9ns/test/adv_ns_process.sh @@ -0,0 +1,202 @@ +#!/usr/bin/env bash +# Adversarial regression tests for 9ns/src/ns.zig and 9ns/src/main.zig: process, +# namespace, signal and CLI handling. Real namespaces, real FUSE. +# Usage: bash 9ns/test/adv_ns_process.sh <9ns> <9proc-demo> (part of zig build 9ns-adv) +# Exit 0 on success (or when the machine cannot run the tests), 1 on failure. +set -u + +NS=$(realpath "${1:?path to 9ns}") +PROC=$(realpath "${2:?path to 9proc-demo}") +# Unix socket paths are limited to ~107 bytes; keep the temp dir short. +TMP=$(mktemp -d "${TMPDIR:-/tmp}/9padv.XXXXXX") +PIDS=() +FAILED=0 +PASSED=0 + +cleanup() { + for p in "${PIDS[@]:-}"; do [ -n "$p" ] && kill "$p" 2>/dev/null; done + rm -rf "$TMP" +} +trap cleanup EXIT + +if ! unshare -Urm true 2>/dev/null; then echo "SKIP: unprivileged user namespaces unavailable"; exit 0; fi +if [ ! -c /dev/fuse ]; then echo "SKIP: /dev/fuse missing"; exit 0; fi + +pass() { PASSED=$((PASSED + 1)); echo "ok - $1"; } +fail() { FAILED=$((FAILED + 1)); echo "FAIL - $1"; shift; [ $# -gt 0 ] && printf ' %s\n' "$@"; } +expect_eq() { if [ "$2" = "$3" ]; then pass "$1"; else fail "$1" "expected: $(printf %q "$2")" "actual: $(printf %q "$3")"; fi; } +expect_contains() { case "$3" in *"$2"*) pass "$1" ;; *) fail "$1" "missing: $(printf %q "$2")" "in: $(printf %q "$3")" ;; esac; } + +SOCK=$TMP/s +"$PROC" --unix "$SOCK" & +PIDS+=($!) +for _ in $(seq 1 100); do [ -S "$SOCK" ] && break; sleep 0.05; done +[ -S "$SOCK" ] || { echo "9proc-demo did not create $SOCK"; exit 1; } +MI_BEFORE=$(grep -v " $TMP" /proc/self/mountinfo | sort) + +TIMEOUT=$(command -v timeout) +run() { "$TIMEOUT" 60 "$NS" --unix "$SOCK" "$@"; } + +echo "# CLI" +expect_eq "--help goes to stdout, exit 0" "Usage: 9ns" "$(run --help 2>/dev/null | head -1 | cut -c1-10; )" +expect_eq "--help exit code" "0" "$("$NS" --help >/dev/null 2>&1; echo $?)" +expect_eq "--version on stdout" "9ns" "$("$NS" --version 2>/dev/null | cut -d' ' -f1)" +expect_eq "single-dash typo is a usage error, not a program" "125" "$(run -mount /x -- true 2>/dev/null; echo $?)" +expect_contains "single-dash typo message" "unknown option -mount" "$(run -mount /x -- true 2>&1)" +expect_eq "--unix= empty is a usage error" "125" "$("$NS" --unix= -- true 2>/dev/null; echo $?)" +expect_contains "--unix= message" "socket path" "$("$NS" --unix= -- true 2>&1)" +expect_eq "--mount '' is a usage error" "125" "$(run --mount '' -- true 2>/dev/null; echo $?)" +expect_eq "--msize huge rejected" "125" "$(run --msize 4294967295 -- true 2>/dev/null; echo $?)" +expect_eq "--msize 16 MiB accepted" "ok" "$(run --msize 16777216 -- sh -c 'echo ok')" +expect_contains "empty program name is reported" "empty program name" "$(run -- '' 2>&1)" +expect_eq "empty program name exit" "125" "$(run -- '' 2>/dev/null; echo $?)" +expect_eq "empty \$SHELL falls back to /bin/sh" "0" "$(SHELL= run -- /dev/null 2>&1; echo $?)" +expect_eq "--fd with a closed descriptor fails early" "125" "$("$NS" --fd 987 -- true 2>/dev/null; echo $?)" +expect_contains "--fd bad descriptor message" "--fd 987: EBADF" "$("$NS" --fd 987 -- true 2>&1)" + +echo "# exec failures" +expect_eq "not found is 127" "127" "$(run -- no-such-program-9ns 2>/dev/null; echo $?)" +expect_eq "PATH element that is a file: still 127" "127" "$(PATH=/etc/passwd run -- true 2>/dev/null; echo $?)" +expect_contains "PATH element that is a file: message" "exec true: E" "$(PATH=/etc/passwd run -- true 2>&1)" +printf '#!/bin/sh\necho no\n' >"$TMP/nx"; chmod 644 "$TMP/nx" +expect_eq "non-executable is 126" "126" "$(run -- "$TMP/nx" 2>/dev/null; echo $?)" +mkdir -p "$TMP/p1" "$TMP/p2"; cp "$TMP/nx" "$TMP/p1/prog"; printf '#!/bin/sh\necho right\n' >"$TMP/p2/prog"; chmod 755 "$TMP/p2/prog" +expect_eq "non-executable first in PATH, executable later" "right" "$(PATH=$TMP/p1:$TMP/p2 run -- prog)" +expect_eq "non-executable only in PATH is 126" "126" "$(PATH=$TMP/p1 run -- prog 2>/dev/null; echo $?)" +expect_eq "argv[0] preserved" "sh" "$(run -- sh -c 'echo $0')" +expect_eq "PATH unset uses default" "ok" "$(env -u PATH "$NS" --unix "$SOCK" -- sh -c 'echo ok')" + +echo "# fd hygiene" +# 9ns passes inherited descriptors through untouched, so compare with what a +# plain child of this script sees (the runner may itself hold extra fds). +FD_LIST='ls /proc/self/fd | grep -v "^3$" | sort -n | tr "\n" " " | sed "s/ $//"' +FD_BASE=$(sh -c "$FD_LIST") +expect_eq "no extra fds in the program (unix)" "$FD_BASE" "$(run -- sh -c "$FD_LIST")" +expect_eq "no extra fds in the program (spawn)" "$FD_BASE" "$("$TIMEOUT" 60 "$NS" --spawn "$PROC --stdio" -- sh -c "$FD_LIST")" +expect_eq "--fd transport does not leak into the program" "0 1 2" "$(python3 - "$NS" "$SOCK" <<'EOF' +import socket, subprocess, sys, os +s = socket.socket(socket.AF_UNIX, socket.SOCK_STREAM); s.connect(sys.argv[2]) +r = subprocess.run([sys.argv[1], "--fd", str(s.fileno()), "--", "sh", "-c", + 'ls /proc/self/fd | grep -v "^3$" | sort -n | tr "\n" " " | sed "s/ $//"'], + pass_fds=[s.fileno()], capture_output=True, text=True) +print(r.stdout.strip()) +EOF +)" + +echo "# signals" +expect_eq "SIGTERM forwarded" "143" "$(run -- sh -c 'kill -TERM $PPID; sleep 5; echo alive' >/dev/null 2>&1; echo $?)" +expect_eq "SIGHUP forwarded" "129" "$(run -- sh -c 'kill -HUP $PPID; sleep 5; echo alive' >/dev/null 2>&1; echo $?)" +expect_eq "SIGINT to 9ns is ignored while the child lives" "still-here" "$(run -- sh -c 'kill -INT $PPID; sleep 0.3; echo still-here')" +# Ctrl-C from the tty must not kill the --spawn server (same process group). +expect_eq "Ctrl-C on the tty leaves the --spawn server alive" "ok" "$(timeout 30 python3 - "$NS" "$PROC" <<'EOF' +import os, pty, sys, time, select +P, I = sys.argv[1], sys.argv[2] +prog = ["python3", "-c", """ +import os, signal, sys, time +signal.signal(signal.SIGINT, lambda *a: None) +m = os.environ['NINE_MOUNT'] +open(m + '/build/optimize').read() +sys.stdin.readline() +try: + open(m + '/build/optimize').read(); print('ok', flush=True) +except Exception as e: + print('mount dead:', e, flush=True) +"""] +pid, fd = pty.fork() +if pid == 0: + os.execv(P, [P, "--spawn", I + " --stdio", "--"] + prog) +out = b"" +def rd(t): + global out + end = time.time() + t + while time.time() < end: + r, _, _ = select.select([fd], [], [], 0.1) + if r: + try: d = os.read(fd, 4096) + except OSError: return + if not d: return + out += d +rd(1.5); os.write(fd, b"\x03"); rd(0.7); os.write(fd, b"\n"); rd(3) +os.waitpid(pid, 0) +print(out.decode(errors="replace").replace("^C", "").strip().splitlines()[-1] if out.strip() else "no output") +EOF +)" +# The --spawn server dying mid-session is reaped (no zombie) and does not end the session. +OUT=$("$TIMEOUT" 60 "$NS" --spawn "$PROC --stdio" -- sh -c 'srv=$(cat $NINE_MOUNT/runtime/pid); kill -TERM $srv; sleep 0.5; st=$(ps -o stat= -p $srv 2>/dev/null); echo "${st:-gone}"; exit 5' 2>/dev/null); RC=$? +expect_eq "server death mid-session: exit status still the child's, server reaped (no zombie)" "5 gone" "$RC $OUT" +# A server that never answers: once the child is dead, SIGTERM must end 9ns. +cat >"$TMP/hang.py" <<'EOF' +import struct, os, sys, time +def rd(n): + b = b"" + while len(b) < n: + c = os.read(0, n - len(b)) + if not c: sys.exit(0) + b += c + return b +while True: + size, = struct.unpack("/dev/null & +HP=$! +sleep 1; kill -TERM $HP +START=$(date +%s) +for _ in $(seq 1 100); do kill -0 $HP 2>/dev/null || break; sleep 0.1; done +if kill -0 $HP 2>/dev/null; then kill -KILL $HP; RC=hung; else wait $HP; RC=$?; fi +expect_eq "hung server: one SIGTERM ends 9ns once the child is dead (watchdog)" "143" "$RC" +expect_eq "hung server: exit was prompt" "yes" "$([ $(( $(date +%s) - START )) -lt 8 ] && echo yes)" +pkill -f "$TMP/hang.py" 2>/dev/null + +echo "# child/parent protocol" +if command -v strace >/dev/null 2>&1 && strace -qq -e trace=none true 2>/dev/null; then + expect_eq "child killed before handoff" "125" "$(strace -f -qq -e trace=unshare -e inject=unshare:signal=KILL -o /dev/null timeout 20 "$NS" --unix "$SOCK" -- true 2>/dev/null; echo $?)" + expect_contains "child killed before handoff: message" "child exited before reporting" "$(strace -f -qq -e trace=unshare -e inject=unshare:signal=KILL -o /dev/null timeout 20 "$NS" --unix "$SOCK" -- true 2>&1)" + expect_eq "status handoff fails" "125" "$(strace -f -qq -e trace=sendmsg -e inject=sendmsg:error=EPIPE -o /dev/null timeout 20 "$NS" --unix "$SOCK" -- true 2>/dev/null; echo $?)" + # recvmsg skipped (returns 1 without the fd): the child must be killed, not exec'd onto a dead mount. + OUT=$(strace -f -qq -e trace=recvmsg -e inject=recvmsg:retval=1:when=1 -o /dev/null timeout 20 "$NS" --unix "$SOCK" -- sh -c 'echo child-ran' 2>&1; echo "rc=$?") + expect_contains "truncated fd handoff: child not exec'd" "rc=125" "$OUT" + expect_eq "truncated fd handoff: program never ran" "no" "$(case "$OUT" in *child-ran*) echo yes;; *) echo no;; esac)" + expect_contains "fuse mount failure is reported" "mount fuse: EPERM" "$(strace -f -qq -e trace=mount -e inject=mount:error=EPERM:when=2 -o /dev/null timeout 20 "$NS" --unix "$SOCK" --mount "$TMP/mp" -- true 2>&1)" +else + echo "skip - strace unavailable (child failure injection)" +fi +expect_contains "fork failure is reported" "fork: E" "$(python3 -c " +import resource, os +resource.setrlimit(resource.RLIMIT_NPROC, (1, 1)) +os.execv('$NS', ['$NS', '--unix', '$SOCK', '--', 'true'])" 2>&1)" + +echo "# mountpoint policy" +ln -s /nonexistent "$TMP/dangling" +expect_contains "dangling symlink mountpoint" "dangling symlink" "$(run --mount "$TMP/dangling" -- true 2>&1)" +expect_eq "refuse to shadow / via /proc/self/root" "125" "$(run --mount /proc/self/root/x9p -- true 2>/dev/null; echo $?)" +expect_contains "refuse to shadow / via /proc/self/root: message" "refusing to shadow /" "$(run --mount /proc/self/root/x9p -- true 2>&1)" +expect_eq "refuse to shadow under /proc" "125" "$(run --mount /proc/self/fd/x9p -- true 2>/dev/null; echo $?)" +if [ "$(ls -A /usr/lib | wc -l)" -gt 4096 ]; then + expect_contains "parent with >4096 entries refused" "more than 4096 entries" "$(run --mount /usr/lib/x9p -- true 2>&1)" +else + echo "skip - no root-owned directory with >4096 entries" +fi +expect_eq "shadowed /run keeps its entries" "$(ls -A /run | sort | tr '\n' ' ')" "$(run --mount /run/x9p -- sh -c 'ls -A /run | grep -v "^x9p$" | sort | tr "\n" " "')" +expect_eq "mountpoint with spaces" "ok" "$(mkdir -p "$TMP/with space" && run --mount "$TMP/with space" -- sh -c '[ -f "$NINE_MOUNT/README" ] && echo ok')" +expect_eq "mountpoint is a file" "125" "$(run --mount "$TMP/nx" -- true 2>/dev/null; echo $?)" + +echo "# leaks" +for i in $(seq 1 30); do run -- sh -c 'cat $NINE_MOUNT/build/optimize >/dev/null' 2>/dev/null; done +BG=(); for i in $(seq 1 10); do ( run -- sh -c 'cat $NINE_MOUNT/build/optimize >/dev/null' 2>/dev/null ) & BG+=($!); done; wait "${BG[@]}" # not a bare wait: that would also wait for the server +sleep 0.3 +expect_eq "no stray 9ns processes" "" "$(pgrep -f "^$NS " | tr '\n' ' ')" +expect_eq "no stray --stdio servers" "" "$(pgrep -f "$PROC --stdio" | tr '\n' ' ')" +expect_eq "host mount table untouched" "same" "$([ "$MI_BEFORE" = "$(grep -v " $TMP" /proc/self/mountinfo | sort)" ] && echo same || echo changed)" + +echo +echo "passed=$PASSED failed=$FAILED" +[ "$FAILED" -eq 0 ] diff --git a/9ns/test/adversarial.sh b/9ns/test/adversarial.sh new file mode 100755 index 0000000..71c6f44 --- /dev/null +++ b/9ns/test/adversarial.sh @@ -0,0 +1,15 @@ +#!/usr/bin/env bash +# Runs every 9ns adversarial suite in sequence (hostile servers, FUSE +# semantics, process/namespace edge cases, stress). The suites aimed at the +# 9proc server itself live in 9proc/test (zig build 9proc-adv). +# Usage: bash 9ns/test/adversarial.sh <9ns> <9proc-demo> (zig build 9ns-adv) +set -u +NS=${1:?path to 9ns} +PROC=${2:?path to 9proc-demo} +HERE=$(cd "$(dirname "$0")" && pwd) +status=0 +for suite in adv_ns_process adv_bridge_hostile adv_bridge_semantics adv_bridge_stress; do + echo "### $suite" + if bash "$HERE/$suite.sh" "$NS" "$PROC"; then echo "### $suite: ok"; else echo "### $suite: FAILED"; status=1; fi +done +exit $status diff --git a/9ns/test/integration.sh b/9ns/test/integration.sh new file mode 100755 index 0000000..5e0ad13 --- /dev/null +++ b/9ns/test/integration.sh @@ -0,0 +1,160 @@ +#!/usr/bin/env bash +# Integration tests for 9ns: real user+mount namespaces, real FUSE, real 9P servers. +# Usage: bash 9ns/test/integration.sh <9ns> <9proc-demo> (zig build 9ns-itest) +# Exit 0 on success (or when the machine cannot run the tests), 1 on failure. +set -u + +NS=$(realpath "${1:?path to 9ns}") +PROC=$(realpath "${2:?path to 9proc-demo}") +TMP=$(mktemp -d "${TMPDIR:-/tmp}/9ns-itest.XXXXXX") +PIDS=() +FAILED=0 +PASSED=0 +M=/mnt/9p + +cleanup() { + for p in "${PIDS[@]:-}"; do [ -n "$p" ] && kill "$p" 2>/dev/null; done + rm -rf "$TMP" +} +trap cleanup EXIT + +if ! unshare -Urm true 2>/dev/null; then + echo "SKIP: unprivileged user namespaces unavailable"; exit 0 +fi +if [ ! -c /dev/fuse ]; then + echo "SKIP: /dev/fuse missing"; exit 0 +fi + +pass() { PASSED=$((PASSED + 1)); echo "ok - $1"; } +fail() { FAILED=$((FAILED + 1)); echo "FAIL - $1"; shift; [ $# -gt 0 ] && printf ' %s\n' "$@"; } +expect_eq() { # name expected actual + if [ "$2" = "$3" ]; then pass "$1"; else fail "$1" "expected: $(printf %q "$2")" "actual: $(printf %q "$3")"; fi +} +expect_contains() { # name needle haystack + case "$3" in *"$2"*) pass "$1" ;; *) fail "$1" "missing: $(printf %q "$2")" "in: $(printf %q "$3")" ;; esac +} + +wait_socket() { # path + for _ in $(seq 1 100); do [ -S "$1" ] && return 0; sleep 0.05; done + return 1 +} + +# run_in "" — run inside a namespace with the current transport ($TRANSPORT array). +run_in() { timeout 60 "$NS" "${TRANSPORT[@]}" -- sh -c "$1" 2>"$TMP/stderr"; } + +# --- scratch battery: works against any writable 9P tree rooted at $1 (relative to mount) --- +scratch_battery() { # label scratchdir + local label=$1 S=$2 + + expect_eq "$label: create+append+read" $'hello\nworld' "$(run_in "echo hello > $M/$S/a && echo world >> $M/$S/a && cat $M/$S/a")" + expect_eq "$label: overwrite" "x" "$(run_in "echo x > $M/$S/a && cat $M/$S/a")" + expect_eq "$label: stat size after overwrite" "2" "$(run_in "stat -c %s $M/$S/a")" + expect_eq "$label: truncate" "0" "$(run_in "truncate -s 0 $M/$S/a && stat -c %s $M/$S/a")" + expect_eq "$label: truncate extend" "10" "$(run_in "truncate -s 10 $M/$S/a && stat -c %s $M/$S/a")" + expect_eq "$label: mkdir -p nested" "directory" "$(run_in "mkdir -p $M/$S/d1/d2/d3 && stat -c %F $M/$S/d1/d2/d3")" + expect_eq "$label: rename within dir" "moved" "$(run_in "echo moved > $M/$S/d1/d2/f && mv $M/$S/d1/d2/f $M/$S/d1/d2/g && cat $M/$S/d1/d2/g")" + # mv(1) silently falls back to copy+delete on EXDEV, so probe rename(2) directly. + expect_contains "$label: rename across dirs is EXDEV" "EXDEV" "$(run_in "python3 -c 'import os,errno +try: os.rename(\"$M/$S/d1/d2/g\", \"$M/$S/d1/g\") +except OSError as e: print(errno.errorcode[e.errno]) +'")" + expect_eq "$label: rm file" "gone" "$(run_in "rm $M/$S/d1/d2/g && [ ! -e $M/$S/d1/d2/g ] && echo gone")" + expect_eq "$label: rmdir non-empty fails" "1" "$(run_in "rmdir $M/$S/d1 2>/dev/null; echo \$?")" + expect_eq "$label: rmdir chain" "ok" "$(run_in "rmdir $M/$S/d1/d2/d3 $M/$S/d1/d2 $M/$S/d1 && echo ok")" + expect_eq "$label: ENOENT" "1" "$(run_in "cat $M/$S/nope 2>/dev/null; echo \$?")" + expect_eq "$label: ENOENT errno text" "No such file or directory" "$(run_in "cat $M/$S/nope 2>&1 | sed 's/.*: //'")" + + head -c 1048576 /dev/urandom >"$TMP/rand" + local sum; sum=$(sha256sum <"$TMP/rand" | cut -d' ' -f1) + expect_eq "$label: 1 MiB round trip (cp)" "$sum" "$(run_in "cp $TMP/rand $M/$S/big && sha256sum < $M/$S/big | cut -d' ' -f1")" + expect_eq "$label: 1 MiB size" "1048576" "$(run_in "stat -c %s $M/$S/big")" + expect_eq "$label: odd block sizes (dd bs=1000)" "$sum" "$(run_in "dd if=$M/$S/big of=$M/$S/big2 bs=1000 status=none && sha256sum < $M/$S/big2 | cut -d' ' -f1")" + expect_eq "$label: partial read at offset" "$(tail -c 12345 "$TMP/rand" | sha256sum | cut -d' ' -f1)" "$(run_in "tail -c 12345 $M/$S/big | sha256sum | cut -d' ' -f1")" + expect_eq "$label: many small files" "200" "$(run_in "mkdir $M/$S/many && for i in \$(seq 1 200); do echo \$i > $M/$S/many/f\$i; done; ls $M/$S/many | wc -l")" + expect_eq "$label: find count" "201" "$(run_in "find $M/$S/many | wc -l")" + expect_eq "$label: readdir contents" "f1 f100 f200" "$(run_in "cd $M/$S/many && ls f1 f100 f200 | tr '\n' ' ' | sed 's/ \$//'")" + expect_eq "$label: cleanup many" "0" "$(run_in "rm -r $M/$S/many $M/$S/big $M/$S/big2 $M/$S/a; ls $M/$S | wc -l")" +} + +# ============================================================================ +echo "# 9proc-demo over a Unix socket" +SOCK=$TMP/9proc.sock +"$PROC" --unix "$SOCK" & +PIDS+=($!) +wait_socket "$SOCK" || { echo "9proc-demo did not create $SOCK"; exit 1; } +TRANSPORT=(--unix "$SOCK") + +expect_eq "zig_version" "$(zig version)" "$(run_in "cat $M/build/zig_version")" +expect_eq "mount exported" "$M" "$(run_in 'echo $NINE_MOUNT')" +expect_contains "root listing" "build" "$(run_in "ls $M")" +expect_contains "root listing has scratch" "scratch" "$(run_in "ls $M")" +expect_eq "README size > 0" "yes" "$(run_in "[ \$(stat -c %s $M/README) -gt 0 ] && echo yes")" +expect_eq "README readable" "yes" "$(run_in "[ -s $M/README ] && head -c 1 $M/README >/dev/null && echo yes")" +expect_eq "fn/now numeric" "num" "$(run_in "cat $M/runtime/fn/now | grep -Eq '^[0-9]+\$' && echo num")" +expect_eq "fn listing from comptime" "yes" "$(run_in "ls $M/runtime/fn | grep -q hostname && echo yes")" +expect_eq "ctl round trip" "5" "$(run_in "echo 'add 2 3' > $M/runtime/ctl && cat $M/runtime/ctl")" +expect_eq "ctl echo" "hi there" "$(run_in "echo 'echo hi there' > $M/runtime/ctl && cat $M/runtime/ctl")" +expect_contains "comptime types" "Qid" "$(run_in "ls $M/comptime/types")" +expect_eq "comptime size of Qid" "16" "$(run_in "cat $M/comptime/types/Qid/size")" +expect_eq "runtime pid is server pid" "${PIDS[-1]}" "$(run_in "cat $M/runtime/pid")" +expect_eq "exit status propagates" "7" "$(run_in 'exit 7'; echo $?)" +expect_eq "mount is fuse" "yes" "$(run_in "grep -q \"^9ns $M fuse\" /proc/mounts && echo yes")" +expect_eq "host /mnt entries still visible" "$(ls -A /mnt | sort | tr '\n' ' ')" "$(run_in "ls -A /mnt | grep -v '^9p\$' | sort | tr '\n' ' '")" +expect_eq "host mount table untouched" "no" "$(grep -q " $M " /proc/self/mountinfo && echo yes || echo no)" +scratch_battery "9proc-demo" scratch + +echo "# nested 9ns" +expect_eq "nested mount" "$(zig version)" "$(run_in "$NS --unix $SOCK --mount $TMP/inner -- sh -c 'cat \$NINE_MOUNT/build/zig_version'")" + +echo "# --mount variants" +mkdir -p "$TMP/mnt" +expect_eq "--mount existing dir" "ok" "$(timeout 60 "$NS" --unix "$SOCK" --mount "$TMP/mnt" -- sh -c "[ -f $TMP/mnt/README ] && echo ok")" +expect_eq "--mount relative" "ok" "$(cd "$TMP" && timeout 60 "$NS" --unix "$SOCK" --mount rel -- sh -c "[ -f $TMP/rel/README ] && echo ok")" +expect_eq "--mount missing under /" "125" "$(timeout 60 "$NS" --unix "$SOCK" --mount /nonexistent-9ns-dir -- true 2>/dev/null; echo $?)" + +echo "# lifecycle" +START=$(date +%s) +expect_eq "background grandchild does not block exit" "3" "$(run_in 'sleep 30 >/dev/null 2>&1 & exit 3'; echo $?)" +expect_eq "exit was prompt" "yes" "$([ $(( $(date +%s) - START )) -lt 10 ] && echo yes)" +expect_eq "SIGTERM forwarded" "143" "$(timeout 60 "$NS" --unix "$SOCK" -- sh -c 'kill -TERM $PPID; sleep 5; echo alive' >/dev/null 2>&1; echo $?)" + +echo "# --spawn transport" +TRANSPORT=(--spawn "$PROC --stdio") +expect_eq "spawn: zig_version" "$(zig version)" "$(run_in "cat $M/build/zig_version")" +expect_eq "spawn: ctl" "7" "$(run_in "echo 'add 3 4' > $M/runtime/ctl && cat $M/runtime/ctl")" +expect_eq "spawn: stateful sequence in one session" 'hello world 12 moved 0' "$(run_in "cd $M/scratch && echo hello > a && echo world >> a && cat a && stat -c %s a && mkdir d && echo moved > d/f && mv d/f d/g && cat d/g && rm d/g && rmdir d && rm a && ls | wc -l" | tr '\n' ' ' | sed 's/ $//')" + +echo "# --tcp transport" +PORT=$(( 20000 + RANDOM % 20000 )) +"$PROC" --tcp "127.0.0.1:$PORT" & +PIDS+=($!) +sleep 0.3 +TRANSPORT=(--tcp "127.0.0.1:$PORT") +expect_eq "tcp: zig_version" "$(zig version)" "$(run_in "cat $M/build/zig_version")" +expect_eq "tcp: scratch" "tcp" "$(run_in "echo tcp > $M/scratch/t && cat $M/scratch/t && rm $M/scratch/t")" + +echo "# server death" +"$PROC" --unix "$TMP/dying.sock" & +DYING=$! +wait_socket "$TMP/dying.sock" +TRANSPORT=(--unix "$TMP/dying.sock") +OUT=$(run_in "cat $M/build/optimize >/dev/null && kill $DYING && sleep 0.3; cat $M/build/optimize 2>&1 >/dev/null | sed 's/.*: //'; echo status=\$?") +expect_contains "server death yields an error, not a hang" "status=0" "$OUT" +expect_eq "server death errno text" "yes" "$(case "$OUT" in *"Input/output error"*|*"Transport endpoint is not connected"*) echo yes;; *) echo "no: $OUT";; esac)" + +echo "# plan9port ramfs (independent 9P2000 implementation)" +if [ -x /usr/lib/plan9/bin/ramfs ]; then + mkdir -p "$TMP/p9ns" + NAMESPACE=$TMP/p9ns /usr/lib/plan9/bin/ramfs -s ramfs + wait_socket "$TMP/p9ns/ramfs" || echo "ramfs socket missing" + PIDS+=($(pgrep -f "9pserve -u unix!$TMP/p9ns/ramfs")) + TRANSPORT=(--unix "$TMP/p9ns/ramfs") + expect_eq "ramfs: mkdir scratch" "ok" "$(run_in "mkdir $M/scratch && echo ok")" + scratch_battery "ramfs" scratch +else + echo "skip - plan9port ramfs not installed" +fi + +echo +echo "passed=$PASSED failed=$FAILED" +[ "$FAILED" -eq 0 ] -- cgit v1.3