From b05abcba3ea09ea106ad28364c6e40a3ec31b890 Mon Sep 17 00:00:00 2001 From: Gabriel Schneider Date: Sat, 19 Sep 2026 21:26:05 -0300 Subject: Add 9player and introspect as programs beside the library 9player/: FUSE mount CLI that mounts a 9P2000 tree into a fresh user+mount namespace and runs a program in it (no root, no libfuse, no libc). introspect/: the 9P debug/introspection library (freestanding core, value renderers, Linux probe with threads/stacks/memory/breakpoints/panics) and its demo server. Each has its own build fragment; the root build.zig wires them behind -D9player/-Dintrospect with namespaced steps (9player-itest, introspect-check-freestanding, programs-test, ...) and exports the introspect module for dependents. This is the layout for related programs. Co-Authored-By: Claude Fable 5.1 --- .gitignore | 1 + 9player/README.md | 156 ++ 9player/build.zig | 71 + 9player/docs/DESIGN.md | 405 +++++ 9player/src/bridge.zig | 971 +++++++++++ 9player/src/fuse.zig | 653 +++++++ 9player/src/main.zig | 444 +++++ 9player/src/nine.zig | 756 ++++++++ 9player/src/ns.zig | 1082 ++++++++++++ 9player/test/adv_bridge_hostile.py | 487 ++++++ 9player/test/adv_bridge_hostile.sh | 181 ++ 9player/test/adv_bridge_semantics.sh | 205 +++ 9player/test/adv_bridge_stress.sh | 64 + 9player/test/adv_ns_process.sh | 202 +++ 9player/test/adversarial.sh | 15 + 9player/test/integration.sh | 160 ++ README.md | 29 + build.zig | 57 + build.zig.zon | 2 +- docs/design.md | 14 + introspect/README.md | 130 ++ introspect/build.zig | 169 ++ introspect/demo/main.zig | 423 +++++ introspect/docs/LIBRARY.md | 287 ++++ introspect/src/core.zig | 2656 +++++++++++++++++++++++++++++ introspect/src/freestanding_check.zig | 71 + introspect/src/linux/debug.zig | 1458 ++++++++++++++++ introspect/src/linux/probe.zig | 829 +++++++++ introspect/src/linux/provider.zig | 604 +++++++ introspect/src/linux/runtime.zig | 90 + introspect/src/root.zig | 24 + introspect/src/scratch.zig | 689 ++++++++ introspect/src/vars.zig | 824 +++++++++ introspect/test/adv_core_hostile.py | 1018 +++++++++++ introspect/test/adv_core_hostile.sh | 10 + introspect/test/adv_introspect_hostile.py | 999 +++++++++++ introspect/test/adv_introspect_hostile.sh | 10 + introspect/test/adv_linux_probe.py | 421 +++++ introspect/test/adv_linux_probe.sh | 10 + introspect/test/adversarial.sh | 19 + introspect/test/debug.sh | 84 + 41 files changed, 16779 insertions(+), 1 deletion(-) create mode 100644 9player/README.md create mode 100644 9player/build.zig create mode 100644 9player/docs/DESIGN.md create mode 100644 9player/src/bridge.zig create mode 100644 9player/src/fuse.zig create mode 100644 9player/src/main.zig create mode 100644 9player/src/nine.zig create mode 100644 9player/src/ns.zig create mode 100755 9player/test/adv_bridge_hostile.py create mode 100755 9player/test/adv_bridge_hostile.sh create mode 100755 9player/test/adv_bridge_semantics.sh create mode 100755 9player/test/adv_bridge_stress.sh create mode 100755 9player/test/adv_ns_process.sh create mode 100755 9player/test/adversarial.sh create mode 100755 9player/test/integration.sh create mode 100644 introspect/README.md create mode 100644 introspect/build.zig create mode 100644 introspect/demo/main.zig create mode 100644 introspect/docs/LIBRARY.md create mode 100644 introspect/src/core.zig create mode 100644 introspect/src/freestanding_check.zig create mode 100644 introspect/src/linux/debug.zig create mode 100644 introspect/src/linux/probe.zig create mode 100644 introspect/src/linux/provider.zig create mode 100644 introspect/src/linux/runtime.zig create mode 100644 introspect/src/root.zig create mode 100644 introspect/src/scratch.zig create mode 100644 introspect/src/vars.zig create mode 100755 introspect/test/adv_core_hostile.py create mode 100755 introspect/test/adv_core_hostile.sh create mode 100755 introspect/test/adv_introspect_hostile.py create mode 100755 introspect/test/adv_introspect_hostile.sh create mode 100755 introspect/test/adv_linux_probe.py create mode 100755 introspect/test/adv_linux_probe.sh create mode 100755 introspect/test/adversarial.sh create mode 100755 introspect/test/debug.sh diff --git a/.gitignore b/.gitignore index af7bb5c..42d714d 100644 --- a/.gitignore +++ b/.gitignore @@ -2,3 +2,4 @@ zig-out/ test/differential/bin/ test/differential/results/ +__pycache__/ diff --git a/9player/README.md b/9player/README.md new file mode 100644 index 0000000..77081d6 --- /dev/null +++ b/9player/README.md @@ -0,0 +1,156 @@ +# 9player + +Mount a 9P2000 file tree into a fresh mount namespace and run a program in it, +as a plain user, without touching the host's mount table. + +```sh +9player --unix /run/user/1000/acme -- fish # a shell that sees the tree at /mnt/9p +9player --tcp 127.0.0.1:564 -- claude # an agent that sees it too +9player --spawn 'introspect --stdio' -- bash # start the server yourself, talk over a socketpair +``` + +Inside, the tree is ordinary files: `ls`, `cat`, `echo x > ctl`, editors, +`find`, `rsync`, whatever. `$NINEPLAYER_MOUNT` tells programs where it is +(default `/mnt/9p`). When the program exits, 9player exits with its status +and the namespace, mount and connection disappear. + +## How it works + +The kernel's own `9p` filesystem cannot be mounted inside an unprivileged user +namespace, so 9player is a small FUSE server that speaks 9P2000 to the real +server. There is no libfuse and no libc: `src/fuse.zig` implements the subset +of the kernel FUSE protocol needed, straight from `linux/fuse.h`. + +``` + program (fish/bash/claude) 9player (parent) 9P server + in a new user+mount namespace │ + /mnt/9p ── FUSE ──▶ kernel ────▶│ fuse.zig ─▶ bridge.zig ─▶ nine.zig ──▶ unix / tcp / socketpair + │ (framing) (translation) (cloud9 Client) +``` + +1. The parent connects to the 9P server (version + attach) so failures are + reported before anything is forked. +2. The child does `unshare(CLONE_NEWUSER|CLONE_NEWNS)`, maps its own uid/gid, + makes every mount private, opens `/dev/fuse` (it must be opened inside the + new user namespace), mounts it on the mountpoint and passes the descriptor + back to the parent over `SCM_RIGHTS`, then execs the program. +3. The parent serves FUSE requests by translating them into 9P transactions + (`Twalk`, `Topen`, `Tread`, `Twrite`, `Tcreate`, `Tremove`, `Tstat`, + `Twstat`) until the child exits or the namespace disappears. + +Files are opened with `FOPEN_DIRECT_IO`, so synthetic files that report length +0 (the 9P convention for control files) still read correctly, and `O_TRUNC` +travels inside the 9P open mode (`OTRUNC`) rather than as a separate +truncate. Repeated lookups of the same qid map to the same inode. + +If `/mnt/9p` does not exist and cannot be created (the normal case), 9player +mounts a tmpfs over `/mnt` *inside the namespace only* and bind-mounts every +existing entry of `/mnt` back into it, so nothing is hidden. Pass `--mount DIR` +to use any other directory. + +## Building and testing + +9player lives in the [cloud9](../) repository as `cloud9/9player/`, beside +the 9P2000 protocol library it is built on, and is wired into cloud9's +`build.zig` through the fragment `9player/build.zig`. Everything is run from +the cloud9 root with Zig 0.16: + +```sh +zig build # zig-out/bin/{9player,introspect,cloud9-http,cloud9-probe} +zig build 9player # build and install only zig-out/bin/9player +zig build 9player-test # unit tests (protocol structs, session, bridge, namespace helpers) +zig build 9player-itest # integration tests: real namespaces, real FUSE, + # introspect over unix/tcp/socketpair, and plan9port's + # ramfs when /usr/lib/plan9/bin/ramfs is installed +zig build 9player-adv # adversarial suites: hostile 9P servers, FUSE semantics, + # namespace/signal edge cases, stress (several minutes) +zig build -Doptimize=ReleaseSafe +zig build -D9player=false # leave 9player out (the default on non-Linux targets) +``` + +The integration suites mount the `introspect` demo server (`../introspect`), +so they need `-Dintrospect=true` (the default on Linux), unprivileged user +namespaces (`kernel.unprivileged_userns_clone=1` on distributions that have +the knob), `/dev/fuse` and Python 3; they skip themselves otherwise. +`zig build programs-test` and `programs-itest` run the unit and integration +steps of every program in the repository. + +## Usage + +``` +9player [options] -- PROGRAM [ARGS...] + +Transport (exactly one): + --unix PATH Unix stream socket + --tcp IP:PORT TCP (IPv4/IPv6 literal) + --fd N an already-connected inherited descriptor + --spawn CMD run CMD via /bin/sh -c with a socketpair on its stdin/stdout + +Options: + --mount PATH mountpoint inside the new namespace (default /mnt/9p) + --uname NAME 9P user name (default $USER) + --aname NAME 9P tree to attach (default "") + --msize BYTES maximum 9P message size to request (default 131072, max 16 MiB) + --cache SECONDS attr/entry cache validity, fractional allowed (default 1) + --no-direct-io let the kernel cache file pages (trusts stat length) + --debug trace FUSE and 9P operations on stderr + --help, --version +``` + +PROGRAM defaults to `$SHELL`. Exit status is the program's (`128+signal` if it +was killed); 125 means 9player itself failed (usage, connect, namespace, +mount); 126/127 are exec failures as usual. + +## introspect: a demo 9P server + +`introspect` is a single-binary 9P2000 server whose file tree is the binary +itself: build-time facts, `comptime` reflection and live runtime state. It is +the demo of the [introspect library](../introspect) (`../introspect/demo/main.zig`), +built and installed by `zig build introspect`. + +``` +/README +/build/{zig_version,target,optimize,time,change} captured by build.zig (jj change id, UTC time) +/comptime/types//{name,size,align,fields} @sizeOf/@alignOf/@typeInfo, generated at comptime +/comptime/decls pub declarations of the server module +/runtime/{pid,ppid,uptime,argv,cwd,env,clients} +/runtime/fn/ reading calls a Zig function (hostname, now, random, uname, fib30); + the directory is generated from @typeInfo of the Fns struct +/runtime/ctl write "fib N" | "add A B" | "echo TEXT" | "sleep-ms N", read the result +/scratch/ in-memory read/write tree +``` + +`/runtime/env` exposes the server's whole environment, so serve introspect on +a Unix socket or loopback only. + +```sh +zig-out/bin/introspect --unix /tmp/intro.sock & +zig-out/bin/9player --unix /tmp/intro.sock -- sh -c ' + cat $NINEPLAYER_MOUNT/build/zig_version; echo + cat $NINEPLAYER_MOUNT/comptime/types/Qid/fields + echo "fib 20" > $NINEPLAYER_MOUNT/runtime/ctl; cat $NINEPLAYER_MOUNT/runtime/ctl' +``` + +The same command with `claude -p "explore /mnt/9p ..."` as the program gives an +agent a live, file-shaped view into a running process; that is the intended +use. + +## Limitations + +* One 9P request is in flight at a time; a server read that blocks (event + files) stalls the mount until it returns. +* Base 9P2000 only: no symlinks, ownership, xattrs or locks. Every file is + reported as owned by the invoking user. Cross-directory rename is `EXDEV`. +* No PID namespace and no `/proc` remount. `--tcp` takes IP literals only + (no libc, no resolver). +* Linux only. + +## Relation to cloud9 + +9player consumes cloud9 as the module `cloud9` and keeps all mounting, +namespace and process policy on its side, which is what cloud9's design asks +of applications. It ships from the cloud9 repository as the sibling directory +`9player/` (sources in `src/`, suites in `test/`, this README and +`docs/DESIGN.md`) with a build fragment that the root `build.zig` enables with +`-D9player` on Linux targets; nothing in the code depends on that layout. +`docs/DESIGN.md` has the full module contracts. diff --git a/9player/build.zig b/9player/build.zig new file mode 100644 index 0000000..f3d5c1b --- /dev/null +++ b/9player/build.zig @@ -0,0 +1,71 @@ +//! Build fragment for 9player: mount a 9P2000 tree into a fresh mount +//! namespace via FUSE and run a program in it (Linux only, no libc). It is +//! `@import`ed by the root build.zig and called with the root builder, so +//! every `b.path(...)` here is relative to the cloud9 root (hence the +//! `9player/` prefix), every option is defined by the root (no +//! `standardTargetOptions` here) and every step it registers lands in the +//! root's step list under the `9player` prefix. +//! +//! Steps: 9player, 9player-test, 9player-itest, 9player-adv. +const std = @import("std"); + +/// What the root passes in. The root owns target/optimize resolution and the +/// cloud9 module; the integration suites also need a 9P server to mount, +/// which is the introspect demo built by the sibling fragment. +pub const Context = struct { + target: std.Build.ResolvedTarget, + optimize: std.builtin.OptimizeMode, + cloud9: *std.Build.Module, + /// The `introspect` demo server; null when introspect is disabled, in + /// which case the end-to-end steps exist but fail with a notice. + introspect_demo: ?*std.Build.Step.Compile, +}; + +pub const Artifacts = struct { + /// The 9player executable (also installed by the plain `zig build`). + exe: *std.Build.Step.Compile, + /// `9player-test`, `9player-itest`, `9player-adv`. + test_step: *std.Build.Step, + itest_step: *std.Build.Step, + adv_step: *std.Build.Step, +}; + +pub fn add(b: *std.Build, ctx: Context) Artifacts { + const target = ctx.target; + const optimize = ctx.optimize; + std.debug.assert(target.result.os.tag == .linux); // the root only enables 9player on Linux + + const player_mod = b.createModule(.{ + .root_source_file = b.path("9player/src/main.zig"), + .target = target, + .optimize = optimize, + .imports = &.{.{ .name = "cloud9", .module = ctx.cloud9 }}, + }); + const player = b.addExecutable(.{ .name = "9player", .root_module = player_mod }); + const install = b.addInstallArtifact(player, .{}); + b.getInstallStep().dependOn(&install.step); + b.step("9player", "Build and install only the 9player binary").dependOn(&install.step); + + // Unit tests: protocol structs, session, bridge, namespace helpers (main.zig + // reaches every module). + const test_step = b.step("9player-test", "Run 9player's unit tests"); + test_step.dependOn(&b.addRunArtifact(b.addTest(.{ .root_module = player_mod })).step); + + // End-to-end suites: real namespaces, real FUSE, a real 9P server. + const itest = b.step("9player-itest", "Run 9player/test/integration.sh (needs unprivileged user namespaces and /dev/fuse)"); + const adv = b.step("9player-adv", "Run 9player/test/adversarial.sh (hostile servers, namespaces, stress; several minutes)"); + const demo = ctx.introspect_demo orelse { + const fail = b.addFail("9player-itest and 9player-adv need the introspect demo server (build with -Dintrospect=true)"); + itest.dependOn(&fail.step); + adv.dependOn(&fail.step); + return .{ .exe = player, .test_step = test_step, .itest_step = itest, .adv_step = adv }; + }; + inline for (.{ .{ itest, "integration" }, .{ adv, "adversarial" } }) |pair| { + const run = b.addSystemCommand(&.{"bash"}); + run.addFileArg(b.path("9player/test/" ++ pair[1] ++ ".sh")); + run.addArtifactArg(player); + run.addArtifactArg(demo); + pair[0].dependOn(&run.step); + } + return .{ .exe = player, .test_step = test_step, .itest_step = itest, .adv_step = adv }; +} diff --git a/9player/docs/DESIGN.md b/9player/docs/DESIGN.md new file mode 100644 index 0000000..b9b2fbe --- /dev/null +++ b/9player/docs/DESIGN.md @@ -0,0 +1,405 @@ +# 9player design + +`9player` mounts a 9P2000 file tree served over a Unix or TCP stream socket +into a **fresh mount namespace** and runs a program inside it. The program +(fish, bash, `claude`, anything) sees the 9P tree as ordinary files, without +root and without touching the host's mount table. + +## Why FUSE + +The kernel's own `9p` filesystem is not mountable inside an unprivileged user +namespace (it lacks `FS_USERNS_MOUNT`) and loading it needs root. FUSE has been +user-namespace mountable since Linux 4.18, and `/dev/fuse` is world read/write. +So 9player is a tiny FUSE server that speaks 9P2000 to the real server: + +``` + program (fish/bash/claude) 9player (parent) 9P server + in new user+mount namespace │ (introspect, + /mnt/9p ─── FUSE ───▶ kernel ──▶│ fuse.zig ──▶ bridge.zig ──▶ nine.zig ──▶ ramfs, ...) + │ (framing) (translation) (cloud9 Client) +``` + +No libfuse: `src/fuse.zig` implements the small subset of the kernel FUSE +protocol we need directly against `/usr/include/linux/fuse.h`. + +## Toolchain facts (Zig 0.16) + +* Zig 0.16.0 at `/usr/bin/zig`, std at `/usr/lib/zig/std`. **Grep the std + tree before assuming an API exists**; 0.16 moved a lot of process/fs code + behind `std.Io`. Raw Linux syscalls in `std.os.linux` (`fork`, `execve`, + `mount`, `unshare`, `waitpid`, `pipe2`, `socketpair`, `poll`, `read`, `write`, + `open`, `openat`, `getdents64`, `sigaction`, `kill`, `readlinkat`, `mkdirat`, + `symlinkat`) are the intended low-level path. They return `usize`; decode + with `std.os.linux.errno(rc)` (an `E` enum, `.SUCCESS` when ok). +* `std.posix.poll`, `std.posix.sigaction`, `std.posix.read`, `std.posix.kill` + exist. `std.posix.fork/execve/waitpid/pipe2/socketpair` do **not**. +* `pub fn main() !void` and `pub fn main(init: std.process.Init) !void` are + both supported. Prefer `main(init: std.process.Init)`; `init.gpa` is a + general purpose allocator, `init.arena` an arena, `init.minimal.args` the + argv (`toSlice(allocator)`), `init.minimal.environ.block` the envp block. +* No libc is linked. Do not use `std.c.*`. Hostname lookups are therefore out + of scope: `--tcp` takes IP literals only. +* 9player lives in the cloud9 repository as `cloud9/9player/` and is built by + the root `build.zig` through the fragment `9player/build.zig` (steps + `9player`, `9player-test`, `9player-itest`, `9player-adv`; toggle + `-D9player`). cloud9 itself is imported as module `cloud9` + (`@import("cloud9")`). Read `../src/client.zig`, `Server.zig`, `wire.zig` + and `../docs/design.md`. Its core is allocation-free and caller-driven: you + push bytes in, take results out. The demo 9P server the tests mount is the + sibling program `../introspect` (`zig build introspect`). +* Standalone module tests while other files are missing (from the cloud9 + root): + `zig test --dep cloud9 -Mroot=9player/src/.zig -Mcloud9=src/root.zig`. +* Format everything with `zig fmt`. + +## Process model + +``` +9player [options] -- PROGRAM [ARGS...] +``` + +1. Parent parses args, probes that `/dev/fuse` exists, connects to the 9P + server, negotiates `version` and `attach`es (fid 0 = root). Connection + failures are reported before anything is forked. +2. Parent forks with a `socketpair` status channel. **Child**: + 1. `unshare(CLONE_NEWUSER | CLONE_NEWNS)`. + 2. Writes `/proc/self/setgroups` = `deny`, `/proc/self/uid_map` = + `" 1"`, `/proc/self/gid_map` = `" 1"` (same ids + inside as outside; the child creating the namespace holds full + capabilities in it until exec). + 3. `mount(NULL, "/", NULL, MS_REC|MS_PRIVATE, NULL)` so nothing propagates. + 4. Ensures the mountpoint exists (see below). + 5. Opens `/dev/fuse` (`O_RDWR|O_CLOEXEC`). The kernel refuses to mount a + fuse descriptor opened from a different user namespace than the mount + ("wrong user namespace for fuse device"), so this must happen here, not + in the parent. + 6. `mount("9player", mountpoint, "fuse", MS_NOSUID|MS_NODEV, + "fd=,rootmode=40000,user_id=,group_id=,max_read=")`. + 7. Sends the fuse fd to the parent over the status socket (`SCM_RIGHTS`). + 8. `statx` of the mountpoint: this forces one GETATTR, which the parent + serves. Without it the kernel keeps the root inode's initial uid 0 + (unmapped in the namespace) and every create in the root gets `EACCES`. + 9. Sets `NINEPLAYER_MOUNT=` in the environment. + 10. `execve` of PROGRAM with PATH search (implemented by hand; no libc). + Exec failures are reported through the `CLOEXEC` status socket + (errno + message); the parent prints them after the serve loop ends. +3. **Parent** receives the fuse fd, then runs the FUSE loop (`bridge.serve`) + until either the child exits (SIGCHLD via self-pipe) or the FUSE fd reports + `ENODEV` (last process in the namespace gone, mount destroyed). It then + closes the fuse fd and exits with the child's status (`128+sig` if + signalled). The self-pipe is also watched by the 9P session while a reply + is outstanding (`Session.stop_fd` → `error.Stopped`), so a server that + never answers cannot keep 9player alive after the child is gone; a 3 s + watchdog armed from the SIGCHLD handler is the last resort. +4. Signals in the parent: `SIGINT`/`SIGQUIT` ignored (the child owns the tty + and gets them itself); `SIGTERM`/`SIGHUP` forwarded to the child; + `SIGPIPE` ignored; `SIGCHLD` → self-pipe. + +The FUSE fd is shared with the child only until exec (CLOEXEC); the parent's +copy keeps the connection alive. + +### Mountpoint policy + +Default mountpoint: `/mnt/9p`. A relative `--mount` is resolved against cwd. + +* If the path is a directory: use it. +* Else try `mkdir`. If that fails with `EACCES`/`EPERM`/`EROFS` (the normal + case for `/mnt/9p` as a plain user), **shadow the parent directory**: + open an fd to the parent, mount a `tmpfs` over it, then recreate every + existing entry inside the tmpfs: directories → `mkdir` + bind mount from + `/proc/self/fd//`; symlinks → `readlinkat` + `symlink`; anything + else → empty regular file + bind mount. Then `mkdir` the target inside. + Refuse (with a clear message) if the parent has more than 4096 entries or + is `/`. This only affects the new namespace. +* Else fail with the errno and a hint to pass `--mount` an existing dir. + +## Module contracts + +### `src/fuse.zig` — kernel FUSE protocol (no policy) + +Extern structs mirroring `linux/fuse.h`, with `comptime` size asserts: +`InHeader` (40), `OutHeader` (16), `Attr` (88), `EntryOut` (128), +`AttrOut` (104), `GetattrIn` (16), `SetattrIn` (88), `OpenIn` (8), +`OpenOut` (16), `ReleaseIn` (24), `FlushIn` (24), `ReadIn` (40), +`WriteIn` (40), `WriteOut` (8), `CreateIn` (16), `MkdirIn` (8), +`RenameIn` (8), `Rename2In` (16), `ForgetIn` (8), `BatchForgetIn` (8), +`ForgetOne` (16), `FsyncIn` (16), `AccessIn` (8), `InterruptIn` (8), +`Kstatfs` (80), `StatfsOut` (80), `InitIn` (64), `InitOut` (64), +`Dirent` (24 header, name padded to 8), `LseekIn` (24). + +`pub const Opcode = enum(u32) { lookup = 1, forget = 2, getattr = 3, setattr = 4, +readlink = 5, symlink = 6, mknod = 8, mkdir = 9, unlink = 10, rmdir = 11, +rename = 12, link = 13, open = 14, read = 15, write = 16, statfs = 17, +release = 18, fsync = 20, setxattr = 21, getxattr = 22, listxattr = 23, +removexattr = 24, flush = 25, init = 26, opendir = 27, readdir = 28, +releasedir = 29, fsyncdir = 30, getlk = 31, setlk = 32, setlkw = 33, +access = 34, create = 35, interrupt = 36, bmap = 37, destroy = 38, +ioctl = 39, poll = 40, notify_reply = 41, batch_forget = 42, fallocate = 43, +readdirplus = 44, rename2 = 45, lseek = 46, copy_file_range = 47, +setupmapping = 48, removemapping = 49, syncfs = 50, tmpfile = 51, statx = 52, _ }` + +Constants: `kernel_version = 7`, `kernel_minor = 31` (what we answer; the +kernel adapts to the lower minor), `FOPEN_DIRECT_IO = 1`, `FOPEN_KEEP_CACHE = 2`, +`FOPEN_NONSEEKABLE = 4`, `FUSE_ASYNC_READ = 1`, `FUSE_MAX_PAGES = 1<<22`, +`FATTR_MODE=1, FATTR_UID=2, FATTR_GID=4, FATTR_SIZE=8, FATTR_ATIME=16, +FATTR_MTIME=32, FATTR_FH=64, FATTR_ATIME_NOW=128, FATTR_MTIME_NOW=256, +FATTR_LOCKOWNER=512, FATTR_CTIME=1024`. `root_id = 1`. + +I/O helpers (blocking fd, no allocation beyond the caller's buffer): + +```zig +pub const Request = struct { header: InHeader, body: []const u8 }; +/// One kernel request. Returns null on ENODEV (unmounted). Retries EINTR/EAGAIN/ENOENT. +pub fn readRequest(fd: i32, buf: []u8) !?Request; +/// Success reply: header + concatenated payload slices, single writev. +pub fn reply(fd: i32, unique: u64, payloads: []const []const u8) !void; +/// Error reply: negative errno. +pub fn replyError(fd: i32, unique: u64, err: std.os.linux.E) !void; +/// Append a fuse_dirent (8-byte padded) to `buf`; returns false if it doesn't fit. +pub fn addDirent(buf: []u8, used: *usize, ino: u64, off: u64, dtype: u32, name: []const u8) bool; +pub fn body(comptime T: type, req: Request) !*const T; // aligned copy-free view, checks size +pub fn nameAfter(comptime T: type, req: Request) ![]const u8; // NUL-terminated name after a struct +``` + +The request buffer must be ≥ `max_write + 4096`; 9player uses 1 MiB + 4 KiB. +Requests with an unknown/unsupported opcode get `ENOSYS`. + +### `src/nine.zig` — synchronous 9P2000 session on a blocking fd + +Thin, synchronous RPC layer over `cloud9.Client` (which is push/take, +non-blocking-agnostic). One outstanding request at a time (the FUSE loop is +single-threaded). Fids are allocated from a free list. + +```zig +pub const Address = union(enum) { unix: []const u8, tcp: struct { host: []const u8, port: u16 }, fd: i32 }; +pub const Session = struct { + pub const Error = error{ Nine, Protocol, Io, Closed, TooLarge, OutOfMemory }; + /// After error.Nine, `ename` holds the server's Rerror text (copied, bounded). + ename: [256]u8, ename_len: usize, + msize: u32, + + pub fn connect(gpa: std.mem.Allocator, address: Address, msize: u32) !Session; // socket+connect, version + pub fn deinit(s: *Session) void; + pub fn attach(s: *Session, fid: u32, uname: []const u8, aname: []const u8) Error!cloud9.Qid; + pub fn allocFid(s: *Session) u32; + pub fn freeFid(s: *Session, fid: u32) void; + /// Generic RPC. Result slices borrow the input buffer until the next call. + pub fn rpc(s: *Session, req: cloud9.Client.Request) Error!cloud9.Client.Result; + // Conveniences (all built on rpc): + pub fn walk(s, fid: u32, newfid: u32, names: []const []const u8) Error!Walk; // Walk = { nwqid, wqid[16] }; partial walk → error.Nine with ename "not found"-ish + pub fn clone(s, fid: u32) Error!u32; // allocFid + walk with no names + pub fn open(s, fid: u32, mode: u8) Error!Open; // Open = { qid, iounit } + pub fn create(s, fid: u32, name: []const u8, perm: u32, mode: u8) Error!Open; + pub fn read(s, fid: u32, offset: u64, buf: []u8) Error!usize; // chunks by maxRead/iounit; stops at short read + pub fn write(s, fid: u32, offset: u64, data: []const u8) Error!usize; // chunks; stops at short write + pub fn stat(s, fid: u32) Error!cloud9.Stat; // strings borrow the input buffer + pub fn wstat(s, fid: u32, st: cloud9.Stat) Error!void; + pub fn clunk(s, fid: u32) Error!void; // frees the fid even on error + pub fn remove(s, fid: u32) Error!void; // frees the fid even on error + pub fn errno(s: *const Session) std.os.linux.E; // map ename → errno (see below) +}; +pub const dontcare = cloud9.Stat{ .type = 0xFFFF, .dev = 0xFFFF_FFFF, .qid = .{ .type = 0xFF, .version = 0xFFFF_FFFF, .path = 0xFFFF_FFFF_FFFF_FFFF }, .mode = 0xFFFF_FFFF, .atime = 0xFFFF_FFFF, .mtime = 0xFFFF_FFFF, .length = 0xFFFF_FFFF_FFFF_FFFF, .name = "", .uid = "", .gid = "", .muid = "" }; +``` + +`connect`: for `.unix` and `.tcp` create a blocking `SOCK_STREAM|SOCK_CLOEXEC` +socket and connect (`TCP_NODELAY` on TCP); for `.fd` adopt it. Buffers of +`msize` bytes for in/out are heap allocated. Then submit `.version`, drain +output to the socket, read until `take()` yields the version result. The +negotiated msize is `result.version.msize`; if the server answered +`"unknown"`, fail with `error.Protocol`. + +`rpc`: submit, write all of `client.output()` (calling `wrote`), then loop: +`take()`; if null, `read` from the fd into a temp buffer and `push` (push +returns how much fit; the frame is at most msize so it always fits after a +`take`). If the fd returns 0 → `error.Closed`. If the client dies → +`error.Protocol`. A `.fail` result copies the ename and returns `error.Nine`. + +Rerror text → errno mapping (case-insensitive substring, in this order): +`"not exist"`, `"not found"`, `"no such"` → `ENOENT`; `"exists"` → `EEXIST`; +`"not empty"` → `ENOTEMPTY`; `"not a dir"` → `ENOTDIR`; +`"is a dir"` → `EISDIR`; `"permission"`, `"denied"` → `EACCES`; +`"read-only"`, `"read only"`, `"readonly"` → `EROFS`; `"no space"` → `ENOSPC`; +`"not allowed"`, `"not permitted"`, `"cannot"` → `EPERM`; +`"fid"` → `EBADF`; `"bad offset"`, `"invalid"`, `"bad "` → `EINVAL`; +`"busy"`, `"in use"` → `EBUSY`; `"too long"` → `ENAMETOOLONG`; +`"not supported"`, `"unsupported"` → `ENOTSUP`; otherwise `EIO`. + +### `src/bridge.zig` — FUSE ↔ 9P translation + +```zig +pub const Options = struct { + uid: u32, gid: u32, // reported owner of every file + attr_timeout_ns: u64 = 1e9, // attr/entry cache validity (0 = none) + direct_io: bool = true, // FOPEN_DIRECT_IO on every regular file + debug: bool = false, // trace to stderr +}; +/// Runs until the FUSE fd reports ENODEV or `stop_fd` becomes readable. +pub fn serve(gpa: std.mem.Allocator, fuse_fd: i32, nine: *nine.Session, root_fid: u32, stop_fd: i32, opts: Options) !void; +``` + +State: + +* `inodes: AutoHashMap(u64 /*nodeid*/, Inode{ fid: u32, qid: Qid, nlookup: u64 })`. + Node 1 is the root (`root_fid`, never forgotten). +* `by_qid: AutoHashMap(u64 /*qid.path*/, u64 /*nodeid*/)` so that repeated + lookups of the same file map to the same inode (the old fid is clunked and + the fresh one kept). Dedupe only merges when the qid type (dir bit) also + matches, so a server reusing a path across a file and a directory cannot + poison an inode. `ino` in attrs is `qid.path` (root, or anything carrying + the root's path: 1). +* `handles: AutoHashMap(u64 /*fh*/, Handle{ fid: u32, dir: ?DirList })`. + `DirList` is the entire directory read at first `READDIR` offset 0: + `[]Entry{ name: []u8, ino: u64, dtype: u32 }` with synthetic `.` and `..` + first. `READDIR` offsets are indices into that list; a `READDIR` at offset 0 + re-reads the directory (rewinddir). + +Op mapping (9P2000 has no symlinks, links, xattrs, locks, mknod): + +| FUSE | 9P | +|---|---| +| INIT | reply `InitOut{ major=7, minor=31, max_readahead=in.max_readahead, flags = FUSE_ASYNC_READ \| FUSE_ATOMIC_O_TRUNC \| FUSE_AUTO_INVAL_DATA \| FUSE_BIG_WRITES (plus FUSE_MAX_PAGES with max_pages=256 if offered), max_background=16, congestion_threshold=12, max_write=1 MiB, time_gran=1 }`. Atomic O_TRUNC matters: without it the kernel truncates via a separate SETATTR(size=0) that synthetic control files reject; with it `O_TRUNC` becomes 9P `OTRUNC` inside the open | +| LOOKUP(parent,name) | `walk(parent.fid → newfid, [name])`; `stat(newfid)`; dedupe by qid; `EntryOut` | +| FORGET / BATCH_FORGET | `nlookup -= n`; at 0 `clunk` and drop (no reply) | +| GETATTR | `stat(inode.fid)` → `AttrOut` | +| SETATTR | `stat` then `wstat` with a *dontcare* Stat: `FATTR_SIZE`→length; `FATTR_MODE`→`(old.mode & ~0o777) \| (mode & 0o777)`; `FATTR_MTIME`→mtime (`FATTR_MTIME_NOW` → now); `FATTR_ATIME` ignored; `FATTR_UID/GID` → `EPERM` unless unchanged; then `stat` again for the reply | +| OPEN | `clone(inode.fid)` then `open(newfid, mode)`; mode from `O_ACCMODE` (`oread/owrite/ordwr`), `O_TRUNC` → `otrunc`; reply `OpenOut{ fh, open_flags = FOPEN_DIRECT_IO }`; on failure clunk | +| OPENDIR | same with `oread`; `fh` with `dir = null` | +| READ | `read(fh.fid, offset, buf[0..min(size, 1 MiB)])`; reply data | +| WRITE | `write(fh.fid, offset, data)`; `WriteOut{ size = n }` | +| READDIR | fill `Dirent`s from the `DirList` starting at `offset`, up to `size` bytes | +| RELEASE / RELEASEDIR | `clunk(fh.fid)`; free DirList | +| FLUSH / FSYNC / FSYNCDIR | ok (no-op) | +| CREATE(parent,name,flags,mode) | `clone(parent)`; `create(fid, name, mode & 0o777, openmode)` → this fid is the **open** file; then `walk(parent → fid2, [name])` + `stat(fid2)` for the inode; reply `EntryOut ++ OpenOut` | +| MKDIR | `clone(parent)`; `create(fid, name, DMDIR \| (mode & 0o777), oread)`; `clunk`; then lookup as above | +| UNLINK / RMDIR | `walk(parent → tmp, [name])`; `remove(tmp)` | +| RENAME / RENAME2 | if `newdir != parent` → `EXDEV`; else `walk(parent → tmp, [oldname])`, `wstat(tmp, dontcare with .name = newname)`, `clunk`. 9P rename never replaces, POSIX does: when the target exists (and `RENAME_NOREPLACE` is not set) a directory target is removed first; a file target is parked under a temporary name, the rename retried, and the parked file removed only after success (restored on failure) | +| STATFS | constant `Kstatfs{ bsize = 4096, namelen = 255, frsize = 4096 }` | +| ACCESS | `ENOSYS` (kernel stops asking; the server enforces permissions on open) | +| READLINK, SYMLINK, LINK, MKNOD, *XATTR, *LK, IOCTL, POLL, BMAP, FALLOCATE, LSEEK, COPY_FILE_RANGE, TMPFILE, STATX | `ENOSYS` | +| INTERRUPT | ignored (reply nothing) | +| DESTROY | return from `serve` | + +Attr mapping from `cloud9.Stat`: `mode = (S_IFDIR if DMDIR else S_IFREG) | +(st.mode & 0o777)`; `nlink = 1`; `size = length`; `blocks = (length+511)/512`; +`blksize = 4096`; `atime/mtime/ctime = st.atime/st.mtime/st.mtime`; +`uid/gid = opts.uid/gid`. `Dirent.type` = `DT_DIR` (4) / `DT_REG` (8). + +Errors: `nine.Session.Error.Nine` → `nine.errno()`; `Closed`/`Protocol`/`Io` +→ `EIO` and, since the session is dead, `serve` returns `error.Closed` after +replying so 9player can report "9P server went away". + +With `direct_io` the kernel never trusts `length` for reads: synthetic files +that report length 0 (very common in 9P) still `cat` correctly, and reads run +until the server returns a short read. With `--no-direct-io` the bridge forces +`attr_timeout_ns = 0`, because a cached stale size truncates reads (observed +data loss on a 4 MiB copy otherwise). + +Hostile-server rules: directory listings are capped at 64 MiB (a server that +ignores read offsets otherwise loops forever); directory records with names +containing `/`, NUL, empty, `.`/`..` or longer than `FUSE_NAME_MAX` are dropped +rather than poisoning the whole READDIR reply; `length` near 2^64 is clamped +to `i64` max; the errno of a failing 9P call is latched before any cleanup +clunk overwrites the session's ename. + +### `src/ns.zig` — namespace and process plumbing + +```zig +pub const Spawn = struct { + argv: []const []const u8, // argv[0] is PATH-searched unless it contains '/' + envp: [*:null]const ?[*:0]const u8, // inherited environment + mountpoint: []const u8, // absolute + fuse_fd: i32, + uid: u32, gid: u32, + max_read: u32, +}; +pub const Child = struct { pid: i32 }; +/// fork; the child sets up the namespace, mounts, and execs. Returns once exec succeeded +/// (status pipe closed) or fails with the child's error (message on stderr). +pub fn spawn(gpa: std.mem.Allocator, s: Spawn) !Child; +pub fn ensureMountpoint(path: [:0]const u8) !void; // the shadowing logic, testable alone +pub fn resolveMountpoint(gpa, path: []const u8) ![:0]u8; // absolute, no trailing slash +pub fn findInPath(gpa, envp, name) ![:0]u8; +``` + +Also exports the signal plumbing used by `main.zig`: +`installSignals(child_pid_ptr: *i32) !i32` returning the SIGCHLD self-pipe +read end (used as `stop_fd` for `bridge.serve`), and +`waitChild(pid) !u8` → exit status (`128+sig` on signal death). + +### `src/main.zig` — CLI + +``` +Usage: 9player [options] -- PROGRAM [ARGS...] +Transport (exactly one): + --unix PATH Unix stream socket + --tcp IP:PORT TCP (IPv4/IPv6 literal) + --fd N already-connected inherited descriptor + --spawn CMD run CMD (via /bin/sh -c) with a socketpair on its stdin/stdout +Options: + --mount PATH mountpoint inside the new namespace (default /mnt/9p) + --uname NAME 9P user name (default $USER, else "none") + --aname NAME 9P tree to attach (default "") + --msize BYTES maximum 9P message size to request (default 131072, max 16 MiB) + --cache SECONDS attr/entry cache validity, may be fractional (default 1) + --no-direct-io let the kernel cache file pages (trusts stat length) + --debug trace FUSE and 9P operations on stderr + --help, --version +PROGRAM defaults to $SHELL (else /bin/sh). The mountpoint is exported as $NINEPLAYER_MOUNT. +``` + +Exit codes: child's status; 125 for 9player's own failures (usage, connect, +mount); 126/127 as usual for exec failures. + +### `../introspect/demo/main.zig` — demo 9P2000 server (binary `introspect`) + +The demo server is a separate program in this repository, built on the +introspect library; see `../introspect/docs/LIBRARY.md` for the library +contract (freestanding core, value renderers, Linux debug probe). The tree it serves keeps the paths the integration tests read +(`/build/*`, `/comptime/types//*`, `/comptime/decls`, `/runtime/fn/*`, +`/runtime/ctl`, `/runtime/{pid,ppid,uptime,argv,cwd,env,clients}`, +`/scratch/`) and adds `/vars`, `/threads`, `/addr`, `/mem`, `/hex`, +`/breakpoints` and `/panic`. + +## Integration test plan (`test/integration.sh`) + +Run by `zig build 9player-itest`; args: path to `9player`, path to `introspect`. +Everything under a temp dir. Skips (exit 0 with a notice) when +`unshare -Urm true` fails or `/dev/fuse` is missing. + +1. introspect on a Unix socket; `9player --unix … -- sh -c` scripts: + `cat /mnt/9p/build/zig_version` == `zig version`; `ls` listings; `stat` + sizes; `/runtime/fn/now` is numeric; `ctl` round trip; `/scratch`: create, + append (`>>`), overwrite, truncate, `mkdir -p a/b/c`, rename within dir, + `mv` across dirs fails with `EXDEV`-ish message, `rm`, `rmdir`, 1 MiB + random file round trip compared with `sha256sum`, `dd` with odd block + sizes, many small files, `find`, exit-status propagation (`exit 7` → 7), + `$NINEPLAYER_MOUNT` set, nested `9player` inside `9player`. +2. `--spawn " --stdio"` variant. +3. `--tcp 127.0.0.1:` variant. +4. If `/usr/lib/plan9/bin/ramfs` exists: `NAMESPACE=$tmp ramfs -s ramfs` + creates `$tmp/ramfs`; run the scratch battery against it. +5. `--mount` with an existing dir, with a relative path, and the default + `/mnt/9p` (exercises parent shadowing; verify `/mnt`'s other entries are + still visible inside). +6. Kill tests: 9player exits when the child exits; server death during use + yields `EIO`, not a hang. + +## Verification + +`zig build 9player-test` (unit), `zig build 9player-itest` (74 end-to-end +checks against introspect over unix/tcp/socketpair and against plan9port's +`ramfs`) and `zig build 9player-adv` (adversarial suites: a scriptable +hostile 9P server with ~30 misbehaviour modes, FUSE semantics through the +bridge, process/namespace/signal edge cases with 51 checks, and stress). The +suites that attack the introspect server itself (a hostile raw-9P client with +181 checks, the core, the Linux layer) moved with it to +`../introspect/test` (`zig build introspect-adv`). All pass in Debug and +ReleaseSafe. + +## Out of scope for v1 (documented, not hidden) + +* One 9P request in flight at a time: a 9P read that blocks (event files) + stalls the whole mount until it returns (but not past the child's exit). +* No 9P2000.u/.L: no symlinks, ownership, or extended attributes. +* No PID namespace, no `/proc` remount. `--tcp` needs an IP literal. +* Cross-directory rename returns `EXDEV` (9P2000 cannot move files). diff --git a/9player/src/bridge.zig b/9player/src/bridge.zig new file mode 100644 index 0000000..387e184 --- /dev/null +++ b/9player/src/bridge.zig @@ -0,0 +1,971 @@ +//! FUSE ↔ 9P2000 translation: the request loop that turns kernel FUSE requests +//! into synchronous 9P calls on a `nine.Session` and sends the replies back. +//! +//! Everything here is single-threaded and one request at a time. State is three +//! tables: inodes (nodeid → fid/qid, deduplicated by qid.path), open handles +//! (fh → fid plus a cached directory listing), and the reverse qid map. +const std = @import("std"); +const cloud9 = @import("cloud9"); +const fuse = @import("fuse.zig"); +const nine = @import("nine.zig"); +const linux = std.os.linux; + +pub const Options = struct { + /// Reported owner of every file. + uid: u32, + gid: u32, + /// attr/entry cache validity (0 = none). + attr_timeout_ns: u64 = 1_000_000_000, + /// FOPEN_DIRECT_IO on every regular file. + direct_io: bool = true, + /// Trace every request, reply and 9P call to stderr. + debug: bool = false, +}; + +/// Largest single READ/WRITE payload we accept from the kernel. +pub const max_write: u32 = 1 << 20; +/// Upper bound on the raw bytes of one directory listing (about a million entries); +/// past it the listing fails with EIO instead of eating memory. +pub const max_dir_bytes: u64 = 64 << 20; +/// The kernel refuses dirents longer than this (FUSE_NAME_MAX) with EIO. +pub const max_name_len: usize = 1024; +/// Request buffer: `max_write` plus room for the header and the largest in-struct. +pub const request_buf_len: usize = max_write + 4096; + +const Inode = struct { + fid: u32, + qid: cloud9.Qid, + nlookup: u64, + /// nodeid of the directory this inode was looked up in (root: itself). Used for "..". + parent: u64, +}; + +pub const Entry = struct { name: []u8, ino: u64, dtype: u32 }; + +pub const DirList = struct { + entries: std.ArrayList(Entry) = .empty, + + pub fn deinit(d: *DirList, gpa: std.mem.Allocator) void { + for (d.entries.items) |e| gpa.free(e.name); + d.entries.deinit(gpa); + } +}; + +const Handle = struct { + fid: u32, + nodeid: u64, + dir: ?DirList, +}; + +/// Errors a request handler may surface. Policy failures are ordinary errors +/// that the dispatcher maps to an errno; `FuseIo` means the kernel side is broken. +const HandlerError = nine.Session.Error || error{ + BadRequest, + NoEntry, + BadHandle, + Exdev, + Perm, + NotSup, + /// A directory listing the server sent could not be parsed (EIO, not fatal). + BadDir, + FuseIo, +}; + +/// Runs until the FUSE fd reports ENODEV, a DESTROY arrives, or `stop_fd` +/// becomes readable (also while a 9P reply is outstanding). Returns +/// `error.Closed` if the 9P server went away. +pub fn serve(gpa: std.mem.Allocator, fuse_fd: i32, session: *nine.Session, root_fid: u32, stop_fd: i32, opts: Options) !void { + var effective = opts; + // With page caching on, a nonzero attr cache lets the kernel trust a stale + // (often zero) size and truncate reads: 9P sizes are authoritative and change + // under us. direct_io ignores the cached size, so the cache is safe only there. + if (!effective.direct_io) effective.attr_timeout_ns = 0; + var b: Bridge = .{ + .gpa = gpa, + .fuse_fd = fuse_fd, + .nine = session, + .opts = effective, + }; + defer b.deinit(); + + b.req_buf = try gpa.alignedAlloc(u8, .@"8", request_buf_len); + b.data_buf = try gpa.alloc(u8, max_write); + + // Abandon any pending 9P reply once the child is gone (stop_fd readable), + // including the initial root stat below: a silent server must not pin us. + session.stop_fd = stop_fd; + defer session.stop_fd = -1; + + // Node 1 is the root; its qid comes from a stat so lookups resolving back to + // it (e.g. via a walk) dedupe onto node 1. + var root_qid: cloud9.Qid = .{ .type = cloud9.qtdir, .version = 0, .path = 0 }; + if (b.stat(root_fid)) |st| { + root_qid = st.qid; + b.root_path = st.qid.path; + try b.by_qid.put(gpa, st.qid.path, fuse.root_id); + } else |e| switch (e) { + error.Nine => {}, + error.Stopped => return, + else => return error.Closed, + } + try b.inodes.put(gpa, fuse.root_id, .{ .fid = root_fid, .qid = root_qid, .nlookup = 1, .parent = fuse.root_id }); + + var pfds = [_]linux.pollfd{ + .{ .fd = fuse_fd, .events = linux.POLL.IN, .revents = 0 }, + .{ .fd = stop_fd, .events = linux.POLL.IN, .revents = 0 }, + }; + while (true) { + pfds[0].revents = 0; + pfds[1].revents = 0; + const rc = linux.poll(&pfds, pfds.len, -1); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .INTR, .AGAIN => continue, + else => return error.Io, + } + if (pfds[1].revents != 0) { + b.trace("stop_fd readable; leaving serve loop", .{}); + return; + } + if (pfds[0].revents == 0) continue; + const req = (fuse.readRequest(fuse_fd, b.req_buf) catch |e| switch (e) { + error.Protocol => return error.FuseProtocol, + else => return error.FuseIo, + }) orelse { + b.trace("fuse fd reports ENODEV; unmounted", .{}); + return; + }; + if (!try b.dispatch(req)) return; + } +} + +const Bridge = struct { + gpa: std.mem.Allocator, + fuse_fd: i32, + nine: *nine.Session, + opts: Options, + req_buf: []align(8) u8 = &.{}, + data_buf: []u8 = &.{}, + inodes: std.AutoHashMapUnmanaged(u64, Inode) = .empty, + by_qid: std.AutoHashMapUnmanaged(u64, u64) = .empty, + handles: std.AutoHashMapUnmanaged(u64, Handle) = .empty, + next_node: u64 = 2, + next_fh: u64 = 1, + /// qid.path of the root, reported as ino 1 wherever it shows up. + root_path: u64 = 0, + /// errno of the most recent Rerror that a handler did not swallow. Kept here + /// because `Session.rpc` clears its ename on every call, and error paths + /// clunk (an rpc) before the dispatcher maps the failure to an errno. + last_err: linux.E = .IO, + + fn deinit(b: *Bridge) void { + var it = b.handles.valueIterator(); + while (it.next()) |h| if (h.dir) |*d| d.deinit(b.gpa); + b.handles.deinit(b.gpa); + b.inodes.deinit(b.gpa); + b.by_qid.deinit(b.gpa); + if (b.req_buf.len != 0) b.gpa.free(b.req_buf); + if (b.data_buf.len != 0) b.gpa.free(b.data_buf); + } + + fn trace(b: *const Bridge, comptime fmt: []const u8, args: anytype) void { + if (b.opts.debug) std.debug.print("9player: " ++ fmt ++ "\n", args); + } + + // -- dispatch -------------------------------------------------------------------- + + /// Handles one request. Returns false when the loop should stop (DESTROY). + /// Fatal errors (dead 9P session, broken FUSE fd) propagate. + fn dispatch(b: *Bridge, req: fuse.Request) !bool { + const h = req.header; + const op = h.op(); + b.trace("<- {s} unique={d} nodeid={d} len={d} (fids={d} inodes={d} handles={d})", .{ opName(op), h.unique, h.nodeid, h.len, b.nine.fidsInUse(), b.inodes.count(), b.handles.count() }); + const wants_reply = switch (op) { + .forget, .batch_forget, .interrupt => false, + else => true, + }; + if (op == .destroy) { + b.reply(h.unique, &.{}) catch {}; + return false; + } + b.handle(req) catch |e| { + const code: linux.E = switch (e) { + error.Nine => b.last_err, + error.BadRequest => .INVAL, + error.NoEntry => .NOENT, + error.BadHandle => .BADF, + error.Exdev => .XDEV, + error.Perm => .PERM, + error.NotSup => .NOSYS, + error.OutOfMemory => .NOMEM, + error.TooLarge => .NAMETOOLONG, + error.BadDir => .IO, + error.Closed, error.Protocol, error.Io, error.Stopped => .IO, + error.FuseIo => return error.FuseIo, + }; + if (wants_reply) try b.replyError(h.unique, code); + switch (e) { + error.Closed, error.Protocol, error.Io => return error.Closed, + error.Stopped => return false, // the child is gone; the mount is being torn down + else => {}, + } + }; + return true; + } + + fn handle(b: *Bridge, req: fuse.Request) HandlerError!void { + const u = req.header.unique; + switch (req.header.op()) { + .init => { + const in = try body(fuse.InitIn, req); + const out = fuse.initReply(in, max_write); + try b.reply(u, &.{std.mem.asBytes(&out)}); + }, + .lookup => { + const name = try nameAfter(void, req); + const entry = try b.lookupEntry(req.header.nodeid, name); + try b.reply(u, &.{std.mem.asBytes(&entry)}); + }, + .forget => { + const in = try body(fuse.ForgetIn, req); + try b.forget(req.header.nodeid, in.nlookup); + }, + .batch_forget => { + const in = try body(fuse.BatchForgetIn, req); + const rest = req.body[@sizeOf(fuse.BatchForgetIn)..]; + const count: usize = in.count; + if (rest.len < count * @sizeOf(fuse.ForgetOne)) return error.BadRequest; + for (0..count) |i| { + const one = std.mem.bytesToValue(fuse.ForgetOne, rest[i * @sizeOf(fuse.ForgetOne) ..][0..@sizeOf(fuse.ForgetOne)]); + try b.forget(one.nodeid, one.nlookup); + } + }, + .getattr => { + const ino = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; + const st = try b.stat(ino.fid); + const out = b.attrOut(st, b.inoOf(req.header.nodeid, ino.qid)); + try b.reply(u, &.{std.mem.asBytes(&out)}); + }, + .setattr => try b.setattr(req), + .open => try b.openFile(req, false), + .opendir => try b.openFile(req, true), + .read => { + const in = try body(fuse.ReadIn, req); + const h = b.handles.get(in.fh) orelse return error.BadHandle; + const want: usize = @min(in.size, max_write); + const n = try b.read(h.fid, in.offset, b.data_buf[0..want]); + try b.reply(u, &.{b.data_buf[0..n]}); + }, + .write => { + const in = try body(fuse.WriteIn, req); + const h = b.handles.get(in.fh) orelse return error.BadHandle; + const rest = req.body[@sizeOf(fuse.WriteIn)..]; + if (rest.len < in.size) return error.BadRequest; + const n = try b.write(h.fid, in.offset, rest[0..in.size]); + const out = fuse.WriteOut{ .size = @intCast(n) }; + try b.reply(u, &.{std.mem.asBytes(&out)}); + }, + .readdir => try b.readdir(req), + .release, .releasedir => { + const in = try body(fuse.ReleaseIn, req); + const kv = b.handles.fetchRemove(in.fh) orelse return error.BadHandle; + var h = kv.value; + if (h.dir) |*d| d.deinit(b.gpa); + try b.clunk(h.fid); + try b.reply(u, &.{}); + }, + .flush, .fsync, .fsyncdir => try b.reply(u, &.{}), + .create => try b.create(req), + .mkdir => { + const in = try body(fuse.MkdirIn, req); + const name = try nameAfter(fuse.MkdirIn, req); + const parent = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; + const fid = try b.clone(parent.fid); + _ = b.create9(fid, name, cloud9.dmdir | (in.mode & 0o777), cloud9.oread) catch |e| { + b.clunkQuiet(fid); + return e; + }; + try b.clunk(fid); + const entry = try b.lookupEntry(req.header.nodeid, name); + try b.reply(u, &.{std.mem.asBytes(&entry)}); + }, + .unlink, .rmdir => { + const name = try nameAfter(void, req); + const parent = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; + const tmp = try b.walkName(parent.fid, name); + try b.remove(tmp); + try b.reply(u, &.{}); + }, + .rename => { + const in = try body(fuse.RenameIn, req); + const old = try nameAfter(fuse.RenameIn, req); + const new = try secondName(req, old, @sizeOf(fuse.RenameIn)); + try b.rename(req.header.nodeid, in.newdir, old, new, 0); + try b.reply(u, &.{}); + }, + .rename2 => { + const in = try body(fuse.Rename2In, req); + const old = try nameAfter(fuse.Rename2In, req); + const new = try secondName(req, old, @sizeOf(fuse.Rename2In)); + try b.rename(req.header.nodeid, in.newdir, old, new, in.flags); + try b.reply(u, &.{}); + }, + .statfs => { + const out = fuse.StatfsOut{ .st = .{ .bsize = 4096, .namelen = 255, .frsize = 4096 } }; + try b.reply(u, &.{std.mem.asBytes(&out)}); + }, + .interrupt => {}, + .destroy => unreachable, // handled in dispatch + .access => return error.NotSup, + else => return error.NotSup, + } + } + + // -- handlers ---------------------------------------------------------------------- + + /// walk(parent → new fid, [name]) + stat, deduplicated by qid.path. Bumps nlookup. + fn lookupEntry(b: *Bridge, parent_id: u64, name: []const u8) HandlerError!fuse.EntryOut { + const parent = b.inodes.get(parent_id) orelse return error.NoEntry; + const newfid = try b.walkName(parent.fid, name); + const st = b.stat(newfid) catch |e| { + b.clunkQuiet(newfid); + return e; + }; + const qid = st.qid; + var nodeid: u64 = undefined; + if (b.by_qid.get(qid.path)) |existing| { + // A directory and a file sharing a qid.path (a server bug) must not + // share a node: the kernel would mark the inode bad, and for the + // root that is fatal for the whole mount. + const merge = if (b.inodes.getPtr(existing)) |ino| (ino.qid.type & cloud9.qtdir) == (qid.type & cloud9.qtdir) else false; + if (merge) { + const ino = b.inodes.getPtr(existing).?; + ino.nlookup += 1; + ino.qid = qid; + nodeid = existing; + if (existing == fuse.root_id) { + b.clunkQuiet(newfid); + } else { + // Keep the fresh fid (it is bound to the current file at this + // name) and retire the older one. + const stale = ino.fid; + ino.fid = newfid; + b.clunkQuiet(stale); + } + } else { + // Stale reverse entry, or a type clash: bind a fresh node to it. + nodeid = try b.newInode(newfid, qid, parent_id); + } + } else { + nodeid = try b.newInode(newfid, qid, parent_id); + } + var out = fuse.EntryOut{ + .nodeid = nodeid, + .generation = 0, + .attr = b.attrFrom(st, b.inoOf(nodeid, qid)), + }; + out.entry_valid = b.opts.attr_timeout_ns / 1_000_000_000; + out.entry_valid_nsec = @intCast(b.opts.attr_timeout_ns % 1_000_000_000); + out.attr_valid = out.entry_valid; + out.attr_valid_nsec = out.entry_valid_nsec; + return out; + } + + fn newInode(b: *Bridge, fid: u32, qid: cloud9.Qid, parent: u64) HandlerError!u64 { + const nodeid = b.next_node; + b.inodes.put(b.gpa, nodeid, .{ .fid = fid, .qid = qid, .nlookup = 1, .parent = parent }) catch |e| { + b.clunkQuiet(fid); + return e; + }; + b.by_qid.put(b.gpa, qid.path, nodeid) catch |e| { + _ = b.inodes.remove(nodeid); + b.clunkQuiet(fid); + return e; + }; + b.next_node += 1; + return nodeid; + } + + fn forget(b: *Bridge, nodeid: u64, n: u64) HandlerError!void { + if (nodeid == fuse.root_id) return; + const ino = b.inodes.getPtr(nodeid) orelse return; + if (ino.nlookup > n) { + ino.nlookup -= n; + return; + } + const fid = ino.fid; + const path = ino.qid.path; + _ = b.inodes.remove(nodeid); + if (b.by_qid.get(path)) |mapped| { + if (mapped == nodeid) _ = b.by_qid.remove(path); + } + b.clunk(fid) catch |e| switch (e) { + error.Nine => {}, + else => return e, + }; + } + + fn setattr(b: *Bridge, req: fuse.Request) HandlerError!void { + const in = try body(fuse.SetattrIn, req); + const ino = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; + const old = try b.stat(ino.fid); + const old_mode = old.mode; + + var st = nine.dontcare; + var changed = false; + if (in.valid & fuse.FATTR_UID != 0 and in.uid != b.opts.uid) return error.Perm; + if (in.valid & fuse.FATTR_GID != 0 and in.gid != b.opts.gid) return error.Perm; + if (in.valid & fuse.FATTR_SIZE != 0) { + st.length = in.size; + changed = true; + } + if (in.valid & fuse.FATTR_MODE != 0) { + st.mode = (old_mode & ~@as(u32, 0o777)) | (in.mode & 0o777); + changed = true; + } + if (in.valid & fuse.FATTR_MTIME_NOW != 0) { + st.mtime = nowSeconds(); + changed = true; + } else if (in.valid & fuse.FATTR_MTIME != 0) { + st.mtime = @truncate(in.mtime); + changed = true; + } + if (changed) try b.wstat(ino.fid, st); + const fresh = try b.stat(ino.fid); + const out = b.attrOut(fresh, b.inoOf(req.header.nodeid, ino.qid)); + try b.reply(req.header.unique, &.{std.mem.asBytes(&out)}); + } + + fn openFile(b: *Bridge, req: fuse.Request, is_dir: bool) HandlerError!void { + const in = try body(fuse.OpenIn, req); + const ino = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; + const mode: u8 = if (is_dir) cloud9.oread else openMode(in.flags); + const fid = try b.clone(ino.fid); + _ = b.open9(fid, mode) catch |e| { + b.clunkQuiet(fid); + return e; + }; + const fh = try b.newHandle(fid, req.header.nodeid); + const out = fuse.OpenOut{ + .fh = fh, + .open_flags = if (!is_dir and b.opts.direct_io) fuse.FOPEN_DIRECT_IO else 0, + }; + try b.reply(req.header.unique, &.{std.mem.asBytes(&out)}); + } + + fn newHandle(b: *Bridge, fid: u32, nodeid: u64) HandlerError!u64 { + const fh = b.next_fh; + b.handles.put(b.gpa, fh, .{ .fid = fid, .nodeid = nodeid, .dir = null }) catch |e| { + b.clunkQuiet(fid); + return e; + }; + b.next_fh += 1; + return fh; + } + + fn create(b: *Bridge, req: fuse.Request) HandlerError!void { + const in = try body(fuse.CreateIn, req); + const name = try nameAfter(fuse.CreateIn, req); + const parent = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; + // The created fid becomes the open file. + const fid = try b.clone(parent.fid); + _ = b.create9(fid, name, in.mode & 0o777, openMode(in.flags)) catch |e| { + b.clunkQuiet(fid); + return e; + }; + const entry = b.lookupEntry(req.header.nodeid, name) catch |e| { + b.clunkQuiet(fid); + return e; + }; + const fh = try b.newHandle(fid, entry.nodeid); + const oo = fuse.OpenOut{ + .fh = fh, + .open_flags = if (b.opts.direct_io) fuse.FOPEN_DIRECT_IO else 0, + }; + try b.reply(req.header.unique, &.{ std.mem.asBytes(&entry), std.mem.asBytes(&oo) }); + } + + fn rename(b: *Bridge, parent_id: u64, newdir: u64, old: []const u8, new: []const u8, flags: u32) HandlerError!void { + if (newdir != parent_id) return error.Exdev; + const rf: linux.RENAME = @bitCast(flags); + if (rf.EXCHANGE or rf.WHITEOUT) return error.BadRequest; + const parent = b.inodes.get(parent_id) orelse return error.NoEntry; + const tmp = try b.walkName(parent.fid, old); + defer b.clunkQuiet(tmp); + var st = nine.dontcare; + st.name = new; + b.wstat(tmp, st) catch |e| { + // 9P2000 rename never replaces an existing name; POSIX rename does. + if (e != error.Nine or rf.NOREPLACE or b.nine.errno() != .EXIST) return e; + try b.renameOver(parent.fid, tmp, new); + }; + } + + /// Replace `new` with the file behind `src`. An (empty) directory target is + /// removed first: it holds no data and the VFS already ruled out mismatched + /// types. A file target is parked under a temporary name so that a failing + /// second rename can put it back instead of having destroyed it. + fn renameOver(b: *Bridge, parent_fid: u32, src: u32, new: []const u8) HandlerError!void { + const victim = try b.walkName(parent_fid, new); + const vst = b.stat(victim) catch |e| { + b.clunkQuiet(victim); + return e; + }; + var st = nine.dontcare; + st.name = new; + if (vst.mode & cloud9.dmdir != 0) { + b.trace(" rename target is a directory; removing it and retrying", .{}); + try b.remove(victim); + return b.wstat(src, st); + } + var park_buf: [48]u8 = undefined; + const park = std.fmt.bufPrint(&park_buf, ".9player-rename-{x}", .{randomU64()}) catch unreachable; + b.trace(" rename target exists; parking it as {s} and retrying", .{park}); + var pst = nine.dontcare; + pst.name = park; + b.wstat(victim, pst) catch |e| { + b.clunkQuiet(victim); + return e; + }; + b.wstat(src, st) catch |e| { + b.trace(" rename still failed; restoring the target", .{}); + const saved = b.last_err; + b.wstat(victim, st) catch {}; + b.last_err = saved; + b.clunkQuiet(victim); + return e; + }; + b.remove(victim) catch b.trace(" could not remove the parked target {s}", .{park}); + } + + fn readdir(b: *Bridge, req: fuse.Request) HandlerError!void { + const in = try body(fuse.ReadIn, req); + const h = b.handles.getPtr(in.fh) orelse return error.BadHandle; + if (in.offset == 0 or h.dir == null) { + if (h.dir) |*d| d.deinit(b.gpa); + h.dir = null; + h.dir = try b.loadDir(h.fid, h.nodeid); + } + const dir = &h.dir.?; + const size: usize = @min(in.size, max_write); + const used = packDirents(dir.entries.items, in.offset, b.data_buf[0..size]); + try b.reply(req.header.unique, &.{b.data_buf[0..used]}); + } + + /// Reads the whole directory and builds its listing, "." and ".." first. + fn loadDir(b: *Bridge, fid: u32, nodeid: u64) HandlerError!DirList { + var list: DirList = .{}; + errdefer list.deinit(b.gpa); + const self_ino = b.inoOfNode(nodeid); + const parent_ino = if (b.inodes.get(nodeid)) |ino| b.inoOfNode(ino.parent) else self_ino; + try list.entries.append(b.gpa, .{ .name = try b.gpa.dupe(u8, "."), .ino = self_ino, .dtype = fuse.DT_DIR }); + try list.entries.append(b.gpa, .{ .name = try b.gpa.dupe(u8, ".."), .ino = parent_ino, .dtype = fuse.DT_DIR }); + + var offset: u64 = 0; + while (true) { + // A server that ignores the offset would otherwise feed us forever. + if (offset >= max_dir_bytes) return error.BadDir; + const n = try b.read(fid, offset, b.data_buf); + if (n == 0) break; + try parseDirRecords(b.gpa, b.data_buf[0..n], &list); + offset += n; + } + // Entries carrying the root's own qid.path get the root's ino (1), as GETATTR would report it. + for (list.entries.items[2..]) |*e| if (e.ino == b.root_path) { + e.ino = fuse.root_id; + }; + return list; + } + + // -- 9P wrappers (tracing) ----------------------------------------------------------- + + fn stat(b: *Bridge, fid: u32) nine.Session.Error!cloud9.Stat { + const st = b.nine.stat(fid) catch |e| return b.nineErr("stat", fid, e); + b.trace(" 9p stat fid={d} -> name={s} mode={o} len={d} qid={x}", .{ fid, st.name, st.mode, st.length, st.qid.path }); + return st; + } + + fn walkName(b: *Bridge, fid: u32, name: []const u8) nine.Session.Error!u32 { + const newfid = b.nine.allocFid(); + _ = b.nine.walk(fid, newfid, &.{name}) catch |e| { + b.nine.freeFid(newfid); + return b.nineErr("walk", fid, e); + }; + b.trace(" 9p walk fid={d} newfid={d} name={s} -> ok", .{ fid, newfid, name }); + return newfid; + } + + fn clone(b: *Bridge, fid: u32) nine.Session.Error!u32 { + const newfid = b.nine.clone(fid) catch |e| return b.nineErr("clone", fid, e); + b.trace(" 9p walk fid={d} newfid={d} (clone) -> ok", .{ fid, newfid }); + return newfid; + } + + fn open9(b: *Bridge, fid: u32, mode: u8) nine.Session.Error!nine.Session.Open { + const o = b.nine.open(fid, mode) catch |e| return b.nineErr("open", fid, e); + b.trace(" 9p open fid={d} mode={d} -> iounit={d}", .{ fid, mode, o.iounit }); + return o; + } + + fn create9(b: *Bridge, fid: u32, name: []const u8, perm: u32, mode: u8) nine.Session.Error!nine.Session.Open { + const o = b.nine.create(fid, name, perm, mode) catch |e| return b.nineErr("create", fid, e); + b.trace(" 9p create fid={d} name={s} perm={o} mode={d} -> iounit={d}", .{ fid, name, perm, mode, o.iounit }); + return o; + } + + fn read(b: *Bridge, fid: u32, offset: u64, buf: []u8) nine.Session.Error!usize { + const n = b.nine.read(fid, offset, buf) catch |e| return b.nineErr("read", fid, e); + b.trace(" 9p read fid={d} offset={d} count={d} -> {d}", .{ fid, offset, buf.len, n }); + return n; + } + + fn write(b: *Bridge, fid: u32, offset: u64, data: []const u8) nine.Session.Error!usize { + const n = b.nine.write(fid, offset, data) catch |e| return b.nineErr("write", fid, e); + b.trace(" 9p write fid={d} offset={d} count={d} -> {d}", .{ fid, offset, data.len, n }); + return n; + } + + fn wstat(b: *Bridge, fid: u32, st: cloud9.Stat) nine.Session.Error!void { + b.nine.wstat(fid, st) catch |e| return b.nineErr("wstat", fid, e); + b.trace(" 9p wstat fid={d} name={s} mode={x} len={x} mtime={x} -> ok", .{ fid, st.name, st.mode, st.length, st.mtime }); + } + + fn clunk(b: *Bridge, fid: u32) nine.Session.Error!void { + b.nine.clunk(fid) catch |e| return b.nineErr("clunk", fid, e); + b.trace(" 9p clunk fid={d} -> ok", .{fid}); + } + + /// Best-effort clunk during error unwinding; a dead session surfaces on the + /// next call. Does not disturb the errno of the failure being unwound. + fn clunkQuiet(b: *Bridge, fid: u32) void { + const saved = b.last_err; + defer b.last_err = saved; + b.clunk(fid) catch {}; + } + + fn remove(b: *Bridge, fid: u32) nine.Session.Error!void { + b.nine.remove(fid) catch |e| return b.nineErr("remove", fid, e); + b.trace(" 9p remove fid={d} -> ok", .{fid}); + } + + fn nineErr(b: *Bridge, what: []const u8, fid: u32, e: nine.Session.Error) nine.Session.Error { + if (e == error.Nine) { + b.last_err = b.nine.errno(); + b.trace(" 9p {s} fid={d} -> Rerror \"{s}\" ({s})", .{ what, fid, b.nine.ename[0..b.nine.ename_len], @tagName(b.nine.errno()) }); + } else { + b.trace(" 9p {s} fid={d} -> {s}", .{ what, fid, @errorName(e) }); + } + return e; + } + + // -- FUSE wrappers (tracing) -------------------------------------------------------- + + fn reply(b: *Bridge, unique: u64, payloads: []const []const u8) error{FuseIo}!void { + var total: usize = 0; + for (payloads) |p| total += p.len; + b.trace("-> unique={d} ok ({d} bytes)", .{ unique, total }); + fuse.reply(b.fuse_fd, unique, payloads) catch return error.FuseIo; + } + + fn replyError(b: *Bridge, unique: u64, code: linux.E) error{FuseIo}!void { + b.trace("-> unique={d} error E{s}", .{ unique, @tagName(code) }); + fuse.replyError(b.fuse_fd, unique, code) catch return error.FuseIo; + } + + // -- attrs --------------------------------------------------------------------------- + + fn inoOf(b: *const Bridge, nodeid: u64, qid: cloud9.Qid) u64 { + return if (nodeid == fuse.root_id or qid.path == b.root_path) fuse.root_id else qid.path; + } + + fn inoOfNode(b: *const Bridge, nodeid: u64) u64 { + if (nodeid == fuse.root_id) return fuse.root_id; + const ino = b.inodes.get(nodeid) orelse return nodeid; + return b.inoOf(nodeid, ino.qid); + } + + fn attrFrom(b: *const Bridge, st: cloud9.Stat, ino: u64) fuse.Attr { + return attrFromStat(st, ino, b.opts.uid, b.opts.gid); + } + + fn attrOut(b: *const Bridge, st: cloud9.Stat, ino: u64) fuse.AttrOut { + return .{ + .attr_valid = b.opts.attr_timeout_ns / 1_000_000_000, + .attr_valid_nsec = @intCast(b.opts.attr_timeout_ns % 1_000_000_000), + .attr = b.attrFrom(st, ino), + }; + } +}; + +// -- pure helpers (unit-tested) ------------------------------------------------------------ + +/// Attr from a 9P Stat: DMDIR → S_IFDIR else S_IFREG, low 9 permission bits kept. +pub fn attrFromStat(st: cloud9.Stat, ino: u64, uid: u32, gid: u32) fuse.Attr { + const ftype: u32 = if (st.mode & cloud9.dmdir != 0) fuse.S_IFDIR else fuse.S_IFREG; + return .{ + .ino = ino, + // The kernel marks an inode bad when size > LLONG_MAX; clamp hostile lengths. + .size = @min(st.length, std.math.maxInt(i64)), + // Saturating: a hostile length of 2^64-1 must not overflow. + .blocks = st.length / 512 + @intFromBool(st.length % 512 != 0), + .atime = st.atime, + .mtime = st.mtime, + .ctime = st.mtime, + .mode = ftype | (st.mode & 0o777), + .nlink = 1, + .uid = uid, + .gid = gid, + .blksize = 4096, + }; +} + +/// Kernel open(2) flags → 9P open mode. O_APPEND has no 9P equivalent and is ignored. +pub fn openMode(flags: u32) u8 { + const o: linux.O = @bitCast(flags); + var mode: u8 = switch (o.ACCMODE) { + .RDONLY => cloud9.oread, + .WRONLY => cloud9.owrite, + .RDWR => cloud9.ordwr, + }; + if (o.TRUNC) mode |= cloud9.otrunc; + return mode; +} + +/// Parses consecutive 9P directory records (2-byte size + Stat) and appends entries. +pub fn parseDirRecords(gpa: std.mem.Allocator, bytes: []const u8, list: *DirList) error{ OutOfMemory, BadDir }!void { + var pos: usize = 0; + while (pos < bytes.len) { + if (bytes.len - pos < 2) return error.BadDir; + const size: usize = std.mem.readInt(u16, bytes[pos..][0..2], .little); + if (bytes.len - pos < 2 + size) return error.BadDir; + const st = cloud9.Stat.decode(bytes[pos..][0 .. 2 + size]) catch return error.BadDir; + pos += 2 + size; + // The kernel rejects a whole READDIR reply (EIO) over one bad name, and + // "." and ".." are synthesised by loadDir: drop such records instead. + if (!validDirentName(st.name)) continue; + const name = try gpa.dupe(u8, st.name); + errdefer gpa.free(name); + try list.entries.append(gpa, .{ + .name = name, + .ino = st.qid.path, + .dtype = if (st.mode & cloud9.dmdir != 0) fuse.DT_DIR else fuse.DT_REG, + }); + } +} + +/// A name the kernel will accept in a dirent and that does not duplicate the synthetic "." / "..". +pub fn validDirentName(name: []const u8) bool { + if (name.len == 0 or name.len > max_name_len) return false; + if (std.mem.indexOfAny(u8, name, "/\x00") != null) return false; + if (std.mem.eql(u8, name, ".") or std.mem.eql(u8, name, "..")) return false; + return true; +} + +/// Packs dirents from `entries[offset..]` into `buf`; each record's `off` is its index + 1. +/// Returns the number of bytes used. +pub fn packDirents(entries: []const Entry, offset: u64, buf: []u8) usize { + var used: usize = 0; + var i: usize = @intCast(@min(offset, entries.len)); + while (i < entries.len) : (i += 1) { + const e = entries[i]; + if (!fuse.addDirent(buf, &used, e.ino, @as(u64, i) + 1, e.dtype, e.name)) break; + } + return used; +} + +fn randomU64() u64 { + var bytes: [8]u8 = undefined; + if (linux.errno(linux.getrandom(&bytes, bytes.len, 0)) == .SUCCESS) return std.mem.readInt(u64, &bytes, .little); + var ts: linux.timespec = undefined; + _ = linux.clock_gettime(.MONOTONIC, &ts); + return @as(u64, @bitCast(ts.nsec)) ^ (@as(u64, @bitCast(ts.sec)) << 32); +} + +fn nowSeconds() u32 { + var ts: linux.timespec = undefined; + if (linux.errno(linux.clock_gettime(.REALTIME, &ts)) != .SUCCESS) return 0; + return @intCast(@as(u64, @intCast(ts.sec)) & 0xFFFF_FFFF); +} + +fn opName(op: fuse.Opcode) []const u8 { + return switch (op) { + _ => "unknown", + else => @tagName(op), + }; +} + +// Thin adapters so fuse.zig's parse errors become HandlerError.BadRequest. +fn body(comptime T: type, req: fuse.Request) error{BadRequest}!*const T { + return fuse.body(T, req) catch error.BadRequest; +} + +fn nameAfter(comptime T: type, req: fuse.Request) error{BadRequest}![]const u8 { + return fuse.nameAfter(T, req) catch error.BadRequest; +} + +fn secondName(req: fuse.Request, first: []const u8, offset: usize) error{BadRequest}![]const u8 { + return fuse.secondName(req, first, offset) catch error.BadRequest; +} + +// -- tests ------------------------------------------------------------------------------ + +const testing = std.testing; + +test { + // Force semantic analysis of `serve` and the whole dispatch path, which no + // unit test can exercise without a FUSE mount. + testing.refAllDecls(@This()); +} + +fn testStat(name: []const u8, mode: u32, length: u64, path: u64) cloud9.Stat { + return .{ + .type = 0, + .dev = 0, + .qid = .{ .type = if (mode & cloud9.dmdir != 0) cloud9.qtdir else 0, .version = 0, .path = path }, + .mode = mode, + .atime = 100, + .mtime = 200, + .length = length, + .name = name, + .uid = "u", + .gid = "g", + .muid = "u", + }; +} + +test "attr mapping: DMDIR → S_IFDIR|perm, length → size/blocks" { + const d = attrFromStat(testStat("d", cloud9.dmdir | 0o755, 0, 9), 9, 1000, 1001); + try testing.expectEqual(fuse.S_IFDIR | 0o755, d.mode); + try testing.expectEqual(@as(u64, 9), d.ino); + try testing.expectEqual(@as(u64, 0), d.size); + try testing.expectEqual(@as(u64, 0), d.blocks); + try testing.expectEqual(@as(u32, 1000), d.uid); + try testing.expectEqual(@as(u32, 1001), d.gid); + try testing.expectEqual(@as(u32, 1), d.nlink); + + const f = attrFromStat(testStat("f", 0o640 | cloud9.dmappend, 1025, 4), 4, 0, 0); + try testing.expectEqual(fuse.S_IFREG | 0o640, f.mode); // dmappend bit not leaked + try testing.expectEqual(@as(u64, 1025), f.size); + try testing.expectEqual(@as(u64, 3), f.blocks); + try testing.expectEqual(@as(u32, 4096), f.blksize); + try testing.expectEqual(@as(u64, 100), f.atime); + try testing.expectEqual(@as(u64, 200), f.mtime); + try testing.expectEqual(@as(u64, 200), f.ctime); + + try testing.expectEqual(@as(u64, 1), attrFromStat(testStat("f", 0o600, 512, 4), 4, 0, 0).blocks); + try testing.expectEqual(@as(u64, 2), attrFromStat(testStat("f", 0o600, 513, 4), 4, 0, 0).blocks); +} + +test "open flag → 9P mode mapping" { + const rdonly: u32 = @bitCast(linux.O{ .ACCMODE = .RDONLY }); + const wronly: u32 = @bitCast(linux.O{ .ACCMODE = .WRONLY }); + const rdwr: u32 = @bitCast(linux.O{ .ACCMODE = .RDWR }); + const trunc: u32 = @bitCast(linux.O{ .TRUNC = true }); + const append: u32 = @bitCast(linux.O{ .APPEND = true }); + const creat: u32 = @bitCast(linux.O{ .CREAT = true }); + try testing.expectEqual(cloud9.oread, openMode(rdonly)); + try testing.expectEqual(cloud9.owrite, openMode(wronly)); + try testing.expectEqual(cloud9.ordwr, openMode(rdwr)); + try testing.expectEqual(cloud9.owrite | cloud9.otrunc, openMode(wronly | trunc)); + try testing.expectEqual(cloud9.ordwr | cloud9.otrunc, openMode(rdwr | trunc | creat)); + try testing.expectEqual(cloud9.owrite, openMode(wronly | append)); // O_APPEND ignored +} + +test "dirlist parsing from two hand-encoded Stat records" { + var buf: [512]u8 = undefined; + const a = try cloud9.Stat.encode(testStat("alpha", 0o644, 10, 0x11), &buf); + const bb = try cloud9.Stat.encode(testStat("beta", cloud9.dmdir | 0o755, 0, 0x22), buf[a.len..]); + const bytes = buf[0 .. a.len + bb.len]; + // Sanity: the record is prefixed by its own 2-byte size. + try testing.expectEqual(a.len - 2, std.mem.readInt(u16, bytes[0..2], .little)); + + var list: DirList = .{}; + defer list.deinit(testing.allocator); + try parseDirRecords(testing.allocator, bytes, &list); + try testing.expectEqual(@as(usize, 2), list.entries.items.len); + try testing.expectEqualStrings("alpha", list.entries.items[0].name); + try testing.expectEqual(@as(u64, 0x11), list.entries.items[0].ino); + try testing.expectEqual(fuse.DT_REG, list.entries.items[0].dtype); + try testing.expectEqualStrings("beta", list.entries.items[1].name); + try testing.expectEqual(@as(u64, 0x22), list.entries.items[1].ino); + try testing.expectEqual(fuse.DT_DIR, list.entries.items[1].dtype); + + // Truncated input is a protocol error and leaves earlier entries intact. + try testing.expectError(error.BadDir, parseDirRecords(testing.allocator, bytes[0 .. bytes.len - 1], &list)); + try testing.expectEqual(@as(usize, 3), list.entries.items.len); +} + +test "readdir packing and offset resumption" { + const names = [_][]const u8{ ".", "..", "one", "two", "three" }; + var entries: [names.len]Entry = undefined; + for (&entries, names, 0..) |*e, n, i| e.* = .{ .name = @constCast(n), .ino = 100 + i, .dtype = if (i < 2) fuse.DT_DIR else fuse.DT_REG }; + + // Everything fits: five records, off = index + 1. + var big: [1024]u8 = undefined; + const used = packDirents(&entries, 0, &big); + var pos: usize = 0; + var idx: usize = 0; + while (pos < used) : (idx += 1) { + const d = std.mem.bytesToValue(fuse.Dirent, big[pos..][0..@sizeOf(fuse.Dirent)]); + try testing.expectEqual(@as(u64, 100 + idx), d.ino); + try testing.expectEqual(@as(u64, idx + 1), d.off); + try testing.expectEqualStrings(names[idx], big[pos + @sizeOf(fuse.Dirent) ..][0..d.namelen]); + pos += (@sizeOf(fuse.Dirent) + d.namelen + 7) & ~@as(usize, 7); + } + try testing.expectEqual(names.len, idx); + + // A buffer that fits exactly two records ("." = 32, ".." = 32) stops there… + var small: [64]u8 = undefined; + const first_used = packDirents(&entries, 0, &small); + try testing.expectEqual(@as(usize, 64), first_used); + const last = std.mem.bytesToValue(fuse.Dirent, small[32..][0..@sizeOf(fuse.Dirent)]); + try testing.expectEqual(@as(u64, 2), last.off); + // …and resuming at the last `off` yields "one" next. + const second_used = packDirents(&entries, last.off, &small); + const next = std.mem.bytesToValue(fuse.Dirent, small[0..@sizeOf(fuse.Dirent)]); + try testing.expectEqualStrings("one", small[@sizeOf(fuse.Dirent)..][0..next.namelen]); + try testing.expectEqual(@as(u64, 3), next.off); + try testing.expect(second_used > 0); + + // Past the end: nothing (EOF for the kernel). + try testing.expectEqual(@as(usize, 0), packDirents(&entries, names.len, &big)); + try testing.expectEqual(@as(usize, 0), packDirents(&entries, 1000, &big)); +} + +test "attr mapping saturates hostile lengths instead of overflowing" { + const a = attrFromStat(testStat("f", 0o600, std.math.maxInt(u64), 4), 4, 0, 0); + try testing.expectEqual(@as(u64, std.math.maxInt(i64)), a.size); + try testing.expectEqual(@as(u64, std.math.maxInt(u64) / 512 + 1), a.blocks); + const b = attrFromStat(testStat("f", 0o600, 1024, 4), 4, 0, 0); + try testing.expectEqual(@as(u64, 2), b.blocks); + try testing.expectEqual(@as(u64, 1024), b.size); +} + +test "dirent names the kernel would reject are dropped from listings" { + try testing.expect(validDirentName("a")); + try testing.expect(validDirentName("x" ** 1024)); + try testing.expect(!validDirentName("")); + try testing.expect(!validDirentName("a/b")); + try testing.expect(!validDirentName("a\x00b")); + try testing.expect(!validDirentName(".")); + try testing.expect(!validDirentName("..")); + try testing.expect(!validDirentName("x" ** 1025)); + + var buf: [4096]u8 = undefined; + var n: usize = 0; + for ([_][]const u8{ ".", "..", "", "a/b", "keep", "x" ** 1025, "also" }) |name| { + n += (try cloud9.Stat.encode(testStat(name, 0o644, 1, 0x30), buf[n..])).len; + } + var list: DirList = .{}; + defer list.deinit(testing.allocator); + try parseDirRecords(testing.allocator, buf[0..n], &list); + try testing.expectEqual(@as(usize, 2), list.entries.items.len); + try testing.expectEqualStrings("keep", list.entries.items[0].name); + try testing.expectEqualStrings("also", list.entries.items[1].name); +} + +test "DirList frees its names" { + var list: DirList = .{}; + try list.entries.append(testing.allocator, .{ .name = try testing.allocator.dupe(u8, "x"), .ino = 1, .dtype = fuse.DT_REG }); + list.deinit(testing.allocator); +} diff --git a/9player/src/fuse.zig b/9player/src/fuse.zig new file mode 100644 index 0000000..ef11873 --- /dev/null +++ b/9player/src/fuse.zig @@ -0,0 +1,653 @@ +//! Kernel FUSE protocol subset (no libfuse, no libc, no policy). +//! +//! Extern structs mirror `/usr/include/linux/fuse.h` (kernel header 7.45); +//! every layout is checked against the header's size at comptime. Only the +//! opcodes and structs 9player needs are here. The I/O helpers are blocking +//! and allocation-free: the caller owns a single request buffer. +//! +//! Wire rules worth remembering: +//! * The kernel delivers exactly one request per `read(2)` on `/dev/fuse`, +//! and a reply must be exactly one `write(2)`/`writev(2)`. +//! * Request bodies start right after the 40-byte `InHeader`; since every +//! in-struct is 8-byte aligned in the header, `body()` requires the caller's +//! buffer to be 8-byte aligned (`std.heap` page allocations and +//! `align(8)` arrays both qualify). +//! * A write that fails with `ENOENT` means the request was interrupted and +//! the kernel already forgot it: the reply is silently dropped. +//! * `ENODEV` on read means the filesystem was unmounted. + +const std = @import("std"); +const linux = std.os.linux; + +pub const kernel_version: u32 = 7; +/// The minor we answer; the kernel adapts to the lower of the two. +pub const kernel_minor: u32 = 31; +pub const root_id: u64 = 1; + +pub const FOPEN_DIRECT_IO: u32 = 1 << 0; +pub const FOPEN_KEEP_CACHE: u32 = 1 << 1; +pub const FOPEN_NONSEEKABLE: u32 = 1 << 2; + +pub const FUSE_ASYNC_READ: u32 = 1 << 0; +/// The kernel passes O_TRUNC in OPEN instead of a separate SETATTR(size=0); 9P has OTRUNC for exactly this. +pub const FUSE_ATOMIC_O_TRUNC: u32 = 1 << 3; +/// Without this the kernel's cached-write path (`--no-direct-io`) sends one 4 KiB WRITE per page. +pub const FUSE_BIG_WRITES: u32 = 1 << 5; +/// Re-fetch a cached inode's size/mtime and drop stale pages when they change. +/// Required for `--no-direct-io` correctness: 9P sizes change under us, and +/// without this the kernel trusts a stale cached size and truncates reads. +pub const FUSE_AUTO_INVAL_DATA: u32 = 1 << 12; +pub const FUSE_MAX_PAGES: u32 = 1 << 22; + +pub const FATTR_MODE: u32 = 1 << 0; +pub const FATTR_UID: u32 = 1 << 1; +pub const FATTR_GID: u32 = 1 << 2; +pub const FATTR_SIZE: u32 = 1 << 3; +pub const FATTR_ATIME: u32 = 1 << 4; +pub const FATTR_MTIME: u32 = 1 << 5; +pub const FATTR_FH: u32 = 1 << 6; +pub const FATTR_ATIME_NOW: u32 = 1 << 7; +pub const FATTR_MTIME_NOW: u32 = 1 << 8; +pub const FATTR_LOCKOWNER: u32 = 1 << 9; +pub const FATTR_CTIME: u32 = 1 << 10; + +/// File type bits for `Attr.mode` and `Dirent.type` (used by the bridge). +pub const S_IFDIR: u32 = linux.S.IFDIR; +pub const S_IFREG: u32 = linux.S.IFREG; +pub const DT_DIR: u32 = linux.DT.DIR; +pub const DT_REG: u32 = linux.DT.REG; + +pub const Opcode = enum(u32) { + lookup = 1, + forget = 2, + getattr = 3, + setattr = 4, + readlink = 5, + symlink = 6, + mknod = 8, + mkdir = 9, + unlink = 10, + rmdir = 11, + rename = 12, + link = 13, + open = 14, + read = 15, + write = 16, + statfs = 17, + release = 18, + fsync = 20, + setxattr = 21, + getxattr = 22, + listxattr = 23, + removexattr = 24, + flush = 25, + init = 26, + opendir = 27, + readdir = 28, + releasedir = 29, + fsyncdir = 30, + getlk = 31, + setlk = 32, + setlkw = 33, + access = 34, + create = 35, + interrupt = 36, + bmap = 37, + destroy = 38, + ioctl = 39, + poll = 40, + notify_reply = 41, + batch_forget = 42, + fallocate = 43, + readdirplus = 44, + rename2 = 45, + lseek = 46, + copy_file_range = 47, + setupmapping = 48, + removemapping = 49, + syncfs = 50, + tmpfile = 51, + statx = 52, + _, +}; + +// --------------------------------------------------------------------------- +// Structs (field order and widths follow linux/fuse.h exactly) +// --------------------------------------------------------------------------- + +pub const InHeader = extern struct { + len: u32, + opcode: u32, + unique: u64, + nodeid: u64, + uid: u32, + gid: u32, + pid: u32, + total_extlen: u16, + padding: u16, + + pub fn op(h: InHeader) Opcode { + return @enumFromInt(h.opcode); + } +}; + +pub const OutHeader = extern struct { + len: u32, + @"error": i32, + unique: u64, +}; + +pub const Attr = extern struct { + ino: u64 = 0, + size: u64 = 0, + blocks: u64 = 0, + atime: u64 = 0, + mtime: u64 = 0, + ctime: u64 = 0, + atimensec: u32 = 0, + mtimensec: u32 = 0, + ctimensec: u32 = 0, + mode: u32 = 0, + nlink: u32 = 0, + uid: u32 = 0, + gid: u32 = 0, + rdev: u32 = 0, + blksize: u32 = 0, + flags: u32 = 0, +}; + +pub const EntryOut = extern struct { + nodeid: u64 = 0, + generation: u64 = 0, + entry_valid: u64 = 0, + attr_valid: u64 = 0, + entry_valid_nsec: u32 = 0, + attr_valid_nsec: u32 = 0, + attr: Attr = .{}, +}; + +pub const AttrOut = extern struct { + attr_valid: u64 = 0, + attr_valid_nsec: u32 = 0, + dummy: u32 = 0, + attr: Attr = .{}, +}; + +pub const GetattrIn = extern struct { getattr_flags: u32, dummy: u32, fh: u64 }; + +pub const SetattrIn = extern struct { + valid: u32, + padding: u32, + fh: u64, + size: u64, + lock_owner: u64, + atime: u64, + mtime: u64, + ctime: u64, + atimensec: u32, + mtimensec: u32, + ctimensec: u32, + mode: u32, + unused4: u32, + uid: u32, + gid: u32, + unused5: u32, +}; + +pub const OpenIn = extern struct { flags: u32, open_flags: u32 }; +pub const OpenOut = extern struct { fh: u64 = 0, open_flags: u32 = 0, backing_id: i32 = 0 }; +pub const ReleaseIn = extern struct { fh: u64, flags: u32, release_flags: u32, lock_owner: u64 }; +pub const FlushIn = extern struct { fh: u64, unused: u32, padding: u32, lock_owner: u64 }; + +pub const ReadIn = extern struct { + fh: u64, + offset: u64, + size: u32, + read_flags: u32, + lock_owner: u64, + flags: u32, + padding: u32, +}; + +pub const WriteIn = extern struct { + fh: u64, + offset: u64, + size: u32, + write_flags: u32, + lock_owner: u64, + flags: u32, + padding: u32, +}; + +pub const WriteOut = extern struct { size: u32, padding: u32 = 0 }; +pub const CreateIn = extern struct { flags: u32, mode: u32, umask: u32, open_flags: u32 }; +pub const MkdirIn = extern struct { mode: u32, umask: u32 }; +pub const RenameIn = extern struct { newdir: u64 }; +pub const Rename2In = extern struct { newdir: u64, flags: u32, padding: u32 }; +pub const ForgetIn = extern struct { nlookup: u64 }; +pub const BatchForgetIn = extern struct { count: u32, dummy: u32 }; +pub const ForgetOne = extern struct { nodeid: u64, nlookup: u64 }; +pub const FsyncIn = extern struct { fh: u64, fsync_flags: u32, padding: u32 }; +pub const AccessIn = extern struct { mask: u32, padding: u32 }; +pub const InterruptIn = extern struct { unique: u64 }; +pub const LseekIn = extern struct { fh: u64, offset: u64, whence: u32, padding: u32 }; + +pub const Kstatfs = extern struct { + blocks: u64 = 0, + bfree: u64 = 0, + bavail: u64 = 0, + files: u64 = 0, + ffree: u64 = 0, + bsize: u32 = 0, + namelen: u32 = 0, + frsize: u32 = 0, + padding: u32 = 0, + spare: [6]u32 = [_]u32{0} ** 6, +}; + +pub const StatfsOut = extern struct { st: Kstatfs = .{} }; + +pub const InitIn = extern struct { + major: u32, + minor: u32, + max_readahead: u32, + flags: u32, + flags2: u32, + unused: [11]u32, +}; + +/// 64 bytes; the kernel accepts this size whenever the answered minor >= 23. +pub const InitOut = extern struct { + major: u32 = kernel_version, + minor: u32 = kernel_minor, + max_readahead: u32 = 0, + flags: u32 = 0, + max_background: u16 = 0, + congestion_threshold: u16 = 0, + max_write: u32 = 0, + time_gran: u32 = 0, + max_pages: u16 = 0, + map_alignment: u16 = 0, + flags2: u32 = 0, + max_stack_depth: u32 = 0, + request_timeout: u16 = 0, + unused: [11]u16 = [_]u16{0} ** 11, +}; + +/// Fixed 24-byte head of `fuse_dirent`; the name follows, padded to 8 bytes. +pub const Dirent = extern struct { ino: u64, off: u64, namelen: u32, type: u32 }; + +comptime { + std.debug.assert(@sizeOf(InHeader) == 40); + std.debug.assert(@sizeOf(OutHeader) == 16); + std.debug.assert(@sizeOf(Attr) == 88); + std.debug.assert(@sizeOf(EntryOut) == 128); + std.debug.assert(@sizeOf(AttrOut) == 104); + std.debug.assert(@sizeOf(GetattrIn) == 16); + std.debug.assert(@sizeOf(SetattrIn) == 88); + std.debug.assert(@sizeOf(OpenIn) == 8); + std.debug.assert(@sizeOf(OpenOut) == 16); + std.debug.assert(@sizeOf(ReleaseIn) == 24); + std.debug.assert(@sizeOf(FlushIn) == 24); + std.debug.assert(@sizeOf(ReadIn) == 40); + std.debug.assert(@sizeOf(WriteIn) == 40); + std.debug.assert(@sizeOf(WriteOut) == 8); + std.debug.assert(@sizeOf(CreateIn) == 16); + std.debug.assert(@sizeOf(MkdirIn) == 8); + std.debug.assert(@sizeOf(RenameIn) == 8); + std.debug.assert(@sizeOf(Rename2In) == 16); + std.debug.assert(@sizeOf(ForgetIn) == 8); + std.debug.assert(@sizeOf(BatchForgetIn) == 8); + std.debug.assert(@sizeOf(ForgetOne) == 16); + std.debug.assert(@sizeOf(FsyncIn) == 16); + std.debug.assert(@sizeOf(AccessIn) == 8); + std.debug.assert(@sizeOf(InterruptIn) == 8); + std.debug.assert(@sizeOf(Kstatfs) == 80); + std.debug.assert(@sizeOf(StatfsOut) == 80); + std.debug.assert(@sizeOf(InitIn) == 64); + std.debug.assert(@sizeOf(InitOut) == 64); + std.debug.assert(@sizeOf(Dirent) == 24); + std.debug.assert(@sizeOf(LseekIn) == 24); +} + +// --------------------------------------------------------------------------- +// Request / reply helpers +// --------------------------------------------------------------------------- + +pub const Error = error{ Protocol, Io, TooManyPayloads }; + +pub const Request = struct { + header: InHeader, + /// Bytes after the header; a slice into the caller's buffer. + body: []const u8, +}; + +/// Reads one kernel request with a single `read(2)`. Returns null on ENODEV +/// (unmounted). Retries EINTR/EAGAIN/ENOENT. `buf` should be at least +/// `max_write + 4096` bytes and 8-byte aligned so `body()` can view it. +pub fn readRequest(fd: i32, buf: []u8) Error!?Request { + while (true) { + const rc = linux.read(fd, buf.ptr, buf.len); + switch (linux.errno(rc)) { + .SUCCESS => { + const n: usize = rc; + if (n < @sizeOf(InHeader)) return error.Protocol; + const header = std.mem.bytesToValue(InHeader, buf[0..@sizeOf(InHeader)]); + if (header.len != n) return error.Protocol; + return .{ .header = header, .body = buf[@sizeOf(InHeader)..n] }; + }, + .INTR, .AGAIN, .NOENT => continue, + .NODEV => return null, + else => return error.Io, + } + } +} + +/// Maximum number of payload slices a single `reply` can carry. +pub const max_payloads = 7; + +/// Success reply: `OutHeader` followed by the concatenated `payloads`, sent in +/// one `writev(2)`. An ENOENT from the kernel means the request was +/// interrupted; the reply is dropped and this returns normally. +pub fn reply(fd: i32, unique: u64, payloads: []const []const u8) Error!void { + if (payloads.len > max_payloads) return error.TooManyPayloads; + var total: usize = @sizeOf(OutHeader); + for (payloads) |p| total += p.len; + if (total > std.math.maxInt(u32)) return error.Protocol; + const header = OutHeader{ .len = @intCast(total), .@"error" = 0, .unique = unique }; + var iov: [max_payloads + 1]std.posix.iovec_const = undefined; + iov[0] = .{ .base = @ptrCast(&header), .len = @sizeOf(OutHeader) }; + for (payloads, 1..) |p, i| iov[i] = .{ .base = p.ptr, .len = p.len }; + return writeAll(fd, &iov, payloads.len + 1, total); +} + +/// Error reply: an `OutHeader` carrying `-errno` and no payload. +pub fn replyError(fd: i32, unique: u64, err: linux.E) Error!void { + const code: i32 = @intCast(@intFromEnum(err)); + const header = OutHeader{ .len = @sizeOf(OutHeader), .@"error" = -code, .unique = unique }; + var iov = [_]std.posix.iovec_const{.{ .base = @ptrCast(&header), .len = @sizeOf(OutHeader) }}; + return writeAll(fd, &iov, 1, @sizeOf(OutHeader)); +} + +fn writeAll(fd: i32, iov: [*]const std.posix.iovec_const, count: usize, total: usize) Error!void { + while (true) { + const rc = linux.writev(fd, iov, count); + switch (linux.errno(rc)) { + .SUCCESS => return if (rc == total) {} else error.Protocol, + .INTR => continue, + .NOENT => return, // request was interrupted; reply dropped + else => return error.Io, + } + } +} + +/// Appends a `fuse_dirent` (head + name, padded to a multiple of 8) at +/// `buf[used.*..]`. Returns false and leaves `buf`/`used` unchanged if the +/// record does not fit. +pub fn addDirent(buf: []u8, used: *usize, ino: u64, off: u64, dtype: u32, name: []const u8) bool { + const raw = @sizeOf(Dirent) + name.len; + const rec = (raw + 7) & ~@as(usize, 7); + if (used.* > buf.len or buf.len - used.* < rec) return false; + const dst = buf[used.*..][0..rec]; + const head = Dirent{ .ino = ino, .off = off, .namelen = @intCast(name.len), .type = dtype }; + @memcpy(dst[0..@sizeOf(Dirent)], std.mem.asBytes(&head)); + @memcpy(dst[@sizeOf(Dirent)..raw], name); + @memset(dst[raw..rec], 0); + used.* += rec; + return true; +} + +/// Views the first `@sizeOf(T)` bytes of `req.body` as `T` (copy-free). +/// Fails with `error.Protocol` if the body is too short or misaligned. +pub fn body(comptime T: type, req: Request) Error!*const T { + if (req.body.len < @sizeOf(T)) return error.Protocol; + if (@intFromPtr(req.body.ptr) % @alignOf(T) != 0) return error.Protocol; + return @ptrCast(@alignCast(req.body.ptr)); +} + +/// The NUL-terminated string at `req.body[offset..]`, without the NUL. +pub fn nameAt(req: Request, offset: usize) Error![]const u8 { + if (offset > req.body.len) return error.Protocol; + const rest = req.body[offset..]; + const end = std.mem.indexOfScalar(u8, rest, 0) orelse return error.Protocol; + return rest[0..end]; +} + +/// The NUL-terminated string following a `T` body (or at offset 0 when +/// `T == void`), e.g. LOOKUP's name (`void`) or MKDIR's name (`MkdirIn`). +pub fn nameAfter(comptime T: type, req: Request) Error![]const u8 { + const offset = if (T == void) 0 else @sizeOf(T); + return nameAt(req, offset); +} + +/// The string that follows `first` (obtained via `nameAt(req, offset)`), +/// for "old\0new\0" pairs such as RENAME's. +pub fn secondName(req: Request, first: []const u8, offset: usize) Error![]const u8 { + return nameAt(req, offset + first.len + 1); +} + +/// Builds the INIT reply per docs/DESIGN.md. +pub fn initReply(in: *const InitIn, max_write: u32) InitOut { + var out = InitOut{ + .major = kernel_version, + .minor = @min(kernel_minor, in.minor), + .max_readahead = in.max_readahead, + .flags = FUSE_ASYNC_READ | FUSE_ATOMIC_O_TRUNC | FUSE_BIG_WRITES | FUSE_AUTO_INVAL_DATA, + .max_background = 16, + .congestion_threshold = 12, + .max_write = max_write, + .time_gran = 1, + }; + if (in.flags & FUSE_MAX_PAGES != 0) { + out.flags |= FUSE_MAX_PAGES; + out.max_pages = 256; + } + return out; +} + +// --------------------------------------------------------------------------- +// Tests +// --------------------------------------------------------------------------- + +const testing = std.testing; + +test "struct sizes match linux/fuse.h" { + // The comptime block above is the real check; this makes it run under + // `zig test` even if the module is otherwise unreferenced. + try testing.expectEqual(@as(usize, 40), @sizeOf(InHeader)); + try testing.expectEqual(@as(usize, 64), @sizeOf(InitOut)); + try testing.expectEqual(@as(usize, 24), @sizeOf(Dirent)); + try testing.expectEqual(@as(u32, 26), @intFromEnum(Opcode.init)); + try testing.expectEqual(Opcode.statx, @as(Opcode, @enumFromInt(52))); +} + +test "addDirent pads records to 8 bytes and refuses when full" { + var buf: [1024]u8 = undefined; + var used: usize = 0; + const name = "abcdefghijklmnopq"; // 17 chars + var expect_total: usize = 0; + var n: usize = 1; + while (n <= 17) : (n += 1) { + const before = used; + try testing.expect(addDirent(&buf, &used, n, n, DT_REG, name[0..n])); + const rec = used - before; + try testing.expectEqual(@as(usize, 0), rec % 8); + try testing.expectEqual((24 + n + 7) & ~@as(usize, 7), rec); + // check head fields and NUL padding + const head = std.mem.bytesToValue(Dirent, buf[before..][0..24]); + try testing.expectEqual(n, head.ino); + try testing.expectEqual(@as(u32, @intCast(n)), head.namelen); + try testing.expectEqualStrings(name[0..n], buf[before + 24 ..][0..n]); + for (buf[before + 24 + n .. used]) |b| try testing.expectEqual(@as(u8, 0), b); + expect_total += rec; + } + try testing.expectEqual(expect_total, used); + + // A record that does not fit leaves everything untouched. + var small: [40]u8 = undefined; + var used2: usize = 0; + try testing.expect(addDirent(&small, &used2, 1, 1, DT_DIR, "0123456789abcdef")); // 24+16 = 40 + try testing.expectEqual(@as(usize, 40), used2); + try testing.expect(!addDirent(&small, &used2, 2, 2, DT_DIR, "x")); + try testing.expectEqual(@as(usize, 40), used2); + var tight: [31]u8 = undefined; + var used3: usize = 0; + try testing.expect(!addDirent(&tight, &used3, 1, 1, DT_REG, "a")); // needs 32 + try testing.expectEqual(@as(usize, 0), used3); +} + +test "body/nameAfter/secondName on hand-built requests" { + var buf: [128]u8 align(8) = undefined; + // LOOKUP(parent=1, "hello") + const name = "hello"; + const hdr = InHeader{ + .len = @intCast(@sizeOf(InHeader) + name.len + 1), + .opcode = @intFromEnum(Opcode.lookup), + .unique = 7, + .nodeid = root_id, + .uid = 1000, + .gid = 1000, + .pid = 42, + .total_extlen = 0, + .padding = 0, + }; + @memcpy(buf[0..40], std.mem.asBytes(&hdr)); + @memcpy(buf[40..45], name); + buf[45] = 0; + const req = Request{ .header = hdr, .body = buf[40..hdr.len] }; + try testing.expectEqual(Opcode.lookup, req.header.op()); + try testing.expectEqualStrings("hello", try nameAfter(void, req)); + try testing.expectError(error.Protocol, body(MkdirIn, Request{ .header = hdr, .body = buf[40..44] })); + + // MKDIR(mode=0o755) + "dir" + const mk = MkdirIn{ .mode = 0o755, .umask = 0o22 }; + @memcpy(buf[40..48], std.mem.asBytes(&mk)); + @memcpy(buf[48..51], "dir"); + buf[51] = 0; + const mreq = Request{ .header = hdr, .body = buf[40..52] }; + const got = try body(MkdirIn, mreq); + try testing.expectEqual(@as(u32, 0o755), got.mode); + try testing.expectEqualStrings("dir", try nameAfter(MkdirIn, mreq)); + + // RENAME(newdir) + "old\0new\0" + const rn = RenameIn{ .newdir = 9 }; + @memcpy(buf[40..48], std.mem.asBytes(&rn)); + @memcpy(buf[48..56], "old\x00new\x00"); + const rreq = Request{ .header = hdr, .body = buf[40..56] }; + try testing.expectEqual(@as(u64, 9), (try body(RenameIn, rreq)).newdir); + const old = try nameAfter(RenameIn, rreq); + try testing.expectEqualStrings("old", old); + try testing.expectEqualStrings("new", try secondName(rreq, old, @sizeOf(RenameIn))); + try testing.expectError(error.Protocol, secondName(rreq, "new", @sizeOf(RenameIn) + 4)); + + // Missing NUL and misalignment are protocol errors. + try testing.expectError(error.Protocol, nameAt(Request{ .header = hdr, .body = buf[48..51] }, 0)); + try testing.expectError(error.Protocol, body(MkdirIn, Request{ .header = hdr, .body = buf[41..57] })); +} + +test "initReply fields" { + var in = InitIn{ .major = 7, .minor = 45, .max_readahead = 131072, .flags = 0, .flags2 = 0, .unused = [_]u32{0} ** 11 }; + const a = initReply(&in, 1 << 20); + try testing.expectEqual(@as(u32, 7), a.major); + try testing.expectEqual(@as(u32, 31), a.minor); + try testing.expectEqual(@as(u32, 131072), a.max_readahead); + try testing.expectEqual(FUSE_ASYNC_READ | FUSE_ATOMIC_O_TRUNC | FUSE_BIG_WRITES | FUSE_AUTO_INVAL_DATA, a.flags); + try testing.expectEqual(@as(u16, 0), a.max_pages); + try testing.expectEqual(@as(u16, 16), a.max_background); + try testing.expectEqual(@as(u16, 12), a.congestion_threshold); + try testing.expectEqual(@as(u32, 1 << 20), a.max_write); + try testing.expectEqual(@as(u32, 1), a.time_gran); + + in.flags = FUSE_MAX_PAGES | FUSE_ASYNC_READ; + in.minor = 27; + const b = initReply(&in, 4096); + try testing.expectEqual(@as(u32, 27), b.minor); + try testing.expectEqual(FUSE_ASYNC_READ | FUSE_ATOMIC_O_TRUNC | FUSE_BIG_WRITES | FUSE_AUTO_INVAL_DATA | FUSE_MAX_PAGES, b.flags); + try testing.expectEqual(@as(u16, 256), b.max_pages); + try testing.expectEqual(@as(u32, 4096), b.max_write); +} + +fn makePipe() ![2]i32 { + var fds: [2]i32 = undefined; + if (linux.errno(linux.pipe2(&fds, .{ .CLOEXEC = true })) != .SUCCESS) return error.Io; + return fds; +} + +fn readExact(fd: i32, out: []u8) !void { + var got: usize = 0; + while (got < out.len) { + const rc = linux.read(fd, out[got..].ptr, out.len - got); + if (linux.errno(rc) != .SUCCESS or rc == 0) return error.Io; + got += rc; + } +} + +test "reply writes header + payloads through a pipe" { + const fds = try makePipe(); + defer _ = linux.close(fds[0]); + defer _ = linux.close(fds[1]); + + const oo = OpenOut{ .fh = 0x1234, .open_flags = FOPEN_DIRECT_IO }; + try reply(fds[1], 99, &.{ std.mem.asBytes(&oo), "tail" }); + + var out: [16 + 16 + 4]u8 = undefined; + try readExact(fds[0], &out); + const h = std.mem.bytesToValue(OutHeader, out[0..16]); + try testing.expectEqual(@as(u32, 36), h.len); + try testing.expectEqual(@as(i32, 0), h.@"error"); + try testing.expectEqual(@as(u64, 99), h.unique); + try testing.expectEqualSlices(u8, std.mem.asBytes(&oo), out[16..32]); + try testing.expectEqualStrings("tail", out[32..36]); + + // Empty payload list: header only. + try reply(fds[1], 5, &.{}); + var only: [16]u8 = undefined; + try readExact(fds[0], &only); + try testing.expectEqual(@as(u32, 16), std.mem.bytesToValue(OutHeader, &only).len); + + var too_many: [max_payloads + 1][]const u8 = undefined; + for (&too_many) |*p| p.* = "x"; + try testing.expectError(error.TooManyPayloads, reply(fds[1], 1, &too_many)); +} + +test "replyError writes a negative errno" { + const fds = try makePipe(); + defer _ = linux.close(fds[0]); + defer _ = linux.close(fds[1]); + + try replyError(fds[1], 0xdead_beef, .NOENT); + var out: [16]u8 = undefined; + try readExact(fds[0], &out); + const h = std.mem.bytesToValue(OutHeader, &out); + try testing.expectEqual(@as(u32, 16), h.len); + try testing.expectEqual(@as(i32, -2), h.@"error"); + try testing.expectEqual(@as(u64, 0xdead_beef), h.unique); + + try replyError(fds[1], 1, .NOSYS); + try readExact(fds[0], &out); + try testing.expectEqual(-@as(i32, @intCast(@intFromEnum(linux.E.NOSYS))), std.mem.bytesToValue(OutHeader, &out).@"error"); +} + +test "readRequest parses one request from a pipe and rejects bad lengths" { + const fds = try makePipe(); + defer _ = linux.close(fds[0]); + defer _ = linux.close(fds[1]); + + var wire: [48]u8 align(8) = undefined; + const hdr = InHeader{ .len = 48, .opcode = @intFromEnum(Opcode.forget), .unique = 3, .nodeid = 2, .uid = 0, .gid = 0, .pid = 0, .total_extlen = 0, .padding = 0 }; + @memcpy(wire[0..40], std.mem.asBytes(&hdr)); + @memcpy(wire[40..48], std.mem.asBytes(&ForgetIn{ .nlookup = 11 })); + try testing.expectEqual(@as(usize, 48), linux.write(fds[1], &wire, wire.len)); + + var buf: [4096]u8 align(8) = undefined; + const req = (try readRequest(fds[0], &buf)) orelse return error.Io; + try testing.expectEqual(Opcode.forget, req.header.op()); + try testing.expectEqual(@as(u64, 2), req.header.nodeid); + try testing.expectEqual(@as(u64, 11), (try body(ForgetIn, req)).nlookup); + + // Header length disagreeing with what was read is a protocol error. + var bad = wire; + std.mem.bytesAsValue(InHeader, bad[0..40]).len = 40; + try testing.expectEqual(@as(usize, 48), linux.write(fds[1], &bad, bad.len)); + try testing.expectError(error.Protocol, readRequest(fds[0], &buf)); +} diff --git a/9player/src/main.zig b/9player/src/main.zig new file mode 100644 index 0000000..24990b9 --- /dev/null +++ b/9player/src/main.zig @@ -0,0 +1,444 @@ +//! 9player: mount a 9P2000 tree into a fresh user+mount namespace via FUSE +//! and run a program inside it. +//! +//! Exit codes: the child's status (128+sig if signalled); 125 for 9player's +//! own failures (usage, connect, attach, namespace/mount); 126/127 for exec +//! failures. + +const std = @import("std"); +const linux = std.os.linux; +const ns = @import("ns.zig"); +const nine = @import("nine.zig"); +const bridge = @import("bridge.zig"); + +const version_string = "9player 0.1.0"; + +const usage_text = + \\Usage: 9player [options] -- PROGRAM [ARGS...] + \\Transport (exactly one): + \\ --unix PATH Unix stream socket + \\ --tcp IP:PORT TCP (IPv4/IPv6 literal) + \\ --fd N already-connected inherited descriptor + \\ --spawn CMD run CMD (via /bin/sh -c) with a socketpair on its stdin/stdout + \\Options: + \\ --mount PATH mountpoint inside the new namespace (default /mnt/9p) + \\ --uname NAME 9P user name (default $USER, else "none") + \\ --aname NAME 9P tree to attach (default "") + \\ --msize BYTES maximum 9P message size to request (default 131072) + \\ --cache SECONDS attr/entry cache validity, may be fractional (default 1) + \\ --no-direct-io let the kernel cache file pages (trusts stat length) + \\ --debug trace FUSE and 9P operations on stderr + \\ --help, --version + \\PROGRAM defaults to $SHELL (else /bin/sh). The mountpoint is exported as $NINEPLAYER_MOUNT. + \\ +; + +const own_failure: u8 = 125; +/// Largest 9P message size we agree to request: the session allocates two +/// buffers of this size up front, before the server negotiates it down. +const max_msize: u32 = 16 * 1024 * 1024; + +/// Write `text` to stdout (informational output such as --help); errors are +/// ignored, there is nowhere better to report them. +fn printStdout(text: []const u8) void { + var off: usize = 0; + while (off < text.len) { + const rc = linux.write(1, text[off..].ptr, text.len - off); + switch (linux.errno(rc)) { + .SUCCESS => off += rc, + .INTR => continue, + else => return, + } + } +} + +const Config = struct { + address: ?nine.Address = null, + spawn_cmd: ?[]const u8 = null, + mount: []const u8 = "/mnt/9p", + uname: ?[]const u8 = null, + aname: []const u8 = "", + msize: u32 = 131072, + cache_ns: u64 = 1_000_000_000, + direct_io: bool = true, + debug: bool = false, + /// Empty means "default program". + program: []const []const u8 = &.{}, +}; + +const ParseResult = union(enum) { + run: Config, + /// Usage error, already reported on stderr; exit with this status. + exit: u8, + /// --help/--version: text for stdout, then exit 0. Printing is left to + /// `main` so that no test path writes to fd 1 (under `zig build test` + /// that is the test runner's protocol pipe). + info: []const u8, +}; + +fn usageError(comptime fmt: []const u8, args: anytype) ParseResult { + std.debug.print("9player: " ++ fmt ++ "\n(try 9player --help)\n", args); + return .{ .exit = own_failure }; +} + +fn parseArgs(arena: std.mem.Allocator, args: []const [:0]const u8) !ParseResult { + var cfg = Config{}; + var transports: usize = 0; + var i: usize = 1; + var program_start: ?usize = null; + while (i < args.len) : (i += 1) { + const arg: []const u8 = args[i]; + if (std.mem.eql(u8, arg, "--")) { + program_start = i + 1; + break; + } + if (!std.mem.startsWith(u8, arg, "--")) { + // A single-dash word is a typo for an option, not a program. + if (arg.len > 1 and arg[0] == '-') return usageError("unknown option {s} (options start with --)", .{arg}); + // A bare word starts PROGRAM, as if "--" were given. + program_start = i; + break; + } + // Split "--opt=value". + var name = arg; + var inline_value: ?[]const u8 = null; + if (std.mem.indexOfScalar(u8, arg, '=')) |eq| { + name = arg[0..eq]; + inline_value = arg[eq + 1 ..]; + } + const Opt = enum { unix, tcp, fd, spawn, mount, uname, aname, msize, cache, @"no-direct-io", debug, help, version, unknown }; + const opt = std.meta.stringToEnum(Opt, name[2..]) orelse .unknown; + switch (opt) { + .@"no-direct-io", .debug, .help, .version => if (inline_value != null) return usageError("{s} takes no value", .{name}), + .unknown => return usageError("unknown option {s}", .{name}), + else => {}, + } + const value: []const u8 = switch (opt) { + .@"no-direct-io", .debug, .help, .version, .unknown => "", + else => inline_value orelse blk: { + i += 1; + if (i >= args.len) return usageError("{s} needs a value", .{name}); + break :blk args[i]; + }, + }; + switch (opt) { + .unix => { + if (value.len == 0) return usageError("--unix wants a socket path", .{}); + cfg.address = .{ .unix = value }; + transports += 1; + }, + .tcp => { + cfg.address = parseTcp(value) orelse return usageError("--tcp wants IP:PORT (IPv6 as [ADDR]:PORT), got '{s}'", .{value}); + transports += 1; + }, + .fd => { + const n = std.fmt.parseInt(i32, value, 10) catch return usageError("--fd wants a number, got '{s}'", .{value}); + if (n < 0) return usageError("--fd wants a non-negative number", .{}); + cfg.address = .{ .fd = n }; + transports += 1; + }, + .spawn => { + if (value.len == 0) return usageError("--spawn wants a command", .{}); + cfg.spawn_cmd = value; + transports += 1; + }, + .mount => { + if (value.len == 0) return usageError("--mount wants a path", .{}); + cfg.mount = value; + }, + .uname => cfg.uname = value, + .aname => cfg.aname = value, + .msize => { + cfg.msize = std.fmt.parseInt(u32, value, 10) catch return usageError("--msize wants a number, got '{s}'", .{value}); + if (cfg.msize < 4096 or cfg.msize > max_msize) return usageError("--msize must be between 4096 and {d}", .{max_msize}); + }, + .cache => { + const secs = std.fmt.parseFloat(f64, value) catch return usageError("--cache wants seconds, got '{s}'", .{value}); + if (!(secs >= 0) or secs > 1e9) return usageError("--cache out of range", .{}); + cfg.cache_ns = @intFromFloat(secs * 1e9); + }, + .@"no-direct-io" => cfg.direct_io = false, + .debug => cfg.debug = true, + .help => return .{ .info = usage_text }, + .version => return .{ .info = version_string ++ "\n" }, + .unknown => unreachable, + } + } + if (transports == 0) return usageError("one transport is required (--unix, --tcp, --fd or --spawn)", .{}); + if (transports > 1) return usageError("exactly one transport is allowed", .{}); + if (program_start) |start| { + const prog = try arena.alloc([]const u8, args.len - start); + for (args[start..], 0..) |a, j| prog[j] = a; + cfg.program = prog; + } + return .{ .run = cfg }; +} + +fn parseTcp(spec: []const u8) ?nine.Address { + const colon = std.mem.lastIndexOfScalar(u8, spec, ':') orelse return null; + var host = spec[0..colon]; + if (host.len >= 2 and host[0] == '[' and host[host.len - 1] == ']') host = host[1 .. host.len - 1]; + if (host.len == 0) return null; + const port = std.fmt.parseInt(u16, spec[colon + 1 ..], 10) catch return null; + return .{ .tcp = .{ .host = host, .port = port } }; +} + +/// `--spawn`: run CMD under /bin/sh with one end of a socketpair as its +/// stdin/stdout; the other end is the 9P transport. +const Server = struct { pid: i32, fd: i32 }; + +fn spawnServer(cmd: [:0]const u8, envp: [*:null]const ?[*:0]const u8) !Server { + var sv: [2]i32 = undefined; + switch (linux.errno(linux.socketpair(linux.AF.UNIX, linux.SOCK.STREAM | linux.SOCK.CLOEXEC, 0, &sv))) { + .SUCCESS => {}, + else => |e| { + std.debug.print("9player: socketpair: E{t}\n", .{e}); + return error.SystemResources; + }, + } + const rc = linux.fork(); + switch (linux.errno(rc)) { + .SUCCESS => {}, + else => |e| { + _ = linux.close(sv[0]); + _ = linux.close(sv[1]); + std.debug.print("9player: fork: E{t}\n", .{e}); + return error.SystemResources; + }, + } + if (rc == 0) { + // Child: dup2 clears CLOEXEC on 0 and 1; everything else is CLOEXEC. + if (linux.errno(linux.dup2(sv[1], 0)) != .SUCCESS or linux.errno(linux.dup2(sv[1], 1)) != .SUCCESS) linux.exit_group(125); + // The server shares our process group, so a Ctrl-C meant for the + // program would kill it and take the mount down with it: ignore the + // tty signals (inherited across exec). SIGPIPE goes back to its + // default, we only ignore it for ourselves. + ignoreSignal(.INT); + ignoreSignal(.QUIT); + defaultSignal(.PIPE); + const argv = [_:null]?[*:0]const u8{ "sh", "-c", cmd.ptr }; + const e = linux.errno(linux.execve("/bin/sh", &argv, envp)); + std.debug.print("9player: --spawn: exec /bin/sh: E{t}\n", .{e}); + linux.exit_group(127); + } + _ = linux.close(sv[1]); + return .{ .pid = @intCast(rc), .fd = sv[0] }; +} + +fn stopServer(server: ?Server) void { + const s = server orelse return; + _ = linux.kill(s.pid, .TERM); + ns.reapAny(s.pid); +} + +/// Fail early (before spawning servers or forking) if /dev/fuse is unusable. +fn probeFuseDevice() bool { + const rc = linux.open("/dev/fuse", .{ .ACCMODE = .RDWR, .CLOEXEC = true }, 0); + switch (linux.errno(rc)) { + .SUCCESS => { + _ = linux.close(@intCast(rc)); + return true; + }, + .NOENT => std.debug.print("9player: /dev/fuse: ENOENT (is the fuse module loaded? try: modprobe fuse)\n", .{}), + else => |e| std.debug.print("9player: open /dev/fuse: E{t}\n", .{e}), + } + return false; +} + +fn ignoreSignal(sig: linux.SIG) void { + const ign = linux.Sigaction{ .handler = .{ .handler = linux.SIG.IGN }, .mask = linux.sigemptyset(), .flags = 0 }; + std.posix.sigaction(sig, &ign, null); +} + +fn defaultSignal(sig: linux.SIG) void { + const dfl = linux.Sigaction{ .handler = .{ .handler = linux.SIG.DFL }, .mask = linux.sigemptyset(), .flags = 0 }; + std.posix.sigaction(sig, &dfl, null); +} + +/// `--fd N`: the descriptor is ours from now on; it must not leak into the +/// program (which could otherwise read 9P replies meant for us). Fails on a +/// bad descriptor, which is the earliest place to report it. +fn adoptFd(fd: i32) bool { + switch (linux.errno(linux.fcntl(fd, linux.F.SETFD, linux.FD_CLOEXEC))) { + .SUCCESS => return true, + else => |e| { + std.debug.print("9player: --fd {d}: E{t}\n", .{ fd, e }); + return false; + }, + } +} + +fn describeAddress(a: nine.Address, buf: []u8) []const u8 { + return switch (a) { + .unix => |p| std.fmt.bufPrint(buf, "unix socket {s}", .{p}) catch "unix socket", + .tcp => |t| std.fmt.bufPrint(buf, "tcp {s}:{d}", .{ t.host, t.port }) catch "tcp", + .fd => |fd| std.fmt.bufPrint(buf, "fd {d}", .{fd}) catch "fd", + }; +} + +pub fn main(init: std.process.Init) !u8 { + const gpa = init.gpa; + const arena = init.arena.allocator(); + const args = try init.minimal.args.toSlice(arena); + const envp: [*:null]const ?[*:0]const u8 = init.minimal.environ.block.slice.ptr; + + var cfg = switch (try parseArgs(arena, args)) { + .exit => |code| return code, + .info => |text| { + printStdout(text); + return 0; + }, + .run => |c| c, + }; + + // Defaults that come from the environment. + if (cfg.program.len == 0) { + const env_shell = ns.getenv(envp, "SHELL") orelse ""; + const shell = if (env_shell.len == 0) "/bin/sh" else env_shell; + cfg.program = try arena.dupe([]const u8, &.{shell}); + } + const uname = cfg.uname orelse ns.getenv(envp, "USER") orelse "none"; + const mountpoint = ns.resolveMountpoint(gpa, cfg.mount) catch |err| { + std.debug.print("9player: --mount {s}: {t}\n", .{ cfg.mount, err }); + return own_failure; + }; + defer gpa.free(mountpoint); + + if (!probeFuseDevice()) return own_failure; + + // Writes to a dead server socket must not kill us. + ignoreSignal(.PIPE); + + var server: ?Server = null; + var address: nine.Address = undefined; + if (cfg.spawn_cmd) |cmd| { + const cmd_z = try arena.dupeZ(u8, cmd); + server = spawnServer(cmd_z, envp) catch return own_failure; + ns.watchServer(server.?.pid); + address = .{ .fd = server.?.fd }; + } else { + address = cfg.address.?; + if (address == .fd and !adoptFd(address.fd)) return own_failure; + } + + var addr_buf: [256]u8 = undefined; + var session = nine.Session.connect(gpa, address, cfg.msize) catch |err| { + std.debug.print("9player: connect to {s}: {t}\n", .{ describeAddress(address, &addr_buf), err }); + stopServer(server); + return own_failure; + }; + defer session.deinit(); + defer stopServer(server); + + _ = session.attach(0, uname, cfg.aname) catch |err| { + switch (err) { + error.Nine => std.debug.print("9player: attach (uname={s}, aname='{s}'): {s}\n", .{ uname, cfg.aname, session.ename[0..session.ename_len] }), + else => std.debug.print("9player: attach: {t}\n", .{err}), + } + return own_failure; + }; + if (cfg.debug) std.debug.print("9player: attached to {s} (msize {d}), mounting on {s}\n", .{ describeAddress(address, &addr_buf), session.msize, mountpoint }); + + var child_pid: i32 = 0; + const stop_fd = ns.installSignals(&child_pid) catch return own_failure; + + const uid = linux.getuid(); + const gid = linux.getgid(); + const child = ns.spawn(gpa, .{ + .argv = cfg.program, + .envp = envp, + .mountpoint = mountpoint, + .uid = uid, + .gid = gid, + .max_read = bridge.max_write, + }) catch return own_failure; + + bridge.serve(gpa, child.fuse_fd, &session, 0, stop_fd, .{ + .uid = uid, + .gid = gid, + .attr_timeout_ns = cfg.cache_ns, + .direct_io = cfg.direct_io, + .debug = cfg.debug, + }) catch |err| switch (err) { + error.Closed => std.debug.print("9player: 9P server connection closed\n", .{}), + else => std.debug.print("9player: fuse: {t}\n", .{err}), + }; + + // Closing the device aborts the FUSE connection: anything still using + // the mount gets ENOTCONN instead of hanging on an unserved request. + _ = linux.close(child.fuse_fd); + + const status = ns.reapIfExited(child.pid) orelse ns.waitChild(child.pid) catch own_failure; + // An exec failure (126/127) is already in `status`; this prints its message. + _ = ns.reportExecFailure(child); + return status; +} + +test "parseTcp" { + const a = parseTcp("127.0.0.1:564").?; + try std.testing.expectEqualStrings("127.0.0.1", a.tcp.host); + try std.testing.expectEqual(@as(u16, 564), a.tcp.port); + const b = parseTcp("[::1]:9999").?; + try std.testing.expectEqualStrings("::1", b.tcp.host); + try std.testing.expectEqual(@as(u16, 9999), b.tcp.port); + try std.testing.expect(parseTcp("nohost") == null); + try std.testing.expect(parseTcp(":564") == null); + try std.testing.expect(parseTcp("1.2.3.4:") == null); + try std.testing.expect(parseTcp("1.2.3.4:70000") == null); +} + +test "parseArgs" { + const arena = std.testing.allocator; + { + const args = [_][:0]const u8{ "9player", "--unix", "/s", "--cache", "0.5", "--msize=8192", "--no-direct-io", "--", "sh", "-c", "x" }; + const r = try parseArgs(arena, &args); + defer arena.free(r.run.program); + try std.testing.expectEqualStrings("/s", r.run.address.?.unix); + try std.testing.expectEqual(@as(u64, 500_000_000), r.run.cache_ns); + try std.testing.expectEqual(@as(u32, 8192), r.run.msize); + try std.testing.expect(!r.run.direct_io); + try std.testing.expectEqual(@as(usize, 3), r.run.program.len); + try std.testing.expectEqualStrings("x", r.run.program[2]); + } + { + const args = [_][:0]const u8{ "9player", "--fd", "3" }; + const r = try parseArgs(arena, &args); + try std.testing.expectEqual(@as(i32, 3), r.run.address.?.fd); + try std.testing.expectEqual(@as(usize, 0), r.run.program.len); + try std.testing.expectEqualStrings("/mnt/9p", r.run.mount); + } + { + // Two transports, no transport, unknown option, missing value: all 125. + const two = [_][:0]const u8{ "9player", "--fd", "3", "--unix", "/s" }; + try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &two)).exit); + const none = [_][:0]const u8{ "9player", "--", "sh" }; + try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &none)).exit); + const unknown = [_][:0]const u8{ "9player", "--bogus" }; + try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &unknown)).exit); + const missing = [_][:0]const u8{ "9player", "--unix" }; + try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &missing)).exit); + const badcache = [_][:0]const u8{ "9player", "--fd", "3", "--cache", "abc" }; + try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &badcache)).exit); + // Empty values, a single-dash typo, and an msize that would allocate gigabytes. + const emptyunix = [_][:0]const u8{ "9player", "--unix=", "--", "sh" }; + try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &emptyunix)).exit); + const emptymount = [_][:0]const u8{ "9player", "--fd", "3", "--mount", "" }; + try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &emptymount)).exit); + const singledash = [_][:0]const u8{ "9player", "--fd", "3", "-mount", "/x" }; + try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &singledash)).exit); + const hugemsize = [_][:0]const u8{ "9player", "--fd", "3", "--msize", "4294967295" }; + try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &hugemsize)).exit); + const okmsize = [_][:0]const u8{ "9player", "--fd", "3", "--msize", "16777216" }; + try std.testing.expectEqual(@as(u32, 16777216), (try parseArgs(arena, &okmsize)).run.msize); + } + { + const ver = [_][:0]const u8{ "9player", "--version" }; + try std.testing.expectEqualStrings(version_string ++ "\n", (try parseArgs(arena, &ver)).info); + const help = [_][:0]const u8{ "9player", "--help" }; + try std.testing.expect(std.mem.startsWith(u8, (try parseArgs(arena, &help)).info, "Usage: 9player")); + } +} + +test { + _ = ns; +} diff --git a/9player/src/nine.zig b/9player/src/nine.zig new file mode 100644 index 0000000..70633e6 --- /dev/null +++ b/9player/src/nine.zig @@ -0,0 +1,756 @@ +//! Synchronous 9P2000 session over a blocking file descriptor. +//! +//! A thin RPC layer over `cloud9.Client` (push/take, allocation-free). One request +//! is outstanding at a time: the FUSE loop that drives this is single-threaded, so +//! every call here blocks until its reply (or the connection's death) arrives. +//! Fids are handed out from a free list; fid 0 is reserved for the root. +const std = @import("std"); +const cloud9 = @import("cloud9"); +const linux = std.os.linux; + +pub const Address = union(enum) { + unix: []const u8, + tcp: struct { host: []const u8, port: u16 }, + fd: i32, +}; + +/// A Stat whose every field means "leave unchanged" in a Twstat. +pub const dontcare = cloud9.Stat{ + .type = 0xFFFF, + .dev = 0xFFFF_FFFF, + .qid = .{ .type = 0xFF, .version = 0xFFFF_FFFF, .path = 0xFFFF_FFFF_FFFF_FFFF }, + .mode = 0xFFFF_FFFF, + .atime = 0xFFFF_FFFF, + .mtime = 0xFFFF_FFFF, + .length = 0xFFFF_FFFF_FFFF_FFFF, + .name = "", + .uid = "", + .gid = "", + .muid = "", +}; + +pub const Session = struct { + pub const Error = error{ Nine, Protocol, Io, Closed, Stopped, TooLarge, OutOfMemory }; + + pub const Walk = struct { nwqid: u16, wqid: [cloud9.max_welem]cloud9.Qid }; + pub const Open = struct { qid: cloud9.Qid, iounit: u32 }; + + gpa: std.mem.Allocator, + fd: i32, + client: cloud9.Client, + in_buf: []u8, + out_buf: []u8, + /// After `error.Nine`, the server's Rerror text (copied, bounded). + ename: [256]u8 = undefined, + ename_len: usize = 0, + /// Negotiated maximum message size. + msize: u32, + next_fid: u32 = 1, + free_fids: std.ArrayList(u32) = .empty, + /// Per-fid iounit learned from open/create (0 = none); used to chunk read/write. + iounits: std.AutoHashMapUnmanaged(u32, u32) = .empty, + /// Optional descriptor watched while waiting for a reply: when it becomes + /// readable (the bridge's "child exited" pipe) the pending rpc fails with + /// `error.Stopped` instead of blocking on a server that never answers. + stop_fd: i32 = -1, + + /// Connect to `address`, then negotiate the protocol version. + /// `msize` is the maximum message size to ask for (0 = the buffers' size). + pub fn connect(gpa: std.mem.Allocator, address: Address, msize: u32) !Session { + const want: u32 = if (msize == 0) 8192 else @max(msize, 24); + const fd = try openTransport(address); + errdefer if (address != .fd) { + _ = linux.close(fd); + }; + + const in_buf = try gpa.alloc(u8, want); + errdefer gpa.free(in_buf); + const out_buf = try gpa.alloc(u8, want); + errdefer gpa.free(out_buf); + + var s: Session = .{ + .gpa = gpa, + .fd = fd, + .client = .init(.{ .in = in_buf, .out = out_buf }), + .in_buf = in_buf, + .out_buf = out_buf, + .msize = want, + }; + const r = try s.rpc(.{ .version = .{ .msize = want } }); + if (!std.mem.eql(u8, r.version.version, "9P2000")) return error.Protocol; + s.msize = r.version.msize; + return s; + } + + /// Closes the descriptor and frees the buffers. Fids are not clunked. + pub fn deinit(s: *Session) void { + _ = linux.close(s.fd); + s.free_fids.deinit(s.gpa); + s.iounits.deinit(s.gpa); + s.gpa.free(s.in_buf); + s.gpa.free(s.out_buf); + s.* = undefined; + } + + pub fn attach(s: *Session, fid: u32, uname: []const u8, aname: []const u8) Error!cloud9.Qid { + const r = try s.rpc(.{ .attach = .{ .fid = fid, .uname = uname, .aname = aname } }); + return r.attach; + } + + /// Fid 0 is never handed out: it belongs to the root attach. + pub fn allocFid(s: *Session) u32 { + if (s.free_fids.pop()) |fid| return fid; + const fid = s.next_fid; + s.next_fid += 1; + return fid; + } + + /// Fids currently bound (excluding fid 0); a debugging aid for leak hunting. + pub fn fidsInUse(s: *const Session) usize { + return (s.next_fid - 1) - s.free_fids.items.len; + } + + pub fn freeFid(s: *Session, fid: u32) void { + _ = s.iounits.remove(fid); + // If the free list cannot grow the fid is simply leaked; the counter keeps going. + s.free_fids.append(s.gpa, fid) catch {}; + } + + /// Generic RPC. Result slices borrow the input buffer until the next call. + pub fn rpc(s: *Session, req: cloud9.Client.Request) Error!cloud9.Client.Result { + s.ename_len = 0; + _ = s.client.submit(req) catch |e| switch (e) { + error.NoTags, error.Handshake, error.Dead => return error.Protocol, + error.NoSpace, error.TooLarge => return error.TooLarge, + error.BadRequest => { + s.setEname("bad request"); + return error.Nine; + }, + }; + try s.flush(); + var tmp: [64 * 1024]u8 = undefined; + while (true) { + if (s.client.take()) |done| { + switch (done.result) { + .fail => |ename| { + s.setEname(ename); + return error.Nine; + }, + else => return done.result, + } + } + if (s.client.dead) return error.Protocol; + // After take() returned null the previous frame is gone, so the free + // space is at least what the pending frame still needs. + const room = s.client.in.len - s.client.in_len; + if (room == 0) return error.Protocol; + const n = try readSome(s.fd, s.stop_fd, tmp[0..@min(room, tmp.len)]); + if (n == 0) return error.Closed; + const pushed = s.client.push(tmp[0..n]); + if (pushed != n) return error.Protocol; + } + } + + /// Walk `names` from `fid` to `newfid`. A partial walk leaves `newfid` unbound + /// (9P semantics) and reports `error.Nine` with ename "file does not exist". + pub fn walk(s: *Session, fid: u32, newfid: u32, names: []const []const u8) Error!Walk { + const r = try s.rpc(.{ .walk = .{ .fid = fid, .newfid = newfid, .names = names } }); + if (r.walk.nwqid < names.len) { + s.setEname("file does not exist"); + return error.Nine; + } + return .{ .nwqid = r.walk.nwqid, .wqid = r.walk.wqid }; + } + + /// allocFid + zero-element walk. The fid is released again on failure. + pub fn clone(s: *Session, fid: u32) Error!u32 { + const newfid = s.allocFid(); + errdefer s.freeFid(newfid); + _ = try s.walk(fid, newfid, &.{}); + return newfid; + } + + pub fn open(s: *Session, fid: u32, mode: u8) Error!Open { + const r = try s.rpc(.{ .open = .{ .fid = fid, .mode = mode } }); + s.noteIounit(fid, r.open.iounit); + return .{ .qid = r.open.qid, .iounit = r.open.iounit }; + } + + pub fn create(s: *Session, fid: u32, name: []const u8, perm: u32, mode: u8) Error!Open { + const r = try s.rpc(.{ .create = .{ .fid = fid, .name = name, .perm = perm, .mode = mode } }); + s.noteIounit(fid, r.create.iounit); + return .{ .qid = r.create.qid, .iounit = r.create.iounit }; + } + + /// Reads into `buf`, chunking by min(maxRead, iounit) and stopping at the first + /// short read. Returns the number of bytes read (0 at end of file). + pub fn read(s: *Session, fid: u32, offset: u64, buf: []u8) Error!usize { + return readWith(s, rpc, fid, offset, buf, s.chunk(fid)); + } + + /// Writes `data`, chunking like `read` and stopping at the first short write. + pub fn write(s: *Session, fid: u32, offset: u64, data: []const u8) Error!usize { + return writeWith(s, rpc, fid, offset, data, s.chunkWrite(fid)); + } + + /// The returned Stat's strings (name/uid/gid/muid) borrow the session's input + /// buffer: they are valid only until the next rpc. Copy what must outlive it. + pub fn stat(s: *Session, fid: u32) Error!cloud9.Stat { + const r = try s.rpc(.{ .stat = .{ .fid = fid } }); + return r.stat; + } + + pub fn wstat(s: *Session, fid: u32, st: cloud9.Stat) Error!void { + _ = try s.rpc(.{ .wstat = .{ .fid = fid, .stat = st } }); + } + + /// Frees the fid locally even when the server reports an error. + pub fn clunk(s: *Session, fid: u32) Error!void { + defer s.freeFid(fid); + _ = try s.rpc(.{ .clunk = .{ .fid = fid } }); + } + + /// Frees the fid locally even when the server reports an error. + pub fn remove(s: *Session, fid: u32) Error!void { + defer s.freeFid(fid); + _ = try s.rpc(.{ .remove = .{ .fid = fid } }); + } + + /// Maps the last Rerror text to an errno (case-insensitive substring match). + pub fn errno(s: *const Session) linux.E { + return enameToErrno(s.ename[0..s.ename_len]); + } + + // -- internals -------------------------------------------------------------- + + fn setEname(s: *Session, text: []const u8) void { + const n = @min(text.len, 255); + @memcpy(s.ename[0..n], text[0..n]); + s.ename_len = n; + } + + fn noteIounit(s: *Session, fid: u32, iounit: u32) void { + if (iounit == 0) { + _ = s.iounits.remove(fid); + } else { + s.iounits.put(s.gpa, fid, iounit) catch {}; + } + } + + fn chunk(s: *Session, fid: u32) u32 { + return chunkSize(s.client.maxRead(), s.iounits.get(fid) orelse 0); + } + + fn chunkWrite(s: *Session, fid: u32) u32 { + return chunkSize(s.client.maxWrite(), s.iounits.get(fid) orelse 0); + } + + /// Writes everything in the client's output buffer to the socket. + fn flush(s: *Session) Error!void { + while (s.client.output().len != 0) { + const out = s.client.output(); + const rc = linux.write(s.fd, out.ptr, out.len); + switch (linux.errno(rc)) { + .SUCCESS => { + if (rc == 0) return error.Closed; + s.client.wrote(rc); + }, + .INTR, .AGAIN => continue, + .PIPE, .CONNRESET => return error.Closed, + else => return error.Io, + } + } + } +}; + +fn chunkSize(max: u32, iounit: u32) u32 { + if (iounit != 0 and iounit < max) return iounit; + return max; +} + +/// Chunked read over any rpc-shaped function (injected so the loop is testable). +fn readWith( + s: anytype, + comptime rpcFn: anytype, + fid: u32, + offset: u64, + buf: []u8, + max_chunk: u32, +) Session.Error!usize { + if (max_chunk == 0) return error.Protocol; + var done: usize = 0; + while (done < buf.len) { + const want: u32 = @intCast(@min(buf.len - done, max_chunk)); + const r = try rpcFn(s, .{ .read = .{ .fid = fid, .offset = offset + done, .count = want } }); + const data = r.read; + @memcpy(buf[done..][0..data.len], data); + done += data.len; + if (data.len < want) break; + } + return done; +} + +/// Chunked write over any rpc-shaped function. +fn writeWith( + s: anytype, + comptime rpcFn: anytype, + fid: u32, + offset: u64, + data: []const u8, + max_chunk: u32, +) Session.Error!usize { + if (max_chunk == 0) return error.Protocol; + var done: usize = 0; + while (done < data.len) { + const want: usize = @min(data.len - done, max_chunk); + const r = try rpcFn(s, .{ .write = .{ .fid = fid, .offset = offset + done, .data = data[done..][0..want] } }); + done += r.write; + if (r.write < want) break; + } + return done; +} + +fn readSome(fd: i32, stop_fd: i32, buf: []u8) Session.Error!usize { + while (true) { + if (stop_fd >= 0) { + var pfds = [_]linux.pollfd{ + .{ .fd = fd, .events = linux.POLL.IN, .revents = 0 }, + .{ .fd = stop_fd, .events = linux.POLL.IN, .revents = 0 }, + }; + const prc = linux.poll(&pfds, pfds.len, -1); + switch (linux.errno(prc)) { + .SUCCESS => {}, + .INTR, .AGAIN => continue, + else => return error.Io, + } + if (pfds[1].revents != 0 and pfds[0].revents == 0) return error.Stopped; + } + const rc = linux.read(fd, buf.ptr, buf.len); + switch (linux.errno(rc)) { + .SUCCESS => return rc, + .INTR, .AGAIN => continue, + .CONNRESET => return error.Closed, + else => return error.Io, + } + } +} + +/// Rerror text → errno, per docs/DESIGN.md (first match wins). +pub fn enameToErrno(ename: []const u8) linux.E { + const Rule = struct { needle: []const u8, err: linux.E }; + const rules = [_]Rule{ + .{ .needle = "not exist", .err = .NOENT }, + .{ .needle = "not found", .err = .NOENT }, + .{ .needle = "no such", .err = .NOENT }, + .{ .needle = "exists", .err = .EXIST }, + .{ .needle = "not empty", .err = .NOTEMPTY }, + .{ .needle = "not a dir", .err = .NOTDIR }, + .{ .needle = "is a dir", .err = .ISDIR }, + .{ .needle = "permission", .err = .ACCES }, + .{ .needle = "denied", .err = .ACCES }, + .{ .needle = "read-only", .err = .ROFS }, + .{ .needle = "read only", .err = .ROFS }, + .{ .needle = "readonly", .err = .ROFS }, + .{ .needle = "no space", .err = .NOSPC }, + .{ .needle = "not allowed", .err = .PERM }, + .{ .needle = "not permitted", .err = .PERM }, + .{ .needle = "cannot", .err = .PERM }, + .{ .needle = "fid", .err = .BADF }, + .{ .needle = "bad offset", .err = .INVAL }, + .{ .needle = "invalid", .err = .INVAL }, + .{ .needle = "bad ", .err = .INVAL }, + .{ .needle = "busy", .err = .BUSY }, + .{ .needle = "in use", .err = .BUSY }, + .{ .needle = "too long", .err = .NAMETOOLONG }, + .{ .needle = "not supported", .err = .OPNOTSUPP }, + .{ .needle = "unsupported", .err = .OPNOTSUPP }, + }; + for (rules) |rule| { + if (std.ascii.findIgnoreCase(ename, rule.needle) != null) return rule.err; + } + return .IO; +} + +// -- transport ------------------------------------------------------------------ + +fn openTransport(address: Address) !i32 { + switch (address) { + .fd => |fd| return fd, + .unix => |path| { + if (path.len == 0 or path.len >= 108) return error.NameTooLong; + var sa: linux.sockaddr.un = .{ .path = @splat(0) }; + @memcpy(sa.path[0..path.len], path); + const fd = try newSocket(linux.AF.UNIX, 0); + errdefer _ = linux.close(fd); + try doConnect(fd, @ptrCast(&sa), @sizeOf(linux.sockaddr.un)); + return fd; + }, + .tcp => |t| { + const ip = std.Io.net.IpAddress.parse(t.host, t.port) catch return error.InvalidAddress; + switch (ip) { + .ip4 => |a| { + const sa: linux.sockaddr.in = .{ + .port = std.mem.nativeToBig(u16, t.port), + .addr = @bitCast(a.bytes), + }; + const fd = try newSocket(linux.AF.INET, linux.IPPROTO.TCP); + errdefer _ = linux.close(fd); + setNodelay(fd); + try doConnect(fd, @ptrCast(&sa), @sizeOf(linux.sockaddr.in)); + return fd; + }, + .ip6 => |a| { + const sa: linux.sockaddr.in6 = .{ + .port = std.mem.nativeToBig(u16, t.port), + .flowinfo = 0, + .addr = a.bytes, + .scope_id = 0, + }; + const fd = try newSocket(linux.AF.INET6, linux.IPPROTO.TCP); + errdefer _ = linux.close(fd); + setNodelay(fd); + try doConnect(fd, @ptrCast(&sa), @sizeOf(linux.sockaddr.in6)); + return fd; + }, + } + }, + } +} + +fn newSocket(domain: u32, protocol: u32) !i32 { + const rc = linux.socket(domain, linux.SOCK.STREAM | linux.SOCK.CLOEXEC, protocol); + switch (linux.errno(rc)) { + .SUCCESS => return @intCast(rc), + .MFILE, .NFILE => return error.ProcessFdQuotaExceeded, + .AFNOSUPPORT, .PROTONOSUPPORT => return error.AddressFamilyNotSupported, + .ACCES => return error.AccessDenied, + .NOMEM, .NOBUFS => return error.SystemResources, + else => return error.Unexpected, + } +} + +fn setNodelay(fd: i32) void { + const one: u32 = 1; + _ = linux.setsockopt(fd, linux.IPPROTO.TCP, linux.TCP.NODELAY, @ptrCast(&one), @sizeOf(u32)); +} + +fn doConnect(fd: i32, addr: *const linux.sockaddr, len: linux.socklen_t) !void { + while (true) { + const rc = linux.connect(fd, addr, len); + switch (linux.errno(rc)) { + .SUCCESS => return, + .INTR => continue, + .CONNREFUSED => return error.ConnectionRefused, + .NOENT, .NOTDIR => return error.FileNotFound, + .ACCES, .PERM => return error.AccessDenied, + .TIMEDOUT => return error.ConnectionTimedOut, + .NETUNREACH, .HOSTUNREACH => return error.NetworkUnreachable, + .ADDRNOTAVAIL => return error.AddressNotAvailable, + .AGAIN, .INPROGRESS => return error.WouldBlock, + else => return error.Unexpected, + } + } +} + +// -- tests ---------------------------------------------------------------------- + +const testing = std.testing; + +test { + testing.refAllDecls(@This()); +} + +test "ename → errno mapping" { + try testing.expectEqual(linux.E.NOENT, enameToErrno("file does not exist")); + try testing.expectEqual(linux.E.NOENT, enameToErrno("No Such File")); + try testing.expectEqual(linux.E.NOENT, enameToErrno("directory entry not found")); + try testing.expectEqual(linux.E.EXIST, enameToErrno("file already exists")); + try testing.expectEqual(linux.E.NOTEMPTY, enameToErrno("directory not empty")); + try testing.expectEqual(linux.E.NOTDIR, enameToErrno("not a directory")); + try testing.expectEqual(linux.E.ISDIR, enameToErrno("is a directory")); + try testing.expectEqual(linux.E.ACCES, enameToErrno("permission denied")); + try testing.expectEqual(linux.E.ACCES, enameToErrno("access denied")); + try testing.expectEqual(linux.E.ROFS, enameToErrno("read-only file system")); + try testing.expectEqual(linux.E.NOSPC, enameToErrno("no space left")); + try testing.expectEqual(linux.E.PERM, enameToErrno("operation not permitted")); + try testing.expectEqual(linux.E.PERM, enameToErrno("cannot remove root")); + try testing.expectEqual(linux.E.BADF, enameToErrno("unknown fid")); + try testing.expectEqual(linux.E.BADF, enameToErrno("fid in use")); // "fid" precedes "in use" + try testing.expectEqual(linux.E.INVAL, enameToErrno("bad offset")); + try testing.expectEqual(linux.E.INVAL, enameToErrno("invalid argument")); + try testing.expectEqual(linux.E.INVAL, enameToErrno("bad request")); + try testing.expectEqual(linux.E.BUSY, enameToErrno("device busy")); + try testing.expectEqual(linux.E.NAMETOOLONG, enameToErrno("name too long")); + try testing.expectEqual(linux.E.OPNOTSUPP, enameToErrno("operation not supported")); + try testing.expectEqual(linux.E.IO, enameToErrno("something odd happened")); + try testing.expectEqual(linux.E.IO, enameToErrno("")); +} + +test "fid allocator recycles and never hands out 0" { + var s: Session = undefined; + s.gpa = testing.allocator; + s.next_fid = 1; + s.free_fids = .empty; + s.iounits = .empty; + defer s.free_fids.deinit(s.gpa); + defer s.iounits.deinit(s.gpa); + + const a = s.allocFid(); + const b = s.allocFid(); + const c = s.allocFid(); + try testing.expectEqual(@as(u32, 1), a); + try testing.expectEqual(@as(u32, 2), b); + try testing.expectEqual(@as(u32, 3), c); + s.freeFid(b); + try testing.expectEqual(b, s.allocFid()); + s.freeFid(a); + s.freeFid(c); + const x = s.allocFid(); + const y = s.allocFid(); + try testing.expect((x == a and y == c) or (x == c and y == a)); + try testing.expectEqual(@as(u32, 4), s.allocFid()); + try testing.expect(a != 0 and b != 0 and c != 0); +} + +test "chunkSize honours iounit only when smaller" { + try testing.expectEqual(@as(u32, 100), chunkSize(100, 0)); + try testing.expectEqual(@as(u32, 40), chunkSize(100, 40)); + try testing.expectEqual(@as(u32, 100), chunkSize(100, 400)); +} + +/// Fake rpc for the chunked read/write loops: a file of `len` bytes where byte i == i & 0xff. +const FakeFile = struct { + len: usize, + calls: usize = 0, + max_count: u32 = 0, + short_write_at: ?usize = null, + scratch: [4096]u8 = undefined, + + fn rpc(f: *FakeFile, req: cloud9.Client.Request) Session.Error!cloud9.Client.Result { + f.calls += 1; + switch (req) { + .read => |r| { + f.max_count = @max(f.max_count, r.count); + if (r.offset >= f.len) return .{ .read = "" }; + const n: usize = @min(@as(usize, r.count), f.len - @as(usize, @intCast(r.offset))); + for (f.scratch[0..n], 0..) |*b, i| b.* = @truncate(r.offset + i); + return .{ .read = f.scratch[0..n] }; + }, + .write => |w| { + f.max_count = @max(f.max_count, @as(u32, @intCast(w.data.len))); + if (f.short_write_at) |at| { + if (w.offset + w.data.len > at) { + const n: usize = if (w.offset >= at) 0 else @intCast(at - w.offset); + return .{ .write = @intCast(n) }; + } + } + return .{ .write = @intCast(w.data.len) }; + }, + else => unreachable, + } + } +}; + +test "read chunks by max_chunk and stops at a short read" { + var f: FakeFile = .{ .len = 2500 }; + var buf: [4000]u8 = undefined; + const n = try readWith(&f, FakeFile.rpc, 7, 0, &buf, 1000); + try testing.expectEqual(@as(usize, 2500), n); + try testing.expectEqual(@as(usize, 3), f.calls); // 1000, 1000, 500 (short → stop) + try testing.expectEqual(@as(u32, 1000), f.max_count); + for (buf[0..n], 0..) |b, i| try testing.expectEqual(@as(u8, @truncate(i)), b); + + // Reading exactly up to a chunk boundary uses one call per chunk and no more. + f = .{ .len = 2000 }; + try testing.expectEqual(@as(usize, 2000), try readWith(&f, FakeFile.rpc, 7, 0, buf[0..2000], 1000)); + try testing.expectEqual(@as(usize, 2), f.calls); + + // Offset past EOF → 0. + f = .{ .len = 10 }; + try testing.expectEqual(@as(usize, 0), try readWith(&f, FakeFile.rpc, 7, 50, &buf, 1000)); +} + +test "write chunks and stops at a short write" { + var f: FakeFile = .{ .len = 0 }; + var data: [2500]u8 = undefined; + for (&data, 0..) |*b, i| b.* = @truncate(i); + try testing.expectEqual(@as(usize, 2500), try writeWith(&f, FakeFile.rpc, 7, 0, &data, 1000)); + try testing.expectEqual(@as(usize, 3), f.calls); + try testing.expectEqual(@as(u32, 1000), f.max_count); + + f = .{ .len = 0, .short_write_at = 1500 }; + try testing.expectEqual(@as(usize, 1500), try writeWith(&f, FakeFile.rpc, 7, 0, &data, 1000)); + try testing.expectEqual(@as(usize, 2), f.calls); +} + +// -- in-process server test --------------------------------------------------------- + +/// A tiny 9P2000 backend on a cloud9.Server: answers version/attach/walk/stat/open/ +/// read/clunk/remove with canned data. Runs in its own thread over a socketpair. +const FakeServer = struct { + fd: i32, + msize: u32, + max_read_count: u32 = 0, + file_len: usize, + + const file_qid: cloud9.Qid = .{ .type = 0, .version = 3, .path = 0x1234 }; + const dir_qid: cloud9.Qid = .{ .type = cloud9.qtdir, .version = 1, .path = 0x1 }; + + fn run(fs: *FakeServer) void { + fs.loop() catch |e| std.debug.print("fake server: {s}\n", .{@errorName(e)}); + _ = linux.close(fs.fd); + } + + fn loop(fs: *FakeServer) !void { + const gpa = testing.allocator; + const in = try gpa.alloc(u8, fs.msize); + defer gpa.free(in); + const out = try gpa.alloc(u8, fs.msize * 2); + defer gpa.free(out); + var srv: cloud9.Server = .init(.{ .in = in, .out = out }); + var tmp: [4096]u8 = undefined; + var data: [8192]u8 = undefined; + while (true) { + while (try srv.receive()) |req| { + const tag = req.tag; + switch (req.msg) { + .tversion => |m| try srv.negotiate(m.msize, m.version), + .tattach => try srv.reply(tag, .{ .rattach = .{ .qid = dir_qid } }), + .twalk => |m| { + var wq: [cloud9.max_welem]cloud9.Qid = @splat(dir_qid); + var n: u16 = 0; + for (m.wname[0..m.nwname]) |name| { + if (std.mem.eql(u8, name, "file")) { + wq[n] = file_qid; + } else if (std.mem.eql(u8, name, "dir")) { + wq[n] = dir_qid; + } else break; + n += 1; + } + if (n == 0 and m.nwname != 0) { + try srv.reply(tag, .{ .rerror = .{ .ename = "file does not exist" } }); + } else { + try srv.reply(tag, .{ .rwalk = .{ .nwqid = n, .wqid = wq } }); + } + }, + .tstat => try srv.reply(tag, .{ .rstat = .{ .stat = .{ + .type = 0, + .dev = 0, + .qid = file_qid, + .mode = 0o644, + .atime = 1, + .mtime = 2, + .length = fs.file_len, + .name = "file", + .uid = "u", + .gid = "g", + .muid = "u", + } } }), + .topen => |m| try srv.reply(tag, .{ .ropen = .{ .qid = file_qid, .iounit = if (m.mode == cloud9.owrite) 700 else 0 } }), + .tread => |m| { + fs.max_read_count = @max(fs.max_read_count, m.count); + var n: usize = 0; + if (m.offset < fs.file_len) n = @min(@as(usize, m.count), fs.file_len - @as(usize, @intCast(m.offset))); + n = @min(n, data.len); + for (data[0..n], 0..) |*b, i| b.* = @truncate(m.offset + i); + try srv.reply(tag, .{ .rread = .{ .data = data[0..n] } }); + }, + .twrite => |m| try srv.reply(tag, .{ .rwrite = .{ .count = @intCast(m.data.len) } }), + .tclunk => try srv.reply(tag, .rclunk), + .tremove => try srv.reply(tag, .{ .rerror = .{ .ename = "permission denied" } }), + .twstat => try srv.reply(tag, .rwstat), + // A flush is the test's "hang up now" signal. + .tflush => return, + else => try srv.reply(tag, .{ .rerror = .{ .ename = "not supported" } }), + } + srv.release(); + } + while (srv.output().len != 0) { + const o = srv.output(); + const rc = linux.write(fs.fd, o.ptr, o.len); + if (linux.errno(rc) != .SUCCESS) return error.Write; + srv.wrote(rc); + } + const rc = linux.read(fs.fd, &tmp, tmp.len); + if (linux.errno(rc) != .SUCCESS) return error.Read; + if (rc == 0) return; + if (srv.push(tmp[0..rc]) != rc) return error.Overflow; + } + } +}; + +test "session against an in-process cloud9.Server" { + var fds: [2]i32 = undefined; + try testing.expectEqual(linux.E.SUCCESS, linux.errno(linux.socketpair(linux.AF.UNIX, linux.SOCK.STREAM | linux.SOCK.CLOEXEC, 0, &fds))); + + var fs: FakeServer = .{ .fd = fds[1], .msize = 8192, .file_len = 20_000 }; + const th = try std.Thread.spawn(.{}, FakeServer.run, .{&fs}); + + var s = try Session.connect(testing.allocator, .{ .fd = fds[0] }, 8192); + defer { + s.deinit(); + th.join(); + } + try testing.expectEqual(@as(u32, 8192), s.msize); + + const root = try s.attach(0, "me", ""); + try testing.expectEqual(FakeServer.dir_qid.path, root.path); + + // Plain rpc + stat borrowing the input buffer. + const fid = s.allocFid(); + const w = try s.walk(0, fid, &.{"file"}); + try testing.expectEqual(@as(u16, 1), w.nwqid); + try testing.expectEqual(FakeServer.file_qid.path, w.wqid[0].path); + const st = try s.stat(fid); + try testing.expectEqualStrings("file", st.name); + try testing.expectEqual(@as(u64, 20_000), st.length); + + // Chunked read: 20000 bytes at maxRead = msize - 11 = 8181 per chunk. + _ = try s.open(fid, cloud9.oread); + const buf = try testing.allocator.alloc(u8, 30_000); + defer testing.allocator.free(buf); + const n = try s.read(fid, 0, buf); + try testing.expectEqual(@as(usize, 20_000), n); + for (buf[0..n], 0..) |b, i| try testing.expectEqual(@as(u8, @truncate(i)), b); + try testing.expectEqual(@as(u32, 8181), fs.max_read_count); + try testing.expectEqual(@as(usize, 0), try s.read(fid, 20_000, buf)); + + // iounit from open bounds the chunk. + const wfid = try s.clone(fid); + _ = try s.open(wfid, cloud9.owrite); + fs.max_read_count = 0; + _ = try s.read(wfid, 0, buf[0..3000]); + try testing.expectEqual(@as(u32, 700), fs.max_read_count); + try testing.expectEqual(@as(usize, 3000), try s.write(wfid, 0, buf[0..3000])); + + // Partial walk → error.Nine with a "not exist" ename → ENOENT. + const pfid = s.allocFid(); + try testing.expectError(error.Nine, s.walk(0, pfid, &.{ "dir", "nope" })); + try testing.expectEqual(linux.E.NOENT, s.errno()); + try testing.expectEqualStrings("file does not exist", s.ename[0..s.ename_len]); + s.freeFid(pfid); + + // Server Rerror → error.Nine, ename copied, fid freed by remove even on error. + try testing.expectError(error.Nine, s.remove(wfid)); + try testing.expectEqual(linux.E.ACCES, s.errno()); + try testing.expectEqual(wfid, s.allocFid()); // recycled + s.freeFid(wfid); + + // Unsupported op → "not supported" → ENOTSUP; a plain wstat succeeds. + try testing.expectError(error.Nine, s.rpc(.{ .auth = .{ .afid = 5, .uname = "me" } })); + try testing.expectEqual(linux.E.OPNOTSUPP, s.errno()); + try s.wstat(fid, dontcare); + try s.clunk(fid); + try testing.expectEqual(fid, s.allocFid()); + s.freeFid(fid); + + // A clone bound to a fid that then fails to walk must release the fid. + const before = s.next_fid; + const cfid = s.allocFid(); + s.freeFid(cfid); + try testing.expectError(error.Nine, s.walk(0, cfid, &.{"nope"})); + try testing.expectEqual(before, s.next_fid); + + // The server hanging up makes the pending rpc fail with error.Closed. + try testing.expectError(error.Closed, s.rpc(.{ .flush = .{ .oldtag = 0 } })); +} diff --git a/9player/src/ns.zig b/9player/src/ns.zig new file mode 100644 index 0000000..2c6f202 --- /dev/null +++ b/9player/src/ns.zig @@ -0,0 +1,1082 @@ +//! Namespace and process plumbing for 9player. +//! +//! Everything here is raw `std.os.linux` syscalls (no libc). The child side +//! of `spawn` runs between `fork` and `execve`; it does not allocate except +//! inside `ensureMountpoint` (the process is single-threaded by then, so the +//! inherited allocator is safe to use). +//! +//! Exit codes produced by the child before exec: 125 for namespace/mount +//! setup failures, 126 when the program was found but is not executable, +//! 127 when it was not found. + +const std = @import("std"); +const builtin = @import("builtin"); +const linux = std.os.linux; +const Allocator = std.mem.Allocator; +const E = linux.E; + +pub const Spawn = struct { + /// argv[0] is PATH-searched unless it contains '/'. + argv: []const []const u8, + /// Inherited environment; `NINEPLAYER_MOUNT` is added or replaced. + envp: [*:null]const ?[*:0]const u8, + /// Absolute mountpoint (see `resolveMountpoint`). + mountpoint: []const u8, + uid: u32, + gid: u32, + max_read: u32, + /// When false the namespace is set up (including mountpoint shadowing) + /// but `/dev/fuse` is not opened and nothing is mounted; `Child.fuse_fd` + /// is then -1. Only for smoke tests. + mount_fuse: bool = true, +}; + +pub const Child = struct { + pid: i32, + /// The `/dev/fuse` connection backing the mount, opened by the child + /// inside its user namespace (the kernel refuses to mount a fuse fd that + /// was opened from another user namespace) and handed back over the + /// status socket with SCM_RIGHTS. Owned by the caller; CLOEXEC. + fuse_fd: i32, + /// Parent end of the status socket. The child reports an exec failure + /// on it (see `reportExecFailure`); it reads EOF once exec succeeded. + status_fd: i32, +}; + +/// Exit status used by the child for setup failures (matches 9player's own). +pub const setup_failure_status: u8 = 125; +/// Refuse to shadow a directory with more entries than this. +pub const max_shadow_entries: usize = 4096; + +const default_path = "/usr/local/bin:/bin:/usr/bin"; +const path_max = 4096; + +// --------------------------------------------------------------------------- +// Mountpoint resolution +// --------------------------------------------------------------------------- + +/// Absolute path (relative paths resolved against cwd), duplicate slashes +/// collapsed, `.` and `..` components resolved lexically, no trailing slash. +/// `/` itself is rejected. +pub fn resolveMountpoint(gpa: Allocator, path: []const u8) ![:0]u8 { + var cwd_buf: [path_max]u8 = undefined; + var cwd: []const u8 = "/"; + if (path.len == 0 or path[0] != '/') { + const rc = linux.getcwd(&cwd_buf, cwd_buf.len); + switch (linux.errno(rc)) { + .SUCCESS => {}, + else => |e| { + std.debug.print("9player: getcwd: E{t}\n", .{e}); + return error.Cwd; + }, + } + // rc counts the terminating NUL. + cwd = cwd_buf[0 .. rc - 1]; + } + return normalizePath(gpa, cwd, path); +} + +/// Pure part of `resolveMountpoint`: `cwd` is only used when `path` is relative. +fn normalizePath(gpa: Allocator, cwd: []const u8, path: []const u8) ![:0]u8 { + if (path.len == 0) return error.InvalidMountpoint; + var out: std.ArrayList(u8) = .empty; + defer out.deinit(gpa); + if (path[0] != '/') try appendComponents(gpa, &out, cwd); + try appendComponents(gpa, &out, path); + if (out.items.len == 0) return error.InvalidMountpoint; // "/" or equivalent + return out.toOwnedSliceSentinel(gpa, 0); +} + +fn appendComponents(gpa: Allocator, out: *std.ArrayList(u8), path: []const u8) !void { + var it = std.mem.tokenizeScalar(u8, path, '/'); + while (it.next()) |comp| { + if (std.mem.eql(u8, comp, ".")) continue; + if (std.mem.eql(u8, comp, "..")) { + // Pop the last component (lexically; "/.." stays "/"). + const idx = std.mem.lastIndexOfScalar(u8, out.items, '/') orelse 0; + out.shrinkRetainingCapacity(idx); + continue; + } + try out.append(gpa, '/'); + try out.appendSlice(gpa, comp); + } +} + +// --------------------------------------------------------------------------- +// Environment helpers +// --------------------------------------------------------------------------- + +/// Look a variable up in a raw envp block. +pub fn getenv(envp: [*:null]const ?[*:0]const u8, name: []const u8) ?[]const u8 { + var i: usize = 0; + while (envp[i]) |entry| : (i += 1) { + const kv = std.mem.span(entry); + if (kv.len > name.len and kv[name.len] == '=' and std.mem.eql(u8, kv[0..name.len], name)) { + return kv[name.len + 1 ..]; + } + } + return null; +} + +/// Every path `execve` should try for `name`, in order: just `name` if it +/// contains a '/', else `/` for each `$PATH` element (an empty +/// element means the current directory; `$PATH` unset falls back to +/// `/usr/local/bin:/bin:/usr/bin`). +pub fn pathCandidates(gpa: Allocator, envp: [*:null]const ?[*:0]const u8, name: []const u8) ![]const [:0]const u8 { + if (name.len == 0) return error.EmptyProgramName; + var list: std.ArrayList([:0]const u8) = .empty; + errdefer { + for (list.items) |c| gpa.free(c); + list.deinit(gpa); + } + if (std.mem.indexOfScalar(u8, name, '/') != null) { + try list.append(gpa, try gpa.dupeZ(u8, name)); + return list.toOwnedSlice(gpa); + } + const path = getenv(envp, "PATH") orelse default_path; + var it = std.mem.splitScalar(u8, path, ':'); + while (it.next()) |dir| { + const d = if (dir.len == 0) "." else dir; + try list.append(gpa, try std.fmt.allocPrintSentinel(gpa, "{s}/{s}", .{ d, name }, 0)); + } + return list.toOwnedSlice(gpa); +} + +/// First PATH candidate that is an executable regular file, or the name +/// itself when it contains a '/'. Provided for completeness; `spawn` simply +/// tries `execve` on every candidate instead. +pub fn findInPath(gpa: Allocator, envp: [*:null]const ?[*:0]const u8, name: []const u8) ![:0]u8 { + const cands = try pathCandidates(gpa, envp, name); + defer { + for (cands) |c| gpa.free(c); + gpa.free(cands); + } + for (cands) |c| { + var stx: linux.Statx = undefined; + const rc = linux.statx(linux.AT.FDCWD, c.ptr, 0, .{ .TYPE = true, .MODE = true }, &stx); + if (linux.errno(rc) != .SUCCESS) continue; + if (stx.mode & linux.S.IFMT != linux.S.IFREG) continue; + if (stx.mode & 0o111 == 0) continue; + return gpa.dupeZ(u8, c); + } + return error.FileNotFound; +} + +/// New envp block: every entry of `envp` except `NINEPLAYER_MOUNT=...`, +/// followed by `NINEPLAYER_MOUNT=`. +fn buildEnvp(gpa: Allocator, envp: [*:null]const ?[*:0]const u8, mountpoint: []const u8) ![:null]?[*:0]const u8 { + const key = "NINEPLAYER_MOUNT="; + var keep: usize = 0; + var i: usize = 0; + while (envp[i]) |entry| : (i += 1) { + if (!std.mem.startsWith(u8, std.mem.span(entry), key)) keep += 1; + } + const out = try gpa.allocSentinel(?[*:0]const u8, keep + 1, null); + errdefer gpa.free(out); + var j: usize = 0; + i = 0; + while (envp[i]) |entry| : (i += 1) { + if (std.mem.startsWith(u8, std.mem.span(entry), key)) continue; + out[j] = entry; + j += 1; + } + const mount_entry = try std.fmt.allocPrintSentinel(gpa, key ++ "{s}", .{mountpoint}, 0); + out[j] = mount_entry.ptr; + return out; +} + +fn buildArgv(gpa: Allocator, argv: []const []const u8) ![:null]?[*:0]const u8 { + const out = try gpa.allocSentinel(?[*:0]const u8, argv.len, null); + for (argv, 0..) |a, i| out[i] = (try gpa.dupeZ(u8, a)).ptr; + return out; +} + +// --------------------------------------------------------------------------- +// Mountpoint policy +// --------------------------------------------------------------------------- + +/// Make sure `path` is a directory, inside the *current* mount namespace: +/// +/// * already a directory → done; +/// * else `mkdir`; on `EACCES`/`EPERM`/`EROFS` shadow the parent directory +/// with a tmpfs that re-exposes every existing entry (bind mounts for +/// directories and files, recreated symlinks) and `mkdir` inside it; +/// * anything else fails with the errno and a hint. +/// +/// Every failure prints `9player: : E` to stderr before +/// returning. Meant to be called in the child of `spawn` (or from a +/// throwaway namespace: `unshare -Urm`). +pub fn ensureMountpoint(gpa: Allocator, path: [:0]const u8) !void { + if (fileType(linux.AT.FDCWD, path, false)) |ft| { + if (ft == .dir) return; + std.debug.print("9player: mountpoint {s}: exists but is not a directory\n", .{path}); + return error.Mountpoint; + } + if (fileType(linux.AT.FDCWD, path, true) == .symlink) { + std.debug.print("9player: mountpoint {s}: dangling symlink\n", .{path}); + return error.Mountpoint; + } + const mk = linux.errno(linux.mkdirat(linux.AT.FDCWD, path, 0o755)); + switch (mk) { + .SUCCESS => return, + .ACCES, .PERM, .ROFS => {}, + else => |e| { + std.debug.print("9player: mkdir {s}: E{t} (pass --mount an existing directory)\n", .{ path, e }); + return error.Mountpoint; + }, + } + const parent = std.fs.path.dirname(path) orelse "/"; + if (std.mem.eql(u8, parent, "/") or isSameDirectory(parent, "/")) { + std.debug.print("9player: mkdir {s}: E{t}; refusing to shadow / (pass --mount an existing directory)\n", .{ path, mk }); + return error.Mountpoint; + } + // The shadow rebuilds entries from /proc/self/fd//; a tmpfs + // over /proc (or a subtree of it) would take that away from itself. + if (std.mem.eql(u8, parent, "/proc") or std.mem.startsWith(u8, parent, "/proc/")) { + std.debug.print("9player: mkdir {s}: E{t}; refusing to shadow {s} (pass --mount an existing directory)\n", .{ path, mk, parent }); + return error.Mountpoint; + } + const parent_z = try gpa.dupeZ(u8, parent); + defer gpa.free(parent_z); + try shadowDirectory(gpa, parent_z); + switch (linux.errno(linux.mkdirat(linux.AT.FDCWD, path, 0o755))) { + .SUCCESS => {}, + else => |e| { + std.debug.print("9player: mkdir {s} (in shadow tmpfs): E{t}\n", .{ path, e }); + return error.Mountpoint; + }, + } +} + +const FileType = enum { dir, symlink, other }; + +/// True when both paths resolve (following symlinks, including magic ones +/// such as /proc/self/root) to the same inode. +fn isSameDirectory(a: []const u8, b: [*:0]const u8) bool { + var a_buf: [path_max]u8 = undefined; + const a_z = std.fmt.bufPrintZ(&a_buf, "{s}", .{a}) catch return false; + var sa: linux.Statx = undefined; + var sb: linux.Statx = undefined; + if (linux.errno(linux.statx(linux.AT.FDCWD, a_z, 0, .{ .INO = true }, &sa)) != .SUCCESS) return false; + if (linux.errno(linux.statx(linux.AT.FDCWD, b, 0, .{ .INO = true }, &sb)) != .SUCCESS) return false; + return sa.ino == sb.ino and sa.dev_major == sb.dev_major and sa.dev_minor == sb.dev_minor; +} + +fn fileType(dirfd: i32, name: [*:0]const u8, nofollow: bool) ?FileType { + var stx: linux.Statx = undefined; + const flags: u32 = if (nofollow) linux.AT.SYMLINK_NOFOLLOW else 0; + const rc = linux.statx(dirfd, name, flags, .{ .TYPE = true }, &stx); + if (linux.errno(rc) != .SUCCESS) return null; + return switch (stx.mode & linux.S.IFMT) { + linux.S.IFDIR => .dir, + linux.S.IFLNK => .symlink, + else => .other, + }; +} + +const Entry = struct { name: [:0]u8, kind: FileType }; + +/// Read every entry of the directory open at `fd` (excluding `.` and `..`). +fn listDir(gpa: Allocator, fd: i32, dirpath: []const u8) ![]Entry { + var list: std.ArrayList(Entry) = .empty; + errdefer { + for (list.items) |e| gpa.free(e.name); + list.deinit(gpa); + } + var buf: [32 * 1024]u8 align(@alignOf(linux.dirent64)) = undefined; + while (true) { + const rc = linux.getdents64(fd, &buf, buf.len); + switch (linux.errno(rc)) { + .SUCCESS => {}, + else => |e| { + std.debug.print("9player: getdents64 {s}: E{t}\n", .{ dirpath, e }); + return error.Mountpoint; + }, + } + if (rc == 0) break; + var off: usize = 0; + while (off < rc) { + const d: *align(1) const linux.dirent64 = @ptrCast(&buf[off]); + const name_ptr: [*:0]const u8 = @ptrCast(&buf[off + @offsetOf(linux.dirent64, "name")]); + const name = std.mem.span(name_ptr); + const dtype = d.type; + off += d.reclen; + if (std.mem.eql(u8, name, ".") or std.mem.eql(u8, name, "..")) continue; + if (list.items.len >= max_shadow_entries) { + std.debug.print("9player: refusing to shadow {s}: more than {d} entries\n", .{ dirpath, max_shadow_entries }); + return error.TooManyEntries; + } + const kind: FileType = switch (dtype) { + linux.DT.DIR => .dir, + linux.DT.LNK => .symlink, + linux.DT.UNKNOWN => fileType(fd, name_ptr, true) orelse .other, + else => .other, + }; + try list.append(gpa, .{ .name = try gpa.dupeZ(u8, name), .kind = kind }); + } + } + return list.toOwnedSlice(gpa); +} + +fn shadowDirectory(gpa: Allocator, parent: [:0]const u8) !void { + const open_rc = linux.open(parent, .{ .ACCMODE = .RDONLY, .DIRECTORY = true, .CLOEXEC = true }, 0); + switch (linux.errno(open_rc)) { + .SUCCESS => {}, + else => |e| { + std.debug.print("9player: open {s}: E{t}\n", .{ parent, e }); + return error.Mountpoint; + }, + } + const pfd: i32 = @intCast(open_rc); + defer _ = linux.close(pfd); + + const entries = try listDir(gpa, pfd, parent); + defer { + for (entries) |e| gpa.free(e.name); + gpa.free(entries); + } + + const tmpfs_opts: [*:0]const u8 = "mode=755"; + switch (linux.errno(linux.mount("tmpfs", parent, "tmpfs", linux.MS.NOSUID | linux.MS.NODEV, @intFromPtr(tmpfs_opts)))) { + .SUCCESS => {}, + else => |e| { + std.debug.print("9player: mount tmpfs on {s}: E{t}\n", .{ parent, e }); + return error.Mountpoint; + }, + } + + // `pfd` still refers to the original directory underneath the tmpfs, so + // `/proc/self/fd//` reaches the hidden entries. + var src_buf: [path_max]u8 = undefined; + var dst_buf: [path_max]u8 = undefined; + var link_buf: [path_max]u8 = undefined; + for (entries) |e| { + const src = std.fmt.bufPrintZ(&src_buf, "/proc/self/fd/{d}/{s}", .{ pfd, e.name }) catch { + std.debug.print("9player: shadow {s}/{s}: name too long (skipped)\n", .{ parent, e.name }); + continue; + }; + const dst = std.fmt.bufPrintZ(&dst_buf, "{s}/{s}", .{ parent, e.name }) catch { + std.debug.print("9player: shadow {s}/{s}: name too long (skipped)\n", .{ parent, e.name }); + continue; + }; + switch (e.kind) { + .dir => { + if (!check("mkdir", dst, linux.mkdirat(linux.AT.FDCWD, dst, 0o755))) continue; + _ = check("bind", dst, linux.mount(src, dst, null, linux.MS.BIND | linux.MS.REC, 0)); + }, + .symlink => { + const rc = linux.readlinkat(pfd, e.name, &link_buf, link_buf.len - 1); + if (!check("readlink", dst, rc)) continue; + link_buf[rc] = 0; + const target: [*:0]const u8 = @ptrCast(&link_buf); + _ = check("symlink", dst, linux.symlinkat(target, linux.AT.FDCWD, dst)); + }, + .other => { + const rc = linux.openat(linux.AT.FDCWD, dst, .{ .ACCMODE = .WRONLY, .CREAT = true, .CLOEXEC = true }, 0o644); + if (!check("create", dst, rc)) continue; + _ = linux.close(@intCast(rc)); + _ = check("bind", dst, linux.mount(src, dst, null, linux.MS.BIND | linux.MS.REC, 0)); + }, + } + } +} + +/// Report a failed per-entry step as a warning (the entry is skipped; the +/// rest of the shadow is still useful). Returns true on success. +fn check(step: []const u8, path: [*:0]const u8, rc: usize) bool { + switch (linux.errno(rc)) { + .SUCCESS => return true, + else => |e| { + std.debug.print("9player: shadow: {s} {s}: E{t} (skipped)\n", .{ step, std.mem.span(path), e }); + return false; + }, + } +} + +// --------------------------------------------------------------------------- +// spawn +// --------------------------------------------------------------------------- + +const ChildArgs = struct { + gpa: Allocator, + status_sock: i32, + mountpoint: [:0]const u8, + fuse_opts_prefix: [:0]const u8, // everything after "fd=," + mount_fuse: bool, + uid_map: []const u8, + gid_map: []const u8, + argv: [:null]?[*:0]const u8, + envp: [:null]?[*:0]const u8, + candidates: []const [:0]const u8, + name: []const u8, +}; + +/// Status channel protocol (child → parent, over a CLOEXEC socketpair): +/// a 0 byte means "namespace and mount are up" and carries the fuse fd as +/// SCM_RIGHTS; a non-zero byte is an exit status followed by a message. +/// EOF ends the conversation (exec succeeded, or the child died). +const ok_byte: u8 = 0; + +/// fork; the child unshares user+mount namespaces, maps its uid/gid, +/// makes `/` private, ensures the mountpoint, opens `/dev/fuse`, mounts it +/// on the mountpoint and sends the fd back. `spawn` returns at that point +/// (with `error.ChildFailed` and a message on stderr if any step failed). +/// The child then stats the mountpoint, which makes the kernel fetch the +/// root's attributes once the parent serves (the kernel seeds the fuse root +/// with uid 0, unmapped in the new user namespace, so nothing could be +/// created in the root until then), sets `NINEPLAYER_MOUNT` and execs +/// `argv`. An exec failure is reported on `Child.status_fd` and ends the +/// child with 126/127; collect it with `reportExecFailure` after +/// `bridge.serve` returns. +/// +/// If `installSignals` was called, the pid is stored into the registered +/// variable as soon as fork returns so no SIGCHLD can be missed. +pub fn spawn(gpa: Allocator, s: Spawn) !Child { + if (s.argv.len == 0 or s.argv[0].len == 0) { + std.debug.print("9player: empty program name\n", .{}); + return error.EmptyProgramName; + } + + const mountpoint = try gpa.dupeZ(u8, s.mountpoint); + defer gpa.free(mountpoint); + const fuse_opts_prefix = try std.fmt.allocPrintSentinel(gpa, "rootmode=40000,user_id={d},group_id={d},max_read={d}", .{ s.uid, s.gid, s.max_read }, 0); + defer gpa.free(fuse_opts_prefix); + var uid_buf: [64]u8 = undefined; + var gid_buf: [64]u8 = undefined; + const uid_map = try std.fmt.bufPrint(&uid_buf, "{d} {d} 1\n", .{ s.uid, s.uid }); + const gid_map = try std.fmt.bufPrint(&gid_buf, "{d} {d} 1\n", .{ s.gid, s.gid }); + const argv = try buildArgv(gpa, s.argv); + defer { + for (argv) |a| gpa.free(std.mem.span(a.?)); + gpa.free(argv); + } + const envp = try buildEnvp(gpa, s.envp, s.mountpoint); + defer { + gpa.free(std.mem.span(envp[envp.len - 1].?)); // the NINEPLAYER_MOUNT entry we created + gpa.free(envp); + } + const candidates = try pathCandidates(gpa, s.envp, s.argv[0]); + defer { + for (candidates) |c| gpa.free(c); + gpa.free(candidates); + } + + var sv: [2]i32 = undefined; + switch (linux.errno(linux.socketpair(linux.AF.UNIX, linux.SOCK.STREAM | linux.SOCK.CLOEXEC, 0, &sv))) { + .SUCCESS => {}, + else => |e| { + std.debug.print("9player: socketpair: E{t}\n", .{e}); + return error.SystemResources; + }, + } + + const child_args = ChildArgs{ + .gpa = gpa, + .status_sock = sv[1], + .mountpoint = mountpoint, + .fuse_opts_prefix = fuse_opts_prefix, + .mount_fuse = s.mount_fuse, + .uid_map = uid_map, + .gid_map = gid_map, + .argv = argv, + .envp = envp, + .candidates = candidates, + .name = s.argv[0], + }; + + const fork_rc = linux.fork(); + switch (linux.errno(fork_rc)) { + .SUCCESS => {}, + else => |e| { + _ = linux.close(sv[0]); + _ = linux.close(sv[1]); + std.debug.print("9player: fork: E{t}\n", .{e}); + return error.SystemResources; + }, + } + if (fork_rc == 0) childMain(&child_args); + + const pid: i32 = @intCast(fork_rc); + if (child_pid_ptr) |p| @atomicStore(i32, p, pid, .seq_cst); + _ = linux.close(sv[1]); + + // First byte: ok (with the fuse fd attached) or a failure status. + var first: [1]u8 = .{ok_byte}; // defined even if recvmsg stores nothing + var fuse_fd: i32 = -1; + var n: usize = 0; + while (true) { + const rc = recvWithFd(sv[0], &first, &fuse_fd, 0); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .INTR => continue, + else => break, + } + n = rc; + break; + } + if (n == 1 and first[0] == ok_byte and (fuse_fd >= 0 or !s.mount_fuse)) { + return .{ .pid = pid, .fuse_fd = fuse_fd, .status_fd = sv[0] }; + } + + // Failure. A status byte means the child is exiting on its own and a + // message follows. Anything else (EOF: the child died before reporting; + // an ok byte without the fd: the SCM_RIGHTS transfer was truncated, e.g. + // EMFILE) is a protocol violation: the child may be about to exec with a + // dead mount, so kill it before waiting rather than reading the status + // socket until an exec'd program eventually exits. + const reported = n == 1 and first[0] != ok_byte; + if (!reported) _ = linux.kill(pid, .KILL); + if (fuse_fd >= 0) _ = linux.close(fuse_fd); + var msg: [512]u8 = undefined; + var len: usize = 0; + while (reported and len < msg.len) { + const rc = linux.read(sv[0], msg[len..].ptr, msg.len - len); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .INTR => continue, + else => break, + } + if (rc == 0) break; + len += rc; + } + _ = linux.close(sv[0]); + if (reported) { + std.debug.print("9player: {s}\n", .{msg[0..len]}); + } else if (n == 1) { + std.debug.print("9player: child handshake failed: no fuse fd received (out of file descriptors?)\n", .{}); + } else { + std.debug.print("9player: child exited before reporting\n", .{}); + } + _ = waitChild(pid) catch {}; + if (child_pid_ptr) |p| @atomicStore(i32, p, 0, .seq_cst); + const status: u8 = if (reported) first[0] else setup_failure_status; + return switch (status) { + 126 => error.ExecPermission, + 127 => error.ExecNotFound, + else => error.ChildFailed, + }; +} + +/// After the child is gone (or the mount is dead): print the exec failure +/// the child reported on `status_fd`, if any, and close it. Returns the +/// status byte the child announced, or null when exec succeeded / nothing +/// was reported. Never blocks. +pub fn reportExecFailure(child: Child) ?u8 { + defer _ = linux.close(child.status_fd); + var msg: [512]u8 = undefined; + var len: usize = 0; + while (len < msg.len) { + var iov = [_]std.posix.iovec{.{ .base = msg[len..].ptr, .len = msg.len - len }}; + var hdr = linux.msghdr{ + .name = null, + .namelen = 0, + .iov = &iov, + .iovlen = 1, + .control = null, + .controllen = 0, + .flags = 0, + }; + const rc = linux.recvmsg(child.status_fd, &hdr, linux.MSG.DONTWAIT); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .INTR => continue, + else => break, + } + if (rc == 0) break; + len += rc; + } + if (len == 0) return null; + std.debug.print("9player: {s}\n", .{msg[1..len]}); + return msg[0]; +} + +const cmsg_fd_len = @sizeOf(linux.cmsghdr) + @sizeOf(i32); +const cmsg_fd_space = std.mem.alignForward(usize, cmsg_fd_len, @sizeOf(usize)); + +/// sendmsg one data byte, optionally with `fd` attached as SCM_RIGHTS. +fn sendWithFd(sock: i32, byte: u8, fd: ?i32) usize { + const data = [_]u8{byte}; + const iov = [_]std.posix.iovec_const{.{ .base = &data, .len = 1 }}; + var cbuf: [cmsg_fd_space]u8 align(@alignOf(linux.cmsghdr)) = @splat(0); + var msg = linux.msghdr_const{ + .name = null, + .namelen = 0, + .iov = &iov, + .iovlen = 1, + .control = null, + .controllen = 0, + .flags = 0, + }; + if (fd) |f| { + const hdr: *linux.cmsghdr = @ptrCast(&cbuf); + hdr.* = .{ .len = cmsg_fd_len, .level = linux.SOL.SOCKET, .type = linux.SCM.RIGHTS }; + @memcpy(cbuf[@sizeOf(linux.cmsghdr)..][0..@sizeOf(i32)], std.mem.asBytes(&f)); + msg.control = &cbuf; + msg.controllen = cmsg_fd_space; + } + return linux.sendmsg(sock, &msg, linux.MSG.NOSIGNAL); +} + +/// recvmsg into `buf`; an SCM_RIGHTS fd, if any, is stored in `fd_out`. +fn recvWithFd(sock: i32, buf: []u8, fd_out: *i32, flags: u32) usize { + var iov = [_]std.posix.iovec{.{ .base = buf.ptr, .len = buf.len }}; + var cbuf: [cmsg_fd_space]u8 align(@alignOf(linux.cmsghdr)) = @splat(0); + var msg = linux.msghdr{ + .name = null, + .namelen = 0, + .iov = &iov, + .iovlen = 1, + .control = &cbuf, + .controllen = cbuf.len, + .flags = 0, + }; + const rc = linux.recvmsg(sock, &msg, linux.MSG.CMSG_CLOEXEC | flags); + if (linux.errno(rc) != .SUCCESS) return rc; + if (msg.controllen >= cmsg_fd_len) { + const hdr: *const linux.cmsghdr = @ptrCast(&cbuf); + if (hdr.level == linux.SOL.SOCKET and hdr.type == linux.SCM.RIGHTS and hdr.len >= cmsg_fd_len) { + var fd: i32 = undefined; + @memcpy(std.mem.asBytes(&fd), cbuf[@sizeOf(linux.cmsghdr)..][0..@sizeOf(i32)]); + fd_out.* = fd; + } + } + return rc; +} + +/// Child side of `spawn`. Never returns. +fn childMain(c: *const ChildArgs) noreturn { + resetSignals(); + + const rc_unshare = linux.errno(linux.unshare(linux.CLONE.NEWUSER | linux.CLONE.NEWNS)); + if (rc_unshare != .SUCCESS) childFail(c, setup_failure_status, "unshare(CLONE_NEWUSER|CLONE_NEWNS)", rc_unshare, true); + writeProcFile(c, "/proc/self/setgroups", "deny", true); + writeProcFile(c, "/proc/self/uid_map", c.uid_map, false); + writeProcFile(c, "/proc/self/gid_map", c.gid_map, false); + + const root: [*:0]const u8 = "/"; + const rc_priv = linux.mount(null, root, null, linux.MS.REC | linux.MS.PRIVATE, 0); + if (linux.errno(rc_priv) != .SUCCESS) childFail(c, setup_failure_status, "mount(/, MS_REC|MS_PRIVATE)", linux.errno(rc_priv), true); + + ensureMountpoint(c.gpa, c.mountpoint) catch { + childFail(c, setup_failure_status, "mountpoint setup failed (pass --mount an existing directory)", .SUCCESS, false); + }; + + var fuse_fd: ?i32 = null; + if (c.mount_fuse) { + // Must be opened here, after unshare: the kernel only mounts a fuse + // device opened from the mount's own user namespace. + const rc_open = linux.open("/dev/fuse", .{ .ACCMODE = .RDWR, .CLOEXEC = true }, 0); + switch (linux.errno(rc_open)) { + .SUCCESS => {}, + .NOENT => childFail(c, setup_failure_status, "open /dev/fuse: ENOENT (is the fuse module loaded? try: modprobe fuse)", .SUCCESS, false), + else => |e| childFail(c, setup_failure_status, "open /dev/fuse", e, true), + } + const fd: i32 = @intCast(rc_open); + var opts_buf: [256]u8 = undefined; + const opts = std.fmt.bufPrintZ(&opts_buf, "fd={d},{s}", .{ fd, c.fuse_opts_prefix }) catch unreachable; + const rc = linux.mount("9player", c.mountpoint, "fuse", linux.MS.NOSUID | linux.MS.NODEV, @intFromPtr(opts.ptr)); + if (linux.errno(rc) != .SUCCESS) childFail(c, setup_failure_status, "mount fuse", linux.errno(rc), true); + fuse_fd = fd; + } + const sent = sendWithFd(c.status_sock, ok_byte, fuse_fd); + if (linux.errno(sent) != .SUCCESS) linux.exit_group(setup_failure_status); + if (fuse_fd) |fd| { + _ = linux.close(fd); // the parent holds the connection now + // Force one GETATTR of the root (served by the parent, which is + // entering its serve loop now); see `spawn`. Errors don't matter. + var stx: linux.Statx = undefined; + _ = linux.statx(linux.AT.FDCWD, c.mountpoint, 0, .{ .TYPE = true }, &stx); + } + + var last: E = .NOENT; + var saw_acces = false; + for (c.candidates) |cand| { + const rc = linux.execve(cand.ptr, c.argv.ptr, c.envp.ptr); + last = linux.errno(rc); + switch (last) { + .NOENT, .NOTDIR, .LOOP, .NAMETOOLONG => continue, + .ACCES => { + saw_acces = true; + continue; + }, + else => break, + } + } + var buf: [512]u8 = undefined; + // "Not found" covers every candidate that could not even be resolved + // (a PATH element that is a file gives ENOTDIR, a symlink loop ELOOP); + // a candidate that existed but was not executable wins over those. + const not_found = switch (last) { + .NOENT, .NOTDIR, .LOOP, .NAMETOOLONG => true, + else => false, + }; + if (not_found and saw_acces) last = .ACCES; + const status: u8 = if (not_found and !saw_acces) 127 else 126; + const text = std.fmt.bufPrint(&buf, "exec {s}", .{c.name}) catch "exec"; + childFail(c, status, text, last, true); +} + +fn writeProcFile(c: *const ChildArgs, path: [*:0]const u8, data: []const u8, ignore_missing: bool) void { + const rc = linux.open(path, .{ .ACCMODE = .WRONLY, .CLOEXEC = true }, 0); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .NOENT => if (ignore_missing) return else childFail(c, setup_failure_status, std.mem.span(path), .NOENT, true), + else => |e| childFail(c, setup_failure_status, std.mem.span(path), e, true), + } + const fd: i32 = @intCast(rc); + const w = linux.write(fd, data.ptr, data.len); + const we = linux.errno(w); + _ = linux.close(fd); + if (we != .SUCCESS) childFail(c, setup_failure_status, std.mem.span(path), we, true); + if (w != data.len) childFail(c, setup_failure_status, std.mem.span(path), .IO, true); +} + +/// Write `[: E]` to the status socket and exit. +fn childFail(c: *const ChildArgs, status: u8, step: []const u8, e: E, with_errno: bool) noreturn { + var buf: [600]u8 = undefined; + buf[0] = status; + const rest = if (with_errno) + std.fmt.bufPrint(buf[1..], "{s}: E{t}", .{ step, e }) catch buf[1..1] + else + std.fmt.bufPrint(buf[1..], "{s}", .{step}) catch buf[1..1]; + const msg = buf[0 .. 1 + rest.len]; + var off: usize = 0; + while (off < msg.len) { + const rc = linux.write(c.status_sock, msg[off..].ptr, msg.len - off); + if (linux.errno(rc) == .INTR) continue; + if (linux.errno(rc) != .SUCCESS) break; + off += rc; + } + linux.exit_group(status); +} + +// --------------------------------------------------------------------------- +// Signals +// --------------------------------------------------------------------------- + +var child_pid_ptr: ?*i32 = null; +var chld_pipe_w: i32 = -1; +var reaped = std.atomic.Value(bool).init(false); +var reaped_status = std.atomic.Value(u32).init(0); +/// A second child (the `--spawn` server) that the SIGCHLD handler reaps so +/// it does not linger as a zombie when it dies mid-session. Its exit does +/// not stop the serve loop. 0 = none. +var server_pid = std.atomic.Value(i32).init(0); + +/// Register the `--spawn` server for reaping by the SIGCHLD handler. +pub fn watchServer(pid: i32) void { + server_pid.store(pid, .seq_cst); +} + +/// Seconds the serve loop gets to come back after the child died before +/// the watchdog ends the process anyway. +pub const exit_grace_seconds: isize = 3; + +/// The watched child is already dead but the serve loop has not come back +/// (it is stuck in a 9P request the server never answers): a terminal +/// signal, or the watchdog armed by `onChld`, then ends 9player with the +/// child's status instead of hanging. Nothing is lost: the mount is torn +/// down when the process exits. +fn bailIfChildGone() void { + if (!reaped.load(.acquire)) return; + const srv = server_pid.load(.seq_cst); + if (srv > 0) _ = linux.kill(srv, .TERM); + linux.exit_group(decodeStatus(reaped_status.load(.acquire))); +} + +fn armWatchdog() void { + // setitimer takes an itimerval; std declares it with itimerspec, which + // has the same layout on 64-bit targets (the sub-second field is 0). + const t = linux.itimerspec{ + .it_interval = .{ .sec = 0, .nsec = 0 }, + .it_value = .{ .sec = exit_grace_seconds, .nsec = 0 }, + }; + _ = linux.setitimer(@intFromEnum(linux.ITIMER.REAL), &t, null); +} + +fn onAlarm(_: linux.SIG) callconv(.c) void { + bailIfChildGone(); +} + +fn onForward(sig: linux.SIG) callconv(.c) void { + const p = child_pid_ptr orelse return; + const pid = @atomicLoad(i32, p, .seq_cst); + if (pid > 0) _ = linux.kill(pid, sig); + bailIfChildGone(); +} + +/// SIGINT/SIGQUIT: the child owns the tty and gets them itself; we only +/// react when the child is already gone (see `bailIfChildGone`). +fn onTerminal(_: linux.SIG) callconv(.c) void { + bailIfChildGone(); +} + +/// Only the watched child counts: reap it here (WNOHANG), remember its +/// status, forget its pid (so a later SIGTERM cannot hit a recycled pid) +/// and poke the self-pipe. The `--spawn` server is reaped too but does not +/// interrupt `bridge.serve`; SIGCHLD from anything else is ignored. +fn onChld(_: linux.SIG) callconv(.c) void { + const srv = server_pid.load(.seq_cst); + if (srv > 0) { + var sst: u32 = 0; + const src = linux.waitpid(srv, &sst, linux.W.NOHANG); + if (linux.errno(src) == .SUCCESS and src != 0) server_pid.store(0, .seq_cst); + } + const p = child_pid_ptr orelse return; + const pid = @atomicLoad(i32, p, .seq_cst); + if (pid <= 0) return; + var st: u32 = 0; + const rc = linux.waitpid(pid, &st, linux.W.NOHANG); + if (linux.errno(rc) != .SUCCESS or rc == 0) return; + reaped_status.store(st, .release); + reaped.store(true, .release); + @atomicStore(i32, p, 0, .seq_cst); + const b = [_]u8{'c'}; + _ = linux.write(chld_pipe_w, &b, 1); + armWatchdog(); +} + +/// SIGPIPE ignored; SIGINT/SIGQUIT effectively ignored (the child owns the +/// tty) unless the child is already dead; SIGTERM/SIGHUP forwarded to +/// `*child_pid`; SIGCHLD for `*child_pid` reaps it, writes a byte to a +/// nonblocking self-pipe whose read end is returned (use it as `stop_fd`) +/// and arms a watchdog (`exit_grace_seconds`, SIGALRM) that ends the +/// process with the child's status should the serve loop stay blocked. +/// `*child_pid` is filled in by `spawn`. +pub fn installSignals(child_pid: *i32) !i32 { + child_pid_ptr = child_pid; + var fds: [2]i32 = undefined; + switch (linux.errno(linux.pipe2(&fds, .{ .CLOEXEC = true, .NONBLOCK = true }))) { + .SUCCESS => {}, + else => |e| { + std.debug.print("9player: pipe2: E{t}\n", .{e}); + return error.SystemResources; + }, + } + chld_pipe_w = fds[1]; + + const ign = linux.Sigaction{ .handler = .{ .handler = linux.SIG.IGN }, .mask = linux.sigemptyset(), .flags = 0 }; + const term = linux.Sigaction{ .handler = .{ .handler = &onTerminal }, .mask = linux.sigemptyset(), .flags = linux.SA.RESTART }; + const fwd = linux.Sigaction{ .handler = .{ .handler = &onForward }, .mask = linux.sigemptyset(), .flags = linux.SA.RESTART }; + const chld = linux.Sigaction{ .handler = .{ .handler = &onChld }, .mask = linux.sigemptyset(), .flags = linux.SA.RESTART | linux.SA.NOCLDSTOP }; + const alrm = linux.Sigaction{ .handler = .{ .handler = &onAlarm }, .mask = linux.sigemptyset(), .flags = linux.SA.RESTART }; + std.posix.sigaction(.INT, &term, null); + std.posix.sigaction(.QUIT, &term, null); + std.posix.sigaction(.ALRM, &alrm, null); + std.posix.sigaction(.PIPE, &ign, null); + std.posix.sigaction(.TERM, &fwd, null); + std.posix.sigaction(.HUP, &fwd, null); + std.posix.sigaction(.CHLD, &chld, null); + return fds[0]; +} + +/// Restore default dispositions in the child before exec (ignored signals +/// would otherwise survive execve). +fn resetSignals() void { + const dfl = linux.Sigaction{ .handler = .{ .handler = linux.SIG.DFL }, .mask = linux.sigemptyset(), .flags = 0 }; + inline for (.{ linux.SIG.INT, linux.SIG.QUIT, linux.SIG.PIPE, linux.SIG.TERM, linux.SIG.HUP, linux.SIG.CHLD, linux.SIG.ALRM }) |sig| { + _ = linux.sigaction(sig, &dfl, null); + } +} + +// --------------------------------------------------------------------------- +// Waiting +// --------------------------------------------------------------------------- + +fn takeReaped() ?u32 { + if (!reaped.load(.acquire)) return null; + return reaped_status.load(.acquire); +} + +/// waitpid status → exit code (`128+sig` when killed by a signal). +pub fn decodeStatus(st: u32) u8 { + if (linux.W.IFEXITED(st)) return linux.W.EXITSTATUS(st); + if (linux.W.IFSIGNALED(st)) return 128 +% @as(u8, @truncate(@intFromEnum(linux.W.TERMSIG(st)))); + return 1; +} + +/// Block until `pid` exits (the SIGCHLD handler may have reaped it already). +pub fn waitChild(pid: i32) !u8 { + while (true) { + if (takeReaped()) |st| return decodeStatus(st); + var st: u32 = 0; + const rc = linux.waitpid(pid, &st, 0); + switch (linux.errno(rc)) { + .SUCCESS => return decodeStatus(st), + .INTR => continue, + .CHILD => { + if (takeReaped()) |s| return decodeStatus(s); + return error.NoChild; + }, + else => |e| { + std.debug.print("9player: waitpid: E{t}\n", .{e}); + return error.Wait; + }, + } + } +} + +/// Non-blocking: the exit status of `pid` if it has exited, else null. +pub fn reapIfExited(pid: i32) ?u8 { + if (takeReaped()) |st| return decodeStatus(st); + var st: u32 = 0; + const rc = linux.waitpid(pid, &st, linux.W.NOHANG); + switch (linux.errno(rc)) { + .SUCCESS => return if (rc == 0) null else decodeStatus(st), + .CHILD => return if (takeReaped()) |s| decodeStatus(s) else null, + else => return null, + } +} + +/// Reap any child (used for the `--spawn` server at exit). Non-blocking. +pub fn reapAny(pid: i32) void { + var st: u32 = 0; + _ = linux.waitpid(pid, &st, linux.W.NOHANG); +} + +// --------------------------------------------------------------------------- +// Tests (no namespaces needed; `ensureMountpoint` is exercised by +// test/integration.sh through the 9player binary) +// --------------------------------------------------------------------------- + +const testing = std.testing; + +test "normalizePath: absolute paths" { + const gpa = testing.allocator; + const cases = [_]struct { in: []const u8, out: []const u8 }{ + .{ .in = "/mnt/9p", .out = "/mnt/9p" }, + .{ .in = "/mnt/9p/", .out = "/mnt/9p" }, + .{ .in = "//mnt///9p//", .out = "/mnt/9p" }, + .{ .in = "/mnt/./9p/.", .out = "/mnt/9p" }, + .{ .in = "/mnt/x/../9p", .out = "/mnt/9p" }, + .{ .in = "/../mnt/9p", .out = "/mnt/9p" }, + .{ .in = "/a/b/c/../..", .out = "/a" }, + }; + for (cases) |c| { + const got = try normalizePath(gpa, "/cwd", c.in); + defer gpa.free(got); + try testing.expectEqualStrings(c.out, got); + try testing.expectEqual(@as(u8, 0), got[got.len]); + } +} + +test "normalizePath: relative paths use cwd" { + const gpa = testing.allocator; + const cases = [_]struct { cwd: []const u8, in: []const u8, out: []const u8 }{ + .{ .cwd = "/home/me", .in = "mnt", .out = "/home/me/mnt" }, + .{ .cwd = "/home/me", .in = "./mnt/", .out = "/home/me/mnt" }, + .{ .cwd = "/home/me", .in = "../mnt", .out = "/home/mnt" }, + .{ .cwd = "/home/me/", .in = ".", .out = "/home/me" }, + .{ .cwd = "/", .in = "x", .out = "/x" }, + }; + for (cases) |c| { + const got = try normalizePath(gpa, c.cwd, c.in); + defer gpa.free(got); + try testing.expectEqualStrings(c.out, got); + } +} + +test "normalizePath: rejects root and empty" { + const gpa = testing.allocator; + try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/cwd", "/")); + try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/cwd", "///")); + try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/cwd", "/mnt/..")); + try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/cwd", "")); + try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/", "..")); +} + +test "resolveMountpoint: relative resolves against the real cwd" { + const gpa = testing.allocator; + const got = try resolveMountpoint(gpa, "sub/dir"); + defer gpa.free(got); + try testing.expect(got[0] == '/'); + try testing.expect(std.mem.endsWith(u8, got, "/sub/dir")); +} + +test "getenv" { + const env = [_:null]?[*:0]const u8{ "PATH=/a:/b", "X=", "PATHX=no", "NINEPLAYER_MOUNT=/m" }; + const envp: [*:null]const ?[*:0]const u8 = &env; + try testing.expectEqualStrings("/a:/b", getenv(envp, "PATH").?); + try testing.expectEqualStrings("", getenv(envp, "X").?); + try testing.expectEqualStrings("/m", getenv(envp, "NINEPLAYER_MOUNT").?); + try testing.expect(getenv(envp, "NOPE") == null); + try testing.expect(getenv(envp, "PAT") == null); +} + +test "pathCandidates: PATH search" { + const gpa = testing.allocator; + const env = [_:null]?[*:0]const u8{ "PATH=/usr/local/bin::/usr/bin", "HOME=/h" }; + const cands = try pathCandidates(gpa, &env, "fish"); + defer { + for (cands) |c| gpa.free(c); + gpa.free(cands); + } + try testing.expectEqual(@as(usize, 3), cands.len); + try testing.expectEqualStrings("/usr/local/bin/fish", cands[0]); + try testing.expectEqualStrings("./fish", cands[1]); + try testing.expectEqualStrings("/usr/bin/fish", cands[2]); +} + +test "pathCandidates: slash means no search; default PATH" { + const gpa = testing.allocator; + const env = [_:null]?[*:0]const u8{"HOME=/h"}; + { + const cands = try pathCandidates(gpa, &env, "./bin/x"); + defer { + for (cands) |c| gpa.free(c); + gpa.free(cands); + } + try testing.expectEqual(@as(usize, 1), cands.len); + try testing.expectEqualStrings("./bin/x", cands[0]); + } + { + const cands = try pathCandidates(gpa, &env, "sh"); + defer { + for (cands) |c| gpa.free(c); + gpa.free(cands); + } + try testing.expectEqual(@as(usize, 3), cands.len); + try testing.expectEqualStrings("/usr/local/bin/sh", cands[0]); + try testing.expectEqualStrings("/bin/sh", cands[1]); + } + try testing.expectError(error.EmptyProgramName, pathCandidates(gpa, &env, "")); +} + +test "findInPath finds sh" { + const gpa = testing.allocator; + const env = [_:null]?[*:0]const u8{"PATH=/nonexistent:/bin:/usr/bin"}; + const p = try findInPath(gpa, &env, "sh"); + defer gpa.free(p); + try testing.expect(std.mem.endsWith(u8, p, "/sh")); + try testing.expectError(error.FileNotFound, findInPath(gpa, &env, "definitely-not-a-program-9player")); +} + +test "buildEnvp replaces NINEPLAYER_MOUNT" { + const gpa = testing.allocator; + const env = [_:null]?[*:0]const u8{ "A=1", "NINEPLAYER_MOUNT=/old", "B=2" }; + const out = try buildEnvp(gpa, &env, "/mnt/9p"); + defer { + gpa.free(std.mem.span(out[out.len - 1].?)); + gpa.free(out); + } + try testing.expectEqual(@as(usize, 3), out.len); + try testing.expectEqualStrings("A=1", std.mem.span(out[0].?)); + try testing.expectEqualStrings("B=2", std.mem.span(out[1].?)); + try testing.expectEqualStrings("NINEPLAYER_MOUNT=/mnt/9p", std.mem.span(out[2].?)); + try testing.expect(out[3] == null); + try testing.expectEqualStrings("/mnt/9p", getenv(out.ptr, "NINEPLAYER_MOUNT").?); +} + +test "decodeStatus" { + try testing.expectEqual(@as(u8, 0), decodeStatus(0)); + try testing.expectEqual(@as(u8, 7), decodeStatus(7 << 8)); + try testing.expectEqual(@as(u8, 255), decodeStatus(255 << 8)); + try testing.expectEqual(@as(u8, 128 + 9), decodeStatus(9)); // SIGKILL + try testing.expectEqual(@as(u8, 128 + 15), decodeStatus(15)); // SIGTERM +} + +test "ensureMountpoint: existing directory is accepted, plain file rejected" { + const gpa = testing.allocator; + try ensureMountpoint(gpa, "/tmp"); + try testing.expectError(error.Mountpoint, ensureMountpoint(gpa, "/proc/self/status")); +} diff --git a/9player/test/adv_bridge_hostile.py b/9player/test/adv_bridge_hostile.py new file mode 100755 index 0000000..b842a1a --- /dev/null +++ b/9player/test/adv_bridge_hostile.py @@ -0,0 +1,487 @@ +#!/usr/bin/env python3 +"""A scriptable, hostile 9P2000 server on a Unix socket (stdlib only). + +Usage: adv_bridge_hostile.py SOCKET MODE + +Serves a tiny in-memory tree: + /f "hello world\\n" + /d/g "in d\\n" + /fids reading it returns the number of fids currently bound + /big 1 MiB of pseudo-random bytes +plus create/write/remove/wstat so the scratch battery can run in `ok` mode. + +MODE selects one misbehaviour (see MODES below). Everything not covered by +the mode behaves normally, so 9player gets through version/attach/stat(root). +""" +import os +import random +import socket +import struct +import sys +import time + +NOTAG = 0xFFFF +NOFID = 0xFFFFFFFF +QTDIR = 0x80 +DMDIR = 0x80000000 + +Tversion, Rversion = 100, 101 +Tauth, Rauth = 102, 103 +Tattach, Rattach = 104, 105 +Rerror = 107 +Tflush, Rflush = 108, 109 +Twalk, Rwalk = 110, 111 +Topen, Ropen = 112, 113 +Tcreate, Rcreate = 114, 115 +Tread, Rread = 116, 117 +Twrite, Rwrite = 118, 119 +Tclunk, Rclunk = 120, 121 +Tremove, Rremove = 122, 123 +Tstat, Rstat = 124, 125 +Twstat, Rwstat = 126, 127 + +MODES = """ +ok behave (qid paths are recycled LIFO after remove, like many servers) +trunc Rread on /f: send half the frame, then close +short_frame Rread on /f: frame whose size field is 3 +huge_frame Rread on /f: frame whose size field is msize+1 +wrong_tag Rread on /f: reply carries tag+1 +wrong_type Tstat on /f: answer with an Rwalk +rread_big Rread on /f: count = requested+1 +rwalk_many Twalk to f: nwqid = nwname+1 +rwalk_zero Twalk to nope: Rwalk nwqid=0 instead of Rerror +rstat_garbage Tstat on /f: random bytes as the stat +rstat_overlong Tstat on /f: inner stat size disagrees with outer +dir_split Tread on /: a stat record split across two Rreads +dir_forever Tread on /: ignore offset, always return the same records +qid_collide every file and dir shares qid.path 7 (root keeps its own) +qid_zero every qid.path is 0, including the root +name_slash / has an entry "a/b" +name_empty / has an entry "" +name_huge / has an entry with a 60000-byte name +name_dots / lists "." and ".." too +rerror_big Twalk to nope: Rerror with 65535 bytes of text +extra_reply Rread on /f: an unsolicited Rclunk (tag 9) precedes the real reply +never Tread on /f: never reply (hang) +close_mid Tread on /f: close the socket without replying +renegotiate Tread on /f: an unsolicited Rversion precedes the real reply +length_max Tstat on /f: length = 2**64-1 +iounit_one Ropen: iounit = 1 +rwrite_big Rwrite: count = requested+1 +msize_tiny Rversion msize = 64 +version_unknown Rversion "unknown" +rename_fail Twstat with a new name always fails "file already exists" +slow every reply delayed 20 ms (for interrupt tests) +""" + + +def s8(x): return struct.pack('= 4: + n = struct.unpack('= n: + msg, buf = buf[:n], buf[n:] + out = self.handle(msg) + if out is None: + return # hang up / hang + if self.mode == 'slow': + time.sleep(0.02) + conn.sendall(out) + continue + data = conn.recv(65536) + if not data: + return + buf += data + + def handle(self, msg): + typ = msg[4] + tag = struct.unpack(' len(node.content): + node.content.extend(b'\0' * (off - len(node.content))) + node.content[off:off + len(data)] = data + node.mtime = int(time.time()) + n = len(data) + 1 if self.mode == 'rwrite_big' else len(data) + return self.frame(Rwrite, tag, s32(n)) + if typ == Tclunk: + fid = r.u32() + if fid not in self.fids: + return self.err(tag, 'unknown fid') + del self.fids[fid] + return self.frame(Rclunk, tag, b'') + if typ == Tremove: + fid = r.u32() + if fid not in self.fids: + return self.err(tag, 'unknown fid') + node = self.fids[fid][0] + del self.fids[fid] + if node is self.root: + return self.err(tag, 'cannot remove root') + if node.isdir and node.children: + return self.err(tag, 'directory not empty') + parent = self.find_parent(self.root, node) + if parent is not None: + del parent.children[node.name] + self.free_paths.append(node.path) + node.removed = True + return self.frame(Rremove, tag, b'') + if typ == Tstat: + fid = r.u32() + if fid not in self.fids: + return self.err(tag, 'unknown fid') + node = self.fids[fid][0] + if node is self.root.children.get('f'): + m = self.mode + if m == 'wrong_type': + return self.frame(Rwalk, tag, s16(0)) + if m == 'rstat_garbage': + junk = bytes([0xAB] * 60) + return self.frame(Rstat, tag, s16(len(junk)) + junk) + if m == 'rstat_overlong': + st = node.stat_bytes(self) + inner = st[2:] + return self.frame(Rstat, tag, s16(len(inner) + 5) + inner) + if m == 'length_max': + st = node.stat_bytes(self, length=2 ** 64 - 1) + return self.frame(Rstat, tag, s16(len(st)) + st) + st = node.stat_bytes(self) + return self.frame(Rstat, tag, s16(len(st)) + st) + if typ == Twstat: + fid = r.u32() + r.u16() + st = r.bytes(r.u16()) + if fid not in self.fids: + return self.err(tag, 'unknown fid') + node = self.fids[fid][0] + sr = Reader(st) + sr.u16(); sr.u32(); sr.bytes(13) + mode = sr.u32(); sr.u32(); mtime = sr.u32(); length = sr.u64() + name = sr.str() + if name and name != node.name: + if self.mode == 'rename_fail': + return self.err(tag, 'file already exists') + parent = self.find_parent(self.root, node) + if name in parent.children: + return self.err(tag, 'file already exists') + del parent.children[node.name] + node.name = name + parent.children[name] = node + if mode != 0xFFFFFFFF: + node.mode = mode & 0o777 + if mtime != 0xFFFFFFFF: + node.mtime = mtime + if length != 0xFFFFFFFFFFFFFFFF and not node.isdir: + if length < len(node.content): + del node.content[length:] + else: + node.content.extend(b'\0' * (length - len(node.content))) + return self.frame(Rwstat, tag, b'') + return self.err(tag, 'unsupported message') + + def find_parent(self, cur, node): + for c in cur.children.values(): + if c is node: + return cur + if c.isdir: + p = self.find_parent(c, node) + if p is not None: + return p + return None + + def readdir(self, tag, node, off, count): + recs = [] + if node is self.root: + m = self.mode + if m == 'name_slash': + recs.append(node.stat_bytes(self, name='a/b')) + if m == 'name_empty': + recs.append(node.stat_bytes(self, name='')) + if m == 'name_huge': + recs.append(node.stat_bytes(self, name='h' * 60000)) + if m == 'name_dots': + recs.append(node.stat_bytes(self, name='.')) + recs.append(node.stat_bytes(self, name='..')) + for c in node.children.values(): + recs.append(c.stat_bytes(self)) + blob = b''.join(recs) + if node is self.root and self.mode == 'dir_forever': + return self.frame(Rread, tag, s32(len(blob)) + blob) + if node is self.root and self.mode == 'dir_split': + # first read: up to the middle of the second record; second read: the rest + cut = len(recs[0]) + len(recs[1]) // 2 + if off == 0: + data = blob[:cut] + elif off == cut: + data = blob[cut:] + else: + data = b'' + return self.frame(Rread, tag, s32(len(data)) + data) + # 9P rule: offset 0 or previous offset+count; never split a record. + out = b'' + pos = 0 + for rec in recs: + if pos >= off and len(out) + len(rec) <= count: + out += rec + elif pos >= off: + break + pos += len(rec) + return self.frame(Rread, tag, s32(len(out)) + out) + + +class Reader: + def __init__(self, b): + self.b = b + self.i = 0 + + def bytes(self, n): + v = self.b[self.i:self.i + n] + self.i += n + return v + + def u8(self): return struct.unpack(' (introspect unused; part of zig build 9player-adv) +set -u +PLAYER=$(realpath "${1:?path to 9player}") +HERE=$(cd "$(dirname "$0")" && pwd) +SRV=$HERE/adv_bridge_hostile.py +TMP=$(mktemp -d "${TMPDIR:-/tmp}/9player-adv.XXXXXX") +M=/mnt/9p +FAILED=0 +PASSED=0 +SRVPID= + +cleanup() { [ -n "$SRVPID" ] && kill "$SRVPID" 2>/dev/null; pkill -f "adv_bridge_hostile.py $TMP" 2>/dev/null; rm -rf "$TMP"; } +trap cleanup EXIT + +if ! unshare -Urm true 2>/dev/null || [ ! -c /dev/fuse ]; then echo "SKIP: no user namespaces or /dev/fuse"; exit 0; fi + +pass() { PASSED=$((PASSED + 1)); echo "ok - $1"; } +fail() { FAILED=$((FAILED + 1)); echo "FAIL - $1"; shift; [ $# -gt 0 ] && printf ' %s\n' "$@"; } +expect_eq() { if [ "$2" = "$3" ]; then pass "$1"; else fail "$1" "expected: $(printf %q "$2")" "actual: $(printf %q "$3")"; fi; } +expect_contains() { case "$3" in *"$2"*) pass "$1" ;; *) fail "$1" "missing: $(printf %q "$2")" "in: $(printf %q "$3")" ;; esac; } + +start_server() { # mode + [ -n "$SRVPID" ] && { kill "$SRVPID" 2>/dev/null; wait "$SRVPID" 2>/dev/null; } + SOCK=$TMP/$1.sock + rm -f "$SOCK" + python3 "$SRV" "$SOCK" "$1" >"$TMP/$1.srv.out" 2>&1 "$TMP/stderr") + RC=$? + STDERR=$(cat "$TMP/stderr") +} + +# 9player must not die of a signal or panic. RC 124 = timeout(1) fired. +no_crash() { # name + if [ "$RC" -ge 128 ] || [ "$RC" -eq 124 ]; then fail "$1: 9player exit $RC" "$STDERR"; return; fi + case "$STDERR" in *panic*|*"Segmentation"*|*"integer overflow"*|*"reached unreachable"*|*"index out of bounds"*) fail "$1: crash text in stderr" "$STDERR";; *) pass "$1: no crash (exit $RC)";; esac +} + +echo "# sanity: the hostile server behaves in 'ok' mode" +run ok "cat $M/f"; expect_eq "ok: cat f" "hello world" "$OUT" +run ok "cat $M/d/g"; expect_eq "ok: nested" "in d" "$OUT" +run ok "cat $M/nope 2>&1 | sed 's/.*: //'"; expect_eq "ok: ENOENT" "No such file or directory" "$OUT" +run ok "head -c 1048576 $M/big | wc -c | grep -q 1048576 && echo yes"; expect_eq "ok: 1 MiB read matches" "yes" "$OUT" + +echo "# unlink + recreate with a recycled qid.path" +run ok "echo 1 > $M/a; rm $M/a; echo 2 > $M/a; cat $M/a; rm $M/a"; expect_eq "qid reuse: new content, not stale" "2" "$OUT" +run ok "echo 1 > $M/x; rm $M/x; mkdir $M/x; stat -c %F $M/x; rmdir $M/x"; expect_eq "qid reuse: file→dir on the same path" "directory" "$OUT" + +echo "# fids do not grow with the number of operations" +loop='i=0; while [ $i -lt N ]; do echo hi > M/t; cat M/t >/dev/null; mkdir M/dd; rmdir M/dd; rm M/t; i=$((i+1)); done; cat M/fids' +run ok "$(echo "$loop" | sed "s|N|20|; s|M/|$M/|g")"; a=$OUT +run ok "$(echo "$loop" | sed "s|N|200|; s|M/|$M/|g")"; b=$OUT +expect_eq "fids after 20 == after 200 iterations ($a)" "$a" "$b" +floop='i=0; while [ $i -lt N ]; do cat M/nope 2>/dev/null; echo x > M/fids 2>/dev/null; mkdir M/f 2>/dev/null; rm M/d 2>/dev/null; mv M/f M/d 2>/dev/null; i=$((i+1)); done; cat M/fids' +run ok "$(echo "$floop" | sed "s|N|20|; s|M/|$M/|g")"; a=$OUT +run ok "$(echo "$floop" | sed "s|N|200|; s|M/|$M/|g")"; b=$OUT +expect_eq "fids after 20 == after 200 failing iterations ($a)" "$a" "$b" + +echo "# rename over an existing file must not lose the target when the rename fails" +run rename_fail "echo A > $M/a; echo B > $M/b; mv $M/a $M/b 2>/dev/null; echo mv=\$?; cat $M/b; cat $M/a" +expect_contains "rename_fail: mv reports failure" "mv=1" "$OUT" +expect_contains "rename_fail: target b still has its content" "B" "$OUT" +expect_contains "rename_fail: source a still has its content" "A" "$OUT" + +echo "# protocol violations on a data read must yield an error, not a crash" +for mode in trunc short_frame huge_frame wrong_tag rread_big extra_reply close_mid renegotiate rwrite_big; do + if [ "$mode" = rwrite_big ]; then script="dd if=/dev/zero of=$M/f bs=10 count=1 2>&1; echo status=\$?"; else script="cat $M/f 2>&1; echo status=\$?"; fi + run "$mode" "$script" + no_crash "$mode" + expect_contains "$mode: child sees an error" "status=1" "$OUT" + case "$OUT" in *"Input/output error"*|*"not connected"*) pass "$mode: EIO/ENOTCONN";; *) fail "$mode: errno text" "$OUT";; esac +done + +echo "# protocol violations on lookup/stat" +for mode in wrong_type rwalk_many rstat_garbage rstat_overlong; do + run "$mode" "stat -c %s $M/f 2>&1; echo status=\$?" + no_crash "$mode" + expect_contains "$mode: child sees an error" "status=1" "$OUT" +done +run rwalk_zero "cat $M/nope 2>&1; echo status=\$?; cat $M/f 2>&1" +no_crash "rwalk_zero" +expect_contains "rwalk_zero: child sees an error" "status=1" "$OUT" +run rerror_big "cat $M/nope 2>&1; echo status=\$?; cat $M/f" +no_crash "rerror_big" +expect_contains "rerror_big: child sees an error" "status=1" "$OUT" +expect_contains "rerror_big: session survives a 64 KiB Rerror" "hello world" "$OUT" + +echo "# hostile stat contents" +run length_max "stat -c '%s %b' $M/f 2>&1; echo status=\$?" +no_crash "length_max" +expect_contains "length_max: stat succeeds with a saturated block count" "status=0" "$OUT" +run iounit_one "cat $M/f; head -c 3000 $M/big | wc -c" +no_crash "iounit_one" +expect_contains "iounit_one: read still complete" "hello world" "$OUT" +expect_contains "iounit_one: 3000 bytes" "3000" "$OUT" + +echo "# hostile directory listings" +run dir_split "ls $M 2>&1; echo status=\$?; cat $M/f" +no_crash "dir_split" +expect_contains "dir_split: readdir fails with EIO" "Input/output error" "$OUT" +expect_contains "dir_split: session survives" "hello world" "$OUT" +run dir_forever "ls $M 2>&1 | tail -c 200; echo status=\$?; cat $M/f; grep VmRSS /proc/\$PPID/status" +no_crash "dir_forever" +expect_contains "dir_forever: infinite directory is cut off with EIO" "Input/output error" "$OUT" +expect_contains "dir_forever: session survives" "hello world" "$OUT" +for mode in name_slash name_empty name_huge name_dots; do + run "$mode" "ls -a $M | tr '\n' ' '; echo; cat $M/f" + no_crash "$mode" + expect_contains "$mode: listing still works" "big d f fids" "$OUT" + expect_contains "$mode: file readable" "hello world" "$OUT" + case "$mode" in + name_slash) expect_eq "$mode: slash entry dropped" "" "$(printf '%s' "$OUT" | grep -o 'a/b')";; + name_dots) expect_eq "$mode: exactly one . and one .." "1 1" "$(printf '%s %s' "$(printf '%s\n' "$OUT" | head -1 | tr ' ' '\n' | grep -c '^\.$')" "$(printf '%s\n' "$OUT" | head -1 | tr ' ' '\n' | grep -c '^\.\.$')")";; + esac +done + +echo "# qid collisions" +run qid_collide "cat $M/f; cat $M/d/g; ls $M/d; stat -c %i $M/f $M/d 2>&1; echo status=\$?" +no_crash "qid_collide" +expect_contains "qid_collide: reads work" "hello world" "$OUT" +run qid_zero "cat $M/f; ls $M | tr '\n' ' '; echo; cat $M/d/g; echo status=\$?" +no_crash "qid_zero" +expect_contains "qid_zero: file with root's qid.path is still a readable file" "hello world" "$OUT" +expect_contains "qid_zero: root still lists" "big d f fids" "$OUT" +expect_contains "qid_zero: nested file readable" "in d" "$OUT" + +echo "# version negotiation" +run version_unknown "echo ran" +expect_eq "version_unknown: 9player refuses (125)" "125" "$RC" +run msize_tiny "cat $M/f 2>&1; echo status=\$?" +no_crash "msize_tiny" + +echo "# a server that never replies" +start_server never +# SIGTERM is forwarded to the child; once the child is gone 9player must leave the +# pending 9P reply behind and exit even though the server stays silent. +timeout -s TERM 3 "$PLAYER" --unix "$SOCK" -- sh -c "cat $M/f; echo unreachable" >"$TMP/never.out" 2>"$TMP/never.err" & +TPID=$! +sleep 4 +if kill -0 "$TPID" 2>/dev/null; then + fail "never: SIGTERM did not end 9player while a reply was outstanding"; kill -9 "$TPID" +else + pass "never: SIGTERM ends 9player even while the server is silent" +fi +wait "$TPID" 2>/dev/null +# Without a signal the mount hangs (documented v1 limitation) until the server dies. +timeout 30 "$PLAYER" --unix "$SOCK" -- sh -c "cat $M/f; echo unreachable" >"$TMP/never.out" 2>"$TMP/never.err" & +TPID=$! +sleep 1.5 +if kill -0 "$TPID" 2>/dev/null; then + pass "never: mount hangs while the server is silent (documented v1 limitation)" + kill "$SRVPID"; wait "$SRVPID" 2>/dev/null; SRVPID= + for _ in $(seq 1 50); do kill -0 "$TPID" 2>/dev/null || break; sleep 0.1; done + if kill -0 "$TPID" 2>/dev/null; then fail "never: 9player still alive after its server died"; kill -9 "$TPID"; else pass "never: killing the server unblocks 9player"; fi +else + wait "$TPID"; fail "never: 9player exited early ($?)" "$(cat "$TMP/never.err")" +fi + +echo "# interrupting a slow read (INTERRUPT must not confuse reply matching)" +run slow "cat $M/big > /dev/null; cat $M/f; cat $M/fids" --msize 8192 +base=$(printf '%s\n' "$OUT" | tail -1) +run slow "(cat $M/big > /dev/null & sleep 0.3; kill -INT \$!; wait \$!) 2>/dev/null; cat $M/f; cat $M/fids" --msize 8192 +no_crash "slow" +expect_contains "slow: read after interrupted read works" "hello world" "$OUT" +expect_eq "slow: fids after an interrupted read == after a complete one ($base)" "$base" "$(printf '%s\n' "$OUT" | tail -1)" + +echo +echo "passed=$PASSED failed=$FAILED" +[ "$FAILED" -eq 0 ] diff --git a/9player/test/adv_bridge_semantics.sh b/9player/test/adv_bridge_semantics.sh new file mode 100755 index 0000000..f13fd57 --- /dev/null +++ b/9player/test/adv_bridge_semantics.sh @@ -0,0 +1,205 @@ +#!/usr/bin/env bash +# FUSE semantics through the bridge against introspect's /scratch tree. +# Usage: bash 9player/test/adv_bridge_semantics.sh <9player> (part of zig build 9player-adv) +set -u +PLAYER=$(realpath "${1:?path to 9player}") +INTROSPECT=$(realpath "${2:?path to introspect}") +TMP=$(mktemp -d "${TMPDIR:-/tmp}/9player-sem.XXXXXX") +M=/mnt/9p +S=$M/scratch +FAILED=0 +PASSED=0 +SRVPID= +cleanup() { [ -n "$SRVPID" ] && kill "$SRVPID" 2>/dev/null; rm -rf "$TMP"; } +trap cleanup EXIT +if ! unshare -Urm true 2>/dev/null || [ ! -c /dev/fuse ]; then echo "SKIP: no user namespaces or /dev/fuse"; exit 0; fi + +pass() { PASSED=$((PASSED + 1)); echo "ok - $1"; } +fail() { FAILED=$((FAILED + 1)); echo "FAIL - $1"; shift; [ $# -gt 0 ] && printf ' %s\n' "$@"; } +expect_eq() { if [ "$2" = "$3" ]; then pass "$1"; else fail "$1" "expected: $(printf %q "$2")" "actual: $(printf %q "$3")"; fi; } +expect_contains() { case "$3" in *"$2"*) pass "$1" ;; *) fail "$1" "missing: $(printf %q "$2")" "in: $(printf %q "$3")" ;; esac; } + +SOCK=$TMP/i.sock +"$INTROSPECT" --unix "$SOCK" >"$TMP/srv.out" 2>&1 & +SRVPID=$! +for _ in $(seq 1 100); do [ -S "$SOCK" ] && break; sleep 0.02; done +# Each run is a fresh session; state persists in the server, so tests clean up after themselves. +run() { OUT=$(timeout 120 "$PLAYER" --unix "$SOCK" "${EXTRA[@]}" -- sh -c "$1" 2>"$TMP/stderr"); RC=$?; STDERR=$(cat "$TMP/stderr"); } +EXTRA=() +py() { run "python3 - <<'PYEOF' +$1 +PYEOF"; } + +echo "# open/create flags" +py " +import os, errno +p='$S/excl' +fd=os.open(p, os.O_CREAT|os.O_WRONLY, 0o644); os.write(fd, b'x'); os.close(fd) +try: + os.open(p, os.O_CREAT|os.O_EXCL|os.O_WRONLY, 0o644); print('no error') +except OSError as e: print(errno.errorcode[e.errno]) +os.unlink(p) +fd=os.open('$S/ro', os.O_CREAT|os.O_RDONLY, 0o644); print(os.read(fd, 10)); os.close(fd) +print(os.path.exists('$S/ro')); os.unlink('$S/ro') +" +expect_eq "O_EXCL on an existing file is EEXIST" "EEXIST" "$(printf '%s\n' "$OUT" | sed -n 1p)" +expect_eq "create with O_RDONLY works and reads empty" $'b\'\'\nTrue' "$(printf '%s\n' "$OUT" | sed -n 2,3p)" + +run "mkdir -m 700 $S/m7 && stat -c %a $S/m7; chmod 755 $S/m7 && stat -c %a $S/m7; rmdir $S/m7" +expect_eq "mkdir -m 700 then chmod 755" $'700\n755' "$OUT" +run "echo x > $S/c && chmod 600 $S/c && stat -c %a $S/c; chmod 444 $S/c && stat -c %a $S/c; rm -f $S/c" +expect_eq "chmod on a file" $'600\n444' "$OUT" +run "echo x > $S/t && touch -d @1000000000 $S/t && stat -c %Y $S/t; touch $S/t && [ \$(stat -c %Y $S/t) -gt 1000000000 ] && echo now; rm $S/t" +expect_eq "utimes (explicit) and touch (now)" $'1000000000\nnow' "$OUT" +run "echo abc > $S/tr && truncate -s 10 $S/tr && stat -c %s $S/tr && od -An -c $S/tr | tr -s ' ' | tr -d '\n'; echo; rm $S/tr" +expect_eq "truncate to larger zero-fills" $'10\n a b c \\n \\0 \\0 \\0 \\0 \\0 \\0' "$OUT" +run "echo a > $S/ap && echo b >> $S/ap && echo c >> $S/ap && cat $S/ap | tr '\n' ' '; rm $S/ap" +expect_eq "shell append" "a b c " "$OUT" +py " +import os +p='$S/ap2' +f1=os.open(p, os.O_CREAT|os.O_WRONLY|os.O_APPEND, 0o644) +f2=os.open(p, os.O_WRONLY|os.O_APPEND) +os.write(f1, b'one '); os.write(f2, b'two '); os.write(f1, b'three') +os.close(f1); os.close(f2) +print(open(p).read()); os.unlink(p) +" +expect_eq "O_APPEND from two descriptors interleaves in order" "one two three" "$OUT" +run "echo 0123456789 > $S/tt && (echo X > $S/tt) && cat $S/tt && stat -c %s $S/tt; rm $S/tt" +expect_eq "O_TRUNC (atomic_o_trunc) truncates before write" $'X\n2' "$OUT" + +echo "# reads at the edges" +run "printf hello > $S/e; dd if=$S/e bs=1 skip=100 count=5 2>/dev/null | wc -c; head -c 0 $S/e | wc -c; dd if=/dev/null of=$S/e bs=1 count=0 conv=notrunc 2>/dev/null; cat $S/e; echo; rm $S/e" +expect_eq "read past EOF is 0 bytes; 0-byte read/write are no-ops" $'0\n0\nhello' "$OUT" +head -c 4194304 /dev/urandom >"$TMP/four" +SUM=$(sha256sum <"$TMP/four" | cut -d' ' -f1) +run "dd if=$TMP/four of=$S/four bs=4M status=none && dd if=$S/four bs=4M status=none | sha256sum | cut -d' ' -f1; stat -c %s $S/four; rm $S/four" +expect_eq "4 MiB single-request dd round trip" "$SUM"$'\n4194304' "$OUT" +py " +import os +p='$S/lseek' +open(p,'w').write('0123456789') +f=open(p,'rb'); f.seek(0, 2); print(f.tell()); f.seek(-3, 2); print(f.read()); f.close() +os.unlink(p) +" +expect_eq "lseek SEEK_END on a direct_io file" $'10\nb\'789\'' "$OUT" + +echo "# unlink of an open file" +py " +import os +p='$S/unl' +fd=os.open(p, os.O_CREAT|os.O_RDWR, 0o644) +os.write(fd, b'before') +os.unlink(p) +print(os.path.exists(p)) +os.lseek(fd, 0, 0); print(os.read(fd, 100)) +os.write(fd, b'-after'); os.lseek(fd, 0, 0); print(os.read(fd, 100)) +os.close(fd) +" +expect_eq "read/write through the fd after unlink" $'False\nb\'before\'\nb\'before-after\'' "$OUT" + +echo "# rename" +run "echo A > $S/ra; echo B > $S/rb; mv $S/ra $S/rb && cat $S/rb; ls $S | tr '\n' ' '; echo; rm $S/rb" +expect_eq "rename over an existing file replaces it, no leftovers" $'A\nrb ' "$OUT" +run "mkdir $S/rd1 && echo x > $S/rd1/f && mv $S/rd1 $S/rd2 && cat $S/rd2/f && ls $S/rd2; rm -r $S/rd2; ls $S | wc -l" +expect_eq "rename of a directory" $'x\nf\n0' "$OUT" +run "mkdir $S/e1 $S/e2 && mv -T $S/e1 $S/e2 && ls $S | tr '\n' ' '; rmdir $S/e2" +expect_eq "rename dir over an empty dir" "e2 " "$OUT" +run "mkdir $S/n1 $S/n2 && echo x > $S/n2/f && mv -T $S/n1 $S/n2 2>&1 | sed 's/.*: //'; rm -r $S/n1 $S/n2" +expect_eq "rename dir over a non-empty dir is ENOTEMPTY" "Directory not empty" "$OUT" +py " +import os +p='$S/same'; open(p,'w').write('x'); os.rename(p, p); print(open(p).read()); os.unlink(p) +" +expect_eq "rename onto itself is a no-op" "x" "$OUT" +py " +import os, ctypes, errno +libc = ctypes.CDLL(None, use_errno=True) +a, b = b'$S/nra', b'$S/nrb' +open(a,'w').write('A'); open(b,'w').write('B') +r = libc.renameat2(-100, a, -100, b, 1) # RENAME_NOREPLACE +print('rc', r, errno.errorcode.get(ctypes.get_errno())) +print(open(b).read()) +os.unlink(a); os.unlink(b) +" +expect_eq "RENAME_NOREPLACE keeps the target" $'rc -1 EEXIST\nB' "$OUT" + +echo "# directories" +run "mkdir $S/many && cd $S/many && i=0; while [ \$i -lt 5000 ]; do : > f\$i; i=\$((i+1)); done; ls | wc -l; ls -l | wc -l; grep VmRSS /proc/\$PPID/status | awk '{print \$2}' > $TMP/rss1; rm -f $S/many/*; rmdir $S/many; ls | wc -l; grep VmRSS /proc/\$PPID/status | awk '{print \$2}' > $TMP/rss2" +expect_eq "5000 entries: ls and ls -l" $'5000\n5001\n0' "$OUT" +r1=$(cat "$TMP/rss1"); r2=$(cat "$TMP/rss2") +if [ "$r2" -le $((r1 + 2048)) ]; then pass "RSS after cleanup ($r2 KiB) <= after listing ($r1 KiB)+2 MiB"; else fail "RSS grew after cleanup: $r1 -> $r2 KiB"; fi +py " +import os +d='$S/chg'; os.mkdir(d) +for i in range(50): open(f'{d}/a{i}','w').close() +it = os.scandir(d); first = next(it).name +for i in range(3000): open(f'{d}/b{i}','w').close() +rest = [e.name for e in it] +print(first[0], len(rest) >= 49, len(set(rest)) == len(rest)) +for n in os.listdir(d): os.unlink(f'{d}/{n}') +os.rmdir(d) +" +expect_eq "readdir of a directory that changes mid-iteration" "a True True" "$OUT" +run "ls $M/.. > /dev/null && echo ok; stat -c %i $M $M/. $M/scratch/..; cd $M/scratch && ls .. | grep -c scratch" +expect_eq ".. of the root and of a subdir" $'ok\n1\n1\n1\n1' "$OUT" +run "cd $M && find . -type d | wc -l && find . -type f | head -1 && find $S -type f | wc -l" +expect_contains "find -type works" "./README" "$OUT" +run "stat -f -c '%T %S %l' $M; df -P $M | tail -1 | awk '{print \$1}'; sync -f $M && echo synced; sync && echo synced2" +expect_eq "statfs, df, syncfs, sync" $'fuse 4096 255\n9player\nsynced\nsynced2' "$OUT" + +echo "# server refusals keep their errno through the error path" +run "echo x > $M/build/zig_version; a=\$?; mkdir $M/build/x 2>/dev/null; b=\$?; rm $M/README 2>/dev/null; c=\$?; rmdir $M/build 2>/dev/null; d=\$?; echo \$a\$b\$c\$d" 2>/dev/null +expect_eq "open-for-write / mkdir / rm / rmdir on read-only nodes all fail (errno preserved through error path)" "1111" "$(printf '%s\n' "$OUT" | tail -1)" + +echo "# unsupported operations fail cleanly" +run "echo x > $S/l1; ln $S/l1 $S/l2 2>&1 | sed 's/.*: //'; ln -s l1 $S/l3 2>&1 | sed 's/.*: //'; mkfifo $S/p 2>&1 | sed 's/.*: //'; ls $S | tr '\n' ' '; echo; rm $S/l1" +# The kernel turns ENOSYS from LINK into EPERM (fuse_link); symlink/mknod keep ENOSYS. +expect_eq "link/symlink/mknod fail cleanly" $'Operation not permitted\nFunction not implemented\nFunction not implemented\nl1 ' "$OUT" +run "echo x > $S/x1; setfattr -n user.a -v 1 $S/x1 2>&1 | sed 's/.*: //'; getfattr -n user.a $S/x1 2>&1 | sed 's/.*: //'; getfattr -d $S/x1 2>&1 | sed 's/.*: //'; rm $S/x1" +expect_eq "xattr ops are EOPNOTSUPP" $'Operation not supported\nOperation not supported\nOperation not supported' "$OUT" +py " +import os, fcntl, mmap, errno +p='$S/mm'; open(p,'w').write('mapme') +fd=os.open(p, os.O_RDWR) +fcntl.flock(fd, fcntl.LOCK_EX); fcntl.flock(fd, fcntl.LOCK_UN); fcntl.lockf(fd, fcntl.LOCK_EX); fcntl.lockf(fd, fcntl.LOCK_UN); print('locks ok') +try: + m = mmap.mmap(fd, 5); print('shared', bytes(m)); m.close() +except OSError as e: print('shared', errno.errorcode[e.errno]) +try: + m = mmap.mmap(fd, 5, flags=mmap.MAP_PRIVATE, prot=mmap.PROT_READ); print('private', bytes(m)); m.close() +except OSError as e: print('private', errno.errorcode[e.errno]) +os.close(fd); os.unlink(p) +" +expect_contains "flock/lockf work (local locks)" "locks ok" "$OUT" +case "$OUT" in *"shared ENODEV"*|*"shared b'mapme'"*) pass "shared mmap: clean result ($(printf '%s\n' "$OUT" | sed -n 2p))";; *) fail "shared mmap" "$OUT";; esac +expect_contains "private mmap reads the file" "private b'mapme'" "$OUT" + +echo "# tools" +mkdir -p "$TMP/tree/sub/deeper"; echo one > "$TMP/tree/a"; echo two > "$TMP/tree/sub/b"; head -c 70000 /dev/urandom > "$TMP/tree/sub/deeper/blob"; chmod 640 "$TMP/tree/a" +run "cp -a $TMP/tree $S/tree 2>&1; diff -r $TMP/tree $S/tree && echo same; stat -c %a $S/tree/a; cp -a $S/tree $TMP/back && diff -r $TMP/tree $TMP/back && echo back; rm -r $S/tree" +expect_eq "cp -a there and back" $'same\n640\nback' "$OUT" +run "cd $TMP && tar cf $S/t.tar tree && cd $S && mkdir tx && tar xf t.tar -C tx && diff -r $TMP/tree tx/tree && echo tar-ok; rm -r $S/tx $S/t.tar" +expect_eq "tar into and out of the mount" "tar-ok" "$OUT" +run "rsync -a $TMP/tree/ $S/rs/ && diff -r $TMP/tree $S/rs && echo rsync-ok; sleep 1.1; echo mod > $TMP/tree/a; rsync -a $TMP/tree/ $S/rs/ && cat $S/rs/a; rm -r $S/rs" +expect_eq "rsync -a twice" $'rsync-ok\nmod' "$OUT" +run "cd $S && mkdir repo && cd repo && git init -q . && git config user.email a@b && git config user.name n && echo hi > f && git add f && git commit -qm init && git log --oneline | wc -l && git status --porcelain | wc -l; cd $S && rm -rf repo; ls $S | wc -l" +expect_eq "git init/add/commit inside the mount" $'1\n0\n0' "$OUT" + +echo "# --no-direct-io" +EXTRA=(--no-direct-io) +run "cp $TMP/four $S/nd && cmp $TMP/four $S/nd && echo same; stat -c %s $S/nd; rm $S/nd" +expect_eq "no-direct-io: 4 MiB round trip" $'same\n4194304' "$OUT" +[ "$RC" -eq 0 ] || echo " stderr: $STDERR" +# Buffered writes are per-page without a writeback cache (kernel behaviour); the +# point here is only that a large buffered write is delivered intact. +EXTRA=(--no-direct-io) +head -c 262144 /dev/urandom > "$TMP/w" +WSUM=$(sha256sum <"$TMP/w" | cut -d' ' -f1) +run "cp $TMP/w $S/w && sha256sum < $S/w | cut -d' ' -f1; stat -c %s $S/w; rm $S/w" +expect_eq "no-direct-io: 256 KiB buffered write is intact" "$WSUM"$'\n262144' "$OUT" +EXTRA=() + +echo +echo "passed=$PASSED failed=$FAILED" +[ "$FAILED" -eq 0 ] diff --git a/9player/test/adv_bridge_stress.sh b/9player/test/adv_bridge_stress.sh new file mode 100755 index 0000000..3d1203a --- /dev/null +++ b/9player/test/adv_bridge_stress.sh @@ -0,0 +1,64 @@ +#!/usr/bin/env bash +# Resource and concurrency stress through the bridge against introspect. +# Usage: bash 9player/test/adv_bridge_stress.sh <9player> (~1-2 min; part of zig build 9player-adv) +set -u +PLAYER=$(realpath "${1:?path to 9player}") +INTROSPECT=$(realpath "${2:?path to introspect}") +TMP=$(mktemp -d "${TMPDIR:-/tmp}/9player-stress.XXXXXX") +M=/mnt/9p +S=$M/scratch +FAILED=0 +PASSED=0 +SRVPID= +cleanup() { [ -n "$SRVPID" ] && kill "$SRVPID" 2>/dev/null; rm -rf "$TMP"; } +trap cleanup EXIT +if ! unshare -Urm true 2>/dev/null || [ ! -c /dev/fuse ]; then echo "SKIP: no user namespaces or /dev/fuse"; exit 0; fi +pass() { PASSED=$((PASSED + 1)); echo "ok - $1"; } +fail() { FAILED=$((FAILED + 1)); echo "FAIL - $1"; shift; [ $# -gt 0 ] && printf ' %s\n' "$@"; } +expect_eq() { if [ "$2" = "$3" ]; then pass "$1"; else fail "$1" "expected: $(printf %q "$2")" "actual: $(printf %q "$3")"; fi; } + +SOCK=$TMP/i.sock +"$INTROSPECT" --unix "$SOCK" >"$TMP/srv.out" 2>&1 & +SRVPID=$! +for _ in $(seq 1 100); do [ -S "$SOCK" ] && break; sleep 0.02; done +run() { OUT=$(timeout 600 "$PLAYER" --unix "$SOCK" -- sh -c "$1" 2>"$TMP/stderr"); RC=$?; STDERR=$(cat "$TMP/stderr"); } + +echo "# 100k+ 9P operations in one session; RSS must plateau" +# Each iteration: create+write+close, open+read+close, unlink, plus a failing lookup: ~15 RPCs. +run "rss() { grep VmRSS /proc/\$PPID/status | awk '{print \$2}'; } +i=0; while [ \$i -lt 8000 ]; do echo \$i > $S/s; cat $S/s > /dev/null; rm $S/s; cat $S/none 2>/dev/null; i=\$((i+1)); if [ \$i -eq 2000 ]; then rss; fi; done; rss; ls $S | wc -l" +r1=$(printf '%s\n' "$OUT" | sed -n 1p); r2=$(printf '%s\n' "$OUT" | sed -n 2p); left=$(printf '%s\n' "$OUT" | sed -n 3p) +expect_eq "scratch left clean" "0" "$left" +if [ -n "$r1" ] && [ -n "$r2" ] && [ "$r2" -le $((r1 + 4096)) ]; then pass "RSS at 2000 iterations = $r1 KiB, at 8000 = $r2 KiB"; else fail "RSS grows: $r1 -> $r2 KiB" "$STDERR"; fi +case "$STDERR" in *leak*) fail "allocator reported leaks" "$STDERR";; *) pass "no leak report from the debug allocator";; esac + +echo "# eight processes hammering the mount concurrently" +run "mkdir $S/par; for p in 1 2 3 4 5 6 7 8; do ( + d=$S/par/p\$p; mkdir \$d; i=0; bad=0 + while [ \$i -lt 300 ]; do + printf '%s-%s' \$p \$i > \$d/f\$((i % 7)); v=\$(cat \$d/f\$((i % 7))); [ \"\$v\" = \"\$p-\$i\" ] || bad=\$((bad+1)) + mkdir \$d/dd; rmdir \$d/dd; ls \$d > /dev/null; i=\$((i+1)) + done; rm -r \$d; echo \$p:\$bad ) & done; wait; ls $S/par | wc -l; rmdir $S/par" +expect_eq "all workers verified their own data" "1:0 2:0 3:0 4:0 5:0 6:0 7:0 8:0" "$(printf '%s\n' "$OUT" | grep ':' | sort | tr '\n' ' ' | sed 's/ $//')" +expect_eq "parallel tree fully removed" "0" "$(printf '%s\n' "$OUT" | grep -v ':')" + +echo "# a process killed mid-read and mid-write" +run "head -c 8000000 /dev/urandom > $S/kb; (cat $S/kb > /dev/null & sleep 0.05; kill -9 \$!; wait \$!) 2>/dev/null; (cat /dev/zero > $S/kw & sleep 0.05; kill -9 \$!; wait \$!) 2>/dev/null; sha256sum < $S/kb | cut -c1-8 > /dev/null && echo readable; [ -f $S/kw ] && echo written; rm $S/kb $S/kw; ls $S | wc -l" +expect_eq "survives SIGKILL mid-read/mid-write" $'readable\nwritten\n0' "$OUT" + +echo "# many open handles at once (fh counter, fid table)" +run "python3 - <<'EOF' +import os +d='$S/fh'; os.mkdir(d) +fds=[] +for i in range(1500): + fd=os.open(f'{d}/h{i%50}', os.O_CREAT|os.O_RDWR, 0o644); os.write(fd, b'z'); fds.append(fd) +for fd in fds: os.close(fd) +for i in range(50): os.unlink(f'{d}/h{i}') +os.rmdir(d); print('ok') +EOF" +expect_eq "1500 simultaneous handles" "ok" "$OUT" + +echo +echo "passed=$PASSED failed=$FAILED" +[ "$FAILED" -eq 0 ] diff --git a/9player/test/adv_ns_process.sh b/9player/test/adv_ns_process.sh new file mode 100755 index 0000000..4ce0ae8 --- /dev/null +++ b/9player/test/adv_ns_process.sh @@ -0,0 +1,202 @@ +#!/usr/bin/env bash +# Adversarial regression tests for 9player/src/ns.zig and 9player/src/main.zig: process, +# namespace, signal and CLI handling. Real namespaces, real FUSE. +# Usage: bash 9player/test/adv_ns_process.sh <9player> (part of zig build 9player-adv) +# Exit 0 on success (or when the machine cannot run the tests), 1 on failure. +set -u + +PLAYER=$(realpath "${1:?path to 9player}") +INTROSPECT=$(realpath "${2:?path to introspect}") +# Unix socket paths are limited to ~107 bytes; keep the temp dir short. +TMP=$(mktemp -d "${TMPDIR:-/tmp}/9padv.XXXXXX") +PIDS=() +FAILED=0 +PASSED=0 + +cleanup() { + for p in "${PIDS[@]:-}"; do [ -n "$p" ] && kill "$p" 2>/dev/null; done + rm -rf "$TMP" +} +trap cleanup EXIT + +if ! unshare -Urm true 2>/dev/null; then echo "SKIP: unprivileged user namespaces unavailable"; exit 0; fi +if [ ! -c /dev/fuse ]; then echo "SKIP: /dev/fuse missing"; exit 0; fi + +pass() { PASSED=$((PASSED + 1)); echo "ok - $1"; } +fail() { FAILED=$((FAILED + 1)); echo "FAIL - $1"; shift; [ $# -gt 0 ] && printf ' %s\n' "$@"; } +expect_eq() { if [ "$2" = "$3" ]; then pass "$1"; else fail "$1" "expected: $(printf %q "$2")" "actual: $(printf %q "$3")"; fi; } +expect_contains() { case "$3" in *"$2"*) pass "$1" ;; *) fail "$1" "missing: $(printf %q "$2")" "in: $(printf %q "$3")" ;; esac; } + +SOCK=$TMP/s +"$INTROSPECT" --unix "$SOCK" & +PIDS+=($!) +for _ in $(seq 1 100); do [ -S "$SOCK" ] && break; sleep 0.05; done +[ -S "$SOCK" ] || { echo "introspect did not create $SOCK"; exit 1; } +MI_BEFORE=$(grep -v " $TMP" /proc/self/mountinfo | sort) + +TIMEOUT=$(command -v timeout) +run() { "$TIMEOUT" 60 "$PLAYER" --unix "$SOCK" "$@"; } + +echo "# CLI" +expect_eq "--help goes to stdout, exit 0" "Usage: 9player" "$(run --help 2>/dev/null | head -1 | cut -c1-14; )" +expect_eq "--help exit code" "0" "$("$PLAYER" --help >/dev/null 2>&1; echo $?)" +expect_eq "--version on stdout" "9player" "$("$PLAYER" --version 2>/dev/null | cut -d' ' -f1)" +expect_eq "single-dash typo is a usage error, not a program" "125" "$(run -mount /x -- true 2>/dev/null; echo $?)" +expect_contains "single-dash typo message" "unknown option -mount" "$(run -mount /x -- true 2>&1)" +expect_eq "--unix= empty is a usage error" "125" "$("$PLAYER" --unix= -- true 2>/dev/null; echo $?)" +expect_contains "--unix= message" "socket path" "$("$PLAYER" --unix= -- true 2>&1)" +expect_eq "--mount '' is a usage error" "125" "$(run --mount '' -- true 2>/dev/null; echo $?)" +expect_eq "--msize huge rejected" "125" "$(run --msize 4294967295 -- true 2>/dev/null; echo $?)" +expect_eq "--msize 16 MiB accepted" "ok" "$(run --msize 16777216 -- sh -c 'echo ok')" +expect_contains "empty program name is reported" "empty program name" "$(run -- '' 2>&1)" +expect_eq "empty program name exit" "125" "$(run -- '' 2>/dev/null; echo $?)" +expect_eq "empty \$SHELL falls back to /bin/sh" "0" "$(SHELL= run -- /dev/null 2>&1; echo $?)" +expect_eq "--fd with a closed descriptor fails early" "125" "$("$PLAYER" --fd 987 -- true 2>/dev/null; echo $?)" +expect_contains "--fd bad descriptor message" "--fd 987: EBADF" "$("$PLAYER" --fd 987 -- true 2>&1)" + +echo "# exec failures" +expect_eq "not found is 127" "127" "$(run -- no-such-program-9player 2>/dev/null; echo $?)" +expect_eq "PATH element that is a file: still 127" "127" "$(PATH=/etc/passwd run -- true 2>/dev/null; echo $?)" +expect_contains "PATH element that is a file: message" "exec true: E" "$(PATH=/etc/passwd run -- true 2>&1)" +printf '#!/bin/sh\necho no\n' >"$TMP/nx"; chmod 644 "$TMP/nx" +expect_eq "non-executable is 126" "126" "$(run -- "$TMP/nx" 2>/dev/null; echo $?)" +mkdir -p "$TMP/p1" "$TMP/p2"; cp "$TMP/nx" "$TMP/p1/prog"; printf '#!/bin/sh\necho right\n' >"$TMP/p2/prog"; chmod 755 "$TMP/p2/prog" +expect_eq "non-executable first in PATH, executable later" "right" "$(PATH=$TMP/p1:$TMP/p2 run -- prog)" +expect_eq "non-executable only in PATH is 126" "126" "$(PATH=$TMP/p1 run -- prog 2>/dev/null; echo $?)" +expect_eq "argv[0] preserved" "sh" "$(run -- sh -c 'echo $0')" +expect_eq "PATH unset uses default" "ok" "$(env -u PATH "$PLAYER" --unix "$SOCK" -- sh -c 'echo ok')" + +echo "# fd hygiene" +# 9player passes inherited descriptors through untouched, so compare with what a +# plain child of this script sees (the runner may itself hold extra fds). +FD_LIST='ls /proc/self/fd | grep -v "^3$" | sort -n | tr "\n" " " | sed "s/ $//"' +FD_BASE=$(sh -c "$FD_LIST") +expect_eq "no extra fds in the program (unix)" "$FD_BASE" "$(run -- sh -c "$FD_LIST")" +expect_eq "no extra fds in the program (spawn)" "$FD_BASE" "$("$TIMEOUT" 60 "$PLAYER" --spawn "$INTROSPECT --stdio" -- sh -c "$FD_LIST")" +expect_eq "--fd transport does not leak into the program" "0 1 2" "$(python3 - "$PLAYER" "$SOCK" <<'EOF' +import socket, subprocess, sys, os +s = socket.socket(socket.AF_UNIX, socket.SOCK_STREAM); s.connect(sys.argv[2]) +r = subprocess.run([sys.argv[1], "--fd", str(s.fileno()), "--", "sh", "-c", + 'ls /proc/self/fd | grep -v "^3$" | sort -n | tr "\n" " " | sed "s/ $//"'], + pass_fds=[s.fileno()], capture_output=True, text=True) +print(r.stdout.strip()) +EOF +)" + +echo "# signals" +expect_eq "SIGTERM forwarded" "143" "$(run -- sh -c 'kill -TERM $PPID; sleep 5; echo alive' >/dev/null 2>&1; echo $?)" +expect_eq "SIGHUP forwarded" "129" "$(run -- sh -c 'kill -HUP $PPID; sleep 5; echo alive' >/dev/null 2>&1; echo $?)" +expect_eq "SIGINT to 9player is ignored while the child lives" "still-here" "$(run -- sh -c 'kill -INT $PPID; sleep 0.3; echo still-here')" +# Ctrl-C from the tty must not kill the --spawn server (same process group). +expect_eq "Ctrl-C on the tty leaves the --spawn server alive" "ok" "$(timeout 30 python3 - "$PLAYER" "$INTROSPECT" <<'EOF' +import os, pty, sys, time, select +P, I = sys.argv[1], sys.argv[2] +prog = ["python3", "-c", """ +import os, signal, sys, time +signal.signal(signal.SIGINT, lambda *a: None) +m = os.environ['NINEPLAYER_MOUNT'] +open(m + '/build/optimize').read() +sys.stdin.readline() +try: + open(m + '/build/optimize').read(); print('ok', flush=True) +except Exception as e: + print('mount dead:', e, flush=True) +"""] +pid, fd = pty.fork() +if pid == 0: + os.execv(P, [P, "--spawn", I + " --stdio", "--"] + prog) +out = b"" +def rd(t): + global out + end = time.time() + t + while time.time() < end: + r, _, _ = select.select([fd], [], [], 0.1) + if r: + try: d = os.read(fd, 4096) + except OSError: return + if not d: return + out += d +rd(1.5); os.write(fd, b"\x03"); rd(0.7); os.write(fd, b"\n"); rd(3) +os.waitpid(pid, 0) +print(out.decode(errors="replace").replace("^C", "").strip().splitlines()[-1] if out.strip() else "no output") +EOF +)" +# The --spawn server dying mid-session is reaped (no zombie) and does not end the session. +OUT=$("$TIMEOUT" 60 "$PLAYER" --spawn "$INTROSPECT --stdio" -- sh -c 'srv=$(cat $NINEPLAYER_MOUNT/runtime/pid); kill -TERM $srv; sleep 0.5; st=$(ps -o stat= -p $srv 2>/dev/null); echo "${st:-gone}"; exit 5' 2>/dev/null); RC=$? +expect_eq "server death mid-session: exit status still the child's, server reaped (no zombie)" "5 gone" "$RC $OUT" +# A server that never answers: once the child is dead, SIGTERM must end 9player. +cat >"$TMP/hang.py" <<'EOF' +import struct, os, sys, time +def rd(n): + b = b"" + while len(b) < n: + c = os.read(0, n - len(b)) + if not c: sys.exit(0) + b += c + return b +while True: + size, = struct.unpack("/dev/null & +HP=$! +sleep 1; kill -TERM $HP +START=$(date +%s) +for _ in $(seq 1 100); do kill -0 $HP 2>/dev/null || break; sleep 0.1; done +if kill -0 $HP 2>/dev/null; then kill -KILL $HP; RC=hung; else wait $HP; RC=$?; fi +expect_eq "hung server: one SIGTERM ends 9player once the child is dead (watchdog)" "143" "$RC" +expect_eq "hung server: exit was prompt" "yes" "$([ $(( $(date +%s) - START )) -lt 8 ] && echo yes)" +pkill -f "$TMP/hang.py" 2>/dev/null + +echo "# child/parent protocol" +if command -v strace >/dev/null 2>&1 && strace -qq -e trace=none true 2>/dev/null; then + expect_eq "child killed before handoff" "125" "$(strace -f -qq -e trace=unshare -e inject=unshare:signal=KILL -o /dev/null timeout 20 "$PLAYER" --unix "$SOCK" -- true 2>/dev/null; echo $?)" + expect_contains "child killed before handoff: message" "child exited before reporting" "$(strace -f -qq -e trace=unshare -e inject=unshare:signal=KILL -o /dev/null timeout 20 "$PLAYER" --unix "$SOCK" -- true 2>&1)" + expect_eq "status handoff fails" "125" "$(strace -f -qq -e trace=sendmsg -e inject=sendmsg:error=EPIPE -o /dev/null timeout 20 "$PLAYER" --unix "$SOCK" -- true 2>/dev/null; echo $?)" + # recvmsg skipped (returns 1 without the fd): the child must be killed, not exec'd onto a dead mount. + OUT=$(strace -f -qq -e trace=recvmsg -e inject=recvmsg:retval=1:when=1 -o /dev/null timeout 20 "$PLAYER" --unix "$SOCK" -- sh -c 'echo child-ran' 2>&1; echo "rc=$?") + expect_contains "truncated fd handoff: child not exec'd" "rc=125" "$OUT" + expect_eq "truncated fd handoff: program never ran" "no" "$(case "$OUT" in *child-ran*) echo yes;; *) echo no;; esac)" + expect_contains "fuse mount failure is reported" "mount fuse: EPERM" "$(strace -f -qq -e trace=mount -e inject=mount:error=EPERM:when=2 -o /dev/null timeout 20 "$PLAYER" --unix "$SOCK" --mount "$TMP/mp" -- true 2>&1)" +else + echo "skip - strace unavailable (child failure injection)" +fi +expect_contains "fork failure is reported" "fork: E" "$(python3 -c " +import resource, os +resource.setrlimit(resource.RLIMIT_NPROC, (1, 1)) +os.execv('$PLAYER', ['$PLAYER', '--unix', '$SOCK', '--', 'true'])" 2>&1)" + +echo "# mountpoint policy" +ln -s /nonexistent "$TMP/dangling" +expect_contains "dangling symlink mountpoint" "dangling symlink" "$(run --mount "$TMP/dangling" -- true 2>&1)" +expect_eq "refuse to shadow / via /proc/self/root" "125" "$(run --mount /proc/self/root/x9p -- true 2>/dev/null; echo $?)" +expect_contains "refuse to shadow / via /proc/self/root: message" "refusing to shadow /" "$(run --mount /proc/self/root/x9p -- true 2>&1)" +expect_eq "refuse to shadow under /proc" "125" "$(run --mount /proc/self/fd/x9p -- true 2>/dev/null; echo $?)" +if [ "$(ls -A /usr/lib | wc -l)" -gt 4096 ]; then + expect_contains "parent with >4096 entries refused" "more than 4096 entries" "$(run --mount /usr/lib/x9p -- true 2>&1)" +else + echo "skip - no root-owned directory with >4096 entries" +fi +expect_eq "shadowed /run keeps its entries" "$(ls -A /run | sort | tr '\n' ' ')" "$(run --mount /run/x9p -- sh -c 'ls -A /run | grep -v "^x9p$" | sort | tr "\n" " "')" +expect_eq "mountpoint with spaces" "ok" "$(mkdir -p "$TMP/with space" && run --mount "$TMP/with space" -- sh -c '[ -f "$NINEPLAYER_MOUNT/README" ] && echo ok')" +expect_eq "mountpoint is a file" "125" "$(run --mount "$TMP/nx" -- true 2>/dev/null; echo $?)" + +echo "# leaks" +for i in $(seq 1 30); do run -- sh -c 'cat $NINEPLAYER_MOUNT/build/optimize >/dev/null' 2>/dev/null; done +BG=(); for i in $(seq 1 10); do ( run -- sh -c 'cat $NINEPLAYER_MOUNT/build/optimize >/dev/null' 2>/dev/null ) & BG+=($!); done; wait "${BG[@]}" # not a bare wait: that would also wait for the server +sleep 0.3 +expect_eq "no stray 9player processes" "" "$(pgrep -f "^$PLAYER " | tr '\n' ' ')" +expect_eq "no stray --stdio servers" "" "$(pgrep -f "$INTROSPECT --stdio" | tr '\n' ' ')" +expect_eq "host mount table untouched" "same" "$([ "$MI_BEFORE" = "$(grep -v " $TMP" /proc/self/mountinfo | sort)" ] && echo same || echo changed)" + +echo +echo "passed=$PASSED failed=$FAILED" +[ "$FAILED" -eq 0 ] diff --git a/9player/test/adversarial.sh b/9player/test/adversarial.sh new file mode 100755 index 0000000..857af14 --- /dev/null +++ b/9player/test/adversarial.sh @@ -0,0 +1,15 @@ +#!/usr/bin/env bash +# Runs every 9player adversarial suite in sequence (hostile servers, FUSE +# semantics, process/namespace edge cases, stress). The suites aimed at the +# introspect server itself live in introspect/test (zig build introspect-adv). +# Usage: bash 9player/test/adversarial.sh <9player> (zig build 9player-adv) +set -u +PLAYER=${1:?path to 9player} +INTROSPECT=${2:?path to introspect} +HERE=$(cd "$(dirname "$0")" && pwd) +status=0 +for suite in adv_ns_process adv_bridge_hostile adv_bridge_semantics adv_bridge_stress; do + echo "### $suite" + if bash "$HERE/$suite.sh" "$PLAYER" "$INTROSPECT"; then echo "### $suite: ok"; else echo "### $suite: FAILED"; status=1; fi +done +exit $status diff --git a/9player/test/integration.sh b/9player/test/integration.sh new file mode 100755 index 0000000..e12cb5f --- /dev/null +++ b/9player/test/integration.sh @@ -0,0 +1,160 @@ +#!/usr/bin/env bash +# Integration tests for 9player: real user+mount namespaces, real FUSE, real 9P servers. +# Usage: bash 9player/test/integration.sh <9player> (zig build 9player-itest) +# Exit 0 on success (or when the machine cannot run the tests), 1 on failure. +set -u + +PLAYER=$(realpath "${1:?path to 9player}") +INTROSPECT=$(realpath "${2:?path to introspect}") +TMP=$(mktemp -d "${TMPDIR:-/tmp}/9player-itest.XXXXXX") +PIDS=() +FAILED=0 +PASSED=0 +M=/mnt/9p + +cleanup() { + for p in "${PIDS[@]:-}"; do [ -n "$p" ] && kill "$p" 2>/dev/null; done + rm -rf "$TMP" +} +trap cleanup EXIT + +if ! unshare -Urm true 2>/dev/null; then + echo "SKIP: unprivileged user namespaces unavailable"; exit 0 +fi +if [ ! -c /dev/fuse ]; then + echo "SKIP: /dev/fuse missing"; exit 0 +fi + +pass() { PASSED=$((PASSED + 1)); echo "ok - $1"; } +fail() { FAILED=$((FAILED + 1)); echo "FAIL - $1"; shift; [ $# -gt 0 ] && printf ' %s\n' "$@"; } +expect_eq() { # name expected actual + if [ "$2" = "$3" ]; then pass "$1"; else fail "$1" "expected: $(printf %q "$2")" "actual: $(printf %q "$3")"; fi +} +expect_contains() { # name needle haystack + case "$3" in *"$2"*) pass "$1" ;; *) fail "$1" "missing: $(printf %q "$2")" "in: $(printf %q "$3")" ;; esac +} + +wait_socket() { # path + for _ in $(seq 1 100); do [ -S "$1" ] && return 0; sleep 0.05; done + return 1 +} + +# run_in "" — run inside a namespace with the current transport ($TRANSPORT array). +run_in() { timeout 60 "$PLAYER" "${TRANSPORT[@]}" -- sh -c "$1" 2>"$TMP/stderr"; } + +# --- scratch battery: works against any writable 9P tree rooted at $1 (relative to mount) --- +scratch_battery() { # label scratchdir + local label=$1 S=$2 + + expect_eq "$label: create+append+read" $'hello\nworld' "$(run_in "echo hello > $M/$S/a && echo world >> $M/$S/a && cat $M/$S/a")" + expect_eq "$label: overwrite" "x" "$(run_in "echo x > $M/$S/a && cat $M/$S/a")" + expect_eq "$label: stat size after overwrite" "2" "$(run_in "stat -c %s $M/$S/a")" + expect_eq "$label: truncate" "0" "$(run_in "truncate -s 0 $M/$S/a && stat -c %s $M/$S/a")" + expect_eq "$label: truncate extend" "10" "$(run_in "truncate -s 10 $M/$S/a && stat -c %s $M/$S/a")" + expect_eq "$label: mkdir -p nested" "directory" "$(run_in "mkdir -p $M/$S/d1/d2/d3 && stat -c %F $M/$S/d1/d2/d3")" + expect_eq "$label: rename within dir" "moved" "$(run_in "echo moved > $M/$S/d1/d2/f && mv $M/$S/d1/d2/f $M/$S/d1/d2/g && cat $M/$S/d1/d2/g")" + # mv(1) silently falls back to copy+delete on EXDEV, so probe rename(2) directly. + expect_contains "$label: rename across dirs is EXDEV" "EXDEV" "$(run_in "python3 -c 'import os,errno +try: os.rename(\"$M/$S/d1/d2/g\", \"$M/$S/d1/g\") +except OSError as e: print(errno.errorcode[e.errno]) +'")" + expect_eq "$label: rm file" "gone" "$(run_in "rm $M/$S/d1/d2/g && [ ! -e $M/$S/d1/d2/g ] && echo gone")" + expect_eq "$label: rmdir non-empty fails" "1" "$(run_in "rmdir $M/$S/d1 2>/dev/null; echo \$?")" + expect_eq "$label: rmdir chain" "ok" "$(run_in "rmdir $M/$S/d1/d2/d3 $M/$S/d1/d2 $M/$S/d1 && echo ok")" + expect_eq "$label: ENOENT" "1" "$(run_in "cat $M/$S/nope 2>/dev/null; echo \$?")" + expect_eq "$label: ENOENT errno text" "No such file or directory" "$(run_in "cat $M/$S/nope 2>&1 | sed 's/.*: //'")" + + head -c 1048576 /dev/urandom >"$TMP/rand" + local sum; sum=$(sha256sum <"$TMP/rand" | cut -d' ' -f1) + expect_eq "$label: 1 MiB round trip (cp)" "$sum" "$(run_in "cp $TMP/rand $M/$S/big && sha256sum < $M/$S/big | cut -d' ' -f1")" + expect_eq "$label: 1 MiB size" "1048576" "$(run_in "stat -c %s $M/$S/big")" + expect_eq "$label: odd block sizes (dd bs=1000)" "$sum" "$(run_in "dd if=$M/$S/big of=$M/$S/big2 bs=1000 status=none && sha256sum < $M/$S/big2 | cut -d' ' -f1")" + expect_eq "$label: partial read at offset" "$(tail -c 12345 "$TMP/rand" | sha256sum | cut -d' ' -f1)" "$(run_in "tail -c 12345 $M/$S/big | sha256sum | cut -d' ' -f1")" + expect_eq "$label: many small files" "200" "$(run_in "mkdir $M/$S/many && for i in \$(seq 1 200); do echo \$i > $M/$S/many/f\$i; done; ls $M/$S/many | wc -l")" + expect_eq "$label: find count" "201" "$(run_in "find $M/$S/many | wc -l")" + expect_eq "$label: readdir contents" "f1 f100 f200" "$(run_in "cd $M/$S/many && ls f1 f100 f200 | tr '\n' ' ' | sed 's/ \$//'")" + expect_eq "$label: cleanup many" "0" "$(run_in "rm -r $M/$S/many $M/$S/big $M/$S/big2 $M/$S/a; ls $M/$S | wc -l")" +} + +# ============================================================================ +echo "# introspect over a Unix socket" +SOCK=$TMP/introspect.sock +"$INTROSPECT" --unix "$SOCK" & +PIDS+=($!) +wait_socket "$SOCK" || { echo "introspect did not create $SOCK"; exit 1; } +TRANSPORT=(--unix "$SOCK") + +expect_eq "zig_version" "$(zig version)" "$(run_in "cat $M/build/zig_version")" +expect_eq "mount exported" "$M" "$(run_in 'echo $NINEPLAYER_MOUNT')" +expect_contains "root listing" "build" "$(run_in "ls $M")" +expect_contains "root listing has scratch" "scratch" "$(run_in "ls $M")" +expect_eq "README size > 0" "yes" "$(run_in "[ \$(stat -c %s $M/README) -gt 0 ] && echo yes")" +expect_eq "README readable" "yes" "$(run_in "[ -s $M/README ] && head -c 1 $M/README >/dev/null && echo yes")" +expect_eq "fn/now numeric" "num" "$(run_in "cat $M/runtime/fn/now | grep -Eq '^[0-9]+\$' && echo num")" +expect_eq "fn listing from comptime" "yes" "$(run_in "ls $M/runtime/fn | grep -q hostname && echo yes")" +expect_eq "ctl round trip" "5" "$(run_in "echo 'add 2 3' > $M/runtime/ctl && cat $M/runtime/ctl")" +expect_eq "ctl echo" "hi there" "$(run_in "echo 'echo hi there' > $M/runtime/ctl && cat $M/runtime/ctl")" +expect_contains "comptime types" "Qid" "$(run_in "ls $M/comptime/types")" +expect_eq "comptime size of Qid" "16" "$(run_in "cat $M/comptime/types/Qid/size")" +expect_eq "runtime pid is server pid" "${PIDS[-1]}" "$(run_in "cat $M/runtime/pid")" +expect_eq "exit status propagates" "7" "$(run_in 'exit 7'; echo $?)" +expect_eq "mount is fuse" "yes" "$(run_in "grep -q \"^9player $M fuse\" /proc/mounts && echo yes")" +expect_eq "host /mnt entries still visible" "$(ls -A /mnt | sort | tr '\n' ' ')" "$(run_in "ls -A /mnt | grep -v '^9p\$' | sort | tr '\n' ' '")" +expect_eq "host mount table untouched" "no" "$(grep -q " $M " /proc/self/mountinfo && echo yes || echo no)" +scratch_battery "introspect" scratch + +echo "# nested 9player" +expect_eq "nested mount" "$(zig version)" "$(run_in "$PLAYER --unix $SOCK --mount $TMP/inner -- sh -c 'cat \$NINEPLAYER_MOUNT/build/zig_version'")" + +echo "# --mount variants" +mkdir -p "$TMP/mnt" +expect_eq "--mount existing dir" "ok" "$(timeout 60 "$PLAYER" --unix "$SOCK" --mount "$TMP/mnt" -- sh -c "[ -f $TMP/mnt/README ] && echo ok")" +expect_eq "--mount relative" "ok" "$(cd "$TMP" && timeout 60 "$PLAYER" --unix "$SOCK" --mount rel -- sh -c "[ -f $TMP/rel/README ] && echo ok")" +expect_eq "--mount missing under /" "125" "$(timeout 60 "$PLAYER" --unix "$SOCK" --mount /nonexistent-9player-dir -- true 2>/dev/null; echo $?)" + +echo "# lifecycle" +START=$(date +%s) +expect_eq "background grandchild does not block exit" "3" "$(run_in 'sleep 30 >/dev/null 2>&1 & exit 3'; echo $?)" +expect_eq "exit was prompt" "yes" "$([ $(( $(date +%s) - START )) -lt 10 ] && echo yes)" +expect_eq "SIGTERM forwarded" "143" "$(timeout 60 "$PLAYER" --unix "$SOCK" -- sh -c 'kill -TERM $PPID; sleep 5; echo alive' >/dev/null 2>&1; echo $?)" + +echo "# --spawn transport" +TRANSPORT=(--spawn "$INTROSPECT --stdio") +expect_eq "spawn: zig_version" "$(zig version)" "$(run_in "cat $M/build/zig_version")" +expect_eq "spawn: ctl" "7" "$(run_in "echo 'add 3 4' > $M/runtime/ctl && cat $M/runtime/ctl")" +expect_eq "spawn: stateful sequence in one session" 'hello world 12 moved 0' "$(run_in "cd $M/scratch && echo hello > a && echo world >> a && cat a && stat -c %s a && mkdir d && echo moved > d/f && mv d/f d/g && cat d/g && rm d/g && rmdir d && rm a && ls | wc -l" | tr '\n' ' ' | sed 's/ $//')" + +echo "# --tcp transport" +PORT=$(( 20000 + RANDOM % 20000 )) +"$INTROSPECT" --tcp "127.0.0.1:$PORT" & +PIDS+=($!) +sleep 0.3 +TRANSPORT=(--tcp "127.0.0.1:$PORT") +expect_eq "tcp: zig_version" "$(zig version)" "$(run_in "cat $M/build/zig_version")" +expect_eq "tcp: scratch" "tcp" "$(run_in "echo tcp > $M/scratch/t && cat $M/scratch/t && rm $M/scratch/t")" + +echo "# server death" +"$INTROSPECT" --unix "$TMP/dying.sock" & +DYING=$! +wait_socket "$TMP/dying.sock" +TRANSPORT=(--unix "$TMP/dying.sock") +OUT=$(run_in "cat $M/build/optimize >/dev/null && kill $DYING && sleep 0.3; cat $M/build/optimize 2>&1 >/dev/null | sed 's/.*: //'; echo status=\$?") +expect_contains "server death yields an error, not a hang" "status=0" "$OUT" +expect_eq "server death errno text" "yes" "$(case "$OUT" in *"Input/output error"*|*"Transport endpoint is not connected"*) echo yes;; *) echo "no: $OUT";; esac)" + +echo "# plan9port ramfs (independent 9P2000 implementation)" +if [ -x /usr/lib/plan9/bin/ramfs ]; then + mkdir -p "$TMP/p9ns" + NAMESPACE=$TMP/p9ns /usr/lib/plan9/bin/ramfs -s ramfs + wait_socket "$TMP/p9ns/ramfs" || echo "ramfs socket missing" + PIDS+=($(pgrep -f "9pserve -u unix!$TMP/p9ns/ramfs")) + TRANSPORT=(--unix "$TMP/p9ns/ramfs") + expect_eq "ramfs: mkdir scratch" "ok" "$(run_in "mkdir $M/scratch && echo ok")" + scratch_battery "ramfs" scratch +else + echo "skip - plan9port ramfs not installed" +fi + +echo +echo "passed=$PASSED failed=$FAILED" +[ "$FAILED" -eq 0 ] diff --git a/README.md b/README.md index 57de62e..d1e52f9 100644 --- a/README.md +++ b/README.md @@ -77,5 +77,34 @@ answer Tversion with `negotiate()`. Answer other requests with `reply()`. The backend implements its filesystem and fid lifecycle; cloud9 checks reply types, tags, counts, negotiated frame sizes, and flush completion. +## Programs + +Programs built on cloud9 live in sibling directories of `src/`, each with its +own sources, tests, README and a `build.zig` fragment that the root +`build.zig` imports and enables with a toggle (`zig build --help` lists the +steps). New related programs follow the same layout. + +* [`introspect/`](introspect/README.md) — a 9P debug/introspection server as + a library (module `introspect`, exported next to `cloud9`; freestanding + core, Linux debug layer) and its demo binary `introspect`. + `-Dintrospect=[bool]`; steps `introspect`, `introspect-test`, + `introspect-check-freestanding`, `introspect-debug-test`, + `introspect-debug-itest`, `introspect-adv`. +* [`9player/`](9player/README.md) — mount a 9P tree into a fresh user+mount + namespace via FUSE and run a program in it, without root (Linux, no libc). + `-D9player=[bool]`; steps `9player`, `9player-test`, `9player-itest`, + `9player-adv`. + +Both toggles default to on for Linux targets and off elsewhere; enabled +programs are installed by the plain `zig build`, and `programs-test` / +`programs-itest` run every enabled program's unit and integration steps. A +dependent package gets both modules from the one dependency: + +```zig +const cloud9_dep = b.dependency("cloud9", .{ .target = target, .optimize = optimize }); +app.root_module.addImport("cloud9", cloud9_dep.module("cloud9")); +app.root_module.addImport("introspect", cloud9_dep.module("introspect")); +``` + See [design and ownership contracts](docs/design.md), [specification references](docs/spec.md), and [validation](docs/validation.md). diff --git a/build.zig b/build.zig index 3228cf2..63d9735 100644 --- a/build.zig +++ b/build.zig @@ -124,4 +124,61 @@ pub fn build(b: *std.Build) void { const run_fuzz = b.addRunArtifact(fuzz); if (b.args) |args| run_fuzz.addArgs(args); b.step("fuzz", "Run deterministic local decoder and server probes (seed, iterations)").dependOn(&run_fuzz.step); + + // ------------------------------------------------------------------ + // Programs related to cloud9. Each lives in its own directory beside + // src/ (introspect/, 9player/) with a build.zig *fragment* exposing + // `add(b, ctx)`; it is imported here and runs with this builder, so its + // b.path() calls are relative to this root and its steps are namespaced + // by the program's name. New related programs follow the same pattern: + // add a directory, a fragment, a toggle here, and the path to + // build.zig.zon. + // + // -Dintrospect=[bool] library module (any target) + freestanding check; + // demo server and Linux suites when the target is Linux + // -D9player=[bool] the FUSE mount CLI (Linux only) + // + // Both default to true on a Linux target and false elsewhere. Enabled + // programs are installed by the plain `zig build` next to cloud9-http + // and cloud9-probe. + const is_linux = target.result.os.tag == .linux; + const want_introspect = b.option(bool, "introspect", "Build the introspect library module and demo (default: target is Linux)") orelse is_linux; + const want_9player = b.option(bool, "9player", "Build the 9player FUSE mount CLI (default: target is Linux; Linux only)") orelse is_linux; + if (want_9player and !is_linux) { + std.debug.print("error: -D9player=true needs a Linux target (got {s})\n", .{@tagName(target.result.os.tag)}); + std.process.exit(1); + } + const programs_test = b.step("programs-test", "Run every enabled program's unit tests and checks"); + const programs_itest = b.step("programs-itest", "Run every enabled program's integration suites"); + + const introspect_build = @import("introspect/build.zig"); + const nineplayer_build = @import("9player/build.zig"); + + const introspect: ?introspect_build.Artifacts = if (want_introspect) introspect_build.add(b, .{ + .target = target, + .optimize = optimize, + .cloud9 = module, + }) else null; + if (introspect) |i| { + programs_test.dependOn(i.test_step); + programs_test.dependOn(i.check_step); + if (i.debug_test_step) |s| programs_test.dependOn(s); + } + + const nineplayer: ?nineplayer_build.Artifacts = if (want_9player) nineplayer_build.add(b, .{ + .target = target, + .optimize = optimize, + .cloud9 = module, + .introspect_demo = if (introspect) |i| i.demo else null, + }) else null; + if (nineplayer) |p| { + programs_test.dependOn(p.test_step); + programs_itest.dependOn(p.itest_step); + } + + // introspect's end-to-end suites drive the demo through a 9player mount, + // so they are wired once both fragments have run. + if (introspect) |i| { + if (introspect_build.addPlayerTests(b, i, if (nineplayer) |p| p.exe else null)) |s| programs_itest.dependOn(s); + } } diff --git a/build.zig.zon b/build.zig.zon index 1afd382..c170b9d 100644 --- a/build.zig.zon +++ b/build.zig.zon @@ -3,5 +3,5 @@ .fingerprint = 0xc8b2d5eaa3adfca, .version = "0.1.0", .minimum_zig_version = "0.16.0", - .paths = .{ "build.zig", "build.zig.zon", "src", "app", "test", "docs", "README.md" }, + .paths = .{ "build.zig", "build.zig.zon", "src", "app", "test", "docs", "README.md", "9player", "introspect" }, } diff --git a/docs/design.md b/docs/design.md index d8d55c4..d12b92c 100644 --- a/docs/design.md +++ b/docs/design.md @@ -66,6 +66,20 @@ identity verification must provide that policy before using it across a trust boundary. OpenSSL allocations and handshake costs are outside the allocation-free protocol core. Accepted connections must close before their shared listener. +# Related programs + +Programs built on the library ship from this repository in sibling +directories of `src/` (`introspect/`, `9player/`), never inside it: each has +its own `src/`, `test/`, `docs/`, README and a `build.zig` fragment +(`pub fn add(b, ctx)`) that the root `build.zig` imports, passes the resolved +target, optimize mode and the `cloud9` module to, and enables with a +`-D` toggle. Fragments register namespaced steps (``, +`-test`, ...), never call `standardTargetOptions`, and use root-relative +`b.path("/...")`. The library keeps its contract: mounting, namespaces, +threads, allocation and process policy stay in the program. `introspect` is +also exported as a module next to `cloud9` for dependents. New related +programs follow this layout. + # Pardes integration Pardes consumes the sibling package through build.zig.zon. Its `src/9p.zig` now diff --git a/introspect/README.md b/introspect/README.md new file mode 100644 index 0000000..ba7120b --- /dev/null +++ b/introspect/README.md @@ -0,0 +1,130 @@ +# introspect + +A 9P2000 debug/introspection server as a Zig 0.16 library, built on +[cloud9](../): a debugger-shaped interface where the protocol is just files. +Anything that can read a filesystem (a shell, an agent, an editor, `9p`, +[9player](../9player)) can inspect a running program: build facts, comptime +type layouts, live values, threads and their stacks, memory, breakpoints, +panics. + +The core (`core`, `vars`) is freestanding: no allocator, no OS, no threads, +caller-owned static `Storage`, fixed-capacity tables sized at comptime. It +compiles for `riscv32-freestanding-none`. `scratch` (an in-memory read/write +tree) takes an allocator; `linux` is the platform layer (listeners, a poll +loop on one background thread, threads/stacks/registers, memory, breakpoints +and panics via `std.debug`). [docs/LIBRARY.md](docs/LIBRARY.md) has the full +contract. + +``` +introspect/ + build.zig fragment imported by cloud9's root build.zig (steps below) + src/root.zig pub const core, vars, scratch, linux; Config, Server(cfg), Provider + src/core.zig Tree/Server engine on cloud9.Server: fids, walks, dir reads, providers + src/vars.zig comptime value renderers (@typeInfo) for /vars + src/scratch.zig in-memory read/write tree provider (takes an Allocator) + src/freestanding_check.zig root for the riscv32-freestanding-none compile check + src/linux/probe.zig background thread + poll loop + unix/tcp/fd listeners + src/linux/debug.zig threads, stacks, registers, addr→source, memory, breakpoints, panic + src/linux/provider.zig the debug provider (/threads, /addr, /mem, /hex, /breakpoints, /panic) + src/linux/runtime.zig /runtime generators (pid, uptime, argv, cwd, env, clients) + demo/main.zig the `introspect` binary (below) + test/ debug.sh, adv_introspect_hostile, adv_core_hostile, adv_linux_probe, adversarial.sh + docs/LIBRARY.md design rules and the module contracts +``` + +## Using the library + +cloud9's `build.zig` exports two modules: `cloud9` (the protocol) and +`introspect` (this library, which imports `cloud9` itself). A package that +depends on cloud9 takes both from the one dependency: + +```zig +const cloud9_dep = b.dependency("cloud9", .{ .target = target, .optimize = optimize }); +exe.root_module.addImport("cloud9", cloud9_dep.module("cloud9")); +exe.root_module.addImport("introspect", cloud9_dep.module("introspect")); +``` + +Embedding the core is three static objects and a push/step/output loop, the +same shape as cloud9's `Server`: + +```zig +const introspect = @import("introspect"); + +const State = struct { ticks: u32, phase: enum { idle, busy } }; +const cfg: introspect.Config = .{ .name = "fw", .types = &.{State}, .msize = 2048, .max_fids = 16 }; +const S = introspect.Server(cfg); + +var state: State = .{ .ticks = 0, .phase = .idle }; +var storage: S.Storage = undefined; // per connection: in/out frames + snapshot slots +var shared: S.Shared = undefined; // once: providers and exposed variables + +pub fn main() void { + shared = .init(&state); + shared.expose("state", &state) catch unreachable; // /vars/state/{value,type,size,addr,raw,f/...} + var conn: S.Conn = .init(&shared, &storage, cfg.msize); + // transport loop: conn.push(bytes) ... while (try conn.step()) {} ... send conn.output(), conn.wrote(n) +} +``` + +On Linux, `introspect.linux.Probe` runs that loop for you on one background +thread over a Unix, TCP or inherited listener, and adds the debug provider; +`pub const panic = std.debug.FullPanic(introspect.linux.debug.panicHook);` +in the root module publishes panics under `/panic`. `demo/main.zig` shows +every piece together. + +## The demo (`zig build introspect`, binary `introspect`) + +A single-binary 9P2000 server whose file tree is the binary itself: build-time +facts, `comptime` reflection, live runtime state, a worker thread whose state +is exposed under `/vars`, and the Linux debug layer. + +``` +/README +/build/{zig_version,target,optimize,time,change} captured by introspect/build.zig (jj change id, UTC time) +/comptime/types//{name,size,align,fields} @sizeOf/@alignOf/@typeInfo, generated at comptime +/comptime/decls pub declarations of the server module +/runtime/{pid,ppid,uptime,argv,cwd,env,clients} +/runtime/fn/ reading calls a Zig function (hostname, now, random, uname, fib30) +/runtime/ctl write "fib N" | "add A B" | "echo TEXT" | "sleep-ms N" | "trap" | "panic" +/scratch/ in-memory read/write tree +/vars/state/... the worker's State (readable and writable leaves) +/threads//{name,stat,stack,regs} /addr/ /mem/{maps,} /hex/ +/breakpoints//{stack,regs,ctl} /panic/{message,stack,ctl} +``` + +`/runtime/env` exposes the server's whole environment, so serve it on a Unix +socket or loopback only. + +```sh +zig-out/bin/introspect --unix /tmp/intro.sock & # or --tcp IP:PORT, --stdio, --no-hold +zig-out/bin/9player --unix /tmp/intro.sock -- sh -c ' + cat $NINEPLAYER_MOUNT/build/zig_version; echo + cat $NINEPLAYER_MOUNT/comptime/types/Qid/fields + echo "fib 20" > $NINEPLAYER_MOUNT/runtime/ctl; cat $NINEPLAYER_MOUNT/runtime/ctl + cat $NINEPLAYER_MOUNT/threads/*/stack' +``` + +## Building and testing + +introspect lives in the cloud9 repository as `cloud9/introspect/` and is +wired into cloud9's `build.zig` through the fragment `introspect/build.zig`. +Everything is run from the cloud9 root: + +```sh +zig build # installs zig-out/bin/introspect with the other binaries +zig build introspect # build and install only the demo +zig build introspect-test # library unit tests (core, vars, scratch, linux) and the demo's +zig build introspect-check-freestanding # compile the core for riscv32-freestanding-none +zig build introspect-debug-test # src/linux/debug.zig unit tests +zig build introspect-debug-itest # test/debug.sh: threads, stacks, breakpoints, panic through 9player +zig build introspect-adv # hostile raw-9P clients against the demo and the core, + # the Linux layer through a 9player mount (several minutes) +zig build -Dintrospect=false # leave introspect out +zig build introspect-check-freestanding -Dtarget=riscv32-freestanding-none -Dintrospect=true +``` + +`-Dintrospect` (default: on for Linux targets) enables the module and the +freestanding check on any target; the demo and the Linux suites are added +only when the target OS is Linux. The end-to-end suites also need 9player +(`-D9player=true`, the Linux default), unprivileged user namespaces, +`/dev/fuse` and Python 3, and skip themselves otherwise. diff --git a/introspect/build.zig b/introspect/build.zig new file mode 100644 index 0000000..6662151 --- /dev/null +++ b/introspect/build.zig @@ -0,0 +1,169 @@ +//! Build fragment for introspect: the 9P debug/introspection library (module +//! `introspect`), its freestanding check, the `introspect` demo server and the +//! test suites under introspect/test. It is `@import`ed by the root build.zig +//! and called with the root builder, so every `b.path(...)` here is relative +//! to the cloud9 root (hence the `introspect/` prefix), every option is +//! defined by the root (no `standardTargetOptions` here) and every step it +//! registers lands in the root's step list under the `introspect` prefix. +//! +//! Steps: introspect, introspect-test, introspect-check-freestanding, +//! introspect-debug-test, introspect-debug-itest, introspect-adv. +const std = @import("std"); + +/// What the root passes in. The root owns target/optimize resolution and the +/// cloud9 module; this fragment derives everything else from them. +pub const Context = struct { + target: std.Build.ResolvedTarget, + optimize: std.builtin.OptimizeMode, + /// The cloud9 library module for `target`. Its `root_source_file` is also + /// used to instantiate cloud9 for the freestanding check target. + cloud9: *std.Build.Module, +}; + +pub const Artifacts = struct { + /// The `introspect` module, exported with `b.addModule` so dependents of + /// cloud9 can `.module("introspect")`. Built for any target; the Linux + /// layer is compiled in only when the target OS is Linux. + module: *std.Build.Module, + /// The demo 9P2000 server (binary `introspect`); null when the target is + /// not Linux. 9player's integration tests use it as their server. + demo: ?*std.Build.Step.Compile, + /// `introspect-test`: library (and demo) unit tests. + test_step: *std.Build.Step, + /// `introspect-check-freestanding`: the core compiled for riscv32-freestanding-none. + check_step: *std.Build.Step, + /// `introspect-debug-test`: linux/debug.zig unit tests; null when not Linux. + debug_test_step: ?*std.Build.Step, + /// `introspect-adv`: the hostile-client suites are attached by `add`, the + /// Linux-layer suite (which needs a 9player mount) by `addPlayerTests`. + adv_step: *std.Build.Step, +}; + +pub fn add(b: *std.Build, ctx: Context) Artifacts { + const target = ctx.target; + const optimize = ctx.optimize; + const is_linux = target.result.os.tag == .linux; + + // Build-time facts embedded into the demo (/build/*): jj change id, UTC + // time, optimize mode, target triple. `jj` is pointed at the build root + // (the cloud9 checkout) explicitly, so the result does not depend on the + // directory `zig build` was invoked from. + const build_options = b.addOptions(); + var code: u8 = 0; + const repo = b.build_root.path orelse "."; + const change_id = b.runAllowFail(&.{ "jj", "-R", repo, "log", "--no-graph", "-r", "@", "-T", "change_id.short()", "--ignore-working-copy" }, &code, .ignore) catch "unknown"; + build_options.addOption([]const u8, "change_id", std.mem.trim(u8, change_id, " \n\r\t")); + const stamp = b.runAllowFail(&.{ "date", "-u", "+%Y-%m-%dT%H:%M:%SZ" }, &code, .ignore) catch "unknown"; + build_options.addOption([]const u8, "build_time", std.mem.trim(u8, stamp, " \n\r\t")); + build_options.addOption([]const u8, "optimize", @tagName(optimize)); + build_options.addOption([]const u8, "target", target.result.zigTriple(b.allocator) catch "unknown"); + + // The library: freestanding core + vars, scratch (allocator), Linux layer. + // Exported under the name `introspect` for packages that depend on cloud9. + const lib_mod = b.addModule("introspect", .{ + .root_source_file = b.path("introspect/src/root.zig"), + .target = target, + .optimize = optimize, + .imports = &.{.{ .name = "cloud9", .module = ctx.cloud9 }}, + }); + + const test_step = b.step("introspect-test", "Run the introspect library's unit tests (and the demo's)"); + test_step.dependOn(&b.addRunArtifact(b.addTest(.{ .root_module = lib_mod })).step); + + // The core must compile without an OS (rule 1 of introspect/docs/LIBRARY.md). + // cloud9 is re-instantiated for that target from the same root source. + const fs_target = b.resolveTargetQuery(.{ .cpu_arch = .riscv32, .os_tag = .freestanding, .abi = .none }); + const fs_cloud9 = b.createModule(.{ + .root_source_file = ctx.cloud9.root_source_file.?, + .target = fs_target, + .optimize = optimize, + }); + const fs_check = b.addObject(.{ + .name = "introspect-freestanding", + .root_module = b.createModule(.{ + .root_source_file = b.path("introspect/src/freestanding_check.zig"), + .target = fs_target, + .optimize = optimize, + .imports = &.{.{ .name = "cloud9", .module = fs_cloud9 }}, + }), + }); + const check_step = b.step("introspect-check-freestanding", "Compile the introspect core for riscv32-freestanding-none"); + check_step.dependOn(&fs_check.step); + + // `introspect-adv` exists on every target so the step list is stable; its + // suites are attached below (Linux only) and by addPlayerTests. + const adv_step = b.step("introspect-adv", "Run introspect's adversarial suites (hostile clients, Linux layer; several minutes)"); + + if (!is_linux) { + adv_step.dependOn(&b.addFail("introspect-adv needs a Linux target (the demo server is Linux-only)").step); + return .{ .module = lib_mod, .demo = null, .test_step = test_step, .check_step = check_step, .debug_test_step = null, .adv_step = adv_step }; + } + + // The demo: a 9P2000 server whose tree is the binary itself (build-time, + // comptime and runtime facts, a worker thread, the Linux debug layer). + const demo_mod = b.createModule(.{ + .root_source_file = b.path("introspect/demo/main.zig"), + .target = target, + .optimize = optimize, + .imports = &.{ + .{ .name = "cloud9", .module = ctx.cloud9 }, + .{ .name = "build_options", .module = build_options.createModule() }, + .{ .name = "introspect", .module = lib_mod }, + }, + }); + const demo = b.addExecutable(.{ .name = "introspect", .root_module = demo_mod }); + const install_demo = b.addInstallArtifact(demo, .{}); + b.getInstallStep().dependOn(&install_demo.step); + b.step("introspect", "Build and install only the introspect demo server").dependOn(&install_demo.step); + test_step.dependOn(&b.addRunArtifact(b.addTest(.{ .root_module = demo_mod })).step); + + // Linux debug facilities (threads, stacks, breakpoints, panic): self-contained unit tests. + const debug_mod = b.createModule(.{ + .root_source_file = b.path("introspect/src/linux/debug.zig"), + .target = target, + .optimize = optimize, + }); + const debug_test_step = b.step("introspect-debug-test", "Run the introspect/src/linux/debug.zig unit tests"); + debug_test_step.dependOn(&b.addRunArtifact(b.addTest(.{ .root_module = debug_mod })).step); + + // Hostile raw-9P clients against the demo (framing, tags, floods; the core's + // /vars tree, snapshots, fid table). `--fast` as in the umbrella script. + inline for (.{ "adv_introspect_hostile", "adv_core_hostile" }) |suite| { + const run = b.addSystemCommand(&.{"bash"}); + run.addFileArg(b.path("introspect/test/" ++ suite ++ ".sh")); + run.addArtifactArg(demo); + run.addArg("--fast"); + adv_step.dependOn(&run.step); + } + + return .{ .module = lib_mod, .demo = demo, .test_step = test_step, .check_step = check_step, .debug_test_step = debug_test_step, .adv_step = adv_step }; +} + +/// The suites that drive the demo through a 9player mount: test/debug.sh +/// (threads, stacks, breakpoints, panic end to end) and +/// test/adv_linux_probe.sh (memory endpoints, signal machinery, poll loop). +/// Called by the root after the 9player fragment; `player` is null when +/// 9player is disabled, in which case the steps exist but fail with a notice. +/// Returns the `introspect-debug-itest` step (null when the target is not Linux). +pub fn addPlayerTests(b: *std.Build, arts: Artifacts, player: ?*std.Build.Step.Compile) ?*std.Build.Step { + const demo = arts.demo orelse return null; // not Linux: nothing to drive + const debug_itest = b.step("introspect-debug-itest", "Run introspect/test/debug.sh (threads, stacks, breakpoints, panic through 9player)"); + const exe = player orelse { + const fail = b.addFail("introspect-debug-itest and the Linux-layer adversarial suite need 9player (build with -D9player=true)"); + debug_itest.dependOn(&fail.step); + arts.adv_step.dependOn(&fail.step); + return debug_itest; + }; + const dbg = b.addSystemCommand(&.{"bash"}); + dbg.addFileArg(b.path("introspect/test/debug.sh")); + dbg.addArtifactArg(exe); + dbg.addArtifactArg(demo); + debug_itest.dependOn(&dbg.step); + + const adv_linux = b.addSystemCommand(&.{"bash"}); + adv_linux.addFileArg(b.path("introspect/test/adv_linux_probe.sh")); + adv_linux.addArtifactArg(exe); + adv_linux.addArtifactArg(demo); + arts.adv_step.dependOn(&adv_linux.step); + return debug_itest; +} diff --git a/introspect/demo/main.zig b/introspect/demo/main.zig new file mode 100644 index 0000000..b1c2c57 --- /dev/null +++ b/introspect/demo/main.zig @@ -0,0 +1,423 @@ +//! introspect: the demo 9P2000 server, built on the introspect library. +//! +//! /README, /build/*, /comptime/{types,decls}, /runtime/{pid,ppid,uptime,argv,cwd,env,clients}, +//! /runtime/fn/{fib30,hostname,now,random,uname}, /runtime/ctl (echo|fib|sleep-ms|add|trap|panic), +//! /scratch (in-memory tree), /vars/state (the worker's exposed state), +//! /threads, /addr, /mem, /hex, /breakpoints, /panic (the Linux debug layer). +//! +//! A worker thread ("worker") runs `workerLoop`, incrementing `state.ticks` +//! every ~10 ms; `trap` makes it execute `@breakpoint()` on its next tick and +//! `panic` makes it panic from inside `workerLoop`. Panics go through the +//! library's hook, so the message and stack are published under /panic and +//! the process is held there until /panic/ctl says "continue" (`--no-hold` +//! disables the hold). +//! +//! Static memory: every buffer is a global; the only heap user is /scratch +//! (init.gpa, 512 MiB budget). With `max_clients` = 16 and msize = 1 MiB the +//! per-client core Storage is 3 MiB + 8 x 64 KiB snapshots and the Conn's fid +//! table ~1.88 MiB (32768 fids plus their hash index, needed for the 20000-fid +//! adversarial test), so `probe_storage` is ~86 MiB of BSS; untouched pages +//! cost nothing (an idle server has an RSS of ~7 MiB). +const std = @import("std"); +const builtin = @import("builtin"); +const cloud9 = @import("cloud9"); +const build_options = @import("build_options"); +const introspect = @import("introspect"); +const linux = std.os.linux; +const Writer = std.Io.Writer; +const plinux = introspect.linux; +const runtime = plinux.runtime; + +pub const panic = std.debug.FullPanic(plinux.debug.panicHook); + +/// Simultaneous 9P clients; further connections are closed (see probe.zig for +/// the idle-eviction rule). +pub const max_clients = 16; +pub const max_msize: u32 = 1 << 20; + +pub const State = struct { + ticks: u64, + phase: enum { idle, working, trapped }, + last_job: struct { id: u32, cost: f32 }, +}; + +const Build = struct { + pub const zig_version: []const u8 = builtin.zig_version_string; + pub const target: []const u8 = build_options.target; + pub const optimize: []const u8 = build_options.optimize; + pub const time: []const u8 = build_options.build_time; + pub const change: []const u8 = build_options.change_id; +}; + +/// What every generator and the ctl handler see (`Shared.ctx`). +const App = struct { + info: runtime.Info, + probe: *ProbeT, + shared: *S.Shared, +}; + +/// /runtime/fn/: reading the file calls the function. +pub const Fns = struct { + pub fn hostname(_: *anyopaque, w: *Writer) anyerror!void { + var u: linux.utsname = undefined; + if (linux.errno(linux.uname(&u)) != .SUCCESS) return error.Uname; + try w.writeAll(std.mem.sliceTo(&u.nodename, 0)); + } + + pub fn now(_: *anyopaque, w: *Writer) anyerror!void { + try w.print("{d}", .{runtime.realtimeSecs()}); + } + + pub fn random(_: *anyopaque, w: *Writer) anyerror!void { + var b: [8]u8 = undefined; + var got: usize = 0; + while (got < b.len) { + const rc = linux.getrandom(b[got..].ptr, b.len - got, 0); + switch (linux.errno(rc)) { + .SUCCESS => got += rc, + .INTR => continue, + else => return error.Random, + } + } + try w.print("{x:0>16}", .{std.mem.readInt(u64, &b, .little)}); + } + + pub fn uname(_: *anyopaque, w: *Writer) anyerror!void { + var u: linux.utsname = undefined; + if (linux.errno(linux.uname(&u)) != .SUCCESS) return error.Uname; + try w.writeAll(std.mem.sliceTo(&u.release, 0)); + } + + pub fn fib30(_: *anyopaque, w: *Writer) anyerror!void { + try w.print("{d}", .{fib(30)}); + } +}; + +const cfg: introspect.Config = .{ + .name = "introspect", + .build = Build, + .types = &.{ cloud9.Qid, cloud9.Stat, cloud9.Msg, introspect.core.Node, linux.Statx }, + .decls_of = @This(), + .fns = Fns, + .runtime = runtime.Fns(App, "info"), + .ctl = &ctl, + .ctl_dir = .runtime, + .ctl_bytes = 64 * 1024, + .msize = max_msize, + .max_fids = 32768, + .max_providers = 8, + .snapshot_slots = 8, + .snapshot_bytes = 64 * 1024, +}; +const S = introspect.Server(cfg); +const ProbeT = plinux.Probe(S); +const ProbeStorage = ProbeT.Storage(max_clients); + +// -- static state ------------------------------------------------------------- + +pub var state: State = .{ .ticks = 0, .phase = .idle, .last_job = .{ .id = 0, .cost = 0 } }; +var trap_requested: std.atomic.Value(bool) = .init(false); +var panic_requested: std.atomic.Value(bool) = .init(false); + +/// Zero-filled static memory for `T`. (An `= undefined` global is emitted as +/// 0xAA-filled .data in Debug builds, which would make the binary 120 MiB; +/// zeros go to .bss and cost nothing until touched.) +fn Bss(comptime T: type) type { + return struct { + bytes: [@sizeOf(T)]u8 align(@alignOf(T)) = @splat(0), + fn get(b: *@This()) *T { + return @ptrCast(&b.bytes); + } + }; +} +var app_mem: Bss(App) = .{}; +var shared_mem: Bss(S.Shared) = .{}; +var probe_storage_mem: Bss(ProbeStorage) = .{}; +var probe_mem: Bss(ProbeT) = .{}; +var scratch_mem: Bss(introspect.Scratch) = .{}; + +/// Sizes of the static pieces, for the report and `--help`. +pub const static_bytes = @sizeOf(ProbeStorage) + @sizeOf(S.Shared) + @sizeOf(ProbeT); + +// -- the worker --------------------------------------------------------------- + +/// Ticks every ~10 ms; honours `trap` and `panic` requests from /runtime/ctl. +pub noinline fn workerLoop() void { + plinux.setThreadName("worker"); + var job: u32 = 0; + while (true) { + napMs(10); + state.ticks +%= 1; + if (panic_requested.swap(false, .acq_rel)) { + state.phase = .working; + @panic("demo panic requested over 9p"); + } + if (trap_requested.swap(false, .acq_rel)) { + state.phase = .trapped; + @breakpoint(); + state.phase = .idle; + } + if (state.ticks % 100 == 0) { + job +%= 1; + state.phase = .working; + state.last_job = .{ .id = job, .cost = @as(f32, @floatFromInt(job % 7)) * 0.5 }; + state.phase = .idle; + } + } +} + +/// The worker's sleep, issued as a raw syscall from this file so that the +/// thread's innermost frame (the first line of /threads//stack, which +/// introspect/test/debug.sh resolves through /addr) is in demo/main.zig rather than in std. +/// EINTR (a capture signal) just ends the nap early. +inline fn napMs(ms: u64) void { + var req: linux.timespec = .{ .sec = @intCast(ms / 1000), .nsec = @intCast((ms % 1000) * 1_000_000) }; + switch (builtin.cpu.arch) { + .x86_64 => _ = asm volatile ("syscall" + : [ret] "={rax}" (-> usize), + : [number] "{rax}" (@intFromEnum(linux.SYS.nanosleep)), + [arg1] "{rdi}" (@intFromPtr(&req)), + [arg2] "{rsi}" (@as(usize, 0)), + : .{ .rcx = true, .r11 = true, .memory = true }), + .aarch64 => _ = asm volatile ("svc #0" + : [ret] "={x0}" (-> usize), + : [number] "{x8}" (@intFromEnum(linux.SYS.nanosleep)), + [arg1] "{x0}" (@intFromPtr(&req)), + [arg2] "{x1}" (@as(usize, 0)), + : .{ .memory = true }), + else => plinux.sleepMs(ms), + } +} + +// -- /runtime/ctl ------------------------------------------------------------- + +/// The /runtime/ctl handler. The core stages the output and commits it only +/// on success, so a failed command leaves the previous result in place +/// (test/adv_introspect_hostile.py checks that). +fn ctl(ctx: *anyopaque, cmd: []const u8, out: *Writer) anyerror!void { + const a: *App = @ptrCast(@alignCast(ctx)); + const line = std.mem.trim(u8, cmd, " \t\r\n\x00"); + var it = std.mem.tokenizeScalar(u8, line, ' '); + const verb = it.next() orelse return error.BadCommand; + if (std.mem.eql(u8, verb, "echo")) { + try out.writeAll(std.mem.trimStart(u8, line[verb.len..], " \t")); + } else if (std.mem.eql(u8, verb, "fib")) { + const n = std.fmt.parseInt(u32, it.next() orelse return error.BadCommand, 10) catch return error.BadCommand; + if (n > 93) return error.BadCommand; // fib(94) overflows u64 + try out.print("{d}", .{fib(n)}); + } else if (std.mem.eql(u8, verb, "sleep-ms")) { + const n = std.fmt.parseInt(u64, it.next() orelse return error.BadCommand, 10) catch return error.BadCommand; + const ms = @min(n, 10_000); + a.probe.sleepServing(ms); + try out.print("slept {d} ms", .{ms}); + } else if (std.mem.eql(u8, verb, "add")) { + const x = std.fmt.parseInt(i64, it.next() orelse return error.BadCommand, 10) catch return error.BadCommand; + const y = std.fmt.parseInt(i64, it.next() orelse return error.BadCommand, 10) catch return error.BadCommand; + try out.print("{d}", .{x +% y}); + } else if (std.mem.eql(u8, verb, "trap")) { + trap_requested.store(true, .release); + try out.writeAll("trap armed: the worker stops in @breakpoint() on its next tick"); + } else if (std.mem.eql(u8, verb, "panic")) { + panic_requested.store(true, .release); + try out.writeAll("panic armed: the worker panics on its next tick"); + } else return error.BadCommand; +} + +/// fib(n) for n <= 93 (fib(93) is the largest that fits u64). +fn fib(n: u32) u64 { + std.debug.assert(n <= 93); + if (n == 0) return 0; + var a: u64 = 0; + var b: u64 = 1; + for (1..n) |_| { + const c = a + b; + a = b; + b = c; + } + return b; +} + +// -- main --------------------------------------------------------------------- + +const usage_text = + \\usage: introspect [--unix PATH | --tcp IP:PORT | --stdio] [--no-hold] + \\ + \\A demo 9P2000 file server exposing this binary's build-time, comptime and + \\runtime facts, plus a debugger-shaped view of the process (threads, stacks, + \\memory, breakpoints, panics). Default is --stdio (9P on fd 0/1). + \\--no-hold lets a panic abort at once instead of waiting for /panic/ctl. + \\ +; + +pub fn main(init: std.process.Init) !void { + run(init) catch |e| switch (e) { + // Already reported on stderr; no stack trace wanted. + error.Usage, error.Syscall => std.process.exit(1), + else => return e, + }; +} + +fn run(init: std.process.Init) !void { + // Transparent huge pages would back the first touched page of every + // client buffer with 2 MiB. Best effort: ignore failure. + _ = linux.prctl(@intFromEnum(linux.PR.SET_THP_DISABLE), 1, 0, 0, 0); + const arena = init.arena.allocator(); + const args = try init.minimal.args.toSlice(arena); + + const Mode = enum { stdio, unix, tcp }; + var mode: Mode = .stdio; + var address: []const u8 = ""; + var hold = true; + var i: usize = 1; + while (i < args.len) : (i += 1) { + const a = args[i]; + if (std.mem.eql(u8, a, "--stdio")) { + mode = .stdio; + } else if (std.mem.eql(u8, a, "--unix") or std.mem.eql(u8, a, "--tcp")) { + i += 1; + if (i >= args.len) { + std.debug.print("introspect: {s} needs an argument\n{s}", .{ a, usage_text }); + return error.Usage; + } + mode = if (a[2] == 'u') .unix else .tcp; + address = args[i]; + } else if (std.mem.eql(u8, a, "--no-hold")) { + hold = false; + } else if (std.mem.eql(u8, a, "--help") or std.mem.eql(u8, a, "-h")) { + std.debug.print("{s}\nstatic memory: {d} bytes ({d} clients, msize {d})\n", .{ usage_text, static_bytes, max_clients, max_msize }); + return; + } else { + std.debug.print("introspect: unknown argument {s}\n{s}", .{ a, usage_text }); + return error.Usage; + } + } + + // argv, env and cwd are gathered once, into the arena. + var argv_text: std.ArrayList(u8) = .empty; + for (args) |a| { + try argv_text.appendSlice(arena, a); + try argv_text.append(arena, '\n'); + } + var env_text: std.ArrayList(u8) = .empty; + for (init.minimal.environ.block.view().slice) |entry| { + try env_text.appendSlice(arena, std.mem.span(entry)); + try env_text.append(arena, '\n'); + } + var cwd_buf: [4096]u8 = undefined; + const cwd_rc = linux.getcwd(&cwd_buf, cwd_buf.len); + const cwd_text: []const u8 = if (linux.errno(cwd_rc) == .SUCCESS) + try arena.dupe(u8, std.mem.sliceTo(cwd_buf[0..cwd_rc], 0)) + else + ""; + + const app = app_mem.get(); + const shared = shared_mem.get(); + const probe = probe_mem.get(); + const scratch = scratch_mem.get(); + app.* = .{ + .info = .{ .argv = argv_text.items, .env = env_text.items, .cwd = cwd_text, .start_mono = runtime.monotonicSecs() }, + .probe = probe, + .shared = shared, + }; + shared.* = .init(app); + try shared.expose("state", &state); + scratch.* = try introspect.Scratch.init(init.gpa, 512 << 20); + // Freed on the way out so a Debug build's allocator does not report the + // tree as leaked (with a stack trace on stderr) after a clean --stdio EOF. + defer scratch.deinit(); + scratch.max_file = 64 << 20; + scratch.now = &runtime.realtimeSecs; + try shared.addProvider(scratch.provider("scratch")); + + const listen: plinux.Listen = switch (mode) { + .stdio => .{ .client = .{ .in = 0, .out = 1 } }, + .unix => .{ .unix = address }, + .tcp => .{ .tcp = address }, + }; + probe.init(shared, probe_storage_mem.get(), .{ .io = init.io, .listen = listen, .msize = max_msize, .hold_on_panic = hold }) catch |e| { + switch (e) { + error.PathTooLong => std.debug.print("introspect: unix socket path too long\n", .{}), + error.BadAddress => std.debug.print("introspect: --tcp wants an IPv4 literal a.b.c.d:port\n", .{}), + error.Syscall => std.debug.print("introspect: {t} ({t})\n", .{ e, probe.last_errno }), + else => std.debug.print("introspect: {t}\n", .{e}), + } + return if (e == error.PathTooLong or e == error.BadAddress) error.Usage else error.Syscall; + }; + // Failure past this point (thread spawn) still closes the listener and + // restores the signal dispositions. + errdefer probe.stop(); + app.info.clients = probe.clientCounter(); + + const worker = std.Thread.spawn(.{}, workerLoop, .{}) catch |e| { + std.debug.print("introspect: worker thread: {t}\n", .{e}); + return error.Syscall; + }; + worker.detach(); + + switch (mode) { + .unix => std.debug.print("introspect: listening on unix!{s}\n", .{address}), + .tcp => std.debug.print("introspect: listening on tcp!{s}\n", .{address}), + .stdio => {}, + } + probe.start() catch |e| { + std.debug.print("introspect: thread spawn failed: {t}\n", .{e}); + return error.Syscall; + }; + // SIGTERM/SIGINT end the loop cleanly: the socket file is unlinked, the + // signal dispositions restored and the scratch tree freed. + // A disposition of SIG_IGN inherited from the parent is left alone (Unix + // convention): 9player runs a --spawn server with SIGINT ignored so that + // Ctrl-C on the terminal reaches only the program, not its file server. + const term: linux.Sigaction = .{ .handler = .{ .handler = onTerm }, .mask = linux.sigemptyset(), .flags = 0 }; + for ([_]linux.SIG{ .TERM, .INT }) |sig| { + var old: linux.Sigaction = undefined; + _ = linux.sigaction(sig, null, &old); + if (old.handler.handler != linux.SIG.IGN) _ = linux.sigaction(sig, &term, null); + } + probe.wait(); + probe.stop(); +} + +/// Async-signal-safe: an atomic store and one eventfd write. +fn onTerm(_: linux.SIG) callconv(.c) void { + probe_mem.get().requestStop(); +} + +// -- tests -------------------------------------------------------------------- + +test "ctl commands" { + const probe = probe_mem.get(); + const shared = shared_mem.get(); + var a: App = .{ .info = .{}, .probe = probe, .shared = shared }; + probe.thread_tid = .init(0); + probe.serving = .initEmpty(); + probe.nested = 0; + var buf: [128]u8 = undefined; + var w: Writer = .fixed(&buf); + try ctl(&a, "add 2 3\n", &w); + try std.testing.expectEqualStrings("5", w.buffered()); + try std.testing.expectError(error.BadCommand, ctl(&a, "nope", &w)); + w = .fixed(&buf); + try ctl(&a, "fib 93", &w); + try std.testing.expectEqualStrings("12200160415121876738", w.buffered()); + w = .fixed(&buf); + try std.testing.expectError(error.BadCommand, ctl(&a, "fib 94", &w)); + try std.testing.expectError(error.BadCommand, ctl(&a, "frobnicate", &w)); + w = .fixed(&buf); + try ctl(&a, "echo hi there ", &w); + try std.testing.expectEqualStrings("hi there", w.buffered()); + w = .fixed(&buf); + try ctl(&a, "sleep-ms 1", &w); + try std.testing.expectEqualStrings("slept 1 ms", w.buffered()); + w = .fixed(&buf); + try ctl(&a, "trap", &w); + try std.testing.expect(trap_requested.swap(false, .acq_rel)); + w = .fixed(&buf); + try ctl(&a, "panic", &w); + try std.testing.expect(panic_requested.swap(false, .acq_rel)); +} + +test "static footprint is what the file comment says" { + try std.testing.expect(@sizeOf(ProbeStorage) > 16 * (3 << 20)); + try std.testing.expect(@sizeOf(ProbeStorage) < 100 << 20); +} diff --git a/introspect/docs/LIBRARY.md b/introspect/docs/LIBRARY.md new file mode 100644 index 0000000..22840dc --- /dev/null +++ b/introspect/docs/LIBRARY.md @@ -0,0 +1,287 @@ +# introspect: a 9P debug/introspection server as a library + +The demo server that 9player's tests use grows into a library any Zig program +can embed: a debugger-shaped interface where the protocol is just files. +Anything that can read a filesystem (a shell, an agent, an editor, `9p`, +9player) can inspect a running process: build facts, comptime type layouts, +live values, threads and their stacks, memory, breakpoints, panics. + +Design rules (non-negotiable, they mirror cloud9): + +1. **The core is freestanding.** No allocator, no OS, no threads, no `std.Io`. + Caller-owned buffers, fixed-capacity tables sized at comptime. It must + compile for `riscv32-freestanding-none` (the ESP32-P4 firmware target, + `../05-zig-p4`), and `zig build introspect-check-freestanding` proves it. +2. **Every dependency on a runtime is an explicit argument.** Features that + truly need an `Allocator` or an `std.Io` take them in their `init`; nothing + reaches for `std.heap.page_allocator` or a global `Io`. Where memory is + needed it is preferably a caller-provided `[]u8` or a comptime-sized + `Storage` struct the caller places in static memory. +3. **All allocation happens up front**, at init, from what the caller passed. + Steady-state operation does not allocate. +4. **Platform layers are separate modules** (`introspect.linux`) and are the + only places that touch sockets, threads, signals, `/proc` or `std.debug`. + +``` + introspect/src/root.zig pub const core, vars, scratch, linux (linux only), Server(cfg) + introspect/src/core.zig Tree/Server engine on cloud9.Server: fids, walks, dir reads, providers + introspect/src/vars.zig comptime value renderers (@typeInfo) for /vars + introspect/src/scratch.zig in-memory read/write tree provider (takes an Allocator) + introspect/src/linux/probe.zig background thread + poll loop + unix/tcp/fd listeners + introspect/src/linux/debug.zig threads, stacks, registers, addr→source, memory, breakpoints, panic + introspect/demo/main.zig the `introspect` binary: embeds everything, worker thread, exposed vars + +(paths from the cloud9 root; the library is the module `introspect` that +cloud9's `build.zig` exports next to `cloud9`, wired by `introspect/build.zig`) +``` + +## Core (`core.zig`) + +```zig +pub const Config = struct { + name: []const u8 = "introspect", // /README and Stat uid/gid + types: []const type = &.{}, // /comptime/types//... + decls_of: ?type = null, // /comptime/decls lists this type's pub decls + fns: type = struct {}, // /runtime/fn/: pub fn (ctx: *anyopaque, w: *std.Io.Writer) anyerror!void + ctl: ?*const fn (ctx: *anyopaque, cmd: []const u8, out: *std.Io.Writer) anyerror!void = null, // /ctl + max_fids: u16 = 64, + max_providers: u8 = 8, + max_vars: u8 = 32, + /// Dynamic file contents are generated at open time into per-fid snapshot + /// slots so that reads at arbitrary offsets are consistent. + snapshot_slots: u8 = 8, + snapshot_bytes: u32 = 16 * 1024, +}; + +pub fn Server(comptime cfg: Config) type { + return struct { + pub const Storage = struct { // caller places this in static memory + in: [msize]u8, out: [msize]u8, snapshots: [cfg.snapshot_slots][cfg.snapshot_bytes]u8, + }; + pub const Shared = struct { // state common to all connections (providers, vars) + pub fn init(name_ctx: *anyopaque) Shared; + pub fn addProvider(s: *Shared, p: Provider) error{Full}!void; + pub fn expose(s: *Shared, name: []const u8, ptr: anytype) error{Full}!void; // typed value → /vars/ + }; + pub const Conn = struct { // one 9P connection, push/step/output like cloud9 + pub fn init(shared: *Shared, storage: *Storage, msize: u32) Conn; + pub fn push(c: *Conn, bytes: []const u8) usize; // feed transport bytes + pub fn step(c: *Conn) error{Protocol}!bool; // handle ≤ 1 request; false = nothing to do + pub fn output(c: *const Conn) []const u8; // bytes to send + pub fn wrote(c: *Conn, n: usize) void; + pub fn hangup(c: *Conn) void; // drop fids, tell providers + }; + }; +} +``` + +`step` drives `cloud9.Server.receive/reply/negotiate` and the backend: the +static tree (comptime-generated from `cfg`: `/README`, `/build/*` via a +`build_options`-like struct passed in `cfg.build`, `/comptime/types/*`, +`/comptime/decls`, `/runtime/fn/*`, `/ctl`, `/vars/*`) plus **providers**. + +A provider is a runtime vtable mounted at a top-level name. It owns a subtree +with its own naming (dynamic directories such as `/threads/` or +`/addr/` cannot be enumerated at comptime): + +```zig +pub const Provider = struct { + name: []const u8, + ctx: *anyopaque, + vtable: *const VTable, + pub const Handle = u64; // provider-defined node id; 0 = provider root + pub const VTable = struct { + walk: *const fn (ctx, parent: Handle, name: []const u8) Error!Handle, + stat: *const fn (ctx, h: Handle, out: *NodeStat) Error!void, // kind (dir/file), mode, length, mtime + list: *const fn (ctx, dir: Handle, index: usize, out: *NodeStat) Error!bool, // nth entry; false when done + open: *const fn (ctx, h: Handle, mode: u8) Error!void, + read: *const fn (ctx, h: Handle, offset: u64, buf: []u8) Error!usize, + write: *const fn (ctx, h: Handle, offset: u64, data: []const u8) Error!usize, + create: ?*const fn (ctx, dir: Handle, name: []const u8, perm: u32, mode: u8) Error!Handle, + remove: ?*const fn (ctx, h: Handle) Error!void, + wstat: ?*const fn (ctx, h: Handle, st: *const cloud9.Stat) Error!void, + clunk: *const fn (ctx, h: Handle) void, // fid released (also on hangup) + }; + pub const Error = error{ NotFound, Exists, Perm, NotDir, IsDir, NotEmpty, BadOffset, NoSpace, Io, Unsupported }; +}; +``` + +Error → Rerror text mapping lives in one place in the core, using the Plan 9 +strings 9player's bridge already understands (`file does not exist`, +`permission denied`, `file already exists`, `directory not empty`, +`not a directory`, `is a directory`, `bad offset`, `no space`, `i/o error`, +`not supported`). + +Directory reads follow the 9P rule (offset 0 or previous offset+count, never +split a record). Dynamic file reads: on open the content is generated once +into a snapshot slot (`open` runs the generator; `read` serves the slot; a +read at offset 0 regenerates); no free slot → Rerror `too many open dynamic +files`. Stats of dynamic files report length 0. + +Qids: static nodes get comptime paths; provider nodes get +`(provider index << 56) | handle`. + +Static memory: `Server(cfg).Storage` per connection, `Shared` once. No heap. +The core has unit tests driven through `cloud9.Client` in memory (like today). + +## Value renderers (`vars.zig`) + +`expose(name, ptr: anytype)` builds at comptime a `VTable` for +`@TypeOf(ptr.*)`: + +``` +/vars//value rendered text (structs: "field: value" lines, nested indented; unions: tag + payload; + optionals: "null" or the value; enums: tag; ints/floats/bools; []const u8 and [*:0]const u8 + as quoted strings (≤ 256 bytes); other pointers as 0x… never followed; arrays/slices ≤ 64 elements) +/vars//type @typeName +/vars//size @sizeOf +/vars//addr 0x… +/vars//raw the bytes (length = @sizeOf) +/vars//f//... same layout recursively for struct fields (depth ≤ 4), leaves writable: + writing text to a scalar's `value` parses and stores it (ints: decimal/0x, bools, floats, enums by tag) +``` + +Rendering is by a comptime-generated function table; no allocation. +Writes to scalars are plain stores (not atomic; documented). + +## Scratch provider (`scratch.zig`) + +The in-memory read/write tree from the current server, as a provider, with +`init(allocator, budget_bytes)`; the only core-level component that takes an +allocator, and it is optional. + +## Linux layer (`linux/probe.zig`) + +```zig +pub const Probe = struct { + pub const Options = struct { + io: std.Io, // for std.debug symbolization + listen: union(enum) { unix: []const u8, tcp: []const u8, fd: i32 }, + max_clients: u8 = 8, + msize: u32 = 64 * 1024, + hold_on_panic: bool = true, + capture_signal: u8 = SIGRTMIN + 3, // used to snapshot other threads + breakpoints: bool = true, // install the SIGTRAP handler + }; + pub fn Storage(comptime max_clients: u8) type; // static: per-client Server.Storage + poll table + pub fn init(p: *Probe, shared: *Server.Shared, storage: *Storage, opts: Options) !void; // listens, registers the debug provider + pub fn start(p: *Probe) !void; // spawns ONE background thread running a poll loop over listener + clients + pub fn stop(p: *Probe) void; // closes, joins +}; +``` + +One thread, `poll()` over the listener and every connection; each connection +is a core `Conn` fed with `push`/`step`/`output`. No per-connection threads. +Symbolization uses `std.debug.getSelfDebugInfo()` with the `io` passed in and +a caller-provided fixed buffer as the text arena. + +## Debug provider (`linux/debug.zig`) + +Mounted as `/threads`, `/addr`, `/mem`, `/hex`, `/breakpoints`, `/panic`. + +``` +/threads/ one directory per tid, enumerated from /proc/self/task at list time +/threads//name comm +/threads//stat state letter + a few fields from /proc/self/task//stat +/threads//stack "#n 0x in (::)" per frame +/threads//regs " 0x" per general register, from the captured cpu context +/addr/ dynamic dir: walk of any hex address yields a file "fn\nfile:line:col\nmodule\n" +/mem/maps /proc/self/maps served by pread at the requested offset (any size) +/mem/ raw bytes at address+offset via process_vm_readv/writev (never faults); writable +/hex/ hexdump text of 256 bytes at address (+offset), like std.debug.dumpHex +/breakpoints/ directory of tids currently stopped in @breakpoint() +/breakpoints//stack, regs as above +/breakpoints//ctl write "continue" (or "step"? no: continue only) to resume +/panic/message the panic message, empty before any panic +/panic/stack frames of the panicking thread +/panic/ctl write "continue" to let the default panic handler run (abort) +``` + +**Capturing another thread** (`stack`, `regs`): the server thread `tgkill`s +the target with `capture_signal`. The handler (SA_SIGINFO, async-signal-safe: +no allocation, no locks) copies the `cpu_context.Native` obtained through +`std.debug.cpu_context.fromPosixSignalContext` into a slot and futex-waits. +The server thread unwinds with `std.debug.StackIterator.init(&ctx)` while the +target is parked, symbolizes, then releases the slot; the target resumes. The +server's own thread unwinds itself directly. Timeout 250 ms → Rerror +`thread did not respond`. Threads blocked in uninterruptible syscalls simply +time out. A target parked while holding std.debug's `SelfInfo` lock (it was +printing a stack trace itself) cannot be unwound without deadlocking; the +probe detects that with `tryLock`, releases the target and answers +`i/o error` (registers still work). + +**Breakpoints**: `@breakpoint()` raises SIGTRAP on the executing thread only. +The installed handler stores the context in a slot, marks the thread paused, +and futex-waits until `/breakpoints//ctl` receives `continue`. On x86_64 +the saved PC already points past `int3`; on aarch64 the handler advances PC by +4 (`brk`) before returning, but only for kernel-generated traps +(`si_code > 0`); a user-sent `SIGTRAP` (`kill -TRAP`, `tgkill`) parks the +thread exactly where it was, which makes it a usable "pause this thread" +request. Other threads keep running; a slot table (`max_paused`, default 16) +bounds simultaneous pauses and, when it is full, the trapping thread simply +steps over the breakpoint (`traps_skipped` counts these). The probe's own +serving thread is never parked or held: a trap or panic on it goes straight +to the default behaviour, since nobody could write its `ctl` files. + +**Panics**: `pub const panic = introspect.linux.panic;` in the root module +(built with `std.debug.FullPanic`). The first panic records message and a +stack capture (`captureCurrentStackTrace` with `first_address`), publishes +them, and, if `hold_on_panic` and the probe is running, futex-waits until +`/panic/ctl` says `continue`; then `std.debug.defaultPanic` runs (prints the +trace and aborts). A nested or second panic goes straight to the default. + +Signal handlers are installed by `Probe.init` (breakpoints optional) and +restored by `stop`. + +## Demo (`demo/main.zig`, binary `introspect`) + +Keeps every path the existing tests read (`/build/*`, `/comptime/types/Qid/*`, +`/comptime/decls`, `/runtime/fn/now|hostname|…`, `/runtime/ctl` with +`add|echo|fib|sleep-ms`, `/runtime/pid|ppid|uptime|argv|cwd|env|clients`, +`/scratch`), served by the library. Adds: + +* a worker thread running `workerLoop` that increments an exposed + `State { ticks: u64, phase: enum, last_job: Job }` (`/vars/state/...`); +* `/runtime/ctl` commands `trap` (the worker executes `@breakpoint()` on its + next tick) and `panic` (the worker panics with a message); +* `--stdio | --unix PATH | --tcp IP:PORT` as today, `--no-hold` to disable + panic holding. + +`main` passes `init.io` and an explicit allocator to the pieces that need one; +the demo's Storage is a global. + +## Verification + +* Unit tests: core (in-memory client drives every op incl. providers and + snapshots), vars (render/set for every category), scratch, debug (capture + own thread and a helper thread; breakpoint pause/continue on a helper + thread; panic record path without holding). +* `zig build introspect-check-freestanding`: compiles `core.zig` + `vars.zig` + for `riscv32-freestanding-none` with a tiny freestanding root that + instantiates `Server(cfg)` with static Storage. +* `zig build introspect-test` (library and demo unit tests) and + `zig build introspect-debug-test` (linux/debug.zig). +* 9player's `test/integration.sh` unchanged and passing; `test/debug.sh` + (`zig build introspect-debug-itest`) through 9player: read the worker's + stack (contains `workerLoop` and `demo/main.zig:`), + resolve a frame through `/addr`, dump `/hex` of the exposed state, read and + write `/vars/state/f/ticks/value`, trap → `/breakpoints` lists the worker, + its stack shows `workerLoop`, `continue` resumes (ticks keep increasing), + panic → `/panic/message`, `/panic/stack`, `continue` → server exits + non-zero. +* Adversarial pass afterwards (`zig build introspect-adv`: hostile client + against the server and the core, signal races, memory reads of unmapped + addresses, panic while a capture is in flight; `test/adversarial.sh` runs + the same suites by hand). + +## Known upstream issue (Zig 0.16 std.debug) + +`std.debug.SelfInfo` for ELF (`std/debug/SelfInfo/Elf.zig`, `findModule`) +rebuilds its module list whenever it is asked about an address outside every +known module. That frees each module's `Dwarf.Unwind` and CIE list but leaves +`unwind_cache` entries pointing into the freed memory, so later unwinds read +freed data: empty traces, "unwind info invalid", or segfaults once the arena +reuses the block. `linux/debug.zig` records the executable's `PT_LOAD` ranges +at init and refuses to hand std an address outside them (`knownCode`), which +is why `/addr/` of a bogus address renders `?` instead of poisoning the +process. Worth reporting upstream; the guard can go once std clears the cache. diff --git a/introspect/src/core.zig b/introspect/src/core.zig new file mode 100644 index 0000000..2707af2 --- /dev/null +++ b/introspect/src/core.zig @@ -0,0 +1,2656 @@ +//! The freestanding 9P2000 introspection engine: a static tree generated at +//! comptime from a `Config` (README, /build, /comptime, /runtime/fn, /ctl, +//! /vars) plus runtime `Provider`s mounted at the top level, served over a +//! `cloud9.Server` connection. No allocator, no OS, no threads: every buffer is +//! caller-owned (`Storage`, `Shared`, `Conn`), every table is sized at comptime. +//! See docs/LIBRARY.md. +const std = @import("std"); +const builtin = @import("builtin"); +const cloud9 = @import("cloud9"); +const vars = @import("vars.zig"); +const Writer = std.Io.Writer; + +/// Longest file name accepted in a create or rename. +pub const max_name: usize = 255; + +/// A dynamic-file generator (`Config.fns`, `Config.runtime`): writes the file's +/// content into `w` at open time (and again at each read from offset 0). +pub const Gen = *const fn (ctx: *anyopaque, w: *Writer) anyerror!void; +/// The /ctl command handler: `cmd` is the written text, `out` receives the +/// result that later reads of /ctl return. +pub const Ctl = *const fn (ctx: *anyopaque, cmd: []const u8, out: *Writer) anyerror!void; + +pub const Config = struct { + /// Appears in /README and as uid/gid/muid of every Stat. + name: []const u8 = "introspect", + /// A type whose pub decls `zig_version`, `target`, `optimize`, `time` + /// and `change` (all `[]const u8`) become the files of /build. `null` + /// omits /build. + build: ?type = null, + /// /comptime/types//{name,size,align,fields}. + types: []const type = &.{}, + /// /comptime/decls lists this type's pub decls (empty when null). + decls_of: ?type = null, + /// /runtime/fn/: every pub decl is a `fn (ctx: *anyopaque, w: *std.Io.Writer) anyerror!void`. + fns: type = struct {}, + /// /runtime/: same signature as `fns`, one level up (pid, uptime, ...). + runtime: type = struct {}, + /// The /ctl handler; `null` omits /ctl. + ctl: ?Ctl = null, + /// Where /ctl lives: the root or /runtime/ctl. + ctl_dir: enum { root, runtime } = .root, + /// Capacity of the ctl result (in `Shared`). + ctl_bytes: u32 = 4096, + /// Largest negotiable msize; sizes `Storage.in/out/data`. + msize: u32 = 8192, + max_fids: u16 = 64, + max_providers: u8 = 8, + max_vars: u8 = 32, + /// Dynamic file contents are generated at open time into per-fid snapshot + /// slots so that reads at arbitrary offsets are consistent. + snapshot_slots: u8 = 8, + snapshot_bytes: u32 = 16 * 1024, +}; + +/// Attributes of a provider node, filled by `VTable.stat` and `VTable.list`. +pub const NodeStat = struct { + /// Permission bits plus `cloud9.dmdir`/`dmappend`/`dmexcl`. + mode: u32, + length: u64 = 0, + atime: u32 = 0, + mtime: u32 = 0, + /// Becomes the qid version. + version: u32 = 0, + /// The entry name (`list`) or the node's own name (`stat`; ignored for the + /// provider root, whose name is the mount name). Must stay valid until the + /// provider's next call. + name: []const u8 = "", + /// Filled by `list`: the entry's handle. Not retained by the core. + handle: Provider.Handle = 0, + /// A stable identity for the qid path (low 56 bits), for providers whose + /// handles are not stable across the node's life (e.g. memory addresses + /// that an allocator may reuse). 0 means "the handle is the path". + path: u64 = 0, + + pub fn isDir(s: NodeStat) bool { + return s.mode & cloud9.dmdir != 0; + } +}; + +/// A runtime subtree mounted at a top-level name. +/// +/// Handle lifetime: every handle returned by `walk` or `create` is released by +/// the core with exactly one `clunk` (after `close` if the fid was open). The +/// root handle 0 is never obtained through `walk`, so providers must treat +/// `clunk(0)` as a no-op. `walk` must accept "." on any node, file or directory +/// (a fresh reference to the same node; the core clones fids with it), and ".." +/// on directories (except at the root, which the core resolves itself). +pub const Provider = struct { + name: []const u8, + ctx: *anyopaque, + vtable: *const VTable, + + /// Provider-defined node id; 0 = provider root. + pub const Handle = u64; + pub const root: Handle = 0; + + pub const Error = error{ NotFound, Exists, Perm, NotDir, IsDir, NotEmpty, BadOffset, NoSpace, Io, Unsupported, Excl }; + + pub const VTable = struct { + walk: *const fn (ctx: *anyopaque, parent: Handle, name: []const u8) Error!Handle, + stat: *const fn (ctx: *anyopaque, h: Handle, out: *NodeStat) Error!void, + /// The `index`-th entry of `dir`; false when done. + list: *const fn (ctx: *anyopaque, dir: Handle, index: usize, out: *NodeStat) Error!bool, + open: *const fn (ctx: *anyopaque, h: Handle, mode: u8) Error!void, + read: *const fn (ctx: *anyopaque, h: Handle, offset: u64, buf: []u8) Error!usize, + write: *const fn (ctx: *anyopaque, h: Handle, offset: u64, data: []const u8) Error!usize, + /// Returns the new node, already open with `mode`. + create: ?*const fn (ctx: *anyopaque, dir: Handle, name: []const u8, perm: u32, mode: u8) Error!Handle = null, + remove: ?*const fn (ctx: *anyopaque, h: Handle) Error!void = null, + /// Only name, length, mode and mtime can differ from the current stat + /// (the core has already checked the immutable fields and the name). + wstat: ?*const fn (ctx: *anyopaque, h: Handle, st: *const cloud9.Stat) Error!void = null, + /// An open fid on `h` was released (before `clunk`). + close: ?*const fn (ctx: *anyopaque, h: Handle) void = null, + /// A fid holding `h` was released (also on hangup and Tversion). + clunk: *const fn (ctx: *anyopaque, h: Handle) void, + }; +}; + +/// The Plan 9 error string for any error the engine or a provider can raise. +pub fn ename(err: anyerror) []const u8 { + return switch (err) { + error.NotFound, error.NoFile => "file does not exist", + error.Perm => "permission denied", + error.Exists => "file already exists", + error.NotEmpty => "directory not empty", + error.NotDir => "not a directory", + error.IsDir => "is a directory", + error.BadOffset => "bad offset", + error.NoSpace => "no space left on device", + error.Io => "i/o error", + error.Unsupported => "not supported", + error.Excl => "exclusive use file already open", + error.FidInUse => "fid in use", + error.UnknownFid => "unknown fid", + error.NotOpen => "file not open", + error.AlreadyOpen => "file already open", + error.AuthNotRequired => "authentication not required", + error.BadCommand => "bad command", + error.BadName => "bad file name", + error.Invalid, error.BadValue => "bad value", + error.TooManyFids => "too many fids", + error.NoSnapshot => "too many open dynamic files", + error.WriteFailed => "no space in buffer", + error.ReplyTooLarge => "reply too large for msize", + error.OutOfMemory => "out of memory", + else => "i/o error", + }; +} + +/// A "don't care" Twstat: every field left as it is. +pub const stat_dontcare: cloud9.Stat = .{ + .type = 0xFFFF, + .dev = 0xFFFF_FFFF, + .qid = .{ .type = 0xFF, .version = 0xFFFF_FFFF, .path = 0xFFFF_FFFF_FFFF_FFFF }, + .mode = 0xFFFF_FFFF, + .atime = 0xFFFF_FFFF, + .mtime = 0xFFFF_FFFF, + .length = 0xFFFF_FFFF_FFFF_FFFF, + .name = "", + .uid = "", + .gid = "", + .muid = "", +}; + +pub fn validName(name: []const u8) error{BadName}!void { + if (name.len == 0 or name.len > max_name) return error.BadName; + if (std.mem.eql(u8, name, ".") or std.mem.eql(u8, name, "..")) return error.BadName; + if (std.mem.indexOfAny(u8, name, "/\x00") != null) return error.BadName; +} + +/// "YYYY-MM-DDTHH:MM:SSZ" as unix seconds, or null. +pub fn parseIso8601(s: []const u8) ?u32 { + if (s.len != 20 or s[4] != '-' or s[7] != '-' or s[10] != 'T' or s[13] != ':' or s[16] != ':' or s[19] != 'Z') return null; + const y = std.fmt.parseInt(i64, s[0..4], 10) catch return null; + const mo = std.fmt.parseInt(i64, s[5..7], 10) catch return null; + const d = std.fmt.parseInt(i64, s[8..10], 10) catch return null; + const h = std.fmt.parseInt(i64, s[11..13], 10) catch return null; + const mi = std.fmt.parseInt(i64, s[14..16], 10) catch return null; + const sec = std.fmt.parseInt(i64, s[17..19], 10) catch return null; + if (mo < 1 or mo > 12 or d < 1 or d > 31 or h > 23 or mi > 59 or sec > 60) return null; + // Howard Hinnant's days_from_civil. + const yy = if (mo <= 2) y - 1 else y; + const era = @divFloor(yy, 400); + const yoe = yy - era * 400; + const mp = if (mo > 2) mo - 3 else mo + 9; + const doy = @divFloor(153 * mp + 2, 5) + d - 1; + const doe = yoe * 365 + @divFloor(yoe, 4) - @divFloor(yoe, 100) + doy; + const days = era * 146097 + doe - 719468; + const total = days * 86400 + h * 3600 + mi * 60 + sec; + if (total < 0 or total > std.math.maxInt(u32)) return null; + return @intCast(total); +} + +/// The last component of @typeName(T): "wire.Qid" -> "Qid". +pub fn shortTypeName(comptime T: type) []const u8 { + const full = @typeName(T); + const dot = std.mem.lastIndexOfScalar(u8, full, '.') orelse return full; + return full[dot + 1 ..]; +} + +/// The /comptime/types//fields text: "name: type @offset" per line. +pub fn fieldsText(comptime T: type) []const u8 { + comptime { + @setEvalBranchQuota(200_000); + var s: []const u8 = ""; + switch (@typeInfo(T)) { + .@"struct" => |info| for (info.fields) |f| { + if (info.layout == .@"packed") { + s = s ++ std.fmt.comptimePrint("{s}: {s} @{d}b\n", .{ f.name, @typeName(f.type), @bitOffsetOf(T, f.name) }); + } else if (f.is_comptime) { + s = s ++ std.fmt.comptimePrint("{s}: {s} (comptime)\n", .{ f.name, @typeName(f.type) }); + } else { + s = s ++ std.fmt.comptimePrint("{s}: {s} @{d}\n", .{ f.name, @typeName(f.type), @offsetOf(T, f.name) }); + } + }, + .@"union" => |info| for (info.fields) |f| { + s = s ++ f.name ++ ": " ++ @typeName(f.type) ++ "\n"; + }, + .@"enum" => |info| for (info.fields) |f| { + s = s ++ std.fmt.comptimePrint("{s} = {d}\n", .{ f.name, f.value }); + }, + else => s = @typeName(T) ++ "\n", + } + return s; + } +} + +fn declsText(comptime T: type) []const u8 { + comptime { + @setEvalBranchQuota(20_000); + const decls = switch (@typeInfo(T)) { + inline .@"struct", .@"union", .@"enum", .@"opaque" => |info| info.decls, + else => &[_]std.builtin.Type.Declaration{}, + }; + var s: []const u8 = ""; + for (decls) |d| s = s ++ d.name ++ "\n"; + return s; + } +} + +/// Wraps `Fns.` in a function of exactly the `Gen` signature. +fn genFor(comptime Fns: type, comptime name: []const u8) Gen { + return &struct { + fn g(ctx: *anyopaque, w: *Writer) anyerror!void { + return @field(Fns, name)(ctx, w); + } + }.g; +} + +/// A node of the static tree, described at comptime. +pub const Node = struct { + name: []const u8, + kind: Kind, + children: []const Node = &.{}, + content: []const u8 = "", + gen: ?Gen = null, + + pub const Kind = enum(u8) { dir, static, dynamic, ctl, vars }; + + fn isDir(n: Node) bool { + return n.kind == .dir or n.kind == .vars; + } +}; + +fn genNodes(comptime Fns: type) [@typeInfo(Fns).@"struct".decls.len]Node { + const decls = @typeInfo(Fns).@"struct".decls; + var arr: [decls.len]Node = undefined; + for (decls, 0..) |d, i| arr[i] = .{ .name = d.name, .kind = .dynamic, .gen = genFor(Fns, d.name) }; + return arr; +} + +pub fn Server(comptime cfg: Config) type { + return struct { + const Self = @This(); + + // -- the static tree ------------------------------------------------ + + pub const readme_text = std.fmt.comptimePrint( + \\{s}: a 9P2000 introspection server (built on cloud9). + \\ + \\/build facts baked in at build time (zig version, target, optimize, time, change id) + \\/comptime facts computed by the Zig compiler: type layouts under types//, pub decls + \\/runtime live facts; fn/ calls a Zig function on every read + \\/ctl write a command, read the result + \\/vars exposed variables: /{{value,type,size,addr,raw,f//...}} + \\ + \\Other top-level directories are providers mounted at runtime. + \\ + , .{cfg.name}); + + fn typeDir(comptime T: type) Node { + return .{ .name = shortTypeName(T), .kind = .dir, .children = &.{ + .{ .name = "name", .kind = .static, .content = @typeName(T) }, + .{ .name = "size", .kind = .static, .content = std.fmt.comptimePrint("{d}", .{@sizeOf(T)}) }, + .{ .name = "align", .kind = .static, .content = std.fmt.comptimePrint("{d}", .{@alignOf(T)}) }, + .{ .name = "fields", .kind = .static, .content = fieldsText(T) }, + } }; + } + + const type_dirs: [cfg.types.len]Node = blk: { + @setEvalBranchQuota(200_000); + var arr: [cfg.types.len]Node = undefined; + for (cfg.types, 0..) |T, i| arr[i] = typeDir(T); + for (arr, 0..) |a, i| for (arr[i + 1 ..]) |b| { + if (std.mem.eql(u8, a.name, b.name)) @compileError("duplicate short type name " ++ a.name); + }; + break :blk arr; + }; + + const decls_text: []const u8 = if (cfg.decls_of) |T| declsText(T) else ""; + const fn_nodes = genNodes(cfg.fns); + const runtime_nodes = genNodes(cfg.runtime); + const ctl_node: Node = .{ .name = "ctl", .kind = .ctl }; + + const build_nodes: []const Node = if (cfg.build) |B| &[_]Node{ + .{ .name = "zig_version", .kind = .static, .content = B.zig_version }, + .{ .name = "target", .kind = .static, .content = B.target }, + .{ .name = "optimize", .kind = .static, .content = B.optimize }, + .{ .name = "time", .kind = .static, .content = B.time }, + .{ .name = "change", .kind = .static, .content = B.change }, + } else &.{}; + + /// Build time as unix seconds (for static atime/mtime), or 0. + pub const build_secs: u32 = if (cfg.build) |B| (parseIso8601(B.time) orelse 0) else 0; + + const runtime_children: []const Node = blk: { + var list: []const Node = &runtime_nodes; + list = list ++ &[_]Node{.{ .name = "fn", .kind = .dir, .children = &fn_nodes }}; + if (cfg.ctl != null and cfg.ctl_dir == .runtime) list = list ++ &[_]Node{ctl_node}; + break :blk list; + }; + + const root_children: []const Node = blk: { + var list: []const Node = &[_]Node{.{ .name = "README", .kind = .static, .content = readme_text }}; + if (cfg.build != null) list = list ++ &[_]Node{.{ .name = "build", .kind = .dir, .children = build_nodes }}; + list = list ++ &[_]Node{ + .{ .name = "comptime", .kind = .dir, .children = &.{ + .{ .name = "types", .kind = .dir, .children = &type_dirs }, + .{ .name = "decls", .kind = .static, .content = decls_text }, + } }, + .{ .name = "runtime", .kind = .dir, .children = runtime_children }, + }; + if (cfg.ctl != null and cfg.ctl_dir == .root) list = list ++ &[_]Node{ctl_node}; + list = list ++ &[_]Node{.{ .name = "vars", .kind = .vars }}; + break :blk list; + }; + + pub const root_node: Node = .{ .name = "/", .kind = .dir, .children = root_children }; + + /// The static tree flattened so nodes can be referenced by index; the + /// children of a node occupy consecutive slots `first..first+count`. + const Flat = struct { node: Node, parent: u32, first: u32, count: u32 }; + + fn countNodes(n: Node) usize { + var c: usize = 1; + for (n.children) |ch| c += countNodes(ch); + return c; + } + + fn fillFlat(arr: []Flat, next: *usize, idx: usize, n: Node, parent: u32) void { + const first = next.*; + next.* += n.children.len; + arr[idx] = .{ .node = n, .parent = parent, .first = @intCast(first), .count = @intCast(n.children.len) }; + for (n.children, 0..) |ch, i| fillFlat(arr, next, first + i, ch, @intCast(idx)); + } + + pub const flat_len = countNodes(root_node); + pub const flat: [flat_len]Flat = blk: { + @setEvalBranchQuota(100_000); + var arr: [flat_len]Flat = undefined; + var next: usize = 1; + fillFlat(&arr, &next, 0, root_node, 0); + break :blk arr; + }; + const vars_idx: u32 = blk: { + for (flat, 0..) |f, i| if (f.node.kind == .vars) break :blk @intCast(i); + @compileError("no vars node"); + }; + + comptime { + for (flat) |f| if (f.node.kind == .dynamic and f.node.gen == null) @compileError("dynamic node without generator"); + std.debug.assert(cfg.msize >= cloud9.Server.msize_min); + std.debug.assert(cfg.snapshot_slots > 0 and cfg.max_fids > 0); + // Provider index 0xFE/0xFF would collide with the var/static qid tags. + std.debug.assert(cfg.max_providers < 0xFE); + } + + // -- qid paths ------------------------------------------------------ + + const static_tag: u64 = 0xFF << 56; + const var_tag: u64 = 0xFE << 56; + const handle_mask: u64 = (1 << 56) - 1; + + // -- storage -------------------------------------------------------- + + /// Per-connection buffers; the caller places one in static memory. + pub const Storage = struct { + in: [cfg.msize]u8, + out: [cfg.msize]u8, + /// Staging area for read replies (directory records, provider and raw reads). + data: [cfg.msize]u8, + snapshots: [cfg.snapshot_slots][cfg.snapshot_bytes]u8, + }; + + const Var = struct { + name: []const u8, + ptr: *anyopaque, + vt: *const vars.VTable, + }; + + /// State common to all connections: providers, exposed variables, the + /// ctl result. Not internally synchronized: one thread serves all + /// connections (or the caller serializes). + pub const Shared = struct { + ctx: *anyopaque, + providers: [cfg.max_providers]Provider = undefined, + nprov: u8 = 0, + vars: [cfg.max_vars]Var = undefined, + nvars: u8 = 0, + /// The ctl result is double-buffered: a command writes into the + /// buffer that is not current and commits it only on success, so + /// a failed command leaves the previous result intact. + ctl_bufs: [2][cfg.ctl_bytes]u8 = undefined, + ctl_cur: u1 = 0, + /// Length of the current ctl result (in `ctl_bufs[ctl_cur]`). + ctl_len: u32 = 0, + ctl_version: u32 = 0, + /// Entropy for the per-connection fid hash. The core mixes in a + /// connection counter and buffer addresses; a platform layer with a + /// random source may set this once after `init` to make the seed + /// unpredictable even where addresses are static. + hash_seed: u32 = 0, + conn_seq: u32 = 0, + + /// `ctx` is passed to every `fns`/`runtime` generator and to `ctl`. + pub fn init(ctx: *anyopaque) Shared { + return .{ .ctx = ctx }; + } + + /// Mounts `p` at `/`. The name must not collide with a + /// static entry or another provider. + pub fn addProvider(s: *Shared, p: Provider) error{Full}!void { + if (s.nprov == cfg.max_providers) return error.Full; + std.debug.assert(validName(p.name) != error.BadName); + std.debug.assert(s.findProvider(p.name) == null); + std.debug.assert(staticChild(0, p.name) == null); + s.providers[s.nprov] = p; + s.nprov += 1; + } + + /// Publishes `ptr.*` as /vars/. `name` and the pointee must + /// outlive the server. + pub fn expose(s: *Shared, name: []const u8, ptr: anytype) error{Full}!void { + const P = @TypeOf(ptr); + const info = @typeInfo(P); + if (info != .pointer or info.pointer.size != .one or info.pointer.is_const) @compileError("expose wants a *T, got " ++ @typeName(P)); + if (s.nvars == cfg.max_vars) return error.Full; + std.debug.assert(validName(name) != error.BadName); + std.debug.assert(s.findVar(name) == null); + s.vars[s.nvars] = .{ .name = name, .ptr = @ptrCast(ptr), .vt = vars.vtableFor(info.pointer.child) }; + s.nvars += 1; + } + + /// The result of the last successful ctl command. + pub fn ctlResult(s: *const Shared) []const u8 { + return s.ctl_bufs[s.ctl_cur][0..s.ctl_len]; + } + + fn findProvider(s: *const Shared, name: []const u8) ?u8 { + for (s.providers[0..s.nprov], 0..) |p, i| if (std.mem.eql(u8, p.name, name)) return @intCast(i); + return null; + } + + fn findVar(s: *const Shared, name: []const u8) ?u8 { + for (s.vars[0..s.nvars], 0..) |v, i| if (std.mem.eql(u8, v.name, name)) return @intCast(i); + return null; + } + }; + + fn staticChild(idx: u32, name: []const u8) ?u32 { + const f = flat[idx]; + for (f.first..f.first + f.count) |ci| { + if (std.mem.eql(u8, flat[ci].node.name, name)) return @intCast(ci); + } + return null; + } + + // -- connection ----------------------------------------------------- + + const NodeRef = union(enum) { + static: u32, + prov: struct { idx: u8, h: Provider.Handle }, + @"var": struct { idx: u8, node: u32 }, + }; + + const Fid = struct { + id: u32 = 0, + used: bool = false, + node: NodeRef = .{ .static = 0 }, + is_dir: bool = true, + open: bool = false, + mode: u8 = 0, + rclose: bool = false, + dir_offset: u64 = 0, + dir_index: usize = 0, + /// Snapshot slot of an open dynamic file. + snap: ?u8 = null, + /// Free-list link, meaningful while `!used`. + next_free: u16 = no_slot, + }; + + const no_slot: u16 = std.math.maxInt(u16); + /// The fid index is an open-addressing (linear probing) table from fid + /// number to a slot of `Conn.fids`, sized to stay at most half full so + /// that lookups are O(1) with any number of fids. + const index_len: usize = std.math.ceilPowerOfTwoAssert(usize, @as(usize, cfg.max_fids) * 2); + const index_mask: usize = index_len - 1; + const index_shift: u5 = @intCast(32 - @as(usize, std.math.log2_int(usize, index_len))); + + /// MurmurHash3's 32-bit finalizer: every input bit affects every output bit. + fn fmix32(x: u32) u32 { + var h = x; + h ^= h >> 16; + h *%= 0x85EB_CA6B; + h ^= h >> 13; + h *%= 0xC2B2_AE35; + h ^= h >> 16; + return h; + } + + /// Everything the engine needs to know about a node for qid/stat. + const Info = struct { + is_dir: bool, + mode: u32, + length: u64, + atime: u32, + mtime: u32, + version: u32, + path: u64, + name: []const u8, + + fn qid(i: Info) cloud9.Qid { + var t: u8 = if (i.is_dir) cloud9.qtdir else cloud9.qtfile; + if (i.mode & cloud9.dmappend != 0) t |= cloud9.qtappend; + if (i.mode & cloud9.dmexcl != 0) t |= cloud9.qtexcl; + return .{ .type = t, .version = i.version, .path = i.path }; + } + + fn stat(i: Info) cloud9.Stat { + return .{ + .type = 0, + .dev = 0, + .qid = i.qid(), + .mode = i.mode, + .atime = i.atime, + .mtime = i.mtime, + .length = i.length, + .name = i.name, + .uid = cfg.name, + .gid = cfg.name, + .muid = cfg.name, + }; + } + }; + + /// One 9P connection: a cloud9.Server plus a fid table and snapshot slots. + pub const Conn = struct { + shared: *Shared, + storage: *Storage, + server: cloud9.Server, + /// Largest msize this connection negotiates. + msize_cap: u32, + fids: [cfg.max_fids]Fid = @splat(.{}), + /// XORed into every fid number before hashing so that a client + /// cannot precompute fid numbers that collide (which would turn the + /// index back into a linear scan). + hash_seed: u32, + /// fid number -> slot of `fids` (`no_slot` = empty bucket). + index: [index_len]u16 = @splat(no_slot), + /// Head of the free list threaded through `Fid.next_free`. + free_head: u16 = no_slot, + /// Slots `high_water..` have never been used (bump allocation). + high_water: u16 = 0, + nfids: u16 = 0, + slot_used: [cfg.snapshot_slots]bool = @splat(false), + slot_len: [cfg.snapshot_slots]u32 = @splat(0), + name_buf: [max_name]u8 = undefined, + + pub fn init(shared: *Shared, storage: *Storage, msize: u32) Conn { + shared.conn_seq +%= 1; + const addr = @intFromPtr(storage) ^ (@intFromPtr(shared) << 7); + const seed = fmix32(shared.hash_seed ^ (shared.conn_seq *% 0x9E37_79B1) ^ @as(u32, @truncate(addr)) ^ @as(u32, @truncate(addr >> 16))); + return .{ + .shared = shared, + .storage = storage, + .server = .init(.{ .in = &storage.in, .out = &storage.out }), + .msize_cap = @max(@min(msize, cfg.msize), cloud9.Server.msize_min), + .hash_seed = seed, + }; + } + + /// Fibonacci hashing of the (seeded) fid number into `index_len` buckets. + fn fidHome(c: *const Conn, id: u32) usize { + return @intCast(((id ^ c.hash_seed) *% 0x9E37_79B1) >> index_shift); + } + + /// Feeds transport bytes; returns how many were taken. + pub fn push(c: *Conn, bytes: []const u8) usize { + return c.server.push(bytes); + } + + /// Bytes to send to the client. + pub fn output(c: *const Conn) []const u8 { + return c.server.output(); + } + + pub fn wrote(c: *Conn, n: usize) void { + c.server.wrote(n); + } + + /// Drops every fid (telling providers) and kills the session. + pub fn hangup(c: *Conn) void { + c.resetFids(); + c.server.hangup(); + } + + /// Handles at most one request. Returns false when more input (or + /// output drainage) is needed. `error.Protocol` is terminal. + pub fn step(c: *Conn) error{Protocol}!bool { + const req = (c.server.receive() catch return error.Protocol) orelse return false; + defer c.server.release(); + switch (req.msg) { + .tversion => |m| { + c.resetFids(); + c.server.negotiate(@min(m.msize, c.msize_cap), m.version) catch return error.Protocol; + }, + else => { + const reply = c.dispatch(req.msg) catch |e| cloud9.Msg{ .rerror = .{ .ename = ename(e) } }; + c.server.reply(req.tag, reply) catch |e| switch (e) { + // The reply does not fit the negotiated msize (Rstat or a long + // Rwalk at a tiny msize). receive() guarantees room for one + // msize-sized message, so this is never backpressure: answer with + // an Rerror (truncated to fit by cloud9). Rwalk, the only + // variable-size reply that follows a state change, is size-checked + // in walk() before anything is mutated. + error.TooLarge => c.server.reply(req.tag, .{ .rerror = .{ .ename = ename(error.ReplyTooLarge) } }) catch return error.Protocol, + else => return error.Protocol, + }; + }, + } + return true; + } + + /// Number of fids currently held. + pub fn fidCount(c: *const Conn) usize { + return c.nfids; + } + + fn dispatch(c: *Conn, msg: cloud9.Msg) anyerror!cloud9.Msg { + return switch (msg) { + .tauth => error.AuthNotRequired, + .tattach => |m| c.attach(m), + .tflush => .rflush, + .twalk => |m| c.walk(m), + .topen => |m| c.open(m), + .tcreate => |m| c.create(m), + .tread => |m| c.read(m), + .twrite => |m| c.write(m), + .tclunk => |m| c.clunk(m), + .tremove => |m| c.remove(m), + .tstat => |m| c.stat(m), + .twstat => |m| c.wstat(m), + else => error.Protocol, + }; + } + + // -- fid table -- + + /// The index bucket holding `id`, if any. + fn findBucket(c: *const Conn, id: u32) ?usize { + var pos = c.fidHome(id); + while (true) : (pos = (pos + 1) & index_mask) { + const slot = c.index[pos]; + if (slot == no_slot) return null; + if (c.fids[slot].id == id) return pos; + } + } + + fn findFid(c: *Conn, id: u32) ?*Fid { + const pos = c.findBucket(id) orelse return null; + return &c.fids[c.index[pos]]; + } + + fn allocFid(c: *Conn, id: u32) !*Fid { + if (c.findBucket(id) != null) return error.FidInUse; + if (c.nfids >= cfg.max_fids) return error.TooManyFids; + const slot: u16 = if (c.free_head != no_slot) blk: { + const slot = c.free_head; + c.free_head = c.fids[slot].next_free; + break :blk slot; + } else blk: { + const slot = c.high_water; + c.high_water += 1; + break :blk slot; + }; + c.fids[slot] = .{ .id = id, .used = true }; + var pos = c.fidHome(id); + while (c.index[pos] != no_slot) pos = (pos + 1) & index_mask; + c.index[pos] = slot; + c.nfids += 1; + return &c.fids[slot]; + } + + /// Removes `id` from the index (backward-shift deletion: no tombstones). + fn unlinkFid(c: *Conn, id: u32) void { + var i = c.findBucket(id).?; + var j = i; + while (true) { + j = (j + 1) & index_mask; + const slot = c.index[j]; + if (slot == no_slot) break; + const k = c.fidHome(c.fids[slot].id); + // The entry at j may move into the hole at i unless its home + // lies in the cyclic interval (i, j]. + const stays = if (i <= j) (k > i and k <= j) else (k > i or k <= j); + if (!stays) { + c.index[i] = slot; + i = j; + } + } + c.index[i] = no_slot; + } + + /// Releases everything a fid holds; the slot stays allocated. + fn dropContents(c: *Conn, f: *Fid) void { + if (f.snap) |s| c.slot_used[s] = false; + f.snap = null; + if (f.node == .prov) { + const p = c.shared.providers[f.node.prov.idx]; + if (f.open) if (p.vtable.close) |close| close(p.ctx, f.node.prov.h); + if (f.rclose and f.open) if (p.vtable.remove) |rm| rm(p.ctx, f.node.prov.h) catch {}; + p.vtable.clunk(p.ctx, f.node.prov.h); + } + f.open = false; + f.rclose = false; + } + + fn freeFid(c: *Conn, f: *Fid) void { + c.dropContents(f); + c.unlinkFid(f.id); + const slot: u16 = @intCast((@intFromPtr(f) - @intFromPtr(&c.fids)) / @sizeOf(Fid)); + f.* = .{ .next_free = c.free_head }; + c.free_head = slot; + c.nfids -= 1; + } + + fn resetFids(c: *Conn) void { + for (c.fids[0..c.high_water]) |*f| { + if (f.used) c.dropContents(f); + f.* = .{}; + } + @memset(&c.index, no_slot); + c.free_head = no_slot; + c.high_water = 0; + c.nfids = 0; + } + + /// Releases a provider handle that is not held by any fid. + fn releaseRef(c: *Conn, ref: NodeRef) void { + if (ref == .prov) { + const p = c.shared.providers[ref.prov.idx]; + p.vtable.clunk(p.ctx, ref.prov.h); + } + } + + // -- node helpers -- + + fn provider(c: *Conn, idx: u8) Provider { + return c.shared.providers[idx]; + } + + fn varBase(c: *Conn, idx: u8, node: u32) [*]u8 { + const v = c.shared.vars[idx]; + return @as([*]u8, @ptrCast(v.ptr)) + v.vt.nodes[node].offset; + } + + fn info(c: *Conn, ref: NodeRef) !Info { + switch (ref) { + .static => |idx| { + const n = flat[idx].node; + return .{ + .is_dir = n.isDir(), + .mode = switch (n.kind) { + .dir, .vars => cloud9.dmdir | 0o555, + .ctl => 0o666, + else => 0o444, + }, + .length = switch (n.kind) { + .static => n.content.len, + .ctl => c.shared.ctl_len, + else => 0, + }, + .atime = build_secs, + .mtime = build_secs, + .version = if (n.kind == .ctl) c.shared.ctl_version else 0, + .path = static_tag | idx, + .name = n.name, + }; + }, + .@"var" => |v| { + const sv = c.shared.vars[v.idx]; + const n = sv.vt.nodes[v.node]; + return .{ + .is_dir = n.isDir(), + .mode = if (n.isDir()) cloud9.dmdir | 0o555 else if (n.writable()) 0o644 else 0o444, + .length = switch (n.kind) { + .type_name, .size => n.content.len, + .raw => n.size, + else => 0, + }, + .atime = 0, + .mtime = 0, + .version = 0, + .path = var_tag | (@as(u64, v.idx) << 32) | v.node, + .name = if (v.node == 0) sv.name else n.name, + }; + }, + .prov => |p| { + const pr = c.provider(p.idx); + var st: NodeStat = .{ .mode = 0 }; + try pr.vtable.stat(pr.ctx, p.h, &st); + return provInfo(pr, p.idx, p.h, st); + }, + } + } + + fn provInfo(pr: Provider, idx: u8, h: Provider.Handle, st: NodeStat) Info { + return .{ + .is_dir = st.isDir(), + .mode = st.mode, + .length = if (st.isDir()) 0 else st.length, + .atime = st.atime, + .mtime = st.mtime, + .version = st.version, + .path = (@as(u64, idx) << 56) | ((if (st.path != 0) st.path else h) & handle_mask), + .name = if (h == Provider.root) pr.name else st.name, + }; + } + + /// Copies `st.name` into the connection so the reply cannot dangle. + fn pinName(c: *Conn, st: cloud9.Stat) cloud9.Stat { + var out = st; + const n = @min(st.name.len, c.name_buf.len); + @memcpy(c.name_buf[0..n], st.name[0..n]); + out.name = c.name_buf[0..n]; + return out; + } + + const Looked = struct { ref: NodeRef, info: Info }; + + /// Resolves `name` in the directory `ref`. A returned provider ref + /// is a fresh handle the caller must release or retain. + fn lookup(c: *Conn, ref: NodeRef, name: []const u8) !Looked { + switch (ref) { + .static => |idx| { + const dot = std.mem.eql(u8, name, "."); + const dotdot = std.mem.eql(u8, name, ".."); + var next: NodeRef = undefined; + if (dot) { + next = ref; + } else if (dotdot) { + next = .{ .static = flat[idx].parent }; + } else if (flat[idx].node.kind == .vars) { + const vi = c.shared.findVar(name) orelse return error.NotFound; + next = .{ .@"var" = .{ .idx = vi, .node = 0 } }; + } else if (staticChild(idx, name)) |ci| { + next = .{ .static = ci }; + } else if (idx == 0) { + const pi = c.shared.findProvider(name) orelse return error.NotFound; + next = .{ .prov = .{ .idx = pi, .h = Provider.root } }; + } else return error.NotFound; + return .{ .ref = next, .info = try c.info(next) }; + }, + .@"var" => |v| { + const vt = c.shared.vars[v.idx].vt; + var next = ref; + if (std.mem.eql(u8, name, ".")) { + // unchanged + } else if (std.mem.eql(u8, name, "..")) { + next = if (v.node == 0) .{ .static = vars_idx } else .{ .@"var" = .{ .idx = v.idx, .node = vt.nodes[v.node].parent } }; + } else { + const ci = vt.child(v.node, name) orelse return error.NotFound; + next = .{ .@"var" = .{ .idx = v.idx, .node = ci } }; + } + return .{ .ref = next, .info = try c.info(next) }; + }, + .prov => |p| { + if (p.h == Provider.root and std.mem.eql(u8, name, "..")) { + const next: NodeRef = .{ .static = 0 }; + return .{ .ref = next, .info = try c.info(next) }; + } + const pr = c.provider(p.idx); + const h = try pr.vtable.walk(pr.ctx, p.h, name); + const next: NodeRef = .{ .prov = .{ .idx = p.idx, .h = h } }; + errdefer c.releaseRef(next); + return .{ .ref = next, .info = try c.info(next) }; + }, + } + } + + /// The i-th entry of directory `ref` as a Stat, or null past the end. + /// The name borrows either static memory or the provider's NodeStat. + fn entryStat(c: *Conn, ref: NodeRef, i: usize) !?cloud9.Stat { + switch (ref) { + .static => |idx| { + const f = flat[idx]; + if (f.node.kind == .vars) { + if (i >= c.shared.nvars) return null; + return (try c.info(.{ .@"var" = .{ .idx = @intCast(i), .node = 0 } })).stat(); + } + if (i < f.count) return (try c.info(.{ .static = f.first + @as(u32, @intCast(i)) })).stat(); + if (idx == 0) { + const pi = i - f.count; + if (pi >= c.shared.nprov) return null; + return (try c.info(.{ .prov = .{ .idx = @intCast(pi), .h = Provider.root } })).stat(); + } + return null; + }, + .@"var" => |v| { + const n = c.shared.vars[v.idx].vt.nodes[v.node]; + if (i >= n.count) return null; + return (try c.info(.{ .@"var" = .{ .idx = v.idx, .node = n.first + @as(u32, @intCast(i)) } })).stat(); + }, + .prov => |p| { + const pr = c.provider(p.idx); + var st: NodeStat = .{ .mode = 0 }; + if (!try pr.vtable.list(pr.ctx, p.h, i, &st)) return null; + return provInfo(pr, p.idx, st.handle, st).stat(); + }, + } + } + + // -- snapshots -- + + fn takeSlot(c: *Conn) !u8 { + for (&c.slot_used, 0..) |*u, i| if (!u.*) { + u.* = true; + return @intCast(i); + }; + return error.NoSnapshot; + } + + /// (Re)generates the content of a dynamic file into its slot. + fn generate(c: *Conn, f: *Fid) !void { + const s = f.snap.?; + var w: Writer = .fixed(&c.storage.snapshots[s]); + c.slot_len[s] = 0; + switch (f.node) { + .static => |idx| try flat[idx].node.gen.?(c.shared.ctx, &w), + .@"var" => |v| { + const n = c.shared.vars[v.idx].vt.nodes[v.node]; + const base = c.varBase(v.idx, v.node); + switch (n.kind) { + .value => try n.render.?(base, &w), + .addr => try w.print("0x{x}", .{@intFromPtr(base)}), + else => unreachable, + } + }, + .prov => unreachable, + } + c.slot_len[s] = @intCast(w.buffered().len); + } + + fn isDynamic(c: *Conn, ref: NodeRef) bool { + return switch (ref) { + .static => |idx| flat[idx].node.kind == .dynamic, + .@"var" => |v| switch (c.shared.vars[v.idx].vt.nodes[v.node].kind) { + .value, .addr => true, + else => false, + }, + .prov => false, + }; + } + + // -- request handlers -- + + fn attach(c: *Conn, m: anytype) !cloud9.Msg { + const f = try c.allocFid(m.fid); + f.node = .{ .static = 0 }; + f.is_dir = true; + return .{ .rattach = .{ .qid = (try c.info(f.node)).qid() } }; + } + + fn walk(c: *Conn, m: anytype) !cloud9.Msg { + const f = c.findFid(m.fid) orelse return error.UnknownFid; + if (m.newfid != m.fid and c.findFid(m.newfid) != null) return error.FidInUse; + if (m.newfid != m.fid and c.nfids >= cfg.max_fids) return error.TooManyFids; + if (m.nwname > 0 and f.open) return error.AlreadyOpen; + // Cloning a fid onto itself changes nothing; in particular it must not + // close an open fid or discard generated content. + if (m.nwname == 0 and m.newfid == m.fid) return .{ .rwalk = .{ .nwqid = 0 } }; + // A full Rwalk must fit the negotiated msize; check before binding anything. + if (cloud9.header_len + 2 + cloud9.qid_len * @as(usize, m.nwname) > c.server.msize) return error.ReplyTooLarge; + var cur = f.node; + var cur_is_dir = f.is_dir; + var held = false; // cur is a provider handle obtained here, not the fid's + var reply: cloud9.Msg = .{ .rwalk = .{ .nwqid = 0 } }; + const names = m.wname[0..m.nwname]; + for (names, 0..) |name, i| { + if (!cur_is_dir) { + if (i == 0) return error.NotDir; + break; + } + const next = c.lookup(cur, name) catch |e| { + if (i == 0) return e; + break; + }; + if (held) c.releaseRef(cur); + cur = next.ref; + cur_is_dir = next.info.is_dir; + held = cur == .prov; + reply.rwalk.wqid[i] = next.info.qid(); + reply.rwalk.nwqid += 1; + } + if (reply.rwalk.nwqid != names.len) { + if (held) c.releaseRef(cur); + return reply; + } + if (names.len == 0 and cur == .prov) { + // A clone of a provider handle needs its own reference. + const dup = try c.lookup(cur, "."); + cur = dup.ref; + cur_is_dir = dup.info.is_dir; + held = true; + } + const target = if (m.newfid == m.fid) f else c.allocFid(m.newfid) catch |e| { + if (held) c.releaseRef(cur); + return e; + }; + if (target == f) c.dropContents(f); + target.node = cur; + target.is_dir = cur_is_dir; + return reply; + } + + fn open(c: *Conn, m: anytype) !cloud9.Msg { + const f = c.findFid(m.fid) orelse return error.UnknownFid; + if (f.open) return error.AlreadyOpen; + const acc = m.mode & 3; + const want_write = acc == cloud9.owrite or acc == cloud9.ordwr; + const trunc = m.mode & cloud9.otrunc != 0; + if (f.is_dir and (want_write or trunc)) return error.IsDir; + switch (f.node) { + .static => |idx| switch (flat[idx].node.kind) { + .dir, .vars, .ctl => {}, + .static, .dynamic => if (want_write or trunc) return error.Perm, + }, + .@"var" => |v| { + const n = c.shared.vars[v.idx].vt.nodes[v.node]; + if ((want_write or trunc) and !n.writable()) return error.Perm; + }, + .prov => |p| { + const pr = c.provider(p.idx); + try pr.vtable.open(pr.ctx, p.h, m.mode); + }, + } + const qid = (c.info(f.node) catch |e| { + // The provider's open succeeded but its stat did not: undo the open. + if (f.node == .prov) { + const pr = c.provider(f.node.prov.idx); + if (pr.vtable.close) |close| close(pr.ctx, f.node.prov.h); + } + return e; + }).qid(); + if (c.isDynamic(f.node)) { + f.snap = try c.takeSlot(); + c.generate(f) catch |e| { + c.slot_used[f.snap.?] = false; + f.snap = null; + if (f.node == .prov) unreachable; + return e; + }; + } + f.open = true; + f.mode = m.mode; + f.rclose = m.mode & cloud9.orclose != 0; + f.dir_offset = 0; + f.dir_index = 0; + return .{ .ropen = .{ .qid = qid, .iounit = 0 } }; + } + + fn create(c: *Conn, m: anytype) !cloud9.Msg { + const f = c.findFid(m.fid) orelse return error.UnknownFid; + if (f.open) return error.AlreadyOpen; + const p = switch (f.node) { + .prov => |p| p, + else => return error.Perm, + }; + if (!f.is_dir) return error.NotDir; + const pr = c.provider(p.idx); + const create_fn = pr.vtable.create orelse return error.Perm; + try validName(m.name); + const is_dir = m.perm & cloud9.dmdir != 0; + const acc = m.mode & 3; + if (is_dir and (acc != cloud9.oread or m.mode & cloud9.otrunc != 0)) return error.IsDir; + const h = try create_fn(pr.ctx, p.h, m.name, m.perm, m.mode); + const node: NodeRef = .{ .prov = .{ .idx = p.idx, .h = h } }; + const qid = (c.info(node) catch |e| { + if (pr.vtable.close) |close| close(pr.ctx, h); + pr.vtable.clunk(pr.ctx, h); + return e; + }).qid(); + c.dropContents(f); + f.node = node; + f.is_dir = is_dir; + f.open = true; + f.mode = m.mode; + f.rclose = m.mode & cloud9.orclose != 0; + f.dir_offset = 0; + f.dir_index = 0; + return .{ .rcreate = .{ .qid = qid, .iounit = 0 } }; + } + + fn read(c: *Conn, m: anytype) !cloud9.Msg { + const f = c.findFid(m.fid) orelse return error.UnknownFid; + if (!f.open or (f.mode & 3) == cloud9.owrite) return error.NotOpen; + const count: usize = @min(m.count, c.server.msize -| cloud9.iohdrsz, c.storage.data.len); + if (f.is_dir) return c.readDir(f, m.offset, count); + const data = &c.storage.data; + const src: []const u8 = switch (f.node) { + .static => |idx| blk: { + const n = flat[idx].node; + switch (n.kind) { + .static => break :blk n.content, + .ctl => break :blk c.shared.ctlResult(), + .dynamic => { + if (m.offset == 0) try c.generate(f); + break :blk c.storage.snapshots[f.snap.?][0..c.slot_len[f.snap.?]]; + }, + .dir, .vars => unreachable, + } + }, + .@"var" => |v| blk: { + const n = c.shared.vars[v.idx].vt.nodes[v.node]; + switch (n.kind) { + .type_name, .size => break :blk n.content, + .value, .addr => { + if (m.offset == 0) try c.generate(f); + break :blk c.storage.snapshots[f.snap.?][0..c.slot_len[f.snap.?]]; + }, + .raw => break :blk c.varBase(v.idx, v.node)[0..n.size], + .dir, .fields => unreachable, + } + }, + .prov => |p| { + const pr = c.provider(p.idx); + const n = try pr.vtable.read(pr.ctx, p.h, m.offset, data[0..count]); + return .{ .rread = .{ .data = data[0..@min(n, count)] } }; + }, + }; + if (m.offset >= src.len) return .{ .rread = .{ .data = "" } }; + const off: usize = @intCast(m.offset); + const n = @min(count, src.len - off); + if (f.node == .@"var" and c.shared.vars[f.node.@"var".idx].vt.nodes[f.node.@"var".node].kind == .raw) { + // Copy out of the variable so the reply does not read live memory twice. + @memcpy(data[0..n], src[off..][0..n]); + return .{ .rread = .{ .data = data[0..n] } }; + } + return .{ .rread = .{ .data = src[off..][0..n] } }; + } + + fn readDir(c: *Conn, f: *Fid, offset: u64, count: usize) !cloud9.Msg { + if (offset == 0) { + f.dir_offset = 0; + f.dir_index = 0; + } else if (offset != f.dir_offset) return error.BadOffset; + const data = &c.storage.data; + var used: usize = 0; + var i = f.dir_index; + while (try c.entryStat(f.node, i)) |st| : (i += 1) { + const rec = st.encode(data[used..count]) catch |e| switch (e) { + error.NoSpace => break, + else => return error.Io, + }; + used += rec.len; + } + f.dir_offset += used; + f.dir_index = i; + return .{ .rread = .{ .data = data[0..used] } }; + } + + fn write(c: *Conn, m: anytype) !cloud9.Msg { + const f = c.findFid(m.fid) orelse return error.UnknownFid; + const acc = f.mode & 3; + if (!f.open or (acc != cloud9.owrite and acc != cloud9.ordwr)) return error.NotOpen; + if (f.is_dir) return error.IsDir; + switch (f.node) { + .static => |idx| switch (flat[idx].node.kind) { + .ctl => try c.ctlCommand(m.data), + else => return error.Perm, + }, + .@"var" => |v| { + const n = c.shared.vars[v.idx].vt.nodes[v.node]; + const set = n.set orelse return error.Perm; + try set(c.varBase(v.idx, v.node), m.data); + }, + .prov => |p| { + const pr = c.provider(p.idx); + const n = try pr.vtable.write(pr.ctx, p.h, m.offset, m.data); + return .{ .rwrite = .{ .count = @intCast(@min(n, m.data.len)) } }; + }, + } + return .{ .rwrite = .{ .count = @intCast(m.data.len) } }; + } + + /// Runs `cfg.ctl`; on success its output becomes the ctl result. + /// On failure the previous result (and its qid version) survive: + /// the handler writes into the staging half of `ctl_bufs`. + fn ctlCommand(c: *Conn, line: []const u8) !void { + const s = c.shared; + const next = s.ctl_cur ^ 1; + var w: Writer = .fixed(&s.ctl_bufs[next]); + try cfg.ctl.?(s.ctx, line, &w); + s.ctl_cur = next; + s.ctl_len = @intCast(w.buffered().len); + s.ctl_version +%= 1; + } + + fn clunk(c: *Conn, m: anytype) !cloud9.Msg { + const f = c.findFid(m.fid) orelse return error.UnknownFid; + c.freeFid(f); + return .rclunk; + } + + fn remove(c: *Conn, m: anytype) !cloud9.Msg { + const f = c.findFid(m.fid) orelse return error.UnknownFid; + defer c.freeFid(f); // Tremove always clunks + f.rclose = false; + switch (f.node) { + .prov => |p| { + const pr = c.provider(p.idx); + const rm = pr.vtable.remove orelse return error.Perm; + try rm(pr.ctx, p.h); + }, + else => return error.Perm, + } + return .rremove; + } + + fn stat(c: *Conn, m: anytype) !cloud9.Msg { + const f = c.findFid(m.fid) orelse return error.UnknownFid; + return .{ .rstat = .{ .stat = c.pinName((try c.info(f.node)).stat()) } }; + } + + fn wstat(c: *Conn, m: anytype) !cloud9.Msg { + const f = c.findFid(m.fid) orelse return error.UnknownFid; + const p = switch (f.node) { + .prov => |p| p, + else => return error.Perm, + }; + const pr = c.provider(p.idx); + const ws = pr.vtable.wstat orelse return error.Perm; + const cur = try c.info(f.node); + const st = m.stat; + const q = cur.qid(); + // Fields we cannot change must be "don't care" or unchanged. + if (st.type != 0xFFFF and st.type != 0) return error.Perm; + if (st.dev != 0xFFFF_FFFF and st.dev != 0) return error.Perm; + if (st.qid.type != 0xFF and st.qid.type != q.type) return error.Perm; + if (st.qid.version != 0xFFFF_FFFF and st.qid.version != q.version) return error.Perm; + if (st.qid.path != 0xFFFF_FFFF_FFFF_FFFF and st.qid.path != q.path) return error.Perm; + if (st.uid.len != 0 and !std.mem.eql(u8, st.uid, cfg.name)) return error.Perm; + if (st.gid.len != 0 and !std.mem.eql(u8, st.gid, cfg.name)) return error.Perm; + if (st.muid.len != 0 and !std.mem.eql(u8, st.muid, cfg.name)) return error.Perm; + if (st.name.len != 0 and !std.mem.eql(u8, st.name, cur.name)) { + if (p.h == Provider.root) return error.Perm; + try validName(st.name); + } + if (st.length != 0xFFFF_FFFF_FFFF_FFFF and st.length != cur.length and cur.is_dir) return error.IsDir; + if (st.mode != 0xFFFF_FFFF and (st.mode & cloud9.dmdir) != (cur.mode & cloud9.dmdir)) return error.Perm; + try ws(pr.ctx, p.h, &st); + return .rwstat; + } + }; + + // -- in-memory test harness ------------------------------------------- + + /// Drives a `Conn` with a `cloud9.Client` in memory. Test-only (uses + /// std.testing.allocator); never referenced by non-test code. + pub const Harness = struct { + shared: *Shared, + storage: *Storage, + conn: Conn, + client: cloud9.Client, + cin: []u8, + cout: []u8, + + pub fn init(h: *Harness, shared: *Shared, storage: *Storage) !void { + h.shared = shared; + h.storage = storage; + h.conn = .init(shared, storage, cfg.msize); + h.cin = try testing.allocator.alloc(u8, cfg.msize); + errdefer testing.allocator.free(h.cin); + h.cout = try testing.allocator.alloc(u8, cfg.msize); + errdefer testing.allocator.free(h.cout); + h.client = .init(.{ .in = h.cin, .out = h.cout }); + try h.version(cfg.msize); + _ = try h.ok(.{ .attach = .{ .fid = 0, .uname = "tester" } }); + } + + pub fn deinit(h: *Harness) void { + h.conn.hangup(); + testing.allocator.free(h.cin); + testing.allocator.free(h.cout); + } + + pub fn version(h: *Harness, msize: u32) !void { + const v = try h.rpc(.{ .version = .{ .msize = msize } }); + try testing.expectEqual(msize, v.version.msize); + try testing.expectEqualStrings("9P2000", v.version.version); + } + + /// One round trip; the result borrows the client input buffer until the next call. + pub fn rpc(h: *Harness, req: cloud9.Client.Request) !cloud9.Client.Result { + _ = try h.client.submit(req); + while (true) { + var moved = false; + while (h.client.output().len > 0) { + const k = h.conn.push(h.client.output()); + h.client.wrote(k); + moved = moved or k > 0; + while (try h.conn.step()) {} + while (h.conn.output().len > 0) { + const n = h.client.push(h.conn.output()); + h.conn.wrote(n); + moved = moved or n > 0; + } + if (k == 0) break; + } + while (try h.conn.step()) {} + while (h.conn.output().len > 0) { + const n = h.client.push(h.conn.output()); + h.conn.wrote(n); + moved = moved or n > 0; + } + if (h.client.take()) |done| return done.result; + if (!moved) return error.Stuck; + } + } + + pub fn ok(h: *Harness, req: cloud9.Client.Request) !cloud9.Client.Result { + const r = try h.rpc(req); + if (r == .fail) { + std.debug.print("unexpected Rerror: {s}\n", .{r.fail}); + return error.Rerror; + } + return r; + } + + pub fn expectFail(h: *Harness, req: cloud9.Client.Request, msg: []const u8) !void { + const r = try h.rpc(req); + if (r != .fail) return error.ExpectedRerror; + try testing.expectEqualStrings(msg, r.fail); + } + + pub fn walkTo(h: *Harness, newfid: u32, names: []const []const u8) !void { + const r = try h.ok(.{ .walk = .{ .fid = 0, .newfid = newfid, .names = names } }); + try testing.expectEqual(@as(u16, @intCast(names.len)), r.walk.nwqid); + } + + /// Opens `fid` for reading and reads it whole (across consecutive offsets); caller frees. + pub fn readAll(h: *Harness, fid: u32) ![]u8 { + _ = try h.ok(.{ .open = .{ .fid = fid, .mode = cloud9.oread } }); + return h.readOpen(fid); + } + + pub fn readOpen(h: *Harness, fid: u32) ![]u8 { + var acc: std.ArrayList(u8) = .empty; + errdefer acc.deinit(testing.allocator); + while (true) { + const r = try h.ok(.{ .read = .{ .fid = fid, .offset = acc.items.len, .count = 1024 } }); + if (r.read.len == 0) break; + try acc.appendSlice(testing.allocator, r.read); + } + return acc.toOwnedSlice(testing.allocator); + } + + pub fn readPath(h: *Harness, names: []const []const u8) ![]u8 { + try h.walkTo(99, names); + defer _ = h.rpc(.{ .clunk = .{ .fid = 99 } }) catch {}; + return h.readAll(99); + } + + pub fn writePath(h: *Harness, names: []const []const u8, data: []const u8) !void { + try h.walkTo(98, names); + defer _ = h.rpc(.{ .clunk = .{ .fid = 98 } }) catch {}; + _ = try h.ok(.{ .open = .{ .fid = 98, .mode = cloud9.owrite } }); + const w = try h.ok(.{ .write = .{ .fid = 98, .offset = 0, .data = data } }); + try testing.expectEqual(@as(u32, @intCast(data.len)), w.write); + } + + /// Reads a whole directory in `count`-byte reads at consecutive offsets; returns owned names. + pub fn listDir(h: *Harness, fid: u32, count: u32) ![][]u8 { + var names: std.ArrayList([]u8) = .empty; + errdefer { + for (names.items) |n| testing.allocator.free(n); + names.deinit(testing.allocator); + } + var offset: u64 = 0; + while (true) { + const r = try h.ok(.{ .read = .{ .fid = fid, .offset = offset, .count = count } }); + if (r.read.len == 0) break; + offset += r.read.len; + var rest = r.read; + while (rest.len > 0) { + const n = std.mem.readInt(u16, rest[0..2], .little) + 2; + const st = try cloud9.Stat.decode(rest[0..n]); + try names.append(testing.allocator, try testing.allocator.dupe(u8, st.name)); + rest = rest[n..]; + } + } + return names.toOwnedSlice(testing.allocator); + } + + pub fn listPath(h: *Harness, names: []const []const u8) ![][]u8 { + try h.walkTo(97, names); + defer _ = h.rpc(.{ .clunk = .{ .fid = 97 } }) catch {}; + _ = try h.ok(.{ .open = .{ .fid = 97, .mode = cloud9.oread } }); + return h.listDir(97, 1024); + } + + pub fn freeNames(names: [][]u8) void { + for (names) |n| testing.allocator.free(n); + testing.allocator.free(names); + } + + pub fn hasName(names: []const []const u8, want: []const u8) bool { + for (names) |n| if (std.mem.eql(u8, n, want)) return true; + return false; + } + }; + }; +} + +// --------------------------------------------------------------------------- +// Tests +// --------------------------------------------------------------------------- + +const testing = std.testing; + +const TestBuild = struct { + pub const zig_version: []const u8 = builtin.zig_version_string; + pub const target: []const u8 = "test-target"; + pub const optimize: []const u8 = "Debug"; + pub const time: []const u8 = "2023-11-14T22:13:20Z"; + pub const change: []const u8 = "abc123"; +}; + +const Layout = struct { a: u8, b: u32, c: u64 }; +const Decls = struct { + pub const one = 1; + pub const two = 2; + pub fn three() void {} +}; + +/// The context every generator and the ctl handler receive in tests. +const TestCtx = struct { + calls: u32 = 0, + ctl_state: i64 = 0, +}; + +const TestFns = struct { + pub fn counter(ctx: *anyopaque, w: *Writer) anyerror!void { + const t: *TestCtx = @ptrCast(@alignCast(ctx)); + t.calls += 1; + try w.print("{d}", .{t.calls}); + } + pub fn fib30(_: *anyopaque, w: *Writer) anyerror!void { + try w.print("{d}", .{fib(30)}); + } + pub fn failing(_: *anyopaque, _: *Writer) anyerror!void { + return error.BadCommand; + } + pub fn huge(_: *anyopaque, w: *Writer) anyerror!void { + try w.splatByteAll('x', 1 << 20); + } +}; + +const TestRuntime = struct { + pub fn pid(_: *anyopaque, w: *Writer) anyerror!void { + try w.writeAll("4242"); + } +}; + +fn fib(n: u32) u64 { + if (n == 0) return 0; + var a: u64 = 0; + var b: u64 = 1; + for (1..n) |_| { + const c = a + b; + a = b; + b = c; + } + return b; +} + +fn testCtl(ctx: *anyopaque, cmd: []const u8, out: *Writer) anyerror!void { + const t: *TestCtx = @ptrCast(@alignCast(ctx)); + const line = std.mem.trim(u8, cmd, " \t\r\n\x00"); + var it = std.mem.tokenizeScalar(u8, line, ' '); + const verb = it.next() orelse return error.BadCommand; + if (std.mem.eql(u8, verb, "echo")) { + try out.writeAll(std.mem.trimStart(u8, line[verb.len..], " \t")); + } else if (std.mem.eql(u8, verb, "add")) { + const a = std.fmt.parseInt(i64, it.next() orelse return error.BadCommand, 10) catch return error.BadCommand; + const b = std.fmt.parseInt(i64, it.next() orelse return error.BadCommand, 10) catch return error.BadCommand; + t.ctl_state = a +% b; + try out.print("{d}", .{t.ctl_state}); + } else if (std.mem.eql(u8, verb, "partial")) { + try out.writeAll("half-written"); + return error.BadCommand; + } else return error.BadCommand; +} + +const test_cfg: Config = .{ + .name = "tester", + .build = TestBuild, + .types = &.{ Layout, cloud9.Qid }, + .decls_of = Decls, + .fns = TestFns, + .runtime = TestRuntime, + .ctl = &testCtl, + .msize = 8192, + .max_fids = 8, + .max_providers = 2, + .max_vars = 4, + .snapshot_slots = 2, + .snapshot_bytes = 512, +}; + +const TS = Server(test_cfg); + +/// A small in-memory provider: /prov/{hello,dir/{inner}} with create/remove/wstat, +/// counting every handle reference so tests can check clunk discipline. +const TestProv = struct { + const max_nodes = 16; + const Entry = struct { + used: bool = false, + name: [max_name]u8 = undefined, + name_len: u8 = 0, + parent: u32 = 0, + is_dir: bool = false, + mode: u32 = 0o644, + data: [64]u8 = undefined, + len: usize = 0, + refs: u32 = 0, + opens: u32 = 0, + mtime: u32 = 0, + + fn nameSlice(e: *const Entry) []const u8 { + return e.name[0..e.name_len]; + } + }; + nodes: [max_nodes]Entry = @splat(.{}), + total_refs: u32 = 0, + clunks: u32 = 0, + fail_io: bool = false, + fail_stat: bool = false, + + fn init() TestProv { + var p: TestProv = .{}; + p.nodes[0] = .{ .used = true, .is_dir = true, .mode = cloud9.dmdir | 0o755 }; + _ = p.add(0, "hello", false, 0o644); + p.nodes[1].len = 5; + @memcpy(p.nodes[1].data[0..5], "hello"); + const d = p.add(0, "dir", true, cloud9.dmdir | 0o755); + _ = p.add(d, "inner", false, 0o600); + _ = p.add(0, "locked", false, 0o000); + return p; + } + + fn add(p: *TestProv, parent: u32, name: []const u8, is_dir: bool, mode: u32) u32 { + for (&p.nodes, 0..) |*e, i| if (!e.used) { + e.* = .{ .used = true, .parent = parent, .is_dir = is_dir, .mode = mode }; + @memcpy(e.name[0..name.len], name); + e.name_len = @intCast(name.len); + return @intCast(i); + }; + unreachable; + } + + fn self(ctx: *anyopaque) *TestProv { + return @ptrCast(@alignCast(ctx)); + } + + fn node(p: *TestProv, h: Provider.Handle) Provider.Error!*Entry { + if (h >= max_nodes or !p.nodes[h].used) return error.NotFound; + return &p.nodes[h]; + } + + fn retain(p: *TestProv, h: Provider.Handle) Provider.Handle { + if (h != 0) { + p.nodes[h].refs += 1; + p.total_refs += 1; + } + return h; + } + + fn walk(ctx: *anyopaque, parent: Provider.Handle, name: []const u8) Provider.Error!Provider.Handle { + const p = self(ctx); + if (p.fail_io) return error.Io; + const d = try p.node(parent); + if (std.mem.eql(u8, name, ".")) return p.retain(parent); + if (!d.is_dir) return error.NotDir; + if (std.mem.eql(u8, name, "..")) return p.retain(d.parent); + for (p.nodes[0..], 0..) |*e, i| { + if (e.used and e.parent == parent and i != 0 and std.mem.eql(u8, e.nameSlice(), name)) return p.retain(@intCast(i)); + } + return error.NotFound; + } + + fn fillStat(e: *const Entry, h: Provider.Handle, out: *NodeStat) void { + out.* = .{ .mode = e.mode, .length = e.len, .mtime = e.mtime, .name = e.nameSlice(), .handle = h }; + } + + fn stat(ctx: *anyopaque, h: Provider.Handle, out: *NodeStat) Provider.Error!void { + const p = self(ctx); + if (p.fail_stat) return error.Io; + fillStat(try p.node(h), h, out); + } + + fn list(ctx: *anyopaque, dir: Provider.Handle, index: usize, out: *NodeStat) Provider.Error!bool { + const p = self(ctx); + const d = try p.node(dir); + if (!d.is_dir) return error.NotDir; + var k: usize = 0; + for (p.nodes[0..], 0..) |*e, i| { + if (!e.used or e.parent != dir or i == 0) continue; + if (k == index) { + fillStat(e, @intCast(i), out); + return true; + } + k += 1; + } + return false; + } + + fn open(ctx: *anyopaque, h: Provider.Handle, mode: u8) Provider.Error!void { + const p = self(ctx); + const e = try p.node(h); + const acc = mode & 3; + if (acc != cloud9.owrite and e.mode & 0o400 == 0) return error.Perm; + if (acc != cloud9.oread and e.mode & 0o200 == 0) return error.Perm; + if (mode & cloud9.otrunc != 0) e.len = 0; + e.opens += 1; + } + + fn close(ctx: *anyopaque, h: Provider.Handle) void { + const p = self(ctx); + p.nodes[h].opens -= 1; + } + + fn read(ctx: *anyopaque, h: Provider.Handle, offset: u64, buf: []u8) Provider.Error!usize { + const p = self(ctx); + const e = try p.node(h); + if (offset >= e.len) return 0; + const n = @min(buf.len, e.len - @as(usize, @intCast(offset))); + @memcpy(buf[0..n], e.data[@intCast(offset)..][0..n]); + return n; + } + + fn write(ctx: *anyopaque, h: Provider.Handle, offset: u64, data: []const u8) Provider.Error!usize { + const p = self(ctx); + const e = try p.node(h); + if (offset + data.len > e.data.len) return error.NoSpace; + const off: usize = @intCast(offset); + @memcpy(e.data[off..][0..data.len], data); + e.len = @max(e.len, off + data.len); + e.mtime += 1; + return data.len; + } + + fn create(ctx: *anyopaque, dir: Provider.Handle, name: []const u8, perm: u32, mode: u8) Provider.Error!Provider.Handle { + const p = self(ctx); + const d = try p.node(dir); + if (!d.is_dir) return error.NotDir; + for (p.nodes[0..]) |*e| if (e.used and e.parent == dir and std.mem.eql(u8, e.nameSlice(), name)) return error.Exists; + var free: ?u32 = null; + for (p.nodes[0..], 0..) |*e, i| if (!e.used) { + free = @intCast(i); + break; + }; + const idx = free orelse return error.NoSpace; + const h = p.add(@intCast(dir), name, perm & cloud9.dmdir != 0, perm); + std.debug.assert(h == idx); + p.nodes[h].opens = 1; + _ = mode; + return p.retain(h); + } + + fn remove(ctx: *anyopaque, h: Provider.Handle) Provider.Error!void { + const p = self(ctx); + const e = try p.node(h); + if (h == 0) return error.Perm; + for (p.nodes[0..]) |*c| if (c.used and c.parent == h) return error.NotEmpty; + e.used = false; // refs still keep the slot "alive" for clunk accounting + e.used = true; + e.parent = std.math.maxInt(u32); // unlinked + } + + fn wstat(ctx: *anyopaque, h: Provider.Handle, st: *const cloud9.Stat) Provider.Error!void { + const p = self(ctx); + const e = try p.node(h); + if (st.name.len != 0) { + @memcpy(e.name[0..st.name.len], st.name); + e.name_len = @intCast(st.name.len); + } + if (st.length != 0xFFFF_FFFF_FFFF_FFFF) { + if (st.length > e.data.len) return error.NoSpace; + e.len = @intCast(st.length); + } + if (st.mode != 0xFFFF_FFFF) e.mode = st.mode; + if (st.mtime != 0xFFFF_FFFF) e.mtime = st.mtime; + } + + fn clunk(ctx: *anyopaque, h: Provider.Handle) void { + const p = self(ctx); + p.clunks += 1; + if (h != 0) { + p.nodes[h].refs -= 1; + p.total_refs -= 1; + } + } + + const vtable: Provider.VTable = .{ + .walk = &walk, + .stat = &stat, + .list = &list, + .open = &open, + .read = &read, + .write = &write, + .create = &create, + .remove = &remove, + .wstat = &wstat, + .close = &close, + .clunk = &clunk, + }; + + fn provider(p: *TestProv) Provider { + return .{ .name = "prov", .ctx = p, .vtable = &vtable }; + } +}; + +const Inner = struct { x: f32 }; +const Exposed = struct { a: u32, b: bool, name: []const u8, inner: Inner }; + +/// Everything a core test needs, in one place; `harness.init` runs version+attach. +const Fixture = struct { + ctx: TestCtx = .{}, + shared: TS.Shared = undefined, + storage: TS.Storage = undefined, + prov: TestProv = undefined, + exposed: Exposed = .{ .a = 1, .b = true, .name = "hello", .inner = .{ .x = 0.5 } }, + counter: u64 = 7, + h: TS.Harness = undefined, + + fn init(x: *Fixture) !void { + x.shared = .init(&x.ctx); + x.prov = TestProv.init(); + try x.shared.addProvider(x.prov.provider()); + try x.shared.expose("state", &x.exposed); + try x.shared.expose("counter", &x.counter); + try x.h.init(&x.shared, &x.storage); + } + + fn deinit(x: *Fixture) void { + x.h.deinit(); + } +}; + +test "README, /build and the static tree read as expected" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + const readme = try x.h.readPath(&.{"README"}); + defer testing.allocator.free(readme); + try testing.expect(std.mem.startsWith(u8, readme, "tester: a 9P2000 introspection server")); + const zv = try x.h.readPath(&.{ "build", "zig_version" }); + defer testing.allocator.free(zv); + try testing.expectEqualStrings(builtin.zig_version_string, zv); + const ch = try x.h.readPath(&.{ "build", "change" }); + defer testing.allocator.free(ch); + try testing.expectEqualStrings("abc123", ch); + try testing.expectEqual(@as(u32, 1_700_000_000), TS.build_secs); + const names = try x.h.listPath(&.{}); + defer TS.Harness.freeNames(names); + for ([_][]const u8{ "README", "build", "comptime", "runtime", "ctl", "vars", "prov" }) |n| try testing.expect(TS.Harness.hasName(names, n)); + try testing.expectEqual(@as(usize, 7), names.len); + // static files are read-only; the static tree admits no creates or removes + try x.h.walkTo(1, &.{ "build", "target" }); + try x.h.expectFail(.{ .open = .{ .fid = 1, .mode = cloud9.owrite } }, "permission denied"); + try x.h.expectFail(.{ .remove = .{ .fid = 1 } }, "permission denied"); + try x.h.walkTo(2, &.{"build"}); + try x.h.expectFail(.{ .create = .{ .fid = 2, .name = "nope", .perm = 0o644, .mode = cloud9.owrite } }, "permission denied"); + try x.h.expectFail(.{ .wstat = .{ .fid = 2, .stat = stat_dontcare } }, "permission denied"); + const st = try x.h.ok(.{ .stat = .{ .fid = 2 } }); + try testing.expectEqualStrings("build", st.stat.name); + try testing.expectEqualStrings("tester", st.stat.uid); + try testing.expect(st.stat.qid.type & cloud9.qtdir != 0); + try testing.expectEqual(TS.build_secs, st.stat.mtime); + try testing.expectEqual(@as(u32, 0), parseIso8601("1970-01-01T00:00:00Z").?); + try testing.expectEqual(@as(?u32, null), parseIso8601("unknown")); +} + +test "comptime/types fields carry @offsetOf and comptime/decls lists pub decls" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + const names = try x.h.listPath(&.{ "comptime", "types" }); + defer TS.Harness.freeNames(names); + try testing.expectEqual(@as(usize, 2), names.len); + try testing.expect(TS.Harness.hasName(names, "Layout")); + try testing.expect(TS.Harness.hasName(names, "Qid")); + const fields = try x.h.readPath(&.{ "comptime", "types", "Layout", "fields" }); + defer testing.allocator.free(fields); + var expect_buf: [128]u8 = undefined; + const expect = try std.fmt.bufPrint(&expect_buf, "a: u8 @{d}\nb: u32 @{d}\nc: u64 @{d}\n", .{ @offsetOf(Layout, "a"), @offsetOf(Layout, "b"), @offsetOf(Layout, "c") }); + try testing.expectEqualStrings(expect, fields); + const size = try x.h.readPath(&.{ "comptime", "types", "Layout", "size" }); + defer testing.allocator.free(size); + try testing.expectEqualStrings(std.fmt.comptimePrint("{d}", .{@sizeOf(Layout)}), size); + const name = try x.h.readPath(&.{ "comptime", "types", "Qid", "name" }); + defer testing.allocator.free(name); + try testing.expectEqualStrings(@typeName(cloud9.Qid), name); + const decls = try x.h.readPath(&.{ "comptime", "decls" }); + defer testing.allocator.free(decls); + try testing.expectEqualStrings("one\ntwo\nthree\n", decls); +} + +test "runtime/fn calls the function at open and at each read from offset 0" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + const names = try x.h.listPath(&.{ "runtime", "fn" }); + defer TS.Harness.freeNames(names); + try testing.expectEqual(@typeInfo(TestFns).@"struct".decls.len, names.len); + const fib_text = try x.h.readPath(&.{ "runtime", "fn", "fib30" }); + defer testing.allocator.free(fib_text); + try testing.expectEqualStrings("832040", fib_text); + const pid = try x.h.readPath(&.{ "runtime", "pid" }); + defer testing.allocator.free(pid); + try testing.expectEqualStrings("4242", pid); + // the generator runs at open, then again at each read from offset 0, not at offset > 0 + try x.h.walkTo(1, &.{ "runtime", "fn", "counter" }); + _ = try x.h.ok(.{ .open = .{ .fid = 1, .mode = cloud9.oread } }); + try testing.expectEqual(@as(u32, 1), x.ctx.calls); + const r1 = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = 100 } }); + try testing.expectEqualStrings("2", r1.read); + const r2 = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 1, .count = 100 } }); + try testing.expectEqualStrings("", r2.read); + try testing.expectEqual(@as(u32, 2), x.ctx.calls); + // stat of a dynamic file reports length 0 + const st = try x.h.ok(.{ .stat = .{ .fid = 1 } }); + try testing.expectEqual(@as(u64, 0), st.stat.length); + try testing.expectEqual(@as(u32, 0o444), st.stat.mode); + // a generator error is the file's Rerror; a generator that overflows the slot too + try x.h.walkTo(2, &.{ "runtime", "fn", "failing" }); + try x.h.expectFail(.{ .open = .{ .fid = 2, .mode = cloud9.oread } }, "bad command"); + try x.h.walkTo(3, &.{ "runtime", "fn", "huge" }); + try x.h.expectFail(.{ .open = .{ .fid = 3, .mode = cloud9.oread } }, "no space in buffer"); + // a failed open frees its slot: two more dynamic opens still succeed + try x.h.walkTo(4, &.{ "runtime", "fn", "fib30" }); + _ = try x.h.ok(.{ .open = .{ .fid = 4, .mode = cloud9.oread } }); + _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); + _ = try x.h.ok(.{ .walk = .{ .fid = 4, .newfid = 5, .names = &.{} } }); + _ = try x.h.ok(.{ .open = .{ .fid = 5, .mode = cloud9.oread } }); +} + +test "snapshot slot exhaustion is an Rerror and clunk frees the slot" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + try x.h.walkTo(1, &.{ "runtime", "fn", "fib30" }); + try x.h.walkTo(2, &.{ "runtime", "fn", "fib30" }); + try x.h.walkTo(3, &.{ "vars", "counter", "value" }); + _ = try x.h.ok(.{ .open = .{ .fid = 1, .mode = cloud9.oread } }); + _ = try x.h.ok(.{ .open = .{ .fid = 2, .mode = cloud9.oread } }); + try x.h.expectFail(.{ .open = .{ .fid = 3, .mode = cloud9.oread } }, "too many open dynamic files"); + // static and provider files need no slot + const t = try x.h.readPath(&.{ "vars", "counter", "type" }); + defer testing.allocator.free(t); + try testing.expectEqualStrings("u64", t); + _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); + _ = try x.h.ok(.{ .open = .{ .fid = 3, .mode = cloud9.oread } }); + const r = try x.h.ok(.{ .read = .{ .fid = 3, .offset = 0, .count = 100 } }); + try testing.expectEqualStrings("7", r.read); + // cloning an open fid onto itself keeps it open and its content + const w = try x.h.ok(.{ .walk = .{ .fid = 3, .newfid = 3, .names = &.{} } }); + try testing.expectEqual(@as(u16, 0), w.walk.nwqid); + const r2 = try x.h.ok(.{ .read = .{ .fid = 3, .offset = 0, .count = 100 } }); + try testing.expectEqualStrings("7", r2.read); + try x.h.expectFail(.{ .walk = .{ .fid = 3, .newfid = 4, .names = &.{".."} } }, "file already open"); + _ = try x.h.ok(.{ .walk = .{ .fid = 3, .newfid = 4, .names = &.{} } }); + try x.h.expectFail(.{ .read = .{ .fid = 4, .offset = 0, .count = 100 } }, "file not open"); +} + +test "ctl round trip" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + try x.h.walkTo(1, &.{"ctl"}); + _ = try x.h.ok(.{ .open = .{ .fid = 1, .mode = cloud9.ordwr } }); + const w = try x.h.ok(.{ .write = .{ .fid = 1, .offset = 0, .data = "add 2 3\n" } }); + try testing.expectEqual(@as(u32, 8), w.write); + const r = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = 100 } }); + try testing.expectEqualStrings("5", r.read); + try testing.expectEqual(@as(i64, 5), x.ctx.ctl_state); + const st = try x.h.ok(.{ .stat = .{ .fid = 1 } }); + try testing.expectEqual(@as(u64, 1), st.stat.length); + try testing.expectEqualStrings("ctl", st.stat.name); + try testing.expectEqual(@as(u32, 0o666), st.stat.mode); + const v1 = st.stat.qid.version; + _ = try x.h.ok(.{ .write = .{ .fid = 1, .offset = 0, .data = "echo hello world" } }); + const r2 = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = 100 } }); + try testing.expectEqualStrings("hello world", r2.read); + const r3 = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 6, .count = 100 } }); + try testing.expectEqualStrings("world", r3.read); + try testing.expect((try x.h.ok(.{ .stat = .{ .fid = 1 } })).stat.qid.version != v1); + const v2 = (try x.h.ok(.{ .stat = .{ .fid = 1 } })).stat.qid.version; + try x.h.expectFail(.{ .write = .{ .fid = 1, .offset = 0, .data = "frobnicate" } }, "bad command"); + // a failed command leaves the previous result, length and version in place + const r4 = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = 100 } }); + try testing.expectEqualStrings("hello world", r4.read); + const st4 = try x.h.ok(.{ .stat = .{ .fid = 1 } }); + try testing.expectEqual(@as(u64, 11), st4.stat.length); + try testing.expectEqual(v2, st4.stat.qid.version); + // even when the handler wrote part of a result before failing + try x.h.expectFail(.{ .write = .{ .fid = 1, .offset = 0, .data = "partial" } }, "bad command"); + const r5 = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = 100 } }); + try testing.expectEqualStrings("hello world", r5.read); + try testing.expectEqualStrings("hello world", x.shared.ctlResult()); + // the empty result is a legitimate result too + _ = try x.h.ok(.{ .write = .{ .fid = 1, .offset = 0, .data = "echo" } }); + try testing.expectEqualStrings("", (try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = 100 } })).read); + try testing.expectEqual(@as(u64, 0), (try x.h.ok(.{ .stat = .{ .fid = 1 } })).stat.length); + // Tversion resets the ctl fid like any other + try x.h.version(4096); + try x.h.expectFail(.{ .clunk = .{ .fid = 1 } }, "unknown fid"); +} + +test "auth is not required and flush is answered" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + try x.h.expectFail(.{ .auth = .{ .afid = 5, .uname = "tester" } }, "authentication not required"); + const f = try x.h.ok(.{ .flush = .{ .oldtag = 1 } }); + try testing.expect(f == .flush); + try x.h.expectFail(.{ .attach = .{ .fid = 0, .uname = "tester" } }, "fid in use"); +} + +test "vars: value/type/size/addr/raw, fields and writes" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + const names = try x.h.listPath(&.{"vars"}); + defer TS.Harness.freeNames(names); + try testing.expectEqual(@as(usize, 2), names.len); + try testing.expectEqualStrings("state", names[0]); + const entries = try x.h.listPath(&.{ "vars", "state" }); + defer TS.Harness.freeNames(entries); + for ([_][]const u8{ "value", "type", "size", "addr", "raw", "f" }) |n| try testing.expect(TS.Harness.hasName(entries, n)); + try testing.expectEqual(@as(usize, 6), entries.len); + const value = try x.h.readPath(&.{ "vars", "state", "value" }); + defer testing.allocator.free(value); + try testing.expectEqualStrings("a: 1\nb: true\nname: \"hello\"\ninner:\n x: 0.5\n", value); + const tn = try x.h.readPath(&.{ "vars", "state", "type" }); + defer testing.allocator.free(tn); + try testing.expectEqualStrings(@typeName(Exposed), tn); + const size = try x.h.readPath(&.{ "vars", "state", "size" }); + defer testing.allocator.free(size); + try testing.expectEqualStrings(std.fmt.comptimePrint("{d}", .{@sizeOf(Exposed)}), size); + const addr = try x.h.readPath(&.{ "vars", "state", "addr" }); + defer testing.allocator.free(addr); + var addr_buf: [32]u8 = undefined; + try testing.expectEqualStrings(try std.fmt.bufPrint(&addr_buf, "0x{x}", .{@intFromPtr(&x.exposed)}), addr); + const raw = try x.h.readPath(&.{ "vars", "state", "raw" }); + defer testing.allocator.free(raw); + try testing.expectEqualSlices(u8, std.mem.asBytes(&x.exposed), raw); + try x.h.walkTo(1, &.{ "vars", "state", "raw" }); + const raw_st = try x.h.ok(.{ .stat = .{ .fid = 1 } }); + try testing.expectEqual(@as(u64, @sizeOf(Exposed)), raw_st.stat.length); + try testing.expectEqual(@as(u32, 0o444), raw_st.stat.mode); + _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); + // fields + const fnames = try x.h.listPath(&.{ "vars", "state", "f" }); + defer TS.Harness.freeNames(fnames); + try testing.expectEqual(@as(usize, 4), fnames.len); + const a_value = try x.h.readPath(&.{ "vars", "state", "f", "a", "value" }); + defer testing.allocator.free(a_value); + try testing.expectEqualStrings("1", a_value); + const a_type = try x.h.readPath(&.{ "vars", "state", "f", "a", "type" }); + defer testing.allocator.free(a_type); + try testing.expectEqualStrings("u32", a_type); + const xv = try x.h.readPath(&.{ "vars", "state", "f", "inner", "f", "x", "value" }); + defer testing.allocator.free(xv); + try testing.expectEqualStrings("0.5", xv); + const b_raw = try x.h.readPath(&.{ "vars", "state", "f", "b", "raw" }); + defer testing.allocator.free(b_raw); + try testing.expectEqualSlices(u8, &.{1}, b_raw); + // writes + try x.h.writePath(&.{ "vars", "state", "f", "a", "value" }, "42"); + try testing.expectEqual(@as(u32, 42), x.exposed.a); + try x.h.writePath(&.{ "vars", "state", "f", "b", "value" }, "false\n"); + try testing.expect(!x.exposed.b); + try x.h.writePath(&.{ "vars", "state", "f", "inner", "f", "x", "value" }, "2.25"); + try testing.expectEqual(@as(f32, 2.25), x.exposed.inner.x); + try x.h.writePath(&.{ "vars", "counter", "value" }, "0x10"); + try testing.expectEqual(@as(u64, 16), x.counter); + try x.h.walkTo(2, &.{ "vars", "state", "f", "a", "value" }); + _ = try x.h.ok(.{ .open = .{ .fid = 2, .mode = cloud9.ordwr } }); + try x.h.expectFail(.{ .write = .{ .fid = 2, .offset = 0, .data = "abc" } }, "bad value"); + const rd = try x.h.ok(.{ .read = .{ .fid = 2, .offset = 0, .count = 100 } }); + try testing.expectEqualStrings("42", rd.read); + const a_st = try x.h.ok(.{ .stat = .{ .fid = 2 } }); + try testing.expectEqual(@as(u32, 0o644), a_st.stat.mode); + try testing.expectEqualStrings("value", a_st.stat.name); + // non-scalar values, type/size/addr/raw and directories are read-only + try x.h.walkTo(3, &.{ "vars", "state", "value" }); + try x.h.expectFail(.{ .open = .{ .fid = 3, .mode = cloud9.owrite } }, "permission denied"); + _ = try x.h.ok(.{ .clunk = .{ .fid = 3 } }); + try x.h.walkTo(4, &.{ "vars", "state", "f", "name", "value" }); + try x.h.expectFail(.{ .open = .{ .fid = 4, .mode = cloud9.owrite } }, "permission denied"); + _ = try x.h.ok(.{ .clunk = .{ .fid = 4 } }); + try x.h.walkTo(5, &.{ "vars", "state" }); + try x.h.expectFail(.{ .open = .{ .fid = 5, .mode = cloud9.owrite } }, "is a directory"); + try x.h.expectFail(.{ .create = .{ .fid = 5, .name = "z", .perm = 0o644, .mode = cloud9.owrite } }, "permission denied"); + // .. climbs back out of the var tree; unknown names fail + const up = try x.h.ok(.{ .walk = .{ .fid = 5, .newfid = 6, .names = &.{ "f", "inner", "..", "..", "..", "..", "README" } } }); + try testing.expectEqual(@as(u16, 7), up.walk.nwqid); + _ = try x.h.ok(.{ .clunk = .{ .fid = 6 } }); + try x.h.walkTo(7, &.{"vars"}); + try x.h.expectFail(.{ .walk = .{ .fid = 7, .newfid = 8, .names = &.{"nope"} } }, "file does not exist"); + try x.h.expectFail(.{ .walk = .{ .fid = 2, .newfid = 8, .names = &.{"x"} } }, "file already open"); + _ = try x.h.ok(.{ .clunk = .{ .fid = 2 } }); + try x.h.walkTo(2, &.{ "vars", "state", "f", "a", "value" }); + try x.h.expectFail(.{ .walk = .{ .fid = 2, .newfid = 8, .names = &.{"x"} } }, "not a directory"); +} + +test "provider: walk/list/stat/open/read/write/create/remove/wstat/clunk and error mapping" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + const names = try x.h.listPath(&.{"prov"}); + defer TS.Harness.freeNames(names); + try testing.expectEqual(@as(usize, 3), names.len); + try testing.expect(TS.Harness.hasName(names, "hello") and TS.Harness.hasName(names, "dir") and TS.Harness.hasName(names, "locked")); + const hello = try x.h.readPath(&.{ "prov", "hello" }); + defer testing.allocator.free(hello); + try testing.expectEqualStrings("hello", hello); + try testing.expectEqual(@as(u32, 0), x.prov.total_refs); // every temp handle was clunked + // stat and qid scheme + try x.h.walkTo(1, &.{ "prov", "dir", "inner" }); + const st = try x.h.ok(.{ .stat = .{ .fid = 1 } }); + try testing.expectEqualStrings("inner", st.stat.name); + try testing.expectEqual(@as(u32, 0o600), st.stat.mode); + try testing.expectEqual(@as(u64, 3), st.stat.qid.path); // provider 0, handle 3 + try testing.expectEqualStrings("tester", st.stat.gid); + try x.h.walkTo(2, &.{"prov"}); + const root_st = try x.h.ok(.{ .stat = .{ .fid = 2 } }); + try testing.expectEqualStrings("prov", root_st.stat.name); + try testing.expect(root_st.stat.qid.type & cloud9.qtdir != 0); + try testing.expectEqual(@as(u64, 0), root_st.stat.qid.path); + try testing.expectEqual(@as(u32, 1), x.prov.total_refs); // fid 1 holds inner; fid 2 holds root (unref'd) + // write then read back; opens are tracked through close + _ = try x.h.ok(.{ .open = .{ .fid = 1, .mode = cloud9.ordwr } }); + try testing.expectEqual(@as(u32, 1), x.prov.nodes[3].opens); + _ = try x.h.ok(.{ .write = .{ .fid = 1, .offset = 0, .data = "abc" } }); + _ = try x.h.ok(.{ .write = .{ .fid = 1, .offset = 3, .data = "def" } }); + const r = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 1, .count = 100 } }); + try testing.expectEqualStrings("bcdef", r.read); + try x.h.expectFail(.{ .write = .{ .fid = 1, .offset = 100, .data = "z" } }, "no space left on device"); + _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); + try testing.expectEqual(@as(u32, 0), x.prov.nodes[3].opens); + try testing.expectEqual(@as(u32, 0), x.prov.nodes[3].refs); + // permission and kind errors come from the provider + try x.h.walkTo(3, &.{ "prov", "locked" }); + try x.h.expectFail(.{ .open = .{ .fid = 3, .mode = cloud9.oread } }, "permission denied"); + try x.h.walkTo(4, &.{ "prov", "hello" }); + try x.h.expectFail(.{ .walk = .{ .fid = 4, .newfid = 5, .names = &.{"x"} } }, "not a directory"); + try x.h.expectFail(.{ .walk = .{ .fid = 2, .newfid = 5, .names = &.{"missing"} } }, "file does not exist"); + try x.h.expectFail(.{ .open = .{ .fid = 2, .mode = cloud9.owrite } }, "is a directory"); + // create in a provider directory: the fid becomes the new open file + const cr = try x.h.ok(.{ .create = .{ .fid = 2, .name = "new", .perm = 0o644, .mode = cloud9.ordwr } }); + try testing.expectEqual(cloud9.qtfile, cr.create.qid.type); + _ = try x.h.ok(.{ .write = .{ .fid = 2, .offset = 0, .data = "fresh" } }); + const rr = try x.h.ok(.{ .read = .{ .fid = 2, .offset = 0, .count = 100 } }); + try testing.expectEqualStrings("fresh", rr.read); + try x.h.walkTo(6, &.{"prov"}); + try x.h.expectFail(.{ .create = .{ .fid = 6, .name = "new", .perm = 0o644, .mode = cloud9.oread } }, "file already exists"); + const long_name = [_]u8{'n'} ** (max_name + 1); + try x.h.expectFail(.{ .create = .{ .fid = 6, .name = &long_name, .perm = 0o644, .mode = cloud9.oread } }, "bad file name"); + try x.h.expectFail(.{ .create = .{ .fid = 6, .name = "d", .perm = cloud9.dmdir | 0o755, .mode = cloud9.owrite } }, "is a directory"); + const dr = try x.h.ok(.{ .create = .{ .fid = 6, .name = "d", .perm = cloud9.dmdir | 0o755, .mode = cloud9.oread } }); + try testing.expectEqual(cloud9.qtdir, dr.create.qid.type); + // wstat: rename, truncate, mode, mtime; immutable fields are refused + var ws = stat_dontcare; + ws.name = "renamed"; + ws.length = 2; + ws.mode = 0o600; + ws.mtime = 99; + _ = try x.h.ok(.{ .wstat = .{ .fid = 2, .stat = ws } }); + const st2 = try x.h.ok(.{ .stat = .{ .fid = 2 } }); + try testing.expectEqualStrings("renamed", st2.stat.name); + try testing.expectEqual(@as(u64, 2), st2.stat.length); + try testing.expectEqual(@as(u32, 0o600), st2.stat.mode); + try testing.expectEqual(@as(u32, 99), st2.stat.mtime); + ws = stat_dontcare; + ws.uid = "someone-else"; + try x.h.expectFail(.{ .wstat = .{ .fid = 2, .stat = ws } }, "permission denied"); + ws = stat_dontcare; + ws.mode = cloud9.dmdir | 0o755; + try x.h.expectFail(.{ .wstat = .{ .fid = 2, .stat = ws } }, "permission denied"); + ws = stat_dontcare; + ws.name = "bad/name"; + try x.h.expectFail(.{ .wstat = .{ .fid = 2, .stat = ws } }, "bad file name"); + ws.name = ".."; + try x.h.expectFail(.{ .wstat = .{ .fid = 2, .stat = ws } }, "bad file name"); + ws = stat_dontcare; + ws.length = 5; + try x.h.walkTo(7, &.{ "prov", "d" }); + try x.h.expectFail(.{ .wstat = .{ .fid = 7, .stat = ws } }, "is a directory"); + ws = stat_dontcare; + ws.name = "root2"; // the provider root cannot be renamed + try x.h.walkTo(14, &.{"prov"}); + try x.h.expectFail(.{ .wstat = .{ .fid = 14, .stat = ws } }, "permission denied"); + _ = try x.h.ok(.{ .clunk = .{ .fid = 14 } }); + // remove always clunks; a non-empty directory refuses + _ = try x.h.ok(.{ .clunk = .{ .fid = 6 } }); + try x.h.walkTo(8, &.{ "prov", "dir" }); + try x.h.expectFail(.{ .remove = .{ .fid = 8 } }, "directory not empty"); + try x.h.expectFail(.{ .clunk = .{ .fid = 8 } }, "unknown fid"); + _ = try x.h.ok(.{ .remove = .{ .fid = 2 } }); + try x.h.walkTo(9, &.{"prov"}); + try x.h.expectFail(.{ .walk = .{ .fid = 9, .newfid = 15, .names = &.{"renamed"} } }, "file does not exist"); + _ = try x.h.ok(.{ .clunk = .{ .fid = 9 } }); + // an i/o error from the provider maps to "i/o error" + try x.h.walkTo(9, &.{"prov"}); + x.prov.fail_io = true; + try x.h.expectFail(.{ .walk = .{ .fid = 9, .newfid = 15, .names = &.{"hello"} } }, "i/o error"); + x.prov.fail_io = false; + _ = try x.h.ok(.{ .clunk = .{ .fid = 9 } }); + // ORCLOSE removes on clunk + try x.h.walkTo(9, &.{"prov"}); + _ = try x.h.ok(.{ .create = .{ .fid = 9, .name = "tmp", .perm = 0o644, .mode = cloud9.owrite | cloud9.orclose } }); + _ = try x.h.ok(.{ .clunk = .{ .fid = 9 } }); + try x.h.walkTo(9, &.{"prov"}); + try x.h.expectFail(.{ .walk = .{ .fid = 9, .newfid = 15, .names = &.{"tmp"} } }, "file does not exist"); + _ = try x.h.ok(.{ .clunk = .{ .fid = 9 } }); + // walking .. out of the provider root and cloning provider fids keeps refs balanced + const up = try x.h.ok(.{ .walk = .{ .fid = 0, .newfid = 10, .names = &.{ "prov", "dir", "..", "..", "build" } } }); + try testing.expectEqual(@as(u16, 5), up.walk.nwqid); + try x.h.walkTo(11, &.{ "prov", "dir", "inner" }); + _ = try x.h.ok(.{ .walk = .{ .fid = 11, .newfid = 12, .names = &.{} } }); + try testing.expectEqual(@as(u32, 2), x.prov.nodes[3].refs); + _ = try x.h.ok(.{ .clunk = .{ .fid = 11 } }); + try testing.expectEqual(@as(u32, 1), x.prov.nodes[3].refs); + // a partial walk releases the handles it obtained + const part = try x.h.ok(.{ .walk = .{ .fid = 0, .newfid = 13, .names = &.{ "prov", "dir", "nope" } } }); + try testing.expectEqual(@as(u16, 2), part.walk.nwqid); + try x.h.expectFail(.{ .clunk = .{ .fid = 13 } }, "unknown fid"); + _ = try x.h.ok(.{ .clunk = .{ .fid = 12 } }); + for (x.h.conn.fids) |f| { + if (f.used) _ = try x.h.ok(.{ .clunk = .{ .fid = f.id } }); + } + try testing.expectEqual(@as(u32, 0), x.prov.total_refs); +} + +test "directory reads across offsets, bad offset, and records never split" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + try x.h.walkTo(1, &.{}); + _ = try x.h.ok(.{ .open = .{ .fid = 1, .mode = cloud9.oread } }); + const first = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = 4096 } }); + try testing.expect(first.read.len > 0); + try x.h.expectFail(.{ .read = .{ .fid = 1, .offset = 5, .count = 4096 } }, "bad offset"); + // offset 0 restarts; the same bytes come back + const again = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = 4096 } }); + try testing.expectEqual(first.read.len, again.read.len); + // small reads at consecutive offsets return every record exactly once + const names = try x.h.listDir(1, 80); + defer TS.Harness.freeNames(names); + try testing.expectEqual(@as(usize, 7), names.len); + // a count too small for even one record returns nothing rather than splitting it + const tiny = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = 10 } }); + try testing.expectEqual(@as(usize, 0), tiny.read.len); + // the same for provider and var directories + const pn = try x.h.listPath(&.{ "prov", "dir" }); + defer TS.Harness.freeNames(pn); + try testing.expectEqual(@as(usize, 1), pn.len); + try x.h.walkTo(2, &.{ "vars", "state", "f" }); + _ = try x.h.ok(.{ .open = .{ .fid = 2, .mode = cloud9.oread } }); + const vn = try x.h.listDir(2, 100); + defer TS.Harness.freeNames(vn); + try testing.expectEqual(@as(usize, 4), vn.len); + try x.h.expectFail(.{ .read = .{ .fid = 2, .offset = 1, .count = 100 } }, "bad offset"); +} + +test "Tversion mid-session resets fids and clunks every provider handle" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + try x.h.walkTo(1, &.{ "prov", "hello" }); + try x.h.walkTo(2, &.{ "prov", "dir", "inner" }); + _ = try x.h.ok(.{ .open = .{ .fid = 2, .mode = cloud9.oread } }); + try x.h.walkTo(3, &.{ "runtime", "fn", "fib30" }); + _ = try x.h.ok(.{ .open = .{ .fid = 3, .mode = cloud9.oread } }); + try testing.expectEqual(@as(u32, 2), x.prov.total_refs); + try testing.expectEqual(@as(u32, 1), x.prov.nodes[3].opens); + try testing.expectEqual(@as(usize, 4), x.h.conn.fidCount()); + const before = x.prov.clunks; + try x.h.version(4096); + try testing.expectEqual(@as(usize, 0), x.h.conn.fidCount()); + try testing.expectEqual(@as(u32, 0), x.prov.total_refs); + try testing.expectEqual(@as(u32, 0), x.prov.nodes[3].opens); + try testing.expectEqual(before + 2, x.prov.clunks); + try testing.expect(!x.h.conn.slot_used[0] and !x.h.conn.slot_used[1]); + try x.h.expectFail(.{ .clunk = .{ .fid = 1 } }, "unknown fid"); + _ = try x.h.ok(.{ .attach = .{ .fid = 0, .uname = "tester" } }); + try x.h.walkTo(1, &.{ "prov", "hello" }); + // hangup does the same + x.h.conn.hangup(); + try testing.expectEqual(@as(u32, 0), x.prov.total_refs); + try testing.expectEqual(@as(usize, 0), x.h.conn.fidCount()); +} + +test "a reply that does not fit msize is an Rerror, not a dead connection" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + try x.h.version(64); + _ = try x.h.ok(.{ .attach = .{ .fid = 0, .uname = "t" } }); + // Rstat of the root is ~70 bytes. + try x.h.expectFail(.{ .stat = .{ .fid = 0 } }, "reply too large for msize"); + // Rwalk with 5 qids is 74 bytes; the walk must not bind newfid. + try x.h.expectFail(.{ .walk = .{ .fid = 0, .newfid = 1, .names = &.{ ".", ".", ".", ".", "." } } }, "reply too large for msize"); + try x.h.expectFail(.{ .clunk = .{ .fid = 1 } }, "unknown fid"); + try x.h.expectFail(.{ .walk = .{ .fid = 0, .newfid = 0, .names = &.{ ".", ".", ".", ".", "." } } }, "reply too large for msize"); + const r = try x.h.ok(.{ .walk = .{ .fid = 0, .newfid = 1, .names = &.{"README"} } }); + try testing.expectEqual(@as(u16, 1), r.walk.nwqid); + _ = try x.h.ok(.{ .open = .{ .fid = 1, .mode = cloud9.oread } }); + const rd = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = 40 } }); + try testing.expect(rd.read.len > 0 and rd.read.len <= 64 - cloud9.iohdrsz); + _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); +} + +test "fid table is bounded per connection" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + var i: u32 = 1; + while (x.h.conn.fidCount() < test_cfg.max_fids) : (i += 1) { + _ = try x.h.ok(.{ .walk = .{ .fid = 0, .newfid = i, .names = &.{} } }); + } + try x.h.expectFail(.{ .walk = .{ .fid = 0, .newfid = i, .names = &.{} } }, "too many fids"); + try x.h.expectFail(.{ .attach = .{ .fid = i, .uname = "tester" } }, "too many fids"); + // self-walks and clunks still work at the limit + _ = try x.h.ok(.{ .walk = .{ .fid = 0, .newfid = 0, .names = &.{"build"} } }); + _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); + _ = try x.h.ok(.{ .walk = .{ .fid = 0, .newfid = i, .names = &.{} } }); + try x.h.expectFail(.{ .walk = .{ .fid = 0, .newfid = 2, .names = &.{"README"} } }, "fid in use"); + try x.h.expectFail(.{ .walk = .{ .fid = 1234, .newfid = 2, .names = &.{} } }, "unknown fid"); + // a walk into a provider at the limit must not leak the handle + _ = try x.h.ok(.{ .clunk = .{ .fid = 2 } }); + _ = try x.h.ok(.{ .walk = .{ .fid = 0, .newfid = 2, .names = &.{ "..", "prov", "hello" } } }); + try x.h.expectFail(.{ .walk = .{ .fid = 2, .newfid = i + 1, .names = &.{} } }, "too many fids"); + try testing.expectEqual(@as(u32, 1), x.prov.total_refs); +} + +test "Shared refuses more providers or vars than configured" { + var ctx: TestCtx = .{}; + var shared: TS.Shared = .init(&ctx); + var p1 = TestProv.init(); + var p2 = TestProv.init(); + var p3 = TestProv.init(); + try shared.addProvider(.{ .name = "a", .ctx = &p1, .vtable = &TestProv.vtable }); + try shared.addProvider(.{ .name = "b", .ctx = &p2, .vtable = &TestProv.vtable }); + try testing.expectError(error.Full, shared.addProvider(.{ .name = "c", .ctx = &p3, .vtable = &TestProv.vtable })); + var v: [5]u32 = @splat(0); + try shared.expose("v0", &v[0]); + try shared.expose("v1", &v[1]); + try shared.expose("v2", &v[2]); + try shared.expose("v3", &v[3]); + try testing.expectError(error.Full, shared.expose("v4", &v[4])); +} + +/// A server with a large fid table for the index tests. +const big_cfg: Config = .{ + .name = "big", + .msize = 8192, + .max_fids = 4096, + .max_providers = 1, + .max_vars = 1, + .snapshot_slots = 1, + .snapshot_bytes = 256, +}; +const BigS = Server(big_cfg); + +/// Fid numbers chosen to stress the index: dense low ids, ids with only high +/// bits set, and ids counting down from 2^32-1 (all distinct for i < 2^20). +fn adversarialId(i: u32) u32 { + return switch (i % 3) { + 0 => i * 8192 + 1, + 1 => 0x8000_0000 | i, + else => 0xFFFF_FFFF - i, + }; +} + +/// Every index bucket points at a used fid that finds itself, and every used +/// fid is found: the invariant the hostile fid tests check after each phase. +fn checkFidIndex(c: *BigS.Conn) !void { + var indexed: usize = 0; + for (c.index) |slot| { + if (slot == BigS.no_slot) continue; + indexed += 1; + try testing.expect(c.fids[slot].used); + try testing.expectEqual(&c.fids[slot], c.findFid(c.fids[slot].id).?); + } + var used: usize = 0; + for (c.fids[0..c.high_water]) |*f| if (f.used) { + used += 1; + try testing.expectEqual(f, c.findFid(f.id).?); + }; + for (c.fids[c.high_water..]) |*f| try testing.expect(!f.used); + try testing.expectEqual(indexed, used); + try testing.expectEqual(used, c.nfids); +} + +test "fid index: thousands of fids, clunk in hostile orders, reuse, Tversion" { + var ctx: TestCtx = .{}; + var shared: BigS.Shared = .init(&ctx); + var prov = TestProv.init(); + try shared.addProvider(prov.provider()); + const storage = try testing.allocator.create(BigS.Storage); + defer testing.allocator.destroy(storage); + var h: BigS.Harness = undefined; + try h.init(&shared, storage); + defer h.deinit(); + const n: u32 = big_cfg.max_fids - 1; // fid 0 is the attach + var i: u32 = 0; + while (i < n) : (i += 1) { + _ = try h.ok(.{ .walk = .{ .fid = 0, .newfid = adversarialId(i), .names = &.{ "prov", "hello" } } }); + } + try testing.expectEqual(@as(usize, n + 1), h.conn.fidCount()); + try testing.expectEqual(n, prov.total_refs); + try h.expectFail(.{ .walk = .{ .fid = 0, .newfid = 0x7FFF_FFFF, .names = &.{} } }, "too many fids"); + try h.expectFail(.{ .walk = .{ .fid = 0, .newfid = adversarialId(5), .names = &.{} } }, "fid in use"); + try h.expectFail(.{ .attach = .{ .fid = adversarialId(7), .uname = "t" } }, "fid in use"); + try testing.expect(h.conn.findFid(0x7FFF_FFFF) == null); + try testing.expect(h.conn.findFid(adversarialId(n)) == null); + try checkFidIndex(&h.conn); + // clunk every third fid, then the rest from the top: backward-shift deletion under churn + i = 0; + while (i < n) : (i += 3) _ = try h.ok(.{ .clunk = .{ .fid = adversarialId(i) } }); + try checkFidIndex(&h.conn); + i = n; + while (i > 0) { + i -= 1; + if (i % 3 == 0) { + try h.expectFail(.{ .clunk = .{ .fid = adversarialId(i) } }, "unknown fid"); + } else { + _ = try h.ok(.{ .clunk = .{ .fid = adversarialId(i) } }); + } + } + try testing.expectEqual(@as(usize, 1), h.conn.fidCount()); + try testing.expectEqual(@as(u32, 0), prov.total_refs); + try checkFidIndex(&h.conn); + // the whole table is reusable after the churn, through the free list + i = 0; + while (i < n) : (i += 1) _ = try h.ok(.{ .walk = .{ .fid = 0, .newfid = n - i, .names = &.{} } }); + try h.expectFail(.{ .walk = .{ .fid = 0, .newfid = n + 1, .names = &.{} } }, "too many fids"); + try checkFidIndex(&h.conn); + // pseudo-random alloc/free storm with verification + var prng = std.Random.DefaultPrng.init(0x9a11); + const rnd = prng.random(); + var live: [n + 1]bool = @splat(true); + live[0] = false; // never touch the attach fid + var round: usize = 0; + while (round < 20_000) : (round += 1) { + const id = 1 + rnd.uintLessThan(u32, n); + if (live[id]) { + _ = try h.ok(.{ .clunk = .{ .fid = id } }); + } else { + _ = try h.ok(.{ .walk = .{ .fid = 0, .newfid = id, .names = &.{"prov"} } }); + } + live[id] = !live[id]; + if (round % 997 == 0) try checkFidIndex(&h.conn); + } + try checkFidIndex(&h.conn); + // Tversion drops everything and the table starts over, provider refs balanced + try h.version(big_cfg.msize); + try testing.expectEqual(@as(usize, 0), h.conn.fidCount()); + try testing.expectEqual(@as(u32, 0), prov.total_refs); + try testing.expectEqual(@as(u16, 0), h.conn.high_water); + try checkFidIndex(&h.conn); + _ = try h.ok(.{ .attach = .{ .fid = 0xFFFF_FFFE, .uname = "t" } }); + _ = try h.ok(.{ .walk = .{ .fid = 0xFFFF_FFFE, .newfid = 0, .names = &.{} } }); + try checkFidIndex(&h.conn); +} + +test "open: a provider stat failure after a successful open closes the file again" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + try x.h.walkTo(1, &.{ "prov", "hello" }); + x.prov.fail_stat = true; + try x.h.expectFail(.{ .open = .{ .fid = 1, .mode = cloud9.oread } }, "i/o error"); + x.prov.fail_stat = false; + try testing.expectEqual(@as(u32, 0), x.prov.nodes[1].opens); + try x.h.expectFail(.{ .read = .{ .fid = 1, .offset = 0, .count = 10 } }, "file not open"); + _ = try x.h.ok(.{ .open = .{ .fid = 1, .mode = cloud9.oread } }); + try testing.expectEqual(@as(u32, 1), x.prov.nodes[1].opens); + // the same for create: a stat failure after the provider created the node releases it + try x.h.walkTo(2, &.{"prov"}); + x.prov.fail_stat = true; + try x.h.expectFail(.{ .create = .{ .fid = 2, .name = "born", .perm = 0o644, .mode = cloud9.owrite } }, "i/o error"); + x.prov.fail_stat = false; + for (x.prov.nodes) |e| if (e.used and std.mem.eql(u8, e.nameSlice(), "born")) { + try testing.expectEqual(@as(u32, 0), e.opens); + try testing.expectEqual(@as(u32, 0), e.refs); + }; + try testing.expect(!x.h.conn.findFid(2).?.open); + try testing.expectEqual(@as(u32, 1), x.prov.total_refs); // fid 1 only +} + +test "fid state machine: open twice, walk from open, remove/clunk of open provider fids" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + try x.h.walkTo(1, &.{ "prov", "dir", "inner" }); + _ = try x.h.ok(.{ .open = .{ .fid = 1, .mode = cloud9.ordwr } }); + try x.h.expectFail(.{ .open = .{ .fid = 1, .mode = cloud9.oread } }, "file already open"); + try x.h.expectFail(.{ .walk = .{ .fid = 1, .newfid = 2, .names = &.{"."} } }, "file already open"); + try x.h.expectFail(.{ .create = .{ .fid = 1, .name = "z", .perm = 0o644, .mode = cloud9.oread } }, "file already open"); + // a clone of an open fid is a fresh, unopened reference + _ = try x.h.ok(.{ .walk = .{ .fid = 1, .newfid = 2, .names = &.{} } }); + try testing.expectEqual(@as(u32, 2), x.prov.nodes[3].refs); + try testing.expectEqual(@as(u32, 1), x.prov.nodes[3].opens); + // walking newfid == fid with names on an unopened provider fid swaps the handle, refs balanced + try x.h.walkTo(7, &.{ "prov", "dir" }); + try testing.expectEqual(@as(u32, 1), x.prov.nodes[2].refs); + _ = try x.h.ok(.{ .walk = .{ .fid = 7, .newfid = 7, .names = &.{ "..", "dir", "inner", "..", "..", "dir" } } }); + try testing.expectEqual(@as(u32, 1), x.prov.nodes[2].refs); + try testing.expectEqual(@as(u32, 2), x.prov.nodes[3].refs); + _ = try x.h.ok(.{ .clunk = .{ .fid = 7 } }); + try testing.expectEqual(@as(u32, 0), x.prov.nodes[2].refs); + // remove of an open fid: close, then remove, then clunk; refs and opens return to zero + _ = try x.h.ok(.{ .remove = .{ .fid = 1 } }); + try testing.expectEqual(@as(u32, 0), x.prov.nodes[3].opens); + try testing.expectEqual(@as(u32, 1), x.prov.nodes[3].refs); + try x.h.expectFail(.{ .open = .{ .fid = 1, .mode = cloud9.oread } }, "unknown fid"); + _ = try x.h.ok(.{ .clunk = .{ .fid = 2 } }); + try testing.expectEqual(@as(u32, 0), x.prov.total_refs); + // walking "." on a file fid is "not a directory" at the protocol level, without a provider walk + try x.h.walkTo(3, &.{ "prov", "hello" }); + const before = x.prov.clunks; + try x.h.expectFail(.{ .walk = .{ .fid = 3, .newfid = 4, .names = &.{"."} } }, "not a directory"); + try testing.expectEqual(before, x.prov.clunks); + try testing.expectEqual(@as(u32, 1), x.prov.total_refs); + // a partial walk through a file releases the handles it took + const part = try x.h.ok(.{ .walk = .{ .fid = 0, .newfid = 5, .names = &.{ "prov", "hello", "x", "y" } } }); + try testing.expectEqual(@as(u16, 2), part.walk.nwqid); + try testing.expectEqual(@as(u32, 1), x.prov.total_refs); + try x.h.expectFail(.{ .clunk = .{ .fid = 5 } }, "unknown fid"); + // Tremove is always a clunk, even of a static node or when the provider refuses + try x.h.walkTo(6, &.{"README"}); + try x.h.expectFail(.{ .remove = .{ .fid = 6 } }, "permission denied"); + try x.h.expectFail(.{ .clunk = .{ .fid = 6 } }, "unknown fid"); + _ = try x.h.ok(.{ .clunk = .{ .fid = 3 } }); + try testing.expectEqual(@as(u32, 0), x.prov.total_refs); +} + +test "snapshot slots: exhaust, hold, Tversion frees; reads past the end and at huge offsets" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + try x.h.walkTo(1, &.{ "runtime", "fn", "fib30" }); + try x.h.walkTo(2, &.{ "vars", "state", "value" }); + try x.h.walkTo(3, &.{ "vars", "state", "addr" }); + _ = try x.h.ok(.{ .open = .{ .fid = 1, .mode = cloud9.oread } }); + _ = try x.h.ok(.{ .open = .{ .fid = 2, .mode = cloud9.oread } }); + try x.h.expectFail(.{ .open = .{ .fid = 3, .mode = cloud9.oread } }, "too many open dynamic files"); + try testing.expect(!x.h.conn.findFid(3).?.open); + // reads at offsets near 2^64 never trap (counts above msize are a raw-9P + // case: the cloud9 client refuses to send them; test/adv_core_hostile.py covers it) + const max_count = test_cfg.msize - cloud9.iohdrsz; + const r = try x.h.ok(.{ .read = .{ .fid = 1, .offset = std.math.maxInt(u64), .count = max_count } }); + try testing.expectEqualStrings("", r.read); + const r2 = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 1 << 63, .count = 0 } }); + try testing.expectEqualStrings("", r2.read); + const r3 = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = max_count } }); + try testing.expectEqualStrings("832040", r3.read); + // raw beyond @sizeOf is empty; a partial raw read at the tail is bounded + try x.h.walkTo(4, &.{ "vars", "state", "raw" }); + _ = try x.h.ok(.{ .open = .{ .fid = 4, .mode = cloud9.oread } }); + const raw_end = try x.h.ok(.{ .read = .{ .fid = 4, .offset = @sizeOf(Exposed), .count = 100 } }); + try testing.expectEqualStrings("", raw_end.read); + const raw_tail = try x.h.ok(.{ .read = .{ .fid = 4, .offset = @sizeOf(Exposed) - 1, .count = 100 } }); + try testing.expectEqual(@as(usize, 1), raw_tail.read.len); + const raw_huge = try x.h.ok(.{ .read = .{ .fid = 4, .offset = std.math.maxInt(u64) - 1, .count = 100 } }); + try testing.expectEqualStrings("", raw_huge.read); + // Tversion releases the held slots + try x.h.version(test_cfg.msize); + try testing.expect(!x.h.conn.slot_used[0] and !x.h.conn.slot_used[1]); + _ = try x.h.ok(.{ .attach = .{ .fid = 0, .uname = "tester" } }); + try x.h.walkTo(3, &.{ "vars", "state", "addr" }); + _ = try x.h.ok(.{ .open = .{ .fid = 3, .mode = cloud9.oread } }); +} + +test "static and var nodes refuse create, remove and wstat; directories refuse writes" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + const dirs = [_][]const []const u8{ &.{}, &.{"build"}, &.{"comptime"}, &.{ "comptime", "types" }, &.{ "comptime", "types", "Layout" }, &.{"runtime"}, &.{ "runtime", "fn" }, &.{"vars"}, &.{ "vars", "state" }, &.{ "vars", "state", "f" }, &.{ "vars", "state", "f", "inner" } }; + for (dirs, 0..) |d, k| { + const fid: u32 = @intCast(10 + k); + try x.h.walkTo(fid, d); + try x.h.expectFail(.{ .create = .{ .fid = fid, .name = "x", .perm = 0o644, .mode = cloud9.owrite } }, "permission denied"); + try x.h.expectFail(.{ .wstat = .{ .fid = fid, .stat = stat_dontcare } }, "permission denied"); + try x.h.expectFail(.{ .open = .{ .fid = fid, .mode = cloud9.owrite } }, "is a directory"); + try x.h.expectFail(.{ .open = .{ .fid = fid, .mode = cloud9.oread | cloud9.otrunc } }, "is a directory"); + try x.h.expectFail(.{ .remove = .{ .fid = fid } }, "permission denied"); + try x.h.expectFail(.{ .clunk = .{ .fid = fid } }, "unknown fid"); + } + const files = [_][]const []const u8{ &.{"README"}, &.{ "build", "time" }, &.{ "comptime", "decls" }, &.{ "runtime", "pid" }, &.{ "runtime", "fn", "fib30" }, &.{"ctl"}, &.{ "vars", "state", "value" }, &.{ "vars", "state", "raw" }, &.{ "vars", "state", "f", "a", "value" }, &.{ "vars", "counter", "type" } }; + for (files, 0..) |f, k| { + const fid: u32 = @intCast(30 + k); + try x.h.walkTo(fid, f); + try x.h.expectFail(.{ .wstat = .{ .fid = fid, .stat = stat_dontcare } }, "permission denied"); + try x.h.expectFail(.{ .walk = .{ .fid = fid, .newfid = 99, .names = &.{".."} } }, "not a directory"); + try x.h.expectFail(.{ .remove = .{ .fid = fid } }, "permission denied"); + } + // writes to a var value at a non-zero offset and with an empty payload + try x.h.walkTo(1, &.{ "vars", "state", "f", "a", "value" }); + _ = try x.h.ok(.{ .open = .{ .fid = 1, .mode = cloud9.owrite | cloud9.otrunc } }); + try x.h.expectFail(.{ .write = .{ .fid = 1, .offset = 0, .data = "" } }, "bad value"); + try x.h.expectFail(.{ .write = .{ .fid = 1, .offset = 0, .data = "-1" } }, "bad value"); + try x.h.expectFail(.{ .write = .{ .fid = 1, .offset = 0, .data = "1e3" } }, "bad value"); + try x.h.expectFail(.{ .write = .{ .fid = 1, .offset = 0, .data = "99999999999999999999" } }, "bad value"); + try testing.expectEqual(@as(u32, 1), x.exposed.a); + _ = try x.h.ok(.{ .write = .{ .fid = 1, .offset = std.math.maxInt(u64), .data = "77\n" } }); + try testing.expectEqual(@as(u32, 77), x.exposed.a); + // reads of a write-only fid are refused; OEXEC reads like OREAD + try x.h.expectFail(.{ .read = .{ .fid = 1, .offset = 0, .count = 10 } }, "file not open"); + try x.h.walkTo(2, &.{"README"}); + _ = try x.h.ok(.{ .open = .{ .fid = 2, .mode = cloud9.oexec } }); + try testing.expect((try x.h.ok(.{ .read = .{ .fid = 2, .offset = 0, .count = 10 } })).read.len == 10); +} + +test "msize 24: every request that fits is answered, every reply that cannot fit is an Rerror" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + try x.h.version(24); + _ = try x.h.ok(.{ .attach = .{ .fid = 0, .uname = "u" } }); // Tattach 20, Rattach 20 + try x.h.expectFail(.{ .stat = .{ .fid = 0 } }, "reply too large"); // Rerror truncated to fit 24 bytes + const w = try x.h.ok(.{ .walk = .{ .fid = 0, .newfid = 1, .names = &.{"ctl"} } }); // Rwalk 22 + try testing.expectEqual(@as(u16, 1), w.walk.nwqid); + try x.h.expectFail(.{ .walk = .{ .fid = 0, .newfid = 2, .names = &.{ ".", "." } } }, "reply too large"); + try x.h.expectFail(.{ .clunk = .{ .fid = 2 } }, "unknown fid"); + _ = try x.h.ok(.{ .open = .{ .fid = 1, .mode = cloud9.ordwr } }); // Ropen 24 + try x.h.expectFail(.{ .write = .{ .fid = 1, .offset = 0, .data = "e" } }, "bad command"); // Twrite 24 + // the largest read the client may ask for is msize - iohdrsz = 0 bytes + const r = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = 0 } }); + try testing.expectEqual(@as(usize, 0), r.read.len); + _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); + try x.h.walkTo(3, &.{"build"}); + _ = try x.h.ok(.{ .open = .{ .fid = 3, .mode = cloud9.oread } }); + const d = try x.h.ok(.{ .read = .{ .fid = 3, .offset = 0, .count = 0 } }); + try testing.expectEqual(@as(usize, 0), d.read.len); // no record fits in 0 bytes, nothing is split + try x.h.expectFail(.{ .read = .{ .fid = 3, .offset = 1, .count = 0 } }, "bad offset"); +} + +test "Conn.init clamps the msize cap to [msize_min, cfg.msize]" { + var ctx: TestCtx = .{}; + var shared: TS.Shared = .init(&ctx); + var storage: TS.Storage = undefined; + const lo: TS.Conn = .init(&shared, &storage, 0); + try testing.expectEqual(cloud9.Server.msize_min, lo.msize_cap); + const hi: TS.Conn = .init(&shared, &storage, std.math.maxInt(u32)); + try testing.expectEqual(test_cfg.msize, hi.msize_cap); + const mid: TS.Conn = .init(&shared, &storage, 4096); + try testing.expectEqual(@as(u32, 4096), mid.msize_cap); +} + +test "parseIso8601 rejects malformed stamps and never traps" { + try testing.expectEqual(@as(?u32, null), parseIso8601("")); + try testing.expectEqual(@as(?u32, null), parseIso8601("2023-11-14T22:13:20")); + try testing.expectEqual(@as(?u32, null), parseIso8601("2023-13-14T22:13:20Z")); + try testing.expectEqual(@as(?u32, null), parseIso8601("2023-11-32T22:13:20Z")); + try testing.expectEqual(@as(?u32, null), parseIso8601("2023-11-14T24:13:20Z")); + try testing.expectEqual(@as(?u32, null), parseIso8601("2023-11-14T22:60:20Z")); + try testing.expectEqual(@as(?u32, null), parseIso8601("1969-12-31T23:59:59Z")); + try testing.expectEqual(@as(?u32, null), parseIso8601("9999-12-31T23:59:59Z")); + try testing.expectEqual(@as(?u32, null), parseIso8601("20x3-11-14T22:13:20Z")); + try testing.expectEqual(@as(?u32, null), parseIso8601("0000-01-01T00:00:00Z")); + try testing.expectEqual(@as(u32, 1_700_000_000), parseIso8601("2023-11-14T22:13:20Z").?); + try testing.expectEqual(@as(u32, 951_782_400), parseIso8601("2000-02-29T00:00:00Z").?); + try testing.expectEqual(@as(u32, 4_102_444_799), parseIso8601("2099-12-31T23:59:59Z").?); + try testing.expectEqual(@as(u32, std.math.maxInt(u32)), parseIso8601("2106-02-07T06:28:15Z").?); + try testing.expectEqual(@as(?u32, null), parseIso8601("2106-02-07T06:28:16Z")); +} + +/// Multiplicative inverse of an odd 32-bit constant (Newton iteration). +fn inverseMod32(a: u32) u32 { + var x: u32 = a; + for (0..5) |_| x *%= 2 -% a *% x; + return x; +} + +test "fid index: fid numbers crafted to collide under the public hash do not cluster a seeded connection" { + var ctx: TestCtx = .{}; + var shared: BigS.Shared = .init(&ctx); + const storage = try testing.allocator.create(BigS.Storage); + defer testing.allocator.destroy(storage); + var h: BigS.Harness = undefined; + try h.init(&shared, storage); + defer h.deinit(); + // two connections on the same Shared never share a seed + const other: BigS.Conn = .init(&shared, storage, big_cfg.msize); + try testing.expect(other.hash_seed != h.conn.hash_seed); + // ids whose products with the golden ratio share their top bits: all one bucket when unseeded + const inv = inverseMod32(0x9E37_79B1); + try testing.expectEqual(@as(u32, 1), inv *% 0x9E37_79B1); + const n: u32 = big_cfg.max_fids - 1; + const base: u32 = 0x4242_0000; + var i: u32 = 0; + while (i < n) : (i += 1) { + const id = (base + i) *% inv; + try testing.expectEqual(@as(usize, base >> BigS.index_shift), @as(usize, @intCast((id *% 0x9E37_79B1) >> BigS.index_shift))); + _ = try h.ok(.{ .walk = .{ .fid = 0, .newfid = id, .names = &.{} } }); + } + try checkFidIndex(&h.conn); + // the longest probe sequence in the seeded table is short; unseeded it would be ~n + var worst: usize = 0; + i = 0; + while (i < n) : (i += 1) { + const id = (base + i) *% inv; + var pos = h.conn.fidHome(id); + var steps: usize = 0; + while (h.conn.fids[h.conn.index[pos]].id != id) : (pos = (pos + 1) & BigS.index_mask) steps += 1; + worst = @max(worst, steps); + } + try testing.expect(worst < 64); +} diff --git a/introspect/src/freestanding_check.zig b/introspect/src/freestanding_check.zig new file mode 100644 index 0000000..ad88d0e --- /dev/null +++ b/introspect/src/freestanding_check.zig @@ -0,0 +1,71 @@ +//! A tiny freestanding root proving that `core` and `vars` compile without an +//! OS: `zig build introspect-check-freestanding` builds this for riscv32-freestanding-none. +//! It instantiates `Server(cfg)` with static Storage/Shared, exposes one +//! variable, and runs one push/step over a canned Tversion frame. It must not +//! import scratch.zig (allocator) or anything OS-specific. +const std = @import("std"); +const core = @import("core.zig"); +const Writer = std.Io.Writer; + +const Build = struct { + pub const zig_version: []const u8 = @import("builtin").zig_version_string; + pub const target: []const u8 = "riscv32-freestanding-none"; + pub const optimize: []const u8 = "check"; + pub const time: []const u8 = "1970-01-01T00:00:00Z"; + pub const change: []const u8 = "none"; +}; + +const State = struct { ticks: u32, phase: enum { idle, busy }, inner: struct { x: f32 } }; + +const Fns = struct { + pub fn ticks(ctx: *anyopaque, w: *Writer) anyerror!void { + const s: *State = @ptrCast(@alignCast(ctx)); + try w.print("{d}", .{s.ticks}); + } +}; + +fn ctl(ctx: *anyopaque, cmd: []const u8, out: *Writer) anyerror!void { + const s: *State = @ptrCast(@alignCast(ctx)); + if (std.mem.startsWith(u8, cmd, "reset")) s.ticks = 0; + try out.writeAll("ok"); +} + +const cfg: core.Config = .{ + .name = "fw", + .build = Build, + .types = &.{ State, core.NodeStat }, + .decls_of = Fns, + .fns = Fns, + .ctl = &ctl, + .msize = 2048, + .max_fids = 16, + .snapshot_slots = 2, + .snapshot_bytes = 1024, +}; + +const S = core.Server(cfg); + +var state: State = .{ .ticks = 0, .phase = .idle, .inner = .{ .x = 0 } }; +var storage: S.Storage = undefined; +var shared: S.Shared = undefined; +var conn: S.Conn = undefined; + +/// Tversion msize=2048 version="9P2000". +const tversion = [_]u8{ 19, 0, 0, 0, 100, 0xFF, 0xFF, 0, 8, 0, 0, 6, 0, '9', 'P', '2', '0', '0', '0' }; + +/// Runs one Tversion through the engine; returns the number of reply bytes. +pub export fn introspect_check() u32 { + shared = .init(&state); + shared.expose("state", &state) catch unreachable; + conn = .init(&shared, &storage, cfg.msize); + _ = conn.push(&tversion); + _ = conn.step() catch return 0; + const out = conn.output(); + conn.wrote(out.len); + return @intCast(out.len); +} + +pub export fn _start() noreturn { + _ = introspect_check(); + while (true) {} +} diff --git a/introspect/src/linux/debug.zig b/introspect/src/linux/debug.zig new file mode 100644 index 0000000..4d43d5b --- /dev/null +++ b/introspect/src/linux/debug.zig @@ -0,0 +1,1458 @@ +//! Linux debug facilities for the introspect server: threads, stacks, +//! registers, address → source, memory, breakpoints and panics. +//! +//! This file is a pure API; a later adapter turns it into a core `Provider`. +//! All text is written to a `*std.Io.Writer`. Nothing here allocates after +//! `init` except from the caller-provided `text_buf`, which is used as a fixed +//! arena for symbol text and reset before every query. +//! +//! Only one `Debug` may exist per process: the signal handlers and the panic +//! hook find their state through the global `current` pointer set by `init`. +//! +//! Mechanics +//! +//! * Capturing another thread's stack or registers: the calling (server) +//! thread sends `capture_signal` with `tgkill`. The SA_SIGINFO handler copies +//! the interrupted register state (`cpu_context.fromPosixSignalContext`) into +//! the single capture slot and parks on a futex. The server unwinds the +//! parked thread's stack from that context, releases the target, then +//! symbolizes. The handler is async-signal-safe: no allocation, no +//! `std.debug`, no locks other than the futex. A target that does not run +//! the handler within `capture_timeout_ns` (signal masked, thread in D +//! state, ...) yields `error.Timeout`; a late-arriving handler run cannot +//! corrupt a reused slot because it must match the requested tid and win a +//! compare-and-swap from `armed` on the slot state (that pair plays the role +//! of a generation counter: a stale run finds the slot idle, armed for +//! another tid, or armed for itself, in which case its capture is simply the +//! valid answer to the new request). +//! * Breakpoints: `@breakpoint()` raises SIGTRAP on the executing thread only. +//! The handler claims a pause slot, saves the context and parks on a futex +//! until `resumeThread`. On x86_64 the saved PC is already past `int3`; on +//! aarch64 the handler advances PC by 4 in the ucontext before returning +//! (only for a real `brk`, i.e. a kernel-generated si_code; a SIGTRAP sent +//! with kill/tgkill parks the thread where it was). With no free slot the +//! thread steps over the breakpoint and keeps running (`traps_skipped` +//! counts them): the debug layer never kills the process. The server thread +//! itself (`server_tid`) is never parked, a breakpoint there is stepped +//! over, because nobody could resume it. Only a stale handler run after +//! `deinit` (no `current`) falls back to the default disposition. +//! * Panics: `panicHook` records the message and a stack capture, then, if +//! `hold_on_panic` and a `Debug` exists, parks until `panicContinue`; then +//! `std.debug.defaultPanic` runs. A nested or second panic, or a panic on +//! the server thread itself (which could never be continued), goes +//! straight to the default handler. +//! * std.debug's `SelfInfo` guards its state with an `Io.RwLock`. A target +//! parked while holding it (a thread inside a stack-trace dump, say) would +//! deadlock the unwind, so after parking a thread the lock is probed with +//! `tryLock`; a held lock yields `error.Busy` and the target is released. +//! * Known-module guard: `std.debug.SelfInfo` (Zig 0.16) rebuilds its module +//! list whenever it is asked about an address outside every known module, +//! freeing the CIE lists its unwind cache still points into; later unwinds +//! then read freed memory. `init` records the PT_LOAD ranges of the +//! executable (the same source std uses) and every lookup or unwind is +//! first checked against them; addresses outside (unmapped, vDSO, ...) +//! render as "?" and are never handed to std. + +const std = @import("std"); +const builtin = @import("builtin"); +const linux = std.os.linux; +const cpu_context = std.debug.cpu_context; +const Writer = std.Io.Writer; +const Native = cpu_context.Native; +const arch = builtin.cpu.arch; + +pub const Options = struct { + /// Used for `std.debug` symbolization (reading debug info from disk). + io: std.Io, + /// Fixed arena for symbol text. A `FixedBufferAllocator` is placed over it + /// and reset before every query. 16 KiB is plenty; 4 KiB is a sane floor. + text_buf: []u8, + /// Real-time signal used to snapshot other threads. SIGRTMIN is 32 on + /// Linux without libc; the default is SIGRTMIN+3. + capture_signal: u8 = default_capture_signal, + /// How long to wait for a target thread to run the capture handler. + capture_timeout_ns: u64 = 250 * std.time.ns_per_ms, + /// How many threads may be parked in `@breakpoint()` at once (≤ 32). + max_paused: u8 = 16, +}; + +pub const default_capture_signal: u8 = 32 + 3; + +/// Hard upper bound of `Options.max_paused` (slot storage is static). +pub const max_paused_cap = 32; +/// Maximum number of frames written by any stack function. +pub const max_frames = 64; +/// Maximum number of tids enumerated from /proc/self/task. +pub const max_threads = 512; +/// Upper bound of the recorded panic message. +pub const panic_msg_cap = 1024; +/// Maximum number of PT_LOAD ranges recorded by the known-module guard. +pub const max_ranges = 64; + +/// Consulted by `panicHook`: when true and a `Debug` is initialized, the +/// panicking thread is held until `panicContinue`. +pub var hold_on_panic: bool = true; + +/// The one live instance, set by `init`, cleared by `deinit`. +pub var current: ?*Debug = null; + +/// The tid of the thread serving requests (0 = none). That thread is never +/// parked by a breakpoint or held by a panic, since nobody could release it. +pub var server_tid: std.atomic.Value(u32) = .init(0); + +/// Breakpoints stepped over because no pause slot was free, or because they +/// were hit on the server thread. +pub var traps_skipped: std.atomic.Value(u32) = .init(0); + +pub const Error = error{ + /// The target thread did not run the capture handler in time. + Timeout, + /// No thread with that tid exists in this process. + NoThread, + /// The address is not mapped (EFAULT from process_vm_readv/writev). + Unmapped, + /// The thread is not parked in a breakpoint. + NotPaused, + /// No panic has been recorded / is being held. + NoPanic, + /// Another `Debug` already exists in this process. + AlreadyInitialized, + /// The operation is not available on this architecture / kernel. + Unsupported, + /// The target thread is parked inside std.debug (holding its lock); its + /// stack cannot be unwound without deadlocking. Retry later. + Busy, + /// Invalid option value. + InvalidOptions, + /// A syscall or /proc read failed unexpectedly. + Unexpected, + /// The writer failed. + WriteFailed, +}; + +// Capture slot states. +const cap_idle: u32 = 0; +const cap_armed: u32 = 1; +const cap_capturing: u32 = 2; +const cap_captured: u32 = 3; +const cap_failed: u32 = 4; + +// Pause slot states. +const pause_free: u32 = 0; +const pause_claimed: u32 = 1; +const pause_paused: u32 = 2; +const pause_resuming: u32 = 3; + +const CaptureSlot = struct { + state: std.atomic.Value(u32) = .init(cap_idle), + target_tid: std.atomic.Value(u32) = .init(0), + ctx: Native = undefined, +}; + +const PauseSlot = struct { + state: std.atomic.Value(u32) = .init(pause_free), + tid: std.atomic.Value(u32) = .init(0), + ctx: Native = undefined, +}; + +pub const Debug = struct { + io: std.Io, + text_buf: []u8, + capture_signal: linux.SIG, + capture_timeout_ns: u64, + max_paused: u8, + + capture: CaptureSlot = .{}, + paused: [max_paused_cap]PauseSlot = [_]PauseSlot{.{}} ** max_paused_cap, + + old_capture_action: linux.Sigaction = undefined, + old_trap_action: linux.Sigaction = undefined, + breakpoints_enabled: bool = false, + + tids: [max_threads]u32 = undefined, + tid_count: usize = 0, + + ranges: [max_ranges]Range = undefined, + range_count: usize = 0, + + const Range = struct { start: usize, len: usize }; + + /// Installs the capture handler (not the SIGTRAP handler) and publishes + /// `d` as `current`. + pub fn init(d: *Debug, opts: Options) Error!void { + if (current != null) return error.AlreadyInitialized; + if (opts.capture_signal < 32 or opts.capture_signal >= linux.NSIG) return error.InvalidOptions; + if (opts.max_paused == 0 or opts.max_paused > max_paused_cap) return error.InvalidOptions; + if (Native == noreturn) return error.Unsupported; + d.* = .{ + .io = opts.io, + .text_buf = opts.text_buf, + .capture_signal = @enumFromInt(opts.capture_signal), + .capture_timeout_ns = opts.capture_timeout_ns, + .max_paused = opts.max_paused, + }; + d.scanModules(); + const act: linux.Sigaction = .{ + .handler = .{ .sigaction = captureHandler }, + .mask = linux.sigemptyset(), + .flags = linux.SA.SIGINFO | linux.SA.RESTART, + }; + current = d; + if (linux.errno(linux.sigaction(d.capture_signal, &act, &d.old_capture_action)) != .SUCCESS) { + current = null; + return error.Unexpected; + } + } + + /// Restores the signal dispositions and clears `current`. Threads parked + /// in a breakpoint are resumed first. + pub fn deinit(d: *Debug) void { + d.disableBreakpoints(); + _ = linux.sigaction(d.capture_signal, &d.old_capture_action, null); + if (current == d) current = null; + } + + /// Installs the SIGTRAP handler so that `@breakpoint()` parks the thread. + pub fn enableBreakpoints(d: *Debug) Error!void { + if (d.breakpoints_enabled) return; + if (arch != .x86_64 and !arch.isAARCH64()) return error.Unsupported; + const act: linux.Sigaction = .{ + .handler = .{ .sigaction = trapHandler }, + .mask = linux.sigemptyset(), + .flags = linux.SA.SIGINFO | linux.SA.RESTART, + }; + if (linux.errno(linux.sigaction(.TRAP, &act, &d.old_trap_action)) != .SUCCESS) return error.Unexpected; + d.breakpoints_enabled = true; + } + + /// Restores the previous SIGTRAP disposition and resumes every parked thread. + pub fn disableBreakpoints(d: *Debug) void { + if (!d.breakpoints_enabled) return; + _ = linux.sigaction(.TRAP, &d.old_trap_action, null); + d.breakpoints_enabled = false; + for (&d.paused) |*slot| { + if (slot.state.cmpxchgStrong(pause_paused, pause_resuming, .acq_rel, .acquire) == null) + futexWake(&slot.state); + } + } + + // ---------------------------------------------------------------- threads + + /// The nth tid of this process, numerically sorted; null past the end. + /// Index 0 rescans /proc/self/task; higher indices reuse that scan. + pub fn threadAt(d: *Debug, index: usize) ?u32 { + if (index == 0 or d.tid_count == 0) d.scanThreads(); + if (index >= d.tid_count) return null; + return d.tids[index]; + } + + pub fn threadExists(d: *Debug, tid: u32) bool { + _ = d; + var path_buf: [64]u8 = undefined; + const path = std.fmt.bufPrintZ(&path_buf, "/proc/self/task/{d}/comm", .{tid}) catch return false; + var buf: [32]u8 = undefined; + _ = readFile(path, &buf) catch return false; + return true; + } + + /// The thread's comm (without the trailing newline). + pub fn threadName(d: *Debug, tid: u32, w: *Writer) Error!void { + _ = d; + var path_buf: [64]u8 = undefined; + const path = std.fmt.bufPrintZ(&path_buf, "/proc/self/task/{d}/comm", .{tid}) catch return error.Unexpected; + var buf: [64]u8 = undefined; + const text = readFile(path, &buf) catch |err| switch (err) { + error.NotFound => return error.NoThread, + else => return error.Unexpected, + }; + w.writeAll(std.mem.trimEnd(u8, text, "\n")) catch return error.WriteFailed; + } + + /// A few fields of /proc/self/task//stat, one "name value" per line: + /// state, utime, stime, minflt, majflt, priority, nice, processor. + pub fn threadStat(d: *Debug, tid: u32, w: *Writer) Error!void { + _ = d; + var path_buf: [64]u8 = undefined; + const path = std.fmt.bufPrintZ(&path_buf, "/proc/self/task/{d}/stat", .{tid}) catch return error.Unexpected; + var buf: [1024]u8 = undefined; + const text = readFile(path, &buf) catch |err| switch (err) { + error.NotFound => return error.NoThread, + else => return error.Unexpected, + }; + // " () S "; comm may contain spaces and parens. + const close = std.mem.lastIndexOfScalar(u8, text, ')') orelse return error.Unexpected; + var it = std.mem.tokenizeScalar(u8, text[close + 1 ..], ' '); + // Field numbers below are 0-based from `state`. + const wanted = [_]struct { idx: usize, name: []const u8 }{ + .{ .idx = 0, .name = "state" }, + .{ .idx = 11, .name = "utime" }, + .{ .idx = 12, .name = "stime" }, + .{ .idx = 7, .name = "minflt" }, + .{ .idx = 9, .name = "majflt" }, + .{ .idx = 15, .name = "priority" }, + .{ .idx = 16, .name = "nice" }, + .{ .idx = 36, .name = "processor" }, + }; + var fields: [40][]const u8 = undefined; + var n: usize = 0; + while (it.next()) |f| : (n += 1) { + if (n == fields.len) break; + fields[n] = f; + } + for (wanted) |want| { + const value = if (want.idx < n) fields[want.idx] else "?"; + w.print("{s} {s}\n", .{ want.name, value }) catch return error.WriteFailed; + } + } + + /// "#n 0x in (::)" per frame. The calling + /// thread unwinds itself directly; any other thread is captured with the + /// capture signal. + pub fn threadStack(d: *Debug, tid: u32, w: *Writer) Error!void { + var addrs: [max_frames]usize = undefined; + var trace: std.debug.StackTrace = undefined; + if (tid == selfTid()) { + trace = std.debug.captureCurrentStackTrace(.{}, &addrs); + } else { + try d.captureThread(tid); + if (!d.selfInfoFree()) { + d.releaseCapture(); + return error.Busy; + } + trace = d.unwindContext(&d.capture.ctx, &addrs); + d.releaseCapture(); + } + try d.writeFrames(trace.return_addresses, w); + } + + /// " 0x" per general register, plus pc/sp/fp aliases. + pub fn threadRegs(d: *Debug, tid: u32, w: *Writer) Error!void { + if (tid == selfTid()) { + const ctx = Native.current(); + return writeRegs(&ctx, w); + } + try d.captureThread(tid); + const ctx = d.capture.ctx; + d.releaseCapture(); + return writeRegs(&ctx, w); + } + + // ------------------------------------------------------ addresses & memory + + /// "fn\nfile:line:col\nmodule\n", unknown parts as "?". + pub fn resolveAddr(d: *Debug, addr: usize, w: *Writer) Error!void { + if (!d.knownCode(addr)) return w.writeAll("?\n?\n?\n") catch error.WriteFailed; + var fba = std.heap.FixedBufferAllocator.init(d.text_buf); + const alloc = fba.allocator(); + const di = std.debug.getSelfDebugInfo() catch return error.Unsupported; + var sym = std.debug.Symbol.unknown; + var symbols: std.ArrayList(std.debug.Symbol) = .empty; + if (di.getSymbols(d.io, alloc, alloc, addr, true, &symbols)) { + if (symbols.items.len > 0) sym = symbols.items[0]; + } else |_| {} + w.print("{s}\n", .{sym.name orelse "?"}) catch return error.WriteFailed; + if (sym.source_location) |sl| { + w.print("{s}:{d}:{d}\n", .{ sl.file_name, sl.line, sl.column }) catch return error.WriteFailed; + } else { + w.writeAll("?\n") catch return error.WriteFailed; + } + const module = di.getModuleName(d.io, addr) catch "?"; + w.print("{s}\n", .{module}) catch return error.WriteFailed; + } + + /// Reads `buf.len` bytes at `addr` via process_vm_readv on the own + /// process. Never faults. Returns the number of bytes read (short when the + /// range crosses into an unmapped page); `error.Unmapped` when nothing + /// could be read. + pub fn readMem(d: *Debug, addr: usize, buf: []u8) Error!usize { + _ = d; + if (buf.len == 0) return 0; + // Page 0 is never mapped (mmap_min_addr) and a null `iovec.base` is a + // safety-checked cast; the same answer without the trap. + if (addr == 0) return error.Unmapped; + const local = [_]std.posix.iovec{.{ .base = buf.ptr, .len = buf.len }}; + const remote = [_]std.posix.iovec_const{.{ .base = @ptrFromInt(addr), .len = buf.len }}; + const rc = linux.process_vm_readv(linux.getpid(), &local, &remote, 0); + switch (linux.errno(rc)) { + .SUCCESS => return rc, + .FAULT => return error.Unmapped, + .NOSYS, .PERM => return error.Unsupported, + else => return error.Unexpected, + } + } + + /// Writes `data` at `addr` via process_vm_writev. Read-only mappings also + /// report `error.Unmapped` (the kernel says EFAULT for both). + pub fn writeMem(d: *Debug, addr: usize, data: []const u8) Error!usize { + _ = d; + if (data.len == 0) return 0; + if (addr == 0) return error.Unmapped; + const local = [_]std.posix.iovec_const{.{ .base = data.ptr, .len = data.len }}; + const remote = [_]std.posix.iovec_const{.{ .base = @ptrFromInt(addr), .len = data.len }}; + const rc = linux.process_vm_writev(linux.getpid(), &local, &remote, 0); + switch (linux.errno(rc)) { + .SUCCESS => return rc, + .FAULT => return error.Unmapped, + .NOSYS, .PERM => return error.Unsupported, + else => return error.Unexpected, + } + } + + /// Hexdump of `len` bytes at `addr` in the shape of `std.debug.dumpHex` + /// (16 bytes per line, address column, bytes in two groups, ASCII column). + /// Stops early at the first unmapped byte; `error.Unmapped` only when the + /// very first chunk is unreadable. + pub fn hexdump(d: *Debug, addr: usize, len: usize, w: *Writer) Error!void { + var chunk: [256]u8 = undefined; + var done: usize = 0; + while (done < len) { + const want = @min(chunk.len, len - done); + const got = d.readMem(addr +% done, chunk[0..want]) catch |err| switch (err) { + error.Unmapped => if (done == 0) return error.Unmapped else break, + else => return err, + }; + if (got == 0) break; + try writeHexLines(addr +% done, chunk[0..got], w); + done += got; + if (got < want) break; + } + } + + /// Copies /proc/self/maps to `w`. + pub fn maps(d: *Debug, w: *Writer) Error!void { + _ = d; + return streamFile("/proc/self/maps", w); + } + + /// Reads `buf.len` bytes of /proc/self/maps at `offset` (0 at the end). + /// Not a consistent snapshot across reads; a map appearing between two + /// reads shifts the text, like `cat` on /proc itself. + pub fn readMaps(d: *Debug, offset: u64, buf: []u8) Error!usize { + _ = d; + if (offset > std.math.maxInt(i64)) return 0; + return preadFile("/proc/self/maps", offset, buf); + } + + // ------------------------------------------------------------ breakpoints + + /// The nth tid currently parked in `@breakpoint()`. + pub fn pausedAt(d: *Debug, index: usize) ?u32 { + var n: usize = 0; + for (d.paused[0..d.max_paused]) |*slot| { + if (slot.state.load(.acquire) != pause_paused) continue; + if (n == index) return slot.tid.load(.acquire); + n += 1; + } + return null; + } + + pub fn isPaused(d: *Debug, tid: u32) bool { + return d.pausedSlot(tid) != null; + } + + pub fn pausedStack(d: *Debug, tid: u32, w: *Writer) Error!void { + const slot = d.pausedSlot(tid) orelse return error.NotPaused; + if (!d.selfInfoFree()) return error.Busy; + var addrs: [max_frames]usize = undefined; + const trace = d.unwindContext(&slot.ctx, &addrs); + try d.writeFrames(trace.return_addresses, w); + } + + pub fn pausedRegs(d: *Debug, tid: u32, w: *Writer) Error!void { + const slot = d.pausedSlot(tid) orelse return error.NotPaused; + return writeRegs(&slot.ctx, w); + } + + /// Lets a parked thread continue past its breakpoint. + pub fn resumeThread(d: *Debug, tid: u32) Error!void { + const slot = d.pausedSlot(tid) orelse return error.NotPaused; + if (slot.state.cmpxchgStrong(pause_paused, pause_resuming, .acq_rel, .acquire) != null) return error.NotPaused; + futexWake(&slot.state); + } + + fn pausedSlot(d: *Debug, tid: u32) ?*PauseSlot { + for (d.paused[0..d.max_paused]) |*slot| { + if (slot.state.load(.acquire) == pause_paused and slot.tid.load(.acquire) == tid) return slot; + } + return null; + } + + // ------------------------------------------------------------------ panic + + /// The recorded panic message; nothing before any panic. + pub fn panicMessage(d: *Debug, w: *Writer) Error!void { + _ = d; + if (panic_state.load(.acquire) == panic_none) return; + w.writeAll(panic_msg[0..panic_msg_len]) catch return error.WriteFailed; + } + + /// Frames of the panicking thread, symbolized lazily. + pub fn panicStack(d: *Debug, w: *Writer) Error!void { + if (panic_state.load(.acquire) == panic_none) return; + try d.writeFrames(panic_addrs[0..panic_addr_count], w); + } + + /// True while a panicking thread is parked waiting for `panicContinue`. + pub fn panicHeld(d: *Debug) bool { + _ = d; + return panic_state.load(.acquire) == panic_held; + } + + /// Releases the held panicking thread into `std.debug.defaultPanic`. + pub fn panicContinue(d: *Debug) Error!void { + _ = d; + if (panic_state.cmpxchgStrong(panic_held, panic_continued, .acq_rel, .acquire) != null) return error.NoPanic; + futexWake(&panic_state); + } + + // -------------------------------------------------------------- internals + + fn scanThreads(d: *Debug) void { + d.tid_count = 0; + const fd_rc = linux.open("/proc/self/task", .{ .ACCMODE = .RDONLY, .DIRECTORY = true, .CLOEXEC = true }, 0); + if (linux.errno(fd_rc) != .SUCCESS) return; + const fd: i32 = @intCast(fd_rc); + defer _ = linux.close(fd); + var buf: [4096]u8 align(@alignOf(linux.dirent64)) = undefined; + while (true) { + const rc = linux.getdents64(fd, &buf, buf.len); + if (linux.errno(rc) != .SUCCESS or rc == 0) break; + var off: usize = 0; + while (off < rc) { + const ent: *align(1) const linux.dirent64 = @ptrCast(&buf[off]); + const name_ptr: [*:0]const u8 = @ptrCast(&buf[off + @offsetOf(linux.dirent64, "name")]); + const name = std.mem.span(name_ptr); + if (std.fmt.parseInt(u32, name, 10)) |tid| { + if (d.tid_count < max_threads) { + d.tids[d.tid_count] = tid; + d.tid_count += 1; + } + } else |_| {} + off += ent.reclen; + } + } + std.mem.sort(u32, d.tids[0..d.tid_count], {}, std.sort.asc(u32)); + } + + /// Arms the capture slot for `tid`, signals it and waits until the handler + /// has parked with its context copied. On success the caller owns the + /// slot until `releaseCapture`. + fn captureThread(d: *Debug, tid: u32) Error!void { + const slot = &d.capture; + slot.target_tid.store(tid, .release); + slot.state.store(cap_armed, .release); + const rc = linux.tgkill(linux.getpid(), @intCast(tid), d.capture_signal); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .SRCH => { + slot.state.store(cap_idle, .release); + return error.NoThread; + }, + else => { + slot.state.store(cap_idle, .release); + return error.Unexpected; + }, + } + const deadline = monotonicNs() + d.capture_timeout_ns; + while (true) { + const s = slot.state.load(.acquire); + switch (s) { + cap_captured => return, + cap_failed => { + slot.state.store(cap_idle, .release); + return error.Unsupported; + }, + cap_armed => { + const now = monotonicNs(); + if (now >= deadline) { + // Disarm; if the handler raced us it has moved on to + // `capturing` and we simply keep waiting for it. + if (slot.state.cmpxchgStrong(cap_armed, cap_idle, .acq_rel, .acquire) == null) return error.Timeout; + continue; + } + futexWaitNs(&slot.state, cap_armed, deadline - now); + }, + // The handler is copying registers; it finishes promptly. + cap_capturing => futexWaitNs(&slot.state, cap_capturing, 1 * std.time.ns_per_ms), + else => unreachable, + } + } + } + + /// Records the PT_LOAD ranges of every module `dl_iterate_phdr` reports + /// (for a static executable: the executable itself, not the vDSO). + fn scanModules(d: *Debug) void { + d.range_count = 0; + std.posix.dl_iterate_phdr(d, error{}, struct { + fn cb(info: *std.posix.dl_phdr_info, _: usize, ctx: *Debug) error{}!void { + for (info.phdr[0..info.phnum]) |phdr| { + if (phdr.type != .LOAD) continue; + if (ctx.range_count == max_ranges) return; + ctx.ranges[ctx.range_count] = .{ .start = info.addr +% phdr.vaddr, .len = phdr.memsz }; + ctx.range_count += 1; + } + } + }.cb) catch {}; + } + + /// True when `addr` lies in a module `std.debug` already knows about, so + /// that asking it about `addr` cannot trigger a module rescan. + fn knownCode(d: *const Debug, addr: usize) bool { + for (d.ranges[0..d.range_count]) |r| { + if (addr >= r.start and addr - r.start < r.len) return true; + } + return false; + } + + /// Unwinds from a saved context. A pc outside every known module (e.g. a + /// thread inside the vDSO) is reported as a single frame and not unwound, + /// because std would otherwise rescan its module list (see the header). + fn unwindContext(d: *const Debug, ctx: *const Native, addrs: *[max_frames]usize) std.debug.StackTrace { + if (!d.knownCode(ctx.getPc())) { + addrs[0] = ctx.getPc() +| 1; + return .{ .return_addresses = addrs[0..1], .skipped = .unknown }; + } + return std.debug.captureCurrentStackTrace(.{ .context = ctx }, addrs); + } + + /// True when nobody holds std.debug's `SelfInfo` lock right now. Called + /// with the target parked, so a held lock means the *target* (or another + /// live thread, which will let go) holds it; only the former deadlocks, + /// and the caller cannot tell them apart, so both yield `error.Busy`. + fn selfInfoFree(d: *const Debug) bool { + if (comptime !@hasField(std.debug.SelfInfo, "rwlock")) return true; + const di = std.debug.getSelfDebugInfo() catch return true; + if (!di.rwlock.tryLock(d.io)) return false; + di.rwlock.unlock(d.io); + return true; + } + + fn releaseCapture(d: *Debug) void { + d.capture.state.store(cap_idle, .release); + futexWake(&d.capture.state); + } + + fn writeFrames(d: *Debug, addrs: []const usize, w: *Writer) Error!void { + var fba = std.heap.FixedBufferAllocator.init(d.text_buf); + const alloc = fba.allocator(); + const di = std.debug.getSelfDebugInfo() catch return error.Unsupported; + for (addrs, 0..) |ret_addr, i| { + // Return addresses point after the call; the first frame of a + // context capture is stored as pc+1 by std for the same reason. + const addr = ret_addr -| 1; + fba.reset(); + var symbols: std.ArrayList(std.debug.Symbol) = .empty; + var sym = std.debug.Symbol.unknown; + if (d.knownCode(addr)) { + if (di.getSymbols(d.io, alloc, alloc, addr, true, &symbols)) { + if (symbols.items.len > 0) sym = symbols.items[0]; + } else |_| {} + } + w.print("#{d} 0x{x} in {s} (", .{ i, addr, sym.name orelse "?" }) catch return error.WriteFailed; + if (sym.source_location) |sl| { + w.print("{s}:{d}:{d})\n", .{ sl.file_name, sl.line, sl.column }) catch return error.WriteFailed; + } else { + w.writeAll("?)\n") catch return error.WriteFailed; + } + } + } +}; + +// ------------------------------------------------------------------ handlers + +fn selfTid() u32 { + return @intCast(linux.gettid()); +} + +fn captureHandler(_: linux.SIG, _: *const linux.siginfo_t, ctx_ptr: ?*anyopaque) callconv(.c) void { + const d = current orelse return; + const slot = &d.capture; + const me = selfTid(); + if (slot.target_tid.load(.acquire) != me) return; + if (slot.state.cmpxchgStrong(cap_armed, cap_capturing, .acq_rel, .acquire) != null) return; + // The tid check and the swap are not one atomic step: a stale run (a + // signal that stayed pending while its request timed out) may have read + // the old tid and then won the swap of a request re-armed for another + // thread. `target_tid` is fixed while the slot is armed, so re-checking + // after the swap closes the window; hand the slot back untouched. + if (slot.target_tid.load(.acquire) != me) { + slot.state.store(cap_armed, .release); + futexWake(&slot.state); + return; + } + if (cpu_context.fromPosixSignalContext(ctx_ptr)) |ctx| { + slot.ctx = ctx; + slot.state.store(cap_captured, .release); + futexWake(&slot.state); + while (slot.state.load(.acquire) == cap_captured) futexWaitNs(&slot.state, cap_captured, null); + } else { + slot.state.store(cap_failed, .release); + futexWake(&slot.state); + } +} + +/// aarch64 Linux ucontext_t, only as far as `mcontext.pc` (see +/// std.debug.cpu_context's signal_ucontext_t). +const UcontextAarch64 = extern struct { + flags: usize, + link: ?*UcontextAarch64, + stack: linux.stack_t, + sigmask: linux.sigset_t, + unused: [120]u8, + mcontext: extern struct { + fault_address: u64 align(16), + x: [30]u64, + lr: u64, + sp: u64, + pc: u64, + }, +}; + +fn trapHandler(_: linux.SIG, info: *const linux.siginfo_t, ctx_ptr: ?*anyopaque) callconv(.c) void { + const d = current orelse return trapFallback(); + const ctx = cpu_context.fromPosixSignalContext(ctx_ptr) orelse return trapFallback(); + // si_code > 0 is kernel-generated (TRAP_BRKPT for int3/brk); <= 0 is + // kill/tgkill/sigqueue from user space, where PC points at the + // interrupted instruction and must not be touched. + const from_instruction = info.code > 0; + if (comptime arch.isAARCH64()) { + // `brk #imm` does not advance PC; step over it so returning from the + // handler does not re-trap. + if (from_instruction) { + const uc: *UcontextAarch64 = @ptrCast(@alignCast(ctx_ptr.?)); + uc.mcontext.pc += 4; + } + } else if (comptime arch != .x86_64) { + return trapFallback(); + } + const tid = selfTid(); + if (tid == server_tid.load(.acquire)) { + // Nobody could resume the thread that serves /breakpoints: step over. + _ = traps_skipped.fetchAdd(1, .acq_rel); + return; + } + const slot: *PauseSlot = for (d.paused[0..d.max_paused]) |*slot| { + if (slot.state.cmpxchgStrong(pause_free, pause_claimed, .acq_rel, .acquire) == null) break slot; + } else { + _ = traps_skipped.fetchAdd(1, .acq_rel); + return; + }; + slot.ctx = ctx; + slot.tid.store(tid, .release); + slot.state.store(pause_paused, .release); + while (slot.state.load(.acquire) == pause_paused) futexWaitNs(&slot.state, pause_paused, null); + slot.state.store(pause_free, .release); +} + +/// Restores the default SIGTRAP disposition and re-raises it: the signal is +/// blocked while the handler runs, so it is delivered (fatally) on return. +/// Only for a handler run with no `Debug` (a trap in flight during `deinit`) +/// or on an architecture whose context cannot be read. +fn trapFallback() void { + const act: linux.Sigaction = .{ + .handler = .{ .handler = linux.SIG.DFL }, + .mask = linux.sigemptyset(), + .flags = 0, + }; + _ = linux.sigaction(.TRAP, &act, null); + _ = linux.tkill(linux.gettid(), .TRAP); +} + +// --------------------------------------------------------------------- panic + +const panic_none: u32 = 0; +const panic_recording: u32 = 1; +const panic_recorded: u32 = 2; +const panic_held: u32 = 3; +const panic_continued: u32 = 4; + +var panic_state: std.atomic.Value(u32) = .init(panic_none); +var panic_msg: [panic_msg_cap]u8 = undefined; +var panic_msg_len: usize = 0; +var panic_addrs: [max_frames]usize = undefined; +var panic_addr_count: usize = 0; +/// The tid of the panicking thread (0 before any panic). +pub var panic_tid: u32 = 0; + +/// Records the first panic: message (bounded copy) and stack addresses. +/// Returns false if a panic was already recorded (nested or second panic). +pub fn recordPanic(msg: []const u8, first_trace_addr: ?usize) bool { + if (panic_state.cmpxchgStrong(panic_none, panic_recording, .acq_rel, .acquire) != null) return false; + panic_tid = selfTid(); + panic_msg_len = @min(msg.len, panic_msg.len); + @memcpy(panic_msg[0..panic_msg_len], msg[0..panic_msg_len]); + const trace = std.debug.captureCurrentStackTrace(.{ .first_address = first_trace_addr }, &panic_addrs); + panic_addr_count = trace.return_addresses.len; + panic_state.store(panic_recorded, .release); + return true; +} + +/// Parks the panicking thread until `Debug.panicContinue` when holding is +/// enabled and a `Debug` exists; then hands over to `std.debug.defaultPanic`. +pub fn panicHook(msg: []const u8, first_trace_addr: ?usize) noreturn { + @branchHint(.cold); + if (recordPanic(msg, first_trace_addr)) { + // The server thread cannot be held: it is the one that would have to + // serve /panic/ctl. + if (hold_on_panic and current != null and panic_tid != server_tid.load(.acquire)) { + if (panic_state.cmpxchgStrong(panic_recorded, panic_held, .acq_rel, .acquire) == null) { + while (panic_state.load(.acquire) == panic_held) futexWaitNs(&panic_state, panic_held, null); + } + } + } + std.debug.defaultPanic(msg, first_trace_addr); +} + +/// Clears the recorded panic. Only meaningful in tests of the record path. +pub fn resetPanicRecord() void { + panic_msg_len = 0; + panic_addr_count = 0; + panic_tid = 0; + panic_state.store(panic_none, .release); +} + +// ------------------------------------------------------------------- helpers + +fn futexWake(word: *std.atomic.Value(u32)) void { + _ = linux.futex_3arg(&word.raw, .{ .cmd = .WAKE, .private = true }, std.math.maxInt(u32)); +} + +/// Waits while `*word == expect`, at most `timeout_ns` (forever when null). +/// Returns on wake, timeout, value change or EINTR; callers loop. +fn futexWaitNs(word: *std.atomic.Value(u32), expect: u32, timeout_ns: ?u64) void { + var ts: linux.timespec = undefined; + const ts_ptr: ?*const linux.timespec = if (timeout_ns) |ns| blk: { + ts = .{ .sec = @intCast(ns / std.time.ns_per_s), .nsec = @intCast(ns % std.time.ns_per_s) }; + break :blk &ts; + } else null; + _ = linux.futex_4arg(&word.raw, .{ .cmd = .WAIT, .private = true }, expect, ts_ptr); +} + +fn monotonicNs() u64 { + var ts: linux.timespec = undefined; + _ = linux.clock_gettime(.MONOTONIC, &ts); + return @as(u64, @intCast(ts.sec)) * std.time.ns_per_s + @as(u64, @intCast(ts.nsec)); +} + +const FileError = error{ NotFound, Unexpected, TooBig }; + +/// Reads a whole (small) file with raw syscalls. +fn readFile(path: [*:0]const u8, buf: []u8) FileError![]u8 { + const fd_rc = linux.open(path, .{ .ACCMODE = .RDONLY, .CLOEXEC = true }, 0); + switch (linux.errno(fd_rc)) { + .SUCCESS => {}, + .NOENT, .SRCH => return error.NotFound, + else => return error.Unexpected, + } + const fd: i32 = @intCast(fd_rc); + defer _ = linux.close(fd); + var len: usize = 0; + while (len < buf.len) { + const rc = linux.read(fd, buf[len..].ptr, buf.len - len); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .INTR => continue, + .SRCH, .NOENT => return error.NotFound, + else => return error.Unexpected, + } + if (rc == 0) return buf[0..len]; + len += rc; + } + return error.TooBig; +} + +/// One pread of `buf.len` bytes at `offset`; 0 at the end of the file. +fn preadFile(path: [*:0]const u8, offset: u64, buf: []u8) Error!usize { + const fd_rc = linux.open(path, .{ .ACCMODE = .RDONLY, .CLOEXEC = true }, 0); + if (linux.errno(fd_rc) != .SUCCESS) return error.Unexpected; + const fd: i32 = @intCast(fd_rc); + defer _ = linux.close(fd); + var len: usize = 0; + while (len < buf.len) { + const rc = linux.pread(fd, buf[len..].ptr, buf.len - len, @intCast(offset + len)); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .INTR => continue, + else => return error.Unexpected, + } + if (rc == 0) break; + len += rc; + } + return len; +} + +/// Streams a file of any size to `w`. +fn streamFile(path: [*:0]const u8, w: *Writer) Error!void { + const fd_rc = linux.open(path, .{ .ACCMODE = .RDONLY, .CLOEXEC = true }, 0); + if (linux.errno(fd_rc) != .SUCCESS) return error.Unexpected; + const fd: i32 = @intCast(fd_rc); + defer _ = linux.close(fd); + var buf: [4096]u8 = undefined; + while (true) { + const rc = linux.read(fd, &buf, buf.len); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .INTR => continue, + else => return error.Unexpected, + } + if (rc == 0) return; + w.writeAll(buf[0..rc]) catch return error.WriteFailed; + } +} + +fn writeHexLines(base: usize, bytes: []const u8, w: *Writer) Error!void { + var offset: usize = 0; + while (offset < bytes.len) : (offset += 16) { + const line = bytes[offset..@min(offset + 16, bytes.len)]; + w.print("{x:0>[1]} ", .{ base +% offset, @sizeOf(usize) * 2 }) catch return error.WriteFailed; + for (line, 0..) |byte, i| { + w.print("{X:0>2} ", .{byte}) catch return error.WriteFailed; + if (i == 7) w.writeByte(' ') catch return error.WriteFailed; + } + w.writeByte(' ') catch return error.WriteFailed; + if (line.len < 16) { + var missing = (16 - line.len) * 3; + if (line.len < 8) missing += 1; + w.splatByteAll(' ', missing) catch return error.WriteFailed; + } + for (line) |byte| { + w.writeByte(if (std.ascii.isPrint(byte)) byte else '.') catch return error.WriteFailed; + } + w.writeByte('\n') catch return error.WriteFailed; + } +} + +fn writeRegs(ctx: *const Native, w: *Writer) Error!void { + if (comptime arch == .x86_64) { + inline for (@typeInfo(Native.Gpr).@"enum".fields) |f| { + w.print("{s} 0x{x}\n", .{ f.name, ctx.gprs.get(@field(Native.Gpr, f.name)) }) catch return error.WriteFailed; + } + w.print("pc 0x{x}\nsp 0x{x}\nfp 0x{x}\n", .{ + ctx.gprs.get(.rip), ctx.gprs.get(.rsp), ctx.gprs.get(.rbp), + }) catch return error.WriteFailed; + } else if (comptime arch.isAARCH64()) { + for (ctx.x, 0..) |x, i| w.print("x{d} 0x{x}\n", .{ i, x }) catch return error.WriteFailed; + w.print("sp 0x{x}\npc 0x{x}\nfp 0x{x}\nlr 0x{x}\n", .{ + ctx.sp, ctx.pc, ctx.x[29], ctx.x[30], + }) catch return error.WriteFailed; + } else { + w.print("pc 0x{x}\nfp 0x{x}\n", .{ ctx.getPc(), ctx.getFp() }) catch return error.WriteFailed; + } +} + +// --------------------------------------------------------------------- tests + +const testing = std.testing; + +fn testOptions(text_buf: []u8) Options { + return .{ .io = testing.io, .text_buf = text_buf }; +} + +noinline fn sleepMs(ms: u64) void { + var ts: linux.timespec = .{ .sec = @intCast(ms / 1000), .nsec = @intCast((ms % 1000) * std.time.ns_per_ms) }; + _ = linux.nanosleep(&ts, null); +} + +// The test threads use atomic builtins rather than `std.atomic.Value` methods +// so that, in release modes, their pc is never inside an inlined callee: the +// DWARF symbolizer names the innermost inlined function at an address (see +// the notes on `writeFrames`). +const SpinState = struct { + tid: std.atomic.Value(u32) = .init(0), + stop: bool = false, + counter: u32 = 0, + done: bool = false, +}; + +noinline fn spinHere(st: *SpinState) void { + while (!@atomicLoad(bool, &st.stop, .acquire)) { + _ = @atomicRmw(u32, &st.counter, .Add, 1, .monotonic); + } +} + +fn spinThreadMain(st: *SpinState) void { + st.tid.store(selfTid(), .release); + spinHere(st); + @atomicStore(bool, &st.done, true, .release); // keeps the call above from becoming a tail call +} + +fn waitForTid(st: *SpinState) u32 { + var tries: usize = 0; + while (st.tid.load(.acquire) == 0) : (tries += 1) { + if (tries > 2000) return 0; + sleepMs(1); + } + return st.tid.load(.acquire); +} + +test "capture own stack" { + var text_buf: [16 * 1024]u8 = undefined; + var d: Debug = undefined; + try d.init(testOptions(&text_buf)); + defer d.deinit(); + try testing.expect(current == &d); + + var out: Writer.Allocating = .init(testing.allocator); + defer out.deinit(); + try d.threadStack(selfTid(), &out.writer); + const text = out.written(); + try testing.expect(std.mem.indexOf(u8, text, "#0 0x") != null); + try testing.expect(std.mem.indexOf(u8, text, "debug.zig:") != null); + try testing.expect(std.mem.indexOf(u8, text, "test.capture own stack") != null); + + out.clearRetainingCapacity(); + try d.threadRegs(selfTid(), &out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "pc 0x") != null); + try testing.expect(std.mem.indexOf(u8, out.written(), "pc 0x0\n") == null); +} + +test "capture another thread: stack, regs, name, stat" { + var text_buf: [16 * 1024]u8 = undefined; + var d: Debug = undefined; + try d.init(testOptions(&text_buf)); + defer d.deinit(); + + var st: SpinState = .{}; + const th = try std.Thread.spawn(.{}, spinThreadMain, .{&st}); + const tid = waitForTid(&st); + try testing.expect(tid != 0); + + var out: Writer.Allocating = .init(testing.allocator); + defer out.deinit(); + try d.threadStack(tid, &out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "spinHere") != null); + try testing.expect(std.mem.indexOf(u8, out.written(), "spinThreadMain") != null); + + out.clearRetainingCapacity(); + try d.threadRegs(tid, &out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "pc 0x") != null); + try testing.expect(std.mem.indexOf(u8, out.written(), "pc 0x0\n") == null); + + out.clearRetainingCapacity(); + try d.threadName(tid, &out.writer); + try testing.expect(out.written().len > 0); + try testing.expect(std.mem.indexOfScalar(u8, out.written(), '\n') == null); + + out.clearRetainingCapacity(); + try d.threadStat(tid, &out.writer); + try testing.expect(std.mem.startsWith(u8, out.written(), "state ")); + try testing.expect(std.mem.indexOf(u8, out.written(), "\nutime ") != null); + + // Enumeration lists both threads and nothing bogus. + try testing.expect(d.threadExists(tid)); + try testing.expect(d.threadExists(selfTid())); + var found_self = false; + var found_other = false; + var i: usize = 0; + var prev: u32 = 0; + while (d.threadAt(i)) |t| : (i += 1) { + try testing.expect(t > prev); + prev = t; + if (t == tid) found_other = true; + if (t == selfTid()) found_self = true; + } + try testing.expect(found_self and found_other); + + // Repeated captures of the same thread keep working. + var k: usize = 0; + while (k < 5) : (k += 1) { + out.clearRetainingCapacity(); + try d.threadStack(tid, &out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "spinHere") != null); + } + const before = @atomicLoad(u32, &st.counter, .acquire); + sleepMs(2); + try testing.expect(@atomicLoad(u32, &st.counter, .acquire) != before); // the thread is running again + + @atomicStore(bool, &st.stop, true, .release); + th.join(); + try testing.expect(!d.threadExists(tid)); + try testing.expectError(error.NoThread, d.threadStack(tid, &out.writer)); + try testing.expectError(error.NoThread, d.threadName(tid, &out.writer)); +} + +/// The address of the call site in the caller, i.e. inside this file's test. +noinline fn callerAddress() usize { + return @returnAddress() - 1; +} + +test "resolveAddr names this file" { + var text_buf: [16 * 1024]u8 = undefined; + var d: Debug = undefined; + try d.init(testOptions(&text_buf)); + defer d.deinit(); + var out: Writer.Allocating = .init(testing.allocator); + defer out.deinit(); + try d.resolveAddr(callerAddress(), &out.writer); + const text = out.written(); + var lines = std.mem.splitScalar(u8, text, '\n'); + const fn_name = lines.next().?; + const loc = lines.next().?; + const module = lines.next().?; + try testing.expect(fn_name.len > 0 and !std.mem.eql(u8, fn_name, "?")); + try testing.expect(std.mem.indexOf(u8, loc, "debug.zig:") != null); + try testing.expect(module.len > 0); + + out.clearRetainingCapacity(); + try d.resolveAddr(8, &out.writer); + try testing.expectEqualStrings("?\n?\n?\n", out.written()); + + // Regression: an unmapped lookup must not poison std's unwind cache (see + // the header); unwinding afterwards still works. + out.clearRetainingCapacity(); + try d.threadStack(selfTid(), &out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "test.resolveAddr names this file") != null); +} + +test "readMem, writeMem, hexdump" { + var text_buf: [16 * 1024]u8 = undefined; + var d: Debug = undefined; + try d.init(testOptions(&text_buf)); + defer d.deinit(); + + var value: [8]u8 = .{ 1, 2, 3, 4, 5, 6, 7, 8 }; + var got: [8]u8 = undefined; + try testing.expectEqual(@as(usize, 8), try d.readMem(@intFromPtr(&value), &got)); + try testing.expectEqualSlices(u8, &value, &got); + try testing.expectError(error.Unmapped, d.readMem(8, &got)); + + const new = [_]u8{ 0xaa, 0xbb, 0xcc }; + try testing.expectEqual(@as(usize, 3), try d.writeMem(@intFromPtr(&value) + 2, &new)); + try testing.expectEqualSlices(u8, &.{ 1, 2, 0xaa, 0xbb, 0xcc, 6, 7, 8 }, &value); + try testing.expectError(error.Unmapped, d.writeMem(8, &new)); + + var bytes: [19]u8 = .{ 0x00, 0x11, 0x22, 0x33, 0x44, 0x55, 0x66, 0x77, 0x88, 0x99, 0xaa, 0xbb, 0xcc, 0xdd, 0xee, 0xff, 0x01, 0x12, 0x13 }; + var out: Writer.Allocating = .init(testing.allocator); + defer out.deinit(); + try d.hexdump(@intFromPtr(&bytes), bytes.len, &out.writer); + const expected = try std.fmt.allocPrint(testing.allocator, + \\{x:0>[2]} 00 11 22 33 44 55 66 77 88 99 AA BB CC DD EE FF .."3DUfw........ + \\{x:0>[2]} 01 12 13 ... + \\ + , .{ @intFromPtr(&bytes), @intFromPtr(&bytes) + 16, @sizeOf(usize) * 2 }); + defer testing.allocator.free(expected); + try testing.expectEqualStrings(expected, out.written()); + try testing.expectError(error.Unmapped, d.hexdump(8, 16, &out.writer)); + + // Address 0 (also reached by an offset that wraps) must be an error, not a + // safety-checked null pointer cast on the server thread. + try testing.expectError(error.Unmapped, d.readMem(0, &got)); + try testing.expectError(error.Unmapped, d.writeMem(0, &new)); + try testing.expectError(error.Unmapped, d.hexdump(0, 16, &out.writer)); + try testing.expectError(error.Unmapped, d.readMem(std.math.maxInt(usize) - 3, &got)); + try testing.expectError(error.Unmapped, d.hexdump(std.math.maxInt(usize) - 3, 16, &out.writer)); + + out.clearRetainingCapacity(); + try d.maps(&out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "[stack]") != null); + + // readMaps serves the file piecewise at any offset and ends with 0. + var piece: [4096]u8 = undefined; + var total: usize = 0; + while (true) { + const n = try d.readMaps(total, &piece); + if (n == 0) break; + total += n; + } + try testing.expect(total >= out.written().len / 2); + try testing.expectEqual(@as(usize, 0), try d.readMaps(std.math.maxInt(u64), &piece)); +} + +test "breakpoint on the server thread and past the slot table steps over; tgkill SIGTRAP parks" { + if (arch != .x86_64 and !arch.isAARCH64()) return error.SkipZigTest; + var text_buf: [16 * 1024]u8 = undefined; + var d: Debug = undefined; + var opts = testOptions(&text_buf); + opts.max_paused = 1; + try d.init(opts); + defer d.deinit(); + try d.enableBreakpoints(); + defer d.disableBreakpoints(); + var out: Writer.Allocating = .init(testing.allocator); + defer out.deinit(); + + // The "server" thread (this one, for the test) hits a breakpoint: it keeps running. + const skipped0 = traps_skipped.load(.acquire); + server_tid.store(selfTid(), .release); + defer server_tid.store(0, .release); + @breakpoint(); + try testing.expectEqual(skipped0 + 1, traps_skipped.load(.acquire)); + try testing.expect(!d.isPaused(selfTid())); + + // One slot: the first trapping thread parks, the second steps over. + var a: TrapState = .{}; + const ta = try std.Thread.spawn(.{}, trapThreadMain, .{&a}); + var tries: usize = 0; + while (a.tid.load(.acquire) == 0 or !d.isPaused(a.tid.load(.acquire))) : (tries += 1) { + try testing.expect(tries < 5000); + sleepMs(1); + } + var b: TrapState = .{}; + const tb = try std.Thread.spawn(.{}, trapThreadMain, .{&b}); + tb.join(); + try testing.expectEqual(@as(u32, 1), @atomicLoad(u32, &b.counter, .acquire)); + try testing.expectEqual(skipped0 + 2, traps_skipped.load(.acquire)); + try testing.expectEqual(@as(u32, 0), @atomicLoad(u32, &a.counter, .acquire)); + try d.resumeThread(a.tid.load(.acquire)); + ta.join(); + try testing.expectEqual(@as(u32, 1), @atomicLoad(u32, &a.counter, .acquire)); + + // A SIGTRAP sent with tgkill (not an int3/brk) parks the thread where it + // was; resuming it must not skip an instruction: the spinner keeps counting. + var st: SpinState = .{}; + const th = try std.Thread.spawn(.{}, spinThreadMain, .{&st}); + const tid = waitForTid(&st); + try testing.expect(tid != 0); + try testing.expectEqual(linux.E.SUCCESS, linux.errno(linux.tgkill(linux.getpid(), @intCast(tid), .TRAP))); + tries = 0; + while (!d.isPaused(tid)) : (tries += 1) { + try testing.expect(tries < 5000); + sleepMs(1); + } + const frozen = @atomicLoad(u32, &st.counter, .acquire); + sleepMs(5); + try testing.expectEqual(frozen, @atomicLoad(u32, &st.counter, .acquire)); + out.clearRetainingCapacity(); + try d.pausedStack(tid, &out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "spinHere") != null); + try d.resumeThread(tid); + sleepMs(5); + try testing.expect(@atomicLoad(u32, &st.counter, .acquire) != frozen); + @atomicStore(bool, &st.stop, true, .release); + th.join(); +} + +const LockState = struct { + tid: std.atomic.Value(u32) = .init(0), + release: std.atomic.Value(bool) = .init(false), + unlocked: std.atomic.Value(bool) = .init(false), + stop: std.atomic.Value(bool) = .init(false), + io: std.Io, +}; + +fn lockHolderMain(st: *LockState) void { + const di = std.debug.getSelfDebugInfo() catch return; + di.rwlock.lockUncancelable(st.io); + st.tid.store(selfTid(), .release); + while (!st.release.load(.acquire)) sleepMs(1); + di.rwlock.unlock(st.io); + st.unlocked.store(true, .release); + while (!st.stop.load(.acquire)) sleepMs(1); +} + +test "a target parked while holding std.debug's lock is Busy, not a deadlock" { + if (comptime !@hasField(std.debug.SelfInfo, "rwlock")) return error.SkipZigTest; + var text_buf: [16 * 1024]u8 = undefined; + var d: Debug = undefined; + try d.init(testOptions(&text_buf)); + defer d.deinit(); + var st: LockState = .{ .io = testing.io }; + const th = try std.Thread.spawn(.{}, lockHolderMain, .{&st}); + var tries: usize = 0; + while (st.tid.load(.acquire) == 0) : (tries += 1) { + try testing.expect(tries < 2000); + sleepMs(1); + } + const tid = st.tid.load(.acquire); + // No allocation while the holder has the lock: `testing.allocator` + // records a stack trace per allocation, which needs that same lock. + var buf: [16 * 1024]u8 = undefined; + var w: Writer = .fixed(&buf); + try testing.expectError(error.Busy, d.threadStack(tid, &w)); + try testing.expectEqual(cap_idle, d.capture.state.load(.acquire)); + // Registers need no unwind and are still available. + try d.threadRegs(tid, &w); + try testing.expect(std.mem.indexOf(u8, w.buffered(), "pc 0x") != null); + // Handshake, not a sleep: a slow holder would otherwise still hold the + // lock and the next capture would legitimately be Busy again. + st.release.store(true, .release); + tries = 0; + while (!st.unlocked.load(.acquire)) : (tries += 1) { + try testing.expect(tries < 5000); + sleepMs(1); + } + w = .fixed(&buf); + try d.threadStack(tid, &w); + try testing.expect(std.mem.indexOf(u8, w.buffered(), "lockHolderMain") != null); + st.stop.store(true, .release); + th.join(); +} + +const TrapState = struct { + tid: std.atomic.Value(u32) = .init(0), + counter: u32 = 0, +}; + +noinline fn trapThreadMain(st: *TrapState) void { + st.tid.store(selfTid(), .release); + @breakpoint(); + _ = @atomicRmw(u32, &st.counter, .Add, 1, .acq_rel); +} + +test "breakpoint: pause, inspect, resume" { + if (arch != .x86_64 and !arch.isAARCH64()) return error.SkipZigTest; + var text_buf: [16 * 1024]u8 = undefined; + var d: Debug = undefined; + try d.init(testOptions(&text_buf)); + defer d.deinit(); + try d.enableBreakpoints(); + + var st: TrapState = .{}; + const th = try std.Thread.spawn(.{}, trapThreadMain, .{&st}); + var tries: usize = 0; + while (st.tid.load(.acquire) == 0 or !d.isPaused(st.tid.load(.acquire))) : (tries += 1) { + try testing.expect(tries < 5000); + sleepMs(1); + } + const tid = st.tid.load(.acquire); + try testing.expectEqual(@as(?u32, tid), d.pausedAt(0)); + try testing.expectEqual(@as(?u32, null), d.pausedAt(1)); + try testing.expectEqual(@as(u32, 0), @atomicLoad(u32, &st.counter, .acquire)); + + var out: Writer.Allocating = .init(testing.allocator); + defer out.deinit(); + try d.pausedStack(tid, &out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "trapThreadMain") != null); + out.clearRetainingCapacity(); + try d.pausedRegs(tid, &out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "pc 0x") != null); + + // A paused thread can also be captured through the signal path. + out.clearRetainingCapacity(); + try d.threadStack(tid, &out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "#0 0x") != null); + + sleepMs(5); + try testing.expectEqual(@as(u32, 0), @atomicLoad(u32, &st.counter, .acquire)); + try d.resumeThread(tid); + th.join(); + try testing.expectEqual(@as(u32, 1), @atomicLoad(u32, &st.counter, .acquire)); + try testing.expect(!d.isPaused(tid)); + try testing.expectEqual(@as(?u32, null), d.pausedAt(0)); + try testing.expectError(error.NotPaused, d.resumeThread(tid)); + try testing.expectError(error.NotPaused, d.pausedStack(tid, &out.writer)); + d.disableBreakpoints(); +} + +/// Stands in for `FullPanic`'s call: the first trace address is the return +/// address into the panicking function. +noinline fn panicLike(msg: []const u8) bool { + return recordPanic(msg, @returnAddress()); +} + +test "panic record path" { + var text_buf: [16 * 1024]u8 = undefined; + var d: Debug = undefined; + try d.init(testOptions(&text_buf)); + defer d.deinit(); + defer resetPanicRecord(); + + var out: Writer.Allocating = .init(testing.allocator); + defer out.deinit(); + try d.panicMessage(&out.writer); + try testing.expectEqualStrings("", out.written()); + try testing.expect(!d.panicHeld()); + try testing.expectError(error.NoPanic, d.panicContinue()); + + try testing.expect(panicLike("something broke")); + try testing.expect(!recordPanic("nested", null)); + try testing.expectEqual(selfTid(), panic_tid); + + try d.panicMessage(&out.writer); + try testing.expectEqualStrings("something broke", out.written()); + out.clearRetainingCapacity(); + try d.panicStack(&out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "#0 0x") != null); + try testing.expect(std.mem.indexOf(u8, out.written(), "test.panic record path") != null); + try testing.expect(!d.panicHeld()); + try testing.expectError(error.NoPanic, d.panicContinue()); + + // A long message is truncated, not overflowed. + resetPanicRecord(); + const long = [_]u8{'x'} ** (panic_msg_cap + 100); + try testing.expect(recordPanic(&long, null)); + out.clearRetainingCapacity(); + try d.panicMessage(&out.writer); + try testing.expectEqual(@as(usize, panic_msg_cap), out.written().len); +} + +const MaskState = struct { + tid: std.atomic.Value(u32) = .init(0), + unblock: std.atomic.Value(bool) = .init(false), + stop: std.atomic.Value(bool) = .init(false), + signal: linux.SIG, +}; + +fn maskedThreadMain(st: *MaskState) void { + var set = linux.sigemptyset(); + linux.sigaddset(&set, st.signal); + _ = linux.sigprocmask(linux.SIG.BLOCK, &set, null); + st.tid.store(selfTid(), .release); + while (!st.unblock.load(.acquire)) sleepMs(1); + _ = linux.sigprocmask(linux.SIG.UNBLOCK, &set, null); + while (!st.stop.load(.acquire)) sleepMs(1); +} + +test "capture timeout on a thread with the signal masked" { + var text_buf: [16 * 1024]u8 = undefined; + var d: Debug = undefined; + var opts = testOptions(&text_buf); + opts.capture_timeout_ns = 50 * std.time.ns_per_ms; + try d.init(opts); + defer d.deinit(); + + var st: MaskState = .{ .signal = d.capture_signal }; + const th = try std.Thread.spawn(.{}, maskedThreadMain, .{&st}); + var tries: usize = 0; + while (st.tid.load(.acquire) == 0) : (tries += 1) { + try testing.expect(tries < 2000); + sleepMs(1); + } + const masked_tid = st.tid.load(.acquire); + + var out: Writer.Allocating = .init(testing.allocator); + defer out.deinit(); + const t0 = monotonicNs(); + try testing.expectError(error.Timeout, d.threadStack(masked_tid, &out.writer)); + try testing.expect(monotonicNs() - t0 >= 50 * std.time.ns_per_ms); + try testing.expectEqual(cap_idle, d.capture.state.load(.acquire)); + + // The process is healthy: another thread can still be captured... + var spin: SpinState = .{}; + const spinner = try std.Thread.spawn(.{}, spinThreadMain, .{&spin}); + const spin_tid = waitForTid(&spin); + try testing.expect(spin_tid != 0); + out.clearRetainingCapacity(); + try d.threadStack(spin_tid, &out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "spinHere") != null); + + // ...and the late delivery of the pending signal is harmless. + st.unblock.store(true, .release); + sleepMs(20); + out.clearRetainingCapacity(); + try d.threadStack(spin_tid, &out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "spinHere") != null); + out.clearRetainingCapacity(); + try d.threadStack(masked_tid, &out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "maskedThreadMain") != null); + + @atomicStore(bool, &spin.stop, true, .release); + spinner.join(); + st.stop.store(true, .release); + th.join(); +} + +test "options validation and single instance" { + var text_buf: [4096]u8 = undefined; + var d: Debug = undefined; + var opts = testOptions(&text_buf); + opts.capture_signal = 5; + try testing.expectError(error.InvalidOptions, d.init(opts)); + opts = testOptions(&text_buf); + opts.max_paused = max_paused_cap + 1; + try testing.expectError(error.InvalidOptions, d.init(opts)); + try d.init(testOptions(&text_buf)); + defer d.deinit(); + var d2: Debug = undefined; + try testing.expectError(error.AlreadyInitialized, d2.init(testOptions(&text_buf))); +} diff --git a/introspect/src/linux/probe.zig b/introspect/src/linux/probe.zig new file mode 100644 index 0000000..f38c491 --- /dev/null +++ b/introspect/src/linux/probe.zig @@ -0,0 +1,829 @@ +//! The Linux platform layer: one background thread runs a `poll()` loop over +//! a listener and every client connection, feeding each connection's core +//! `Conn` with `push`/`step`/`output`/`wrote`. No per-connection threads, no +//! allocation after `init`; every buffer lives in a caller-placed `Storage`. +//! +//! Also home of the debug facilities (`debug`, `provider`) and the /runtime +//! generators (`runtime`). See docs/LIBRARY.md. +//! +//! Client admission: a new connection takes a free slot. When every slot is +//! taken, the connection that has held a slot without any fid (never +//! attached, or fully clunked) for longer than `evict_idle_ms` is dropped in +//! its favour; if there is none, the new connection is closed ("refused"). +//! Nothing that holds a fid is ever evicted. +//! +//! `sleepServing(ms)` lets a request handler (a ctl command, say) wait +//! without stalling the other clients: called on the probe thread from inside +//! a request it keeps running the poll loop for every client whose request +//! is not in progress until the time is up. Requests served from inside such +//! a wait may wait themselves, up to `max_nested_sleeps` deep (each level is a +//! different client, so the depth is bounded by the client table anyway); the +//! level past that, and any call off the probe thread, is a plain sleep. +const std = @import("std"); +const builtin = @import("builtin"); +const linux = std.os.linux; +const cloud9 = @import("cloud9"); +const core = @import("../core.zig"); + +pub const debug = @import("debug.zig"); +pub const provider = @import("provider.zig"); +pub const runtime = @import("runtime.zig"); +pub const DebugProvider = provider.DebugProvider; + +pub const Listen = union(enum) { + /// A unix socket path (< 108 bytes); a stale socket file is unlinked first. + unix: []const u8, + /// An IPv4 literal "a.b.c.d:port". + tcp: []const u8, + /// An already listening socket, owned by the caller. + fd: i32, + /// One pre-connected client on these descriptors (stdio: 0 and 1). Nothing + /// is accepted and the loop ends when the client hangs up. + client: struct { in: i32, out: i32 }, +}; + +pub const Options = struct { + /// For `std.debug` symbolization. + io: std.Io, + listen: Listen, + /// Largest msize offered to clients (clamped to the server's `cfg.msize`). + msize: u32 = 64 * 1024, + /// Hold a panicking thread until /panic/ctl says "continue". + hold_on_panic: bool = true, + /// Real-time signal used to snapshot other threads. + capture_signal: u8 = debug.default_capture_signal, + /// Install the SIGTRAP handler so `@breakpoint()` parks the thread. + breakpoints: bool = true, + /// Mount /threads, /addr, /mem, /hex, /breakpoints, /panic (six provider slots). + mount_debug: bool = true, +}; + +pub const Error = error{ + /// Another `Debug` (another probe) exists in this process. + AlreadyInitialized, + /// No register capture on this architecture. + Unsupported, + /// `Shared` has fewer than six free provider slots. + TooManyProviders, + PathTooLong, + BadAddress, + /// A syscall failed; `last_errno` says which error. + Syscall, +}; + +/// Idle time without fids after which a slot holder may be evicted. +pub const evict_idle_ms: i64 = 500; +/// Largest number of connections accepted per poll wakeup. +const accept_burst = 64; +/// How deep `sleepServing` may nest (each level keeps a poll round on the stack). +pub const max_nested_sleeps = 8; + +/// Static per-client storage: `max_clients` core `Storage`s and `Conn`s, the +/// poll table, the debug text arena and the debug provider's snapshot pool. +pub fn Storage(comptime max_clients: u8, comptime Srv: type) type { + return Probe(Srv).Storage(max_clients); +} + +pub fn Probe(comptime Srv: type) type { + return struct { + const Self = @This(); + + pub fn Storage(comptime max_clients: u8) type { + comptime std.debug.assert(max_clients > 0); + return struct { + pub const capacity = max_clients; + conns: [max_clients]Srv.Storage, + clients: [max_clients]Client, + /// [0] wake eventfd, [1] listener, [2..] one per client slot. + pollfds: [max_clients + 2]linux.pollfd, + text_buf: [16 * 1024]u8, + dp: DebugProvider, + }; + } + + pub const Client = struct { + conn: Srv.Conn, + in: i32 = -1, + out: i32 = -1, + used: bool = false, + /// Descriptors we opened (accepted) are closed on drop; borrowed ones are not. + owned: bool = false, + /// Send with MSG_NOSIGNAL; falls back to write(2) on ENOTSOCK. + is_socket: bool = true, + /// Monotonic ms of the last byte received. + last_active: i64 = 0, + }; + + shared: *Srv.Shared, + clients: []Client, + conns: []Srv.Storage, + pollfds: []linux.pollfd, + dbg: debug.Debug, + dp: *DebugProvider, + msize: u32, + listen_fd: i32 = -1, + own_listener: bool = false, + is_tcp: bool = false, + single: bool = false, + wake_fd: i32 = -1, + unix_path: [108]u8 = undefined, + unix_len: usize = 0, + thread: ?std.Thread = null, + thread_tid: std.atomic.Value(u32) = .init(0), + nclients: std.atomic.Value(u32) = .init(0), + /// Connections closed because no slot was free. + refused: u64 = 0, + stopping: std.atomic.Value(bool) = .init(false), + /// Slots whose request is being handled (excluded from nested servicing and eviction). + serving: std.StaticBitSet(256) = .initEmpty(), + /// Current `sleepServing` nesting depth. + nested: u8 = 0, + debug_ready: bool = false, + last_errno: linux.E = .SUCCESS, + + // -- lifecycle ------------------------------------------------------- + + /// Installs the debug facilities, mounts the debug providers into + /// `shared` and opens the listener. `storage` is a `*Storage(n)`. On + /// failure `shared` may already hold the debug providers and must be + /// discarded. + pub fn init(p: *Self, shared: *Srv.Shared, storage: anytype, opts: Options) Error!void { + p.* = .{ + .shared = shared, + .clients = &storage.clients, + .conns = &storage.conns, + .pollfds = &storage.pollfds, + .dbg = undefined, + .dp = &storage.dp, + .msize = opts.msize, + }; + for (p.clients) |*c| c.used = false; + shared.hash_seed = randomSeed(); + debug.hold_on_panic = opts.hold_on_panic; + p.dbg.init(.{ .io = opts.io, .text_buf = &storage.text_buf, .capture_signal = opts.capture_signal }) catch |e| return switch (e) { + error.AlreadyInitialized => error.AlreadyInitialized, + error.Unsupported => error.Unsupported, + else => error.Syscall, + }; + p.debug_ready = true; + errdefer { + p.dbg.deinit(); + p.debug_ready = false; + } + if (opts.breakpoints) p.dbg.enableBreakpoints() catch |e| switch (e) { + // No breakpoint support on this architecture: everything else still works. + error.Unsupported => {}, + else => return error.Syscall, + }; + if (opts.mount_debug) { + p.dp.init(&p.dbg); + p.dp.mountAll(shared) catch return error.TooManyProviders; + } + const efd = linux.eventfd(0, linux.EFD.CLOEXEC | linux.EFD.NONBLOCK); + try p.check(efd); + p.wake_fd = @intCast(efd); + errdefer { + _ = linux.close(p.wake_fd); + p.wake_fd = -1; + } + switch (opts.listen) { + .unix => |path| try p.listenUnix(path), + .tcp => |text| try p.listenTcp(text), + .fd => |fd| { + try p.setNonblock(fd); + p.listen_fd = fd; + }, + .client => |c| { + p.single = true; + try p.setNonblock(c.in); + if (c.out != c.in) try p.setNonblock(c.out); + _ = p.addClient(c.in, c.out, false); + }, + } + } + + /// Spawns the poll thread. + pub fn start(p: *Self) std.Thread.SpawnError!void { + std.debug.assert(p.thread == null); + p.stopping.store(false, .release); + p.thread = try std.Thread.spawn(.{}, run, .{p}); + } + + /// Waits for the poll thread to end (only happens by itself in + /// `.client` mode, when the client hangs up). + pub fn wait(p: *Self) void { + if (p.thread) |t| { + t.join(); + p.thread = null; + } + } + + /// Stops the poll thread, drops every client, closes what `init` + /// opened and restores the signal dispositions. + pub fn stop(p: *Self) void { + // Joining the poll thread from itself would hang forever; a + // request handler that wants the server gone uses `requestStop`. + std.debug.assert(p.thread_tid.load(.acquire) != @as(u32, @intCast(linux.gettid()))); + p.stopping.store(true, .release); + p.wakeLoop(); + p.wait(); + for (p.clients, 0..) |*c, i| if (c.used) p.dropClient(i); + if (p.listen_fd >= 0) { + if (p.own_listener) _ = linux.close(p.listen_fd); + p.listen_fd = -1; + } + if (p.unix_len > 0) { + _ = linux.unlink(@ptrCast(&p.unix_path)); + p.unix_len = 0; + } + if (p.wake_fd >= 0) { + _ = linux.close(p.wake_fd); + p.wake_fd = -1; + } + if (p.debug_ready) { + p.dbg.deinit(); + p.debug_ready = false; + } + } + + /// Live client count (for /runtime/clients). + pub fn clientCount(p: *const Self) u32 { + return p.nclients.load(.acquire); + } + + pub fn clientCounter(p: *const Self) *const std.atomic.Value(u32) { + return &p.nclients; + } + + /// Waits `ms` while keeping the other clients served (see the file comment). + pub fn sleepServing(p: *Self, ms: u64) void { + const on_thread = p.thread_tid.load(.acquire) == @as(u32, @intCast(linux.gettid())); + if (!on_thread or p.serving.count() == 0 or p.nested >= max_nested_sleeps) return sleepMs(ms); + p.nested += 1; + defer p.nested -= 1; + const deadline = monotonicMs() + @as(i64, @intCast(@min(ms, std.math.maxInt(i32)))); + while (!p.stopping.load(.acquire)) { + const now = monotonicMs(); + if (now >= deadline) break; + p.pollOnce(@intCast(deadline - now)); + } + } + + /// Asks the poll thread to stop; safe to call from a signal handler + /// (an atomic store and one write to the wake eventfd). `stop` (or + /// `wait`) still has to run afterwards to release everything. + pub fn requestStop(p: *Self) void { + p.stopping.store(true, .release); + p.wakeLoop(); + } + + // -- the loop -------------------------------------------------------- + + fn run(p: *Self) void { + const tid: u32 = @intCast(linux.gettid()); + p.thread_tid.store(tid, .release); + // A breakpoint or panic on this thread must never park it (see debug.zig). + debug.server_tid.store(tid, .release); + setThreadName("introspect"); + while (!p.stopping.load(.acquire)) { + if (p.single and p.clientCount() == 0) break; + p.pollOnce(-1); + } + debug.server_tid.store(0, .release); + p.thread_tid.store(0, .release); + } + + fn wakeLoop(p: *Self) void { + if (p.wake_fd < 0) return; + const one: u64 = 1; + _ = linux.write(p.wake_fd, @ptrCast(&one), 8); + } + + /// One `poll()` round: accept, read, step, write. Slots whose request + /// is in progress (`serving`, only inside `sleepServing`) are left untouched. + fn pollOnce(p: *Self, timeout_ms: i32) void { + p.pollfds[0] = .{ .fd = p.wake_fd, .events = linux.POLL.IN, .revents = 0 }; + p.pollfds[1] = .{ .fd = p.listen_fd, .events = linux.POLL.IN, .revents = 0 }; + for (p.clients, 0..) |*c, i| { + var fd: i32 = -1; + var events: i16 = 0; + if (c.used and !p.serving.isSet(i)) { + fd = c.in; + if (c.conn.output().len > 0) { + fd = c.out; + events = linux.POLL.OUT; + } else if (inputRoom(&c.conn) > 0) { + events = linux.POLL.IN; + } + } + p.pollfds[2 + i] = .{ .fd = fd, .events = events, .revents = 0 }; + } + const rc = linux.poll(p.pollfds.ptr, p.pollfds.len, timeout_ms); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .INTR => return, + else => { + sleepMs(10); + return; + }, + } + if (p.pollfds[0].revents != 0) { + var v: u64 = 0; + _ = linux.read(p.wake_fd, @ptrCast(&v), 8); + } + if (p.stopping.load(.acquire)) return; + if (p.pollfds[1].revents != 0) p.acceptSome(); + for (p.clients, 0..) |*c, i| { + const re = p.pollfds[2 + i].revents; + if (re == 0 or !c.used or p.serving.isSet(i)) continue; + if (re & (linux.POLL.IN | linux.POLL.HUP | linux.POLL.ERR | linux.POLL.NVAL) != 0) { + p.readClient(i, re & linux.POLL.HUP != 0); + } else if (re & linux.POLL.OUT != 0) { + p.service(i); + } + if (p.stopping.load(.acquire)) return; + } + } + + fn acceptSome(p: *Self) void { + var n: usize = 0; + while (n < accept_burst) : (n += 1) { + const rc = linux.accept4(p.listen_fd, null, null, linux.SOCK.NONBLOCK | linux.SOCK.CLOEXEC); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .AGAIN => return, + .INTR, .CONNABORTED => continue, + // Descriptor/memory exhaustion is transient (clients hang up); + // back off instead of spinning on a readable listener. + .MFILE, .NFILE, .NOBUFS, .NOMEM, .PERM => { + sleepMs(100); + return; + }, + else => return, + } + const cfd: i32 = @intCast(rc); + if (p.is_tcp) { + const one: u32 = 1; + _ = linux.setsockopt(cfd, linux.IPPROTO.TCP, linux.TCP.NODELAY, @ptrCast(&one), @sizeOf(u32)); + } + if (p.addClient(cfd, cfd, true) != null) continue; + if (p.evictable()) |victim| { + p.dropClient(victim); + _ = p.addClient(cfd, cfd, true); + continue; + } + p.refused += 1; + // A flood must not flood stderr. + if (p.refused == 1 or p.refused % 1000 == 0) + std.debug.print("introspect: refused connection ({d} clients open, {d} refused so far)\n", .{ p.clients.len, p.refused }); + _ = linux.close(cfd); + } + } + + /// The longest-idle slot holder without fids, if idle long enough. + fn evictable(p: *Self) ?usize { + const now = monotonicMs(); + var best: ?usize = null; + for (p.clients, 0..) |*c, i| { + if (!c.used or !c.owned) continue; + if (p.serving.isSet(i)) continue; + if (c.conn.fidCount() != 0) continue; + if (now - c.last_active < evict_idle_ms) continue; + if (best == null or c.last_active < p.clients[best.?].last_active) best = i; + } + return best; + } + + fn addClient(p: *Self, in: i32, out: i32, owned: bool) ?usize { + for (p.clients, 0..) |*c, i| { + if (c.used) continue; + c.conn = Srv.Conn.init(p.shared, &p.conns[i], p.msize); + c.in = in; + c.out = out; + c.used = true; + c.owned = owned; + c.is_socket = true; + c.last_active = monotonicMs(); + _ = p.nclients.fetchAdd(1, .acq_rel); + return i; + } + return null; + } + + fn dropClient(p: *Self, i: usize) void { + const c = &p.clients[i]; + if (!c.used) return; + c.conn.hangup(); + if (c.owned) { + _ = linux.close(c.in); + if (c.out != c.in) _ = linux.close(c.out); + } + c.used = false; + c.in = -1; + c.out = -1; + _ = p.nclients.fetchSub(1, .acq_rel); + } + + /// Free space in the connection's input buffer (cloud9 keeps one + /// msize-sized frame; `push` copies at most this much). + fn inputRoom(conn: *const Srv.Conn) usize { + return conn.server.in.len - conn.server.in_len; + } + + fn readClient(p: *Self, i: usize, hup: bool) void { + const c = &p.clients[i]; + var buf: [64 * 1024]u8 = undefined; + const room = inputRoom(&c.conn); + if (room == 0) return p.service(i); + const want = @min(room, buf.len); + while (true) { + const rc = linux.read(c.in, &buf, want); + switch (linux.errno(rc)) { + .SUCCESS => { + if (rc == 0) return p.dropClient(i); + const taken = c.conn.push(buf[0..rc]); + std.debug.assert(taken == rc); + c.last_active = monotonicMs(); + return p.service(i); + }, + .INTR => continue, + .AGAIN => { + if (hup) p.dropClient(i); + return; + }, + else => return p.dropClient(i), + } + } + } + + /// Runs requests and drains output until nothing moves. + fn service(p: *Self, i: usize) void { + const c = &p.clients[i]; + std.debug.assert(!p.serving.isSet(i)); + p.serving.set(i); + defer p.serving.unset(i); + while (c.used) { + var moved = false; + while (true) { + const more = c.conn.step() catch return p.dropClient(i); + if (!more) break; + moved = true; + } + const before = c.conn.output().len; + p.flush(c) catch return p.dropClient(i); + if (c.conn.output().len != before) moved = true; + if (!moved) return; + } + } + + fn flush(p: *Self, c: *Client) error{Closed}!void { + _ = p; + while (c.conn.output().len > 0) { + const chunk = c.conn.output(); + const rc = if (c.is_socket) + linux.sendto(c.out, chunk.ptr, chunk.len, linux.MSG.NOSIGNAL, null, 0) + else + linux.write(c.out, chunk.ptr, chunk.len); + switch (linux.errno(rc)) { + .SUCCESS => { + if (rc == 0) return; + c.conn.wrote(rc); + }, + .INTR => continue, + .AGAIN => return, + .NOTSOCK => c.is_socket = false, + else => return error.Closed, + } + } + } + + // -- listeners ------------------------------------------------------- + + fn check(p: *Self, rc: usize) Error!void { + const e = linux.errno(rc); + if (e != .SUCCESS) { + p.last_errno = e; + return error.Syscall; + } + } + + fn setNonblock(p: *Self, fd: i32) Error!void { + const rc = linux.fcntl(fd, linux.F.GETFL, 0); + try p.check(rc); + const nonblock: u32 = @bitCast(linux.O{ .NONBLOCK = true }); + try p.check(linux.fcntl(fd, linux.F.SETFL, rc | nonblock)); + } + + fn listenUnix(p: *Self, path: []const u8) Error!void { + var sa: linux.sockaddr.un = .{ .path = @splat(0) }; + if (path.len == 0 or path.len >= sa.path.len) return error.PathTooLong; + @memcpy(sa.path[0..path.len], path); + const rc = linux.socket(linux.AF.UNIX, linux.SOCK.STREAM | linux.SOCK.CLOEXEC | linux.SOCK.NONBLOCK, 0); + try p.check(rc); + const lfd: i32 = @intCast(rc); + errdefer _ = linux.close(lfd); + // No libc, so no "is it still listening" probe: unlink a stale socket and bind. + _ = linux.unlink(@ptrCast(&sa.path)); + try p.check(linux.bind(lfd, @ptrCast(&sa), @sizeOf(linux.sockaddr.un))); + try p.check(linux.listen(lfd, 128)); + p.listen_fd = lfd; + p.own_listener = true; + p.unix_path = sa.path; + p.unix_len = path.len; + } + + fn listenTcp(p: *Self, text: []const u8) Error!void { + const sa = parseIpv4(text) orelse return error.BadAddress; + const rc = linux.socket(linux.AF.INET, linux.SOCK.STREAM | linux.SOCK.CLOEXEC | linux.SOCK.NONBLOCK, 0); + try p.check(rc); + const lfd: i32 = @intCast(rc); + errdefer _ = linux.close(lfd); + const one: u32 = 1; + _ = linux.setsockopt(lfd, linux.SOL.SOCKET, linux.SO.REUSEADDR, @ptrCast(&one), @sizeOf(u32)); + try p.check(linux.bind(lfd, @ptrCast(&sa), @sizeOf(linux.sockaddr.in))); + try p.check(linux.listen(lfd, 128)); + p.listen_fd = lfd; + p.own_listener = true; + p.is_tcp = true; + } + }; +} + +/// Entropy for the core's fid hash (so fid numbers cannot be chosen to +/// collide); falls back to the clock if getrandom fails. +fn randomSeed() u32 { + var b: [4]u8 = undefined; + if (linux.errno(linux.getrandom(&b, b.len, 0)) == .SUCCESS) return std.mem.readInt(u32, &b, .little); + var ts: linux.timespec = undefined; + _ = linux.clock_gettime(.MONOTONIC, &ts); + return @truncate(@as(u64, @bitCast(ts.nsec)) ^ (@as(u64, @bitCast(ts.sec)) << 20)); +} + +/// "a.b.c.d:port" as a socket address, or null. +pub fn parseIpv4(text: []const u8) ?linux.sockaddr.in { + const colon = std.mem.lastIndexOfScalar(u8, text, ':') orelse return null; + const port = std.fmt.parseInt(u16, text[colon + 1 ..], 10) catch return null; + var octets: [4]u8 = undefined; + var it = std.mem.splitScalar(u8, text[0..colon], '.'); + for (&octets) |*o| o.* = std.fmt.parseInt(u8, it.next() orelse return null, 10) catch return null; + if (it.next() != null) return null; + return .{ .port = std.mem.nativeToBig(u16, port), .addr = @bitCast(octets) }; +} + +/// Names the calling thread (comm, at most 15 bytes) via prctl. +pub fn setThreadName(name: []const u8) void { + var buf: [16]u8 = @splat(0); + const n = @min(name.len, 15); + @memcpy(buf[0..n], name[0..n]); + _ = linux.prctl(@intFromEnum(linux.PR.SET_NAME), @intFromPtr(&buf), 0, 0, 0); +} + +pub fn sleepMs(ms: u64) void { + var req: linux.timespec = .{ .sec = @intCast(ms / 1000), .nsec = @intCast((ms % 1000) * 1_000_000) }; + var rem: linux.timespec = undefined; + while (linux.errno(linux.nanosleep(&req, &rem)) == .INTR) req = rem; +} + +pub fn monotonicMs() i64 { + var ts: linux.timespec = undefined; + _ = linux.clock_gettime(.MONOTONIC, &ts); + return ts.sec * 1000 + @divTrunc(ts.nsec, 1_000_000); +} + +// --------------------------------------------------------------------------- +// Tests: a real unix socket, a cloud9.Client on the other end. +// --------------------------------------------------------------------------- + +const testing = std.testing; + +test { + _ = debug; + _ = provider; + _ = runtime; +} + +const TestBuild = struct { + pub const zig_version: []const u8 = builtin.zig_version_string; + pub const target: []const u8 = "test"; + pub const optimize: []const u8 = "Debug"; + pub const time: []const u8 = "2024-01-01T00:00:00Z"; + pub const change: []const u8 = "none"; +}; + +const test_cfg: core.Config = .{ + .name = "probetest", + .build = TestBuild, + .msize = 8192, + .max_fids = 16, + .max_providers = 6, + .snapshot_slots = 2, + .snapshot_bytes = 1024, +}; +const TS = core.Server(test_cfg); +const TP = Probe(TS); + +/// A blocking client over a connected socket. +const SockClient = struct { + fd: i32, + client: cloud9.Client, + cin: [8192]u8 = undefined, + cout: [8192]u8 = undefined, + + fn connect(sc: *SockClient, path: []const u8) !void { + var sa: linux.sockaddr.un = .{ .path = @splat(0) }; + @memcpy(sa.path[0..path.len], path); + const rc = linux.socket(linux.AF.UNIX, linux.SOCK.STREAM | linux.SOCK.CLOEXEC, 0); + if (linux.errno(rc) != .SUCCESS) return error.Socket; + sc.fd = @intCast(rc); + if (linux.errno(linux.connect(sc.fd, &sa, @sizeOf(linux.sockaddr.un))) != .SUCCESS) return error.Connect; + sc.client = .init(.{ .in = &sc.cin, .out = &sc.cout }); + } + + fn close(sc: *SockClient) void { + _ = linux.close(sc.fd); + } + + /// One round trip; null when the server closed the connection. + fn rpc(sc: *SockClient, req: cloud9.Client.Request) !?cloud9.Client.Result { + _ = try sc.client.submit(req); + while (sc.client.output().len > 0) { + const out = sc.client.output(); + const rc = linux.write(sc.fd, out.ptr, out.len); + switch (linux.errno(rc)) { + .SUCCESS => sc.client.wrote(rc), + .PIPE, .CONNRESET => return null, + else => return error.Write, + } + } + var buf: [8192]u8 = undefined; + while (true) { + if (sc.client.take()) |done| return done.result; + const rc = linux.read(sc.fd, &buf, buf.len); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .CONNRESET => return null, + else => return error.Read, + } + if (rc == 0) return null; + var rest: []const u8 = buf[0..rc]; + while (rest.len > 0) rest = rest[sc.client.push(rest)..]; + } + } + + fn session(sc: *SockClient) !void { + const v = (try sc.rpc(.{ .version = .{ .msize = 8192 } })) orelse return error.Closed; + try testing.expectEqual(@as(u32, 8192), v.version.msize); + const a = (try sc.rpc(.{ .attach = .{ .fid = 0, .uname = "t" } })) orelse return error.Closed; + try testing.expect(a == .attach); + } + + fn readFile(sc: *SockClient, names: []const []const u8, out: []u8) ![]u8 { + const w = (try sc.rpc(.{ .walk = .{ .fid = 0, .newfid = 1, .names = names } })) orelse return error.Closed; + try testing.expectEqual(@as(u16, @intCast(names.len)), w.walk.nwqid); + _ = (try sc.rpc(.{ .open = .{ .fid = 1, .mode = cloud9.oread } })) orelse return error.Closed; + const r = (try sc.rpc(.{ .read = .{ .fid = 1, .offset = 0, .count = @intCast(out.len) } })) orelse return error.Closed; + const n = r.read.len; + @memcpy(out[0..n], r.read); + _ = (try sc.rpc(.{ .clunk = .{ .fid = 1 } })) orelse return error.Closed; + return out[0..n]; + } +}; + +fn testSockPath(buf: []u8, tag: []const u8) ![]const u8 { + return std.fmt.bufPrint(buf, "/tmp/introspect-probe-{d}-{s}.sock", .{ linux.getpid(), tag }); +} + +const TestCtx = struct { info: runtime.Info }; + +test "probe: start, serve a client over a unix socket, stop" { + var ctx: TestCtx = .{ .info = .now() }; + var shared: TS.Shared = .init(&ctx); + const storage = try testing.allocator.create(TP.Storage(2)); + defer testing.allocator.destroy(storage); + var probe: TP = undefined; + var path_buf: [64]u8 = undefined; + const path = try testSockPath(&path_buf, "basic"); + try probe.init(&shared, storage, .{ .io = testing.io, .listen = .{ .unix = path } }); + defer probe.stop(); + try probe.start(); + ctx.info.clients = probe.clientCounter(); + + var sc: SockClient = undefined; + try sc.connect(path); + defer sc.close(); + try sc.session(); + var buf: [1024]u8 = undefined; + const zv = try sc.readFile(&.{ "build", "zig_version" }, &buf); + try testing.expectEqualStrings(builtin.zig_version_string, zv); + try testing.expectEqual(@as(u32, 1), probe.clientCount()); + + // The debug providers are mounted: /threads lists the probe thread by name. + const names = (try sc.rpc(.{ .walk = .{ .fid = 0, .newfid = 2, .names = &.{"threads"} } })) orelse return error.Closed; + try testing.expectEqual(@as(u16, 1), names.walk.nwqid); + _ = (try sc.rpc(.{ .open = .{ .fid = 2, .mode = cloud9.oread } })) orelse return error.Closed; + const dir = (try sc.rpc(.{ .read = .{ .fid = 2, .offset = 0, .count = 4096 } })) orelse return error.Closed; + try testing.expect(dir.read.len > 0); + _ = (try sc.rpc(.{ .clunk = .{ .fid = 2 } })) orelse return error.Closed; + + // A missing file is the Plan 9 error string. + const bad = (try sc.rpc(.{ .walk = .{ .fid = 0, .newfid = 3, .names = &.{"nope"} } })) orelse return error.Closed; + try testing.expect(bad == .fail); + try testing.expectEqualStrings("file does not exist", bad.fail); + + probe.stop(); + // stop() is idempotent and the socket file is gone. + probe.stop(); + var gone: SockClient = undefined; + try testing.expectError(error.Connect, gone.connect(path)); + // The debug facilities can be set up again after stop. + var probe2: TP = undefined; + var shared2: TS.Shared = .init(&ctx); + try probe2.init(&shared2, storage, .{ .io = testing.io, .listen = .{ .unix = path } }); + probe2.stop(); +} + +test "probe: max_clients refusal and idle eviction" { + var ctx: TestCtx = .{ .info = .now() }; + var shared: TS.Shared = .init(&ctx); + const storage = try testing.allocator.create(TP.Storage(2)); + defer testing.allocator.destroy(storage); + var probe: TP = undefined; + var path_buf: [64]u8 = undefined; + const path = try testSockPath(&path_buf, "limit"); + try probe.init(&shared, storage, .{ .io = testing.io, .listen = .{ .unix = path } }); + defer probe.stop(); + try probe.start(); + + // Two attached clients fill the table; a third is accepted then closed. + var a: SockClient = undefined; + try a.connect(path); + defer a.close(); + try a.session(); + var b: SockClient = undefined; + try b.connect(path); + defer b.close(); + try b.session(); + var c: SockClient = undefined; + try c.connect(path); + defer c.close(); + try testing.expectEqual(@as(?cloud9.Client.Result, null), try c.rpc(.{ .version = .{ .msize = 8192 } })); + try testing.expectEqual(@as(u64, 1), probe.refused); + // Attached clients are never evicted, even when idle for long. + sleepMs(evict_idle_ms + 100); + var d: SockClient = undefined; + try d.connect(path); + defer d.close(); + try testing.expectEqual(@as(?cloud9.Client.Result, null), try d.rpc(.{ .version = .{ .msize = 8192 } })); + var buf: [256]u8 = undefined; + _ = try a.readFile(&.{"README"}, &buf); + + // A client without fids that has been idle long enough gives way. + _ = (try b.rpc(.{ .clunk = .{ .fid = 0 } })) orelse return error.Closed; + sleepMs(evict_idle_ms + 100); + var e: SockClient = undefined; + try e.connect(path); + defer e.close(); + try e.session(); + try testing.expectEqual(@as(?cloud9.Client.Result, null), try b.rpc(.{ .version = .{ .msize = 8192 } })); + try testing.expectEqual(@as(u32, 2), probe.clientCount()); +} + +test "probe: single pre-connected client mode ends when the client hangs up" { + var ctx: TestCtx = .{ .info = .now() }; + var shared: TS.Shared = .init(&ctx); + const storage = try testing.allocator.create(TP.Storage(1)); + defer testing.allocator.destroy(storage); + var sv: [2]i32 = undefined; + try testing.expectEqual(linux.E.SUCCESS, linux.errno(linux.socketpair(linux.AF.UNIX, linux.SOCK.STREAM | linux.SOCK.CLOEXEC, 0, &sv))); + var probe: TP = undefined; + try probe.init(&shared, storage, .{ .io = testing.io, .listen = .{ .client = .{ .in = sv[1], .out = sv[1] } } }); + defer probe.stop(); + try probe.start(); + var sc: SockClient = .{ .fd = sv[0], .client = undefined }; + sc.client = .init(.{ .in = &sc.cin, .out = &sc.cout }); + try sc.session(); + var buf: [256]u8 = undefined; + try testing.expect((try sc.readFile(&.{"README"}, &buf)).len > 0); + try testing.expectEqual(@as(u32, 1), probe.clientCount()); + _ = linux.close(sv[0]); + probe.wait(); + try testing.expectEqual(@as(u32, 0), probe.clientCount()); + _ = linux.close(sv[1]); +} + +test "parseIpv4 and sleepServing off the probe thread" { + const sa = parseIpv4("127.0.0.1:564").?; + try testing.expectEqual(std.mem.nativeToBig(u16, 564), sa.port); + try testing.expectEqual(@as(u32, @bitCast([4]u8{ 127, 0, 0, 1 })), sa.addr); + try testing.expect(parseIpv4("localhost:1") == null); + try testing.expect(parseIpv4("1.2.3:1") == null); + try testing.expect(parseIpv4("1.2.3.4") == null); + try testing.expect(parseIpv4("1.2.3.4:70000") == null); + const t0 = monotonicMs(); + var probe: TP = undefined; + probe.thread_tid = .init(0); + probe.serving = .initEmpty(); + probe.nested = 0; + probe.sleepServing(20); + try testing.expect(monotonicMs() - t0 >= 20); +} diff --git a/introspect/src/linux/provider.zig b/introspect/src/linux/provider.zig new file mode 100644 index 0000000..9952398 --- /dev/null +++ b/introspect/src/linux/provider.zig @@ -0,0 +1,604 @@ +//! Adapts `debug.Debug` into core `Provider`s. The core mounts providers at +//! top level only, so one `DebugProvider` registers six of them, all sharing +//! the same `Debug` and the same snapshot pool: +//! +//! /threads//{name,stat,stack,regs} (lists only live tids) +//! /addr/ "fn\nfile:line:col\nmodule\n" +//! /mem/maps, /mem/ /proc/self/maps; raw bytes at address+offset (writable) +//! /hex/ hexdump of 256 bytes at the address +//! /breakpoints//{stack,regs,ctl} ctl accepts "continue" (lists only paused tids) +//! /panic/{message,stack,ctl} ctl accepts "continue" +//! +//! /addr, /mem, /hex and /panic carry a README; /threads and /breakpoints +//! list nothing but tids so that a shell glob over them sees only threads. +//! +//! Handles encode `(kind, tid-or-address)` in 56 bits (the core keeps the +//! low 56 bits of a handle for the qid path): kind in bits 48..55, value in +//! bits 0..47. Handles carry no reference count, so `clunk` is a no-op. +//! +//! The core hands providers a buffer-based `read`, not a writer, so every +//! text file is generated at `open` into one of `snapshot_slots` fixed slots +//! (keyed by handle, reference counted across fids) and served from there; a +//! read at offset 0 regenerates, like the core's own dynamic files. `/mem/` +//! is read and written directly at address+offset and never snapshotted, and +//! `/mem/maps` is read straight from /proc/self/maps at the requested offset +//! (a big process has more mappings than a snapshot slot holds). +const std = @import("std"); +const cloud9 = @import("cloud9"); +const core = @import("../core.zig"); +const debug = @import("debug.zig"); +const Writer = std.Io.Writer; +const Provider = core.Provider; +const Handle = Provider.Handle; +const Error = Provider.Error; +const NodeStat = core.NodeStat; + +pub const snapshot_slots = 8; +pub const snapshot_bytes = 32 * 1024; +/// Bytes shown by /hex/. +pub const hex_bytes = 256; + +pub const Tree = enum(u8) { threads, addr, mem, hex, breakpoints, panic }; +pub const tree_names = [_][]const u8{ "threads", "addr", "mem", "hex", "breakpoints", "panic" }; + +const Kind = enum(u8) { + root = 0, + readme, + thread_dir, + thread_name, + thread_stat, + thread_stack, + thread_regs, + addr_file, + maps, + mem_file, + hex_file, + bp_dir, + bp_stack, + bp_regs, + bp_ctl, + panic_message, + panic_stack, + panic_ctl, + + fn isDir(k: Kind) bool { + return k == .root or k == .thread_dir or k == .bp_dir; + } + + /// Text files generated into a snapshot slot at open. + fn isText(k: Kind) bool { + return switch (k) { + .readme, .thread_name, .thread_stat, .thread_stack, .thread_regs, .addr_file, .hex_file, .bp_stack, .bp_regs, .panic_message, .panic_stack => true, + else => false, + }; + } + + fn isCtl(k: Kind) bool { + return k == .bp_ctl or k == .panic_ctl; + } + + fn mode(k: Kind) u32 { + if (k.isDir()) return cloud9.dmdir | 0o555; + if (k.isCtl()) return 0o222; + if (k == .mem_file) return 0o666; + return 0o444; + } + + fn fixedName(k: Kind) ?[]const u8 { + return switch (k) { + .readme => "README", + .thread_name => "name", + .thread_stat => "stat", + .thread_stack, .bp_stack, .panic_stack => "stack", + .thread_regs, .bp_regs => "regs", + .maps => "maps", + .bp_ctl, .panic_ctl => "ctl", + .panic_message => "message", + else => null, + }; + } +}; + +const value_bits = 48; +const value_mask: u64 = (1 << value_bits) - 1; + +fn mk(kind: Kind, value: u64) Handle { + return (@as(u64, @intFromEnum(kind)) << value_bits) | (value & value_mask); +} + +fn kindOf(h: Handle) Kind { + return @enumFromInt(@as(u8, @truncate(h >> value_bits))); +} + +fn valueOf(h: Handle) u64 { + return h & value_mask; +} + +const readme_threads = + \\One directory per thread of this process, named by tid: + \\ name the thread's comm + \\ stat state and a few fields of /proc/self/task//stat + \\ stack "#n 0x in (::)" per frame + \\ regs general registers captured while the thread was stopped + \\ +; +const readme_addr = + \\Walk any hex address: /addr/ reads as "fn\nfile:line:col\nmodule\n". + \\ +; +const readme_mem = + \\maps /proc/self/maps + \\ raw process memory at that address (+ file offset); writable + \\ +; +const readme_hex = + \\Walk any hex address: /hex/ is a hexdump of the 256 bytes there. + \\ +; +const readme_breakpoints = + \\One directory per thread stopped in @breakpoint(), named by tid: + \\ stack, regs as under /threads + \\ ctl write "continue" to resume the thread + \\ +; +const readme_panic = + \\message the first panic's message (empty before any panic) + \\stack frames of the panicking thread + \\ctl write "continue" to let the default panic handler run + \\ +; + +fn readmeFor(tree: Tree) []const u8 { + return switch (tree) { + .threads => readme_threads, + .addr => readme_addr, + .mem => readme_mem, + .hex => readme_hex, + .breakpoints => readme_breakpoints, + .panic => readme_panic, + }; +} + +const Slot = struct { + handle: Handle = 0, + refs: u32 = 0, + len: u32 = 0, + buf: [snapshot_bytes]u8 = undefined, +}; + +pub const DebugProvider = struct { + d: *debug.Debug, + slots: [snapshot_slots]Slot = @splat(.{}), + /// Backs `NodeStat.name` until the next call. + name_buf: [32]u8 = undefined, + + pub fn init(dp: *DebugProvider, d: *debug.Debug) void { + dp.* = .{ .d = d }; + } + + /// The provider for one tree, to pass to `Shared.addProvider`. + pub fn provider(dp: *DebugProvider, comptime tree: Tree) Provider { + return .{ .name = tree_names[@intFromEnum(tree)], .ctx = dp, .vtable = vtableFor(tree) }; + } + + /// Mounts all six trees; `shared` is a `Server(cfg).Shared`. + pub fn mountAll(dp: *DebugProvider, shared: anytype) error{Full}!void { + inline for (comptime std.meta.tags(Tree)) |tree| try shared.addProvider(dp.provider(tree)); + } + + fn self(ctx: *anyopaque) *DebugProvider { + return @ptrCast(@alignCast(ctx)); + } + + fn vtableFor(comptime tree: Tree) *const Provider.VTable { + return &struct { + const vt: Provider.VTable = .{ + .walk = walkFn, + .stat = statFn, + .list = listFn, + .open = openFn, + .read = readFn, + .write = writeFn, + .close = closeFn, + .clunk = clunkFn, + }; + fn walkFn(ctx: *anyopaque, parent: Handle, name: []const u8) Error!Handle { + return self(ctx).walk(tree, parent, name); + } + fn statFn(ctx: *anyopaque, h: Handle, out: *NodeStat) Error!void { + return self(ctx).stat(tree, h, out); + } + fn listFn(ctx: *anyopaque, dir: Handle, index: usize, out: *NodeStat) Error!bool { + return self(ctx).list(tree, dir, index, out); + } + fn openFn(ctx: *anyopaque, h: Handle, mode: u8) Error!void { + return self(ctx).open(tree, h, mode); + } + fn readFn(ctx: *anyopaque, h: Handle, offset: u64, buf: []u8) Error!usize { + return self(ctx).read(tree, h, offset, buf); + } + fn writeFn(ctx: *anyopaque, h: Handle, offset: u64, data: []const u8) Error!usize { + return self(ctx).write(tree, h, offset, data); + } + fn closeFn(ctx: *anyopaque, h: Handle) void { + self(ctx).close(h); + } + fn clunkFn(_: *anyopaque, _: Handle) void {} + }.vt; + } + + // -- naming ------------------------------------------------------------ + + fn parseTid(name: []const u8) ?u32 { + if (name.len == 0 or name.len > 10) return null; + for (name) |ch| if (!std.ascii.isDigit(ch)) return null; + return std.fmt.parseInt(u32, name, 10) catch null; + } + + fn parseHex(name: []const u8) ?u64 { + const digits = if (std.mem.startsWith(u8, name, "0x")) name[2..] else name; + if (digits.len == 0 or digits.len > 12) return null; + for (digits) |ch| if (!std.ascii.isHex(ch)) return null; + const v = std.fmt.parseInt(u64, digits, 16) catch return null; + if (v > value_mask) return null; + return v; + } + + fn nodeName(dp: *DebugProvider, h: Handle) []const u8 { + const k = kindOf(h); + if (k.fixedName()) |n| return n; + return switch (k) { + .root => "", + .thread_dir, .bp_dir => std.fmt.bufPrint(&dp.name_buf, "{d}", .{valueOf(h)}) catch unreachable, + .addr_file, .mem_file, .hex_file => std.fmt.bufPrint(&dp.name_buf, "{x}", .{valueOf(h)}) catch unreachable, + else => unreachable, + }; + } + + // -- vtable ------------------------------------------------------------ + + fn walk(dp: *DebugProvider, tree: Tree, parent: Handle, name: []const u8) Error!Handle { + const k = kindOf(parent); + if (std.mem.eql(u8, name, ".")) return parent; + if (!k.isDir()) return error.NotDir; + if (std.mem.eql(u8, name, "..")) return Provider.root; + switch (k) { + .root => { + if (tree != .threads and tree != .breakpoints and std.mem.eql(u8, name, "README")) return mk(.readme, 0); + switch (tree) { + .threads => { + const tid = parseTid(name) orelse return error.NotFound; + if (!dp.d.threadExists(tid)) return error.NotFound; + return mk(.thread_dir, tid); + }, + .addr => return mk(.addr_file, parseHex(name) orelse return error.NotFound), + .mem => { + if (std.mem.eql(u8, name, "maps")) return mk(.maps, 0); + return mk(.mem_file, parseHex(name) orelse return error.NotFound); + }, + .hex => return mk(.hex_file, parseHex(name) orelse return error.NotFound), + .breakpoints => { + const tid = parseTid(name) orelse return error.NotFound; + if (!dp.d.isPaused(tid)) return error.NotFound; + return mk(.bp_dir, tid); + }, + .panic => { + if (std.mem.eql(u8, name, "message")) return mk(.panic_message, 0); + if (std.mem.eql(u8, name, "stack")) return mk(.panic_stack, 0); + if (std.mem.eql(u8, name, "ctl")) return mk(.panic_ctl, 0); + return error.NotFound; + }, + } + }, + .thread_dir => { + const tid = valueOf(parent); + if (std.mem.eql(u8, name, "name")) return mk(.thread_name, tid); + if (std.mem.eql(u8, name, "stat")) return mk(.thread_stat, tid); + if (std.mem.eql(u8, name, "stack")) return mk(.thread_stack, tid); + if (std.mem.eql(u8, name, "regs")) return mk(.thread_regs, tid); + return error.NotFound; + }, + .bp_dir => { + const tid = valueOf(parent); + if (std.mem.eql(u8, name, "stack")) return mk(.bp_stack, tid); + if (std.mem.eql(u8, name, "regs")) return mk(.bp_regs, tid); + if (std.mem.eql(u8, name, "ctl")) return mk(.bp_ctl, tid); + return error.NotFound; + }, + else => unreachable, + } + } + + fn fill(dp: *DebugProvider, tree: Tree, h: Handle, out: *NodeStat) void { + const k = kindOf(h); + out.* = .{ + .mode = k.mode(), + .length = if (k == .readme) readmeFor(tree).len else 0, + .name = dp.nodeName(h), + .handle = h, + }; + } + + fn stat(dp: *DebugProvider, tree: Tree, h: Handle, out: *NodeStat) Error!void { + dp.fill(tree, h, out); + } + + fn list(dp: *DebugProvider, tree: Tree, dir: Handle, index: usize, out: *NodeStat) Error!bool { + const k = kindOf(dir); + if (!k.isDir()) return error.NotDir; + const h: Handle = switch (k) { + .root => switch (tree) { + .threads => mk(.thread_dir, dp.d.threadAt(index) orelse return false), + .addr, .hex => if (index == 0) mk(.readme, 0) else return false, + .mem => switch (index) { + 0 => mk(.readme, 0), + 1 => mk(.maps, 0), + else => return false, + }, + .breakpoints => mk(.bp_dir, dp.d.pausedAt(index) orelse return false), + .panic => switch (index) { + 0 => mk(.readme, 0), + 1 => mk(.panic_message, 0), + 2 => mk(.panic_stack, 0), + 3 => mk(.panic_ctl, 0), + else => return false, + }, + }, + .thread_dir => switch (index) { + 0 => mk(.thread_name, valueOf(dir)), + 1 => mk(.thread_stat, valueOf(dir)), + 2 => mk(.thread_stack, valueOf(dir)), + 3 => mk(.thread_regs, valueOf(dir)), + else => return false, + }, + .bp_dir => switch (index) { + 0 => mk(.bp_stack, valueOf(dir)), + 1 => mk(.bp_regs, valueOf(dir)), + 2 => mk(.bp_ctl, valueOf(dir)), + else => return false, + }, + else => unreachable, + }; + dp.fill(tree, h, out); + return true; + } + + fn open(dp: *DebugProvider, tree: Tree, h: Handle, mode: u8) Error!void { + const k = kindOf(h); + const acc = mode & 3; + const wants_write = acc == cloud9.owrite or acc == cloud9.ordwr; + const wants_read = acc != cloud9.owrite; + if (k.isDir()) { + if (wants_write or mode & cloud9.otrunc != 0) return error.IsDir; + return; + } + if (k.isCtl()) { + if (wants_read) return error.Perm; + return; + } + if (k == .mem_file) return; + if (wants_write or mode & cloud9.otrunc != 0) return error.Perm; + if (k == .maps) return; + std.debug.assert(k.isText()); + const slot = dp.takeSlot(h) orelse return error.NoSpace; + errdefer dp.releaseSlot(slot); + try dp.generate(tree, h, slot); + } + + fn close(dp: *DebugProvider, h: Handle) void { + if (!kindOf(h).isText()) return; + if (dp.findSlot(h)) |s| dp.releaseSlot(s); + } + + fn read(dp: *DebugProvider, tree: Tree, h: Handle, offset: u64, buf: []u8) Error!usize { + const k = kindOf(h); + if (k.isDir()) return error.IsDir; + if (k.isCtl()) return error.Perm; + if (k == .mem_file) { + const addr = valueOf(h) +% offset; + return dp.d.readMem(addr, buf) catch |e| mapErr(e); + } + if (k == .maps) return dp.d.readMaps(offset, buf) catch |e| mapErr(e); + const slot = dp.findSlot(h) orelse return error.Io; + if (offset == 0) try dp.generate(tree, h, slot); + if (offset >= slot.len) return 0; + const off: usize = @intCast(offset); + const n = @min(buf.len, slot.len - off); + @memcpy(buf[0..n], slot.buf[off..][0..n]); + return n; + } + + fn write(dp: *DebugProvider, tree: Tree, h: Handle, offset: u64, data: []const u8) Error!usize { + _ = tree; + const k = kindOf(h); + if (k.isDir()) return error.IsDir; + switch (k) { + .mem_file => { + const addr = valueOf(h) +% offset; + return dp.d.writeMem(addr, data) catch |e| mapErr(e); + }, + .bp_ctl, .panic_ctl => { + const cmd = std.mem.trim(u8, data, " \t\r\n\x00"); + if (!std.mem.eql(u8, cmd, "continue")) return error.Unsupported; + if (k == .bp_ctl) { + dp.d.resumeThread(@intCast(valueOf(h))) catch |e| return mapErr(e); + } else { + dp.d.panicContinue() catch |e| return mapErr(e); + } + return data.len; + }, + else => return error.Perm, + } + } + + // -- snapshots --------------------------------------------------------- + + fn findSlot(dp: *DebugProvider, h: Handle) ?*Slot { + for (&dp.slots) |*s| if (s.refs > 0 and s.handle == h) return s; + return null; + } + + fn takeSlot(dp: *DebugProvider, h: Handle) ?*Slot { + if (dp.findSlot(h)) |s| { + s.refs += 1; + return s; + } + for (&dp.slots) |*s| if (s.refs == 0) { + s.* = .{ .handle = h, .refs = 1 }; + return s; + }; + return null; + } + + fn releaseSlot(_: *DebugProvider, s: *Slot) void { + s.refs -= 1; + } + + /// (Re)generates the text of `h` into `slot`. A text that does not fit is + /// truncated, not an error. + fn generate(dp: *DebugProvider, tree: Tree, h: Handle, slot: *Slot) Error!void { + var w: Writer = .fixed(&slot.buf); + slot.len = 0; + dp.render(tree, h, &w) catch |e| switch (e) { + error.WriteFailed => {}, + else => return mapErr(e), + }; + slot.len = @intCast(w.buffered().len); + } + + fn render(dp: *DebugProvider, tree: Tree, h: Handle, w: *Writer) debug.Error!void { + const d = dp.d; + const v = valueOf(h); + switch (kindOf(h)) { + .readme => w.writeAll(readmeFor(tree)) catch return error.WriteFailed, + .thread_name => try d.threadName(@intCast(v), w), + .thread_stat => try d.threadStat(@intCast(v), w), + .thread_stack => try d.threadStack(@intCast(v), w), + .thread_regs => try d.threadRegs(@intCast(v), w), + .addr_file => try d.resolveAddr(@intCast(v), w), + .hex_file => try d.hexdump(@intCast(v), hex_bytes, w), + .bp_stack => try d.pausedStack(@intCast(v), w), + .bp_regs => try d.pausedRegs(@intCast(v), w), + .panic_message => try d.panicMessage(w), + .panic_stack => try d.panicStack(w), + else => unreachable, + } + } + + fn mapErr(e: debug.Error) Error { + return switch (e) { + error.NoThread, error.NotPaused, error.NoPanic => error.NotFound, + error.Unsupported => error.Unsupported, + error.WriteFailed => error.NoSpace, + error.Timeout, error.Busy, error.Unmapped, error.Unexpected, error.AlreadyInitialized, error.InvalidOptions => error.Io, + }; + } +}; + +// --------------------------------------------------------------------------- +// Tests (through the core's in-memory harness) +// --------------------------------------------------------------------------- + +const testing = std.testing; + +const TestCfg: core.Config = .{ .name = "dbgtest", .msize = 8192, .max_fids = 16, .max_providers = 6, .snapshot_slots = 2, .snapshot_bytes = 512 }; +const TS = core.Server(TestCfg); + +test "debug provider: threads, addr, mem, hex, breakpoints, panic through the core" { + var text_buf: [16 * 1024]u8 = undefined; + var d: debug.Debug = undefined; + try d.init(.{ .io = testing.io, .text_buf = &text_buf }); + defer d.deinit(); + var dp: DebugProvider = undefined; + dp.init(&d); + + var dummy: u8 = 0; + var shared: TS.Shared = .init(&dummy); + try dp.mountAll(&shared); + var storage: TS.Storage = undefined; + var h: TS.Harness = undefined; + try h.init(&shared, &storage); + defer h.deinit(); + + // /threads lists tids only, among them this thread. + const names = try h.listPath(&.{"threads"}); + defer TS.Harness.freeNames(names); + try testing.expect(names.len >= 1); + try testing.expect(!TS.Harness.hasName(names, "README")); + const no_readme = try h.ok(.{ .walk = .{ .fid = 0, .newfid = 7, .names = &.{ "threads", "README" } } }); + try testing.expectEqual(@as(u16, 1), no_readme.walk.nwqid); + var tid_buf: [16]u8 = undefined; + const tid = try std.fmt.bufPrint(&tid_buf, "{d}", .{std.os.linux.gettid()}); + try testing.expect(TS.Harness.hasName(names, tid)); + + // Own stack names this test function's file. + const stack = try h.readPath(&.{ "threads", tid, "stack" }); + defer testing.allocator.free(stack); + try testing.expect(std.mem.indexOf(u8, stack, "#0 0x") != null); + + // /addr/ of a function here resolves to this file. + var addr_buf: [32]u8 = undefined; + const addr_name = try std.fmt.bufPrint(&addr_buf, "{x}", .{@intFromPtr(&DebugProvider.parseTid)}); + const resolved = try h.readPath(&.{ "addr", addr_name }); + defer testing.allocator.free(resolved); + try testing.expect(std.mem.indexOf(u8, resolved, "provider.zig") != null); + const partial = try h.ok(.{ .walk = .{ .fid = 0, .newfid = 5, .names = &.{ "addr", "zzz" } } }); + try testing.expectEqual(@as(u16, 1), partial.walk.nwqid); + try h.walkTo(5, &.{"addr"}); + try h.expectFail(.{ .walk = .{ .fid = 5, .newfid = 6, .names = &.{"zzz"} } }, "file does not exist"); + _ = try h.ok(.{ .clunk = .{ .fid = 5 } }); + + // /mem/ reads and writes live memory; unmapped is an error. + var cell: [8]u8 = "abcdefgh".*; + var mem_buf: [32]u8 = undefined; + const mem_name = try std.fmt.bufPrint(&mem_buf, "{x}", .{@intFromPtr(&cell)}); + try h.walkTo(1, &.{ "mem", mem_name }); + _ = try h.ok(.{ .open = .{ .fid = 1, .mode = cloud9.ordwr } }); + const r = try h.ok(.{ .read = .{ .fid = 1, .offset = 2, .count = 4 } }); + try testing.expectEqualStrings("cdef", r.read); + _ = try h.ok(.{ .write = .{ .fid = 1, .offset = 0, .data = "XY" } }); + try testing.expectEqualStrings("XYcdefgh", &cell); + _ = try h.ok(.{ .clunk = .{ .fid = 1 } }); + try h.walkTo(2, &.{ "mem", "8" }); + _ = try h.ok(.{ .open = .{ .fid = 2, .mode = cloud9.oread } }); + try h.expectFail(.{ .read = .{ .fid = 2, .offset = 0, .count = 4 } }, "i/o error"); + _ = try h.ok(.{ .clunk = .{ .fid = 2 } }); + const maps = try h.readPath(&.{ "mem", "maps" }); + defer testing.allocator.free(maps); + try testing.expect(std.mem.indexOf(u8, maps, "r-xp") != null or std.mem.indexOf(u8, maps, "r--p") != null); + + // /hex/ is a hexdump. + const hex = try h.readPath(&.{ "hex", mem_name }); + defer testing.allocator.free(hex); + try testing.expect(std.mem.indexOf(u8, hex, "XYcdefgh") != null); + + // Nothing paused, no panic. + const bps = try h.listPath(&.{"breakpoints"}); + defer TS.Harness.freeNames(bps); + try testing.expectEqual(@as(usize, 0), bps.len); + const msg = try h.readPath(&.{ "panic", "message" }); + defer testing.allocator.free(msg); + try testing.expectEqualStrings("", msg); + try h.walkTo(3, &.{ "panic", "ctl" }); + _ = try h.ok(.{ .open = .{ .fid = 3, .mode = cloud9.owrite } }); + try h.expectFail(.{ .write = .{ .fid = 3, .offset = 0, .data = "continue" } }, "file does not exist"); + try h.expectFail(.{ .write = .{ .fid = 3, .offset = 0, .data = "bogus" } }, "not supported"); + _ = try h.ok(.{ .clunk = .{ .fid = 3 } }); + + // Snapshot slots are released on clunk: open more files than slots, sequentially. + for (0..4) |_| { + const t = try h.readPath(&.{ "threads", tid, "name" }); + testing.allocator.free(t); + } + for (&dp.slots) |s| try testing.expectEqual(@as(u32, 0), s.refs); +} + +test "handle encoding round-trips" { + const h = mk(.mem_file, 0x7fff_dead_beef); + try testing.expectEqual(Kind.mem_file, kindOf(h)); + try testing.expectEqual(@as(u64, 0x7fff_dead_beef), valueOf(h)); + try testing.expect(h < (1 << 56)); + try testing.expectEqual(@as(?u64, null), DebugProvider.parseHex("1_0")); + try testing.expectEqual(@as(?u64, 0x10), DebugProvider.parseHex("0x10")); + try testing.expectEqual(@as(?u32, null), DebugProvider.parseTid("+5")); +} diff --git a/introspect/src/linux/runtime.zig b/introspect/src/linux/runtime.zig new file mode 100644 index 0000000..503d6c2 --- /dev/null +++ b/introspect/src/linux/runtime.zig @@ -0,0 +1,90 @@ +//! Generators for `Config.runtime`: /runtime/{pid,ppid,uptime,argv,cwd,env,clients}. +//! The core passes every generator `Shared.ctx`; `Fns(Ctx, field)` casts it +//! to `*Ctx` and reads the `Info` stored in `@field(ctx, field)`. +const std = @import("std"); +const linux = std.os.linux; +const Writer = std.Io.Writer; + +/// What the generators report. Texts are borrowed for the server's lifetime. +pub const Info = struct { + /// argv, one argument per line. + argv: []const u8 = "", + /// Environment, one KEY=VALUE per line. + env: []const u8 = "", + cwd: []const u8 = "", + /// Monotonic seconds at startup; /runtime/uptime is the difference. + start_mono: i64 = 0, + /// Live client count, published by the probe. + clients: ?*const std.atomic.Value(u32) = null, + + pub fn now() Info { + return .{ .start_mono = monotonicSecs() }; + } +}; + +pub fn monotonicSecs() i64 { + var ts: linux.timespec = undefined; + _ = linux.clock_gettime(.MONOTONIC, &ts); + return ts.sec; +} + +/// Seconds since the epoch, clamped to u32 (for atime/mtime and `fn/now`). +pub fn realtimeSecs() u32 { + var ts: linux.timespec = undefined; + _ = linux.clock_gettime(.REALTIME, &ts); + return @intCast(std.math.clamp(ts.sec, 0, std.math.maxInt(u32))); +} + +/// The `Config.runtime` type: `Ctx` is the type behind `Shared.ctx`, `field` +/// the name of its `Info` field. +pub fn Fns(comptime Ctx: type, comptime field: []const u8) type { + return struct { + fn info(ctx: *anyopaque) *const Info { + const c: *Ctx = @ptrCast(@alignCast(ctx)); + return &@field(c, field); + } + pub fn pid(_: *anyopaque, w: *Writer) anyerror!void { + try w.print("{d}", .{linux.getpid()}); + } + pub fn ppid(_: *anyopaque, w: *Writer) anyerror!void { + try w.print("{d}", .{linux.getppid()}); + } + pub fn uptime(ctx: *anyopaque, w: *Writer) anyerror!void { + try w.print("{d}", .{monotonicSecs() - info(ctx).start_mono}); + } + pub fn argv(ctx: *anyopaque, w: *Writer) anyerror!void { + try w.writeAll(info(ctx).argv); + } + pub fn cwd(ctx: *anyopaque, w: *Writer) anyerror!void { + try w.writeAll(info(ctx).cwd); + } + pub fn env(ctx: *anyopaque, w: *Writer) anyerror!void { + try w.writeAll(info(ctx).env); + } + pub fn clients(ctx: *anyopaque, w: *Writer) anyerror!void { + const n: u32 = if (info(ctx).clients) |c| c.load(.acquire) else 0; + try w.print("{d}", .{n}); + } + }; +} + +test "runtime generators read Info through the context" { + const Ctx = struct { x: u32, info: Info }; + var count: std.atomic.Value(u32) = .init(3); + var ctx: Ctx = .{ .x = 0, .info = .{ .argv = "a\nb\n", .cwd = "/tmp", .env = "K=V\n", .start_mono = monotonicSecs(), .clients = &count } }; + const F = Fns(Ctx, "info"); + var buf: [64]u8 = undefined; + var w: Writer = .fixed(&buf); + try F.clients(&ctx, &w); + try std.testing.expectEqualStrings("3", w.buffered()); + w = .fixed(&buf); + try F.argv(&ctx, &w); + try std.testing.expectEqualStrings("a\nb\n", w.buffered()); + w = .fixed(&buf); + try F.uptime(&ctx, &w); + try std.testing.expect(w.buffered().len >= 1); + w = .fixed(&buf); + try F.pid(&ctx, &w); + try std.testing.expectEqual(linux.getpid(), try std.fmt.parseInt(i32, w.buffered(), 10)); + try std.testing.expectEqual(@as(usize, 7), @typeInfo(F).@"struct".decls.len); +} diff --git a/introspect/src/root.zig b/introspect/src/root.zig new file mode 100644 index 0000000..1dcc892 --- /dev/null +++ b/introspect/src/root.zig @@ -0,0 +1,24 @@ +//! introspect: a 9P2000 debug/introspection server as a library. See +//! docs/LIBRARY.md. `core` and `vars` are freestanding; `scratch` takes an +//! allocator; `linux` is the platform layer (only on Linux). +const std = @import("std"); +const builtin = @import("builtin"); + +pub const core = @import("core.zig"); +pub const vars = @import("vars.zig"); +pub const scratch = @import("scratch.zig"); +pub const linux = if (builtin.os.tag == .linux) @import("linux/probe.zig") else struct {}; + +pub const Config = core.Config; +pub const Server = core.Server; +pub const Provider = core.Provider; +pub const NodeStat = core.NodeStat; +pub const Scratch = scratch.Scratch; + +test { + std.testing.refAllDecls(@This()); + _ = core; + _ = vars; + _ = scratch; + if (builtin.os.tag == .linux) _ = linux; +} diff --git a/introspect/src/scratch.zig b/introspect/src/scratch.zig new file mode 100644 index 0000000..2163a28 --- /dev/null +++ b/introspect/src/scratch.zig @@ -0,0 +1,689 @@ +//! An in-memory read/write tree as a `Provider`: create, write, truncate, +//! rename, remove, mkdir, DMAPPEND, DMEXCL. The one core-level component that +//! takes an `Allocator` (nodes and file contents live on it); it is optional. +//! +//! Nodes are kept alive by `refs` (fids holding a handle) after removal, so a +//! handle stays valid until the core clunks it. Handles are node addresses; +//! the root is handle 0. Not internally synchronized (like `Shared`). +const std = @import("std"); +const cloud9 = @import("cloud9"); +const core = @import("core.zig"); +const Allocator = std.mem.Allocator; +const Provider = core.Provider; +const Handle = Provider.Handle; +const Error = Provider.Error; +const NodeStat = core.NodeStat; + +/// A node of the tree. +pub const Node = struct { + name: []u8, + path: u64, + version: u32 = 0, + mode: u32, + atime: u32, + mtime: u32, + data: std.ArrayList(u8) = .empty, + children: std.ArrayList(*Node) = .empty, + parent: ?*Node, + /// Handles held by the core. + refs: u32 = 0, + /// Fids currently open on this node (DMEXCL admits at most one). + opens: u32 = 0, + removed: bool = false, + + pub fn isDir(n: *const Node) bool { + return n.mode & cloud9.dmdir != 0; + } + + fn find(n: *const Node, name: []const u8) ?*Node { + for (n.children.items) |ch| if (std.mem.eql(u8, ch.name, name)) return ch; + return null; + } +}; + +/// Seconds since the epoch, for atime/mtime; the default clock reports 0. +pub const Clock = *const fn () u32; + +fn zeroClock() u32 { + return 0; +} + +pub const Scratch = struct { + gpa: Allocator, + root: *Node, + /// Qid paths are a counter, never reused: the root is 0 (the provider + /// root handle), so a removed-and-recreated file gets a fresh identity + /// even when the allocator hands back the same address. + next_path: u64 = 0, + /// Sum of all file lengths, bounded by `budget`. + bytes: usize = 0, + /// Largest total of file contents across all files. + budget: usize, + /// Largest single file; defaults to the budget. + max_file: usize, + /// The time source for atime/mtime (a platform layer sets it). + now: Clock = &zeroClock, + + /// The tree's only allocation policy: every node and every file's + /// contents come from `gpa`, and no file content ever exceeds `budget_bytes` + /// in total. + pub fn init(gpa: Allocator, budget_bytes: usize) Allocator.Error!Scratch { + var s: Scratch = .{ .gpa = gpa, .root = undefined, .budget = budget_bytes, .max_file = budget_bytes }; + s.root = try s.newNode("", cloud9.dmdir | 0o777, null); + return s; + } + + pub fn deinit(s: *Scratch) void { + s.destroyTree(s.root); + s.* = undefined; + } + + /// The provider to mount, at `/`. + pub fn provider(s: *Scratch, name: []const u8) Provider { + return .{ .name = name, .ctx = s, .vtable = &vtable }; + } + + pub const vtable: Provider.VTable = .{ + .walk = &walk, + .stat = &stat, + .list = &list, + .open = &open, + .read = &read, + .write = &write, + .create = &create, + .remove = &remove, + .wstat = &wstat, + .close = &close, + .clunk = &clunk, + }; + + // -- node management -- + + fn destroyTree(s: *Scratch, n: *Node) void { + for (n.children.items) |ch| s.destroyTree(ch); + n.children.clearRetainingCapacity(); + n.removed = true; + if (n.refs == 0 or n == s.root) s.destroyNode(n); + } + + fn destroyNode(s: *Scratch, n: *Node) void { + s.bytes -= n.data.items.len; + s.gpa.free(n.name); + n.data.deinit(s.gpa); + n.children.deinit(s.gpa); + s.gpa.destroy(n); + } + + fn newNode(s: *Scratch, name: []const u8, mode: u32, parent: ?*Node) Allocator.Error!*Node { + const n = try s.gpa.create(Node); + errdefer s.gpa.destroy(n); + const t = s.now(); + n.* = .{ + .name = try s.gpa.dupe(u8, name), + .path = s.next_path, + .mode = mode, + .atime = t, + .mtime = t, + .parent = parent, + }; + errdefer s.gpa.free(n.name); + if (parent) |p| try p.children.append(s.gpa, n); + s.next_path += 1; + return n; + } + + /// Sets a file's length, zero-filling growth and charging the budget. + /// Shrinking releases the memory so a truncated file costs nothing. + fn resizeData(s: *Scratch, n: *Node, new_len: usize) Error!void { + const old = n.data.items.len; + if (new_len > old) { + if (new_len > s.max_file) return error.NoSpace; + if (s.bytes + (new_len - old) > s.budget) return error.NoSpace; + n.data.resize(s.gpa, new_len) catch return error.NoSpace; + @memset(n.data.items[old..new_len], 0); + s.bytes += new_len - old; + } else if (new_len < old) { + n.data.shrinkAndFree(s.gpa, new_len); + s.bytes -= old - new_len; + } + } + + fn touch(s: *Scratch, n: *Node) void { + n.version +%= 1; + n.mtime = s.now(); + } + + fn self(ctx: *anyopaque) *Scratch { + return @ptrCast(@alignCast(ctx)); + } + + fn handle(s: *Scratch, n: *Node) Handle { + return if (n == s.root) Provider.root else @intFromPtr(n); + } + + fn node(s: *Scratch, h: Handle) *Node { + return if (h == Provider.root) s.root else @ptrFromInt(@as(usize, @intCast(h))); + } + + /// A handle the core will clunk exactly once. + fn retain(s: *Scratch, n: *Node) Handle { + if (n != s.root) n.refs += 1; + return s.handle(n); + } + + fn release(s: *Scratch, n: *Node) void { + if (n == s.root) return; + n.refs -= 1; + if (n.refs == 0 and n.removed) s.destroyNode(n); + } + + fn fillStat(n: *const Node, h: Handle, out: *NodeStat) void { + out.* = .{ + .mode = n.mode, + .length = if (n.isDir()) 0 else n.data.items.len, + .atime = n.atime, + .mtime = n.mtime, + .version = n.version, + .name = n.name, + .handle = h, + .path = n.path, + }; + } + + // -- the vtable -- + + fn walk(ctx: *anyopaque, parent: Handle, name: []const u8) Error!Handle { + const s = self(ctx); + const p = s.node(parent); + if (std.mem.eql(u8, name, ".")) return s.retain(p); + if (p.removed) return error.NotFound; + if (!p.isDir()) return error.NotDir; + if (std.mem.eql(u8, name, "..")) return s.retain(p.parent orelse s.root); + return s.retain(p.find(name) orelse return error.NotFound); + } + + fn stat(ctx: *anyopaque, h: Handle, out: *NodeStat) Error!void { + const s = self(ctx); + fillStat(s.node(h), h, out); + } + + fn list(ctx: *anyopaque, dir: Handle, index: usize, out: *NodeStat) Error!bool { + const s = self(ctx); + const d = s.node(dir); + if (!d.isDir()) return error.NotDir; + if (index >= d.children.items.len) return false; + const ch = d.children.items[index]; + fillStat(ch, s.handle(ch), out); + return true; + } + + fn open(ctx: *anyopaque, h: Handle, mode: u8) Error!void { + const s = self(ctx); + const n = s.node(h); + if (n.removed) return error.NotFound; + const acc = mode & 3; + const want_write = acc == cloud9.owrite or acc == cloud9.ordwr; + const want_read = !want_write or acc == cloud9.ordwr; + const trunc = mode & cloud9.otrunc != 0; + if (n.isDir()) { + if (want_write or trunc) return error.IsDir; + if (n.mode & 0o400 == 0) return error.Perm; + } else { + if (want_read and n.mode & 0o400 == 0) return error.Perm; + if ((want_write or trunc) and n.mode & 0o200 == 0) return error.Perm; + if (n.mode & cloud9.dmexcl != 0 and n.opens != 0) return error.Excl; + if (trunc and n.mode & cloud9.dmappend == 0) { + s.resizeData(n, 0) catch unreachable; // shrinking cannot fail + s.touch(n); + } + } + n.opens += 1; + } + + fn close(ctx: *anyopaque, h: Handle) void { + const s = self(ctx); + s.node(h).opens -= 1; + } + + fn read(ctx: *anyopaque, h: Handle, offset: u64, buf: []u8) Error!usize { + const s = self(ctx); + const n = s.node(h); + if (n.isDir()) return error.IsDir; + const src = n.data.items; + if (offset >= src.len) return 0; + const off: usize = @intCast(offset); + const len = @min(buf.len, src.len - off); + @memcpy(buf[0..len], src[off..][0..len]); + return len; + } + + fn write(ctx: *anyopaque, h: Handle, offset: u64, data: []const u8) Error!usize { + const s = self(ctx); + const n = s.node(h); + if (n.isDir()) return error.IsDir; + // A zero-length write changes nothing (and must not extend the file). + if (data.len == 0) return 0; + const off: usize = if (n.mode & cloud9.dmappend != 0) n.data.items.len else @intCast(@min(offset, s.max_file)); + const end = off + data.len; + if (end > s.max_file) return error.NoSpace; + if (end > n.data.items.len) try s.resizeData(n, end); + @memcpy(n.data.items[off..end], data); + s.touch(n); + return data.len; + } + + fn create(ctx: *anyopaque, dir: Handle, name: []const u8, perm: u32, mode: u8) Error!Handle { + const s = self(ctx); + const d = s.node(dir); + if (d.removed) return error.NotFound; + if (!d.isDir()) return error.NotDir; + if (d.mode & 0o200 == 0) return error.Perm; + if (d.find(name) != null) return error.Exists; + const is_dir = perm & cloud9.dmdir != 0; + const inherit: u32 = if (is_dir) d.mode & 0o777 else d.mode & 0o666; + const n = s.newNode(name, perm & (~@as(u32, 0o777) | inherit), d) catch return error.NoSpace; + s.touch(d); + n.opens += 1; + _ = mode; + return s.retain(n); + } + + fn remove(ctx: *anyopaque, h: Handle) Error!void { + const s = self(ctx); + const n = s.node(h); + if (n.removed) return error.NotFound; + const parent = n.parent orelse return error.Perm; + if (parent.mode & 0o200 == 0) return error.Perm; + if (n.isDir() and n.children.items.len != 0) return error.NotEmpty; + const i = std.mem.indexOfScalar(*Node, parent.children.items, n) orelse return error.NotFound; + _ = parent.children.orderedRemove(i); + s.touch(parent); + n.removed = true; + if (n.refs == 0) s.destroyNode(n); + } + + fn wstat(ctx: *anyopaque, h: Handle, st: *const cloud9.Stat) Error!void { + const s = self(ctx); + const n = s.node(h); + if (n.removed) return error.NotFound; + // Validate everything before changing anything. + const rename = st.name.len != 0 and !std.mem.eql(u8, st.name, n.name); + if (rename) { + const parent = n.parent orelse return error.Perm; + if (parent.find(st.name) != null) return error.Exists; + } + const cur_len: u64 = if (n.isDir()) 0 else n.data.items.len; + const set_len = st.length != 0xFFFF_FFFF_FFFF_FFFF and st.length != cur_len; + if (set_len) { + if (n.isDir()) return error.IsDir; + if (st.length > s.max_file) return error.NoSpace; + } + const set_mode = st.mode != 0xFFFF_FFFF and st.mode != n.mode; + if (set_mode and (st.mode & cloud9.dmdir) != (n.mode & cloud9.dmdir)) return error.Perm; + const set_mtime = st.mtime != 0xFFFF_FFFF and st.mtime != n.mtime; + if (!(rename or set_len or set_mode or set_mtime)) return; + const new_name: ?[]u8 = if (rename) s.gpa.dupe(u8, st.name) catch return error.NoSpace else null; + errdefer if (new_name) |nn| s.gpa.free(nn); + if (set_len) try s.resizeData(n, @intCast(st.length)); + // Nothing below can fail. + if (new_name) |nn| { + s.gpa.free(n.name); + n.name = nn; + s.touch(n.parent.?); + } + if (set_mode) n.mode = (n.mode & cloud9.dmdir) | (st.mode & ~cloud9.dmdir); + s.touch(n); + if (set_mtime) n.mtime = st.mtime; + } + + fn clunk(ctx: *anyopaque, h: Handle) void { + const s = self(ctx); + s.release(s.node(h)); + } +}; + +// --------------------------------------------------------------------------- +// Tests: the scratch tree mounted at /scratch of a core server. +// --------------------------------------------------------------------------- + +const testing = std.testing; + +const test_cfg: core.Config = .{ .name = "tester", .msize = 8192, .max_fids = 32 }; +const TS = core.Server(test_cfg); + +const budget: usize = 1 << 20; + +const Fixture = struct { + ctx: u8 = 0, + shared: TS.Shared = undefined, + storage: TS.Storage = undefined, + scratch: Scratch = undefined, + h: TS.Harness = undefined, + + fn init(x: *Fixture) !void { + x.shared = .init(&x.ctx); + x.scratch = try Scratch.init(testing.allocator, budget); + errdefer x.scratch.deinit(); + try x.shared.addProvider(x.scratch.provider("scratch")); + try x.h.init(&x.shared, &x.storage); + } + + fn deinit(x: *Fixture) void { + x.h.deinit(); + x.scratch.deinit(); + } + + fn nodeOf(x: *Fixture, fid: u32) *Node { + for (x.h.conn.fids) |f| if (f.used and f.id == fid) return x.scratch.node(f.node.prov.h); + unreachable; + } +}; + +const dontcare = core.stat_dontcare; + +test "scratch create/write/read/rename/truncate/remove" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + try x.h.walkTo(1, &.{"scratch"}); + // create + write + const cr = try x.h.ok(.{ .create = .{ .fid = 1, .name = "x", .perm = 0o644, .mode = cloud9.ordwr } }); + try testing.expectEqual(cloud9.qtfile, cr.create.qid.type); + _ = try x.h.ok(.{ .write = .{ .fid = 1, .offset = 0, .data = "hello" } }); + _ = try x.h.ok(.{ .write = .{ .fid = 1, .offset = 5, .data = " world" } }); + const r = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = 100 } }); + try testing.expectEqualStrings("hello world", r.read); + _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); + // rename x -> y + try x.h.walkTo(2, &.{ "scratch", "x" }); + var st = dontcare; + st.name = "y"; + _ = try x.h.ok(.{ .wstat = .{ .fid = 2, .stat = st } }); + try x.h.walkTo(3, &.{"scratch"}); + try x.h.expectFail(.{ .walk = .{ .fid = 3, .newfid = 30, .names = &.{"x"} } }, "file does not exist"); + _ = try x.h.ok(.{ .clunk = .{ .fid = 3 } }); + try x.h.walkTo(3, &.{ "scratch", "y" }); + // truncate then extend with zero fill + st = dontcare; + st.length = 2; + _ = try x.h.ok(.{ .wstat = .{ .fid = 3, .stat = st } }); + st.length = 4; + _ = try x.h.ok(.{ .wstat = .{ .fid = 3, .stat = st } }); + const text = try x.h.readAll(3); + defer testing.allocator.free(text); + try testing.expectEqualStrings("he\x00\x00", text); + const s3 = try x.h.ok(.{ .stat = .{ .fid = 3 } }); + try testing.expectEqualStrings("y", s3.stat.name); + try testing.expectEqual(@as(u64, 4), s3.stat.length); + try testing.expectEqualStrings("tester", s3.stat.uid); + _ = try x.h.ok(.{ .clunk = .{ .fid = 3 } }); + _ = try x.h.ok(.{ .clunk = .{ .fid = 2 } }); + // mkdir, nested create, remove rules + try x.h.walkTo(4, &.{"scratch"}); + const dr = try x.h.ok(.{ .create = .{ .fid = 4, .name = "d", .perm = cloud9.dmdir | 0o755, .mode = cloud9.oread } }); + try testing.expectEqual(cloud9.qtdir, dr.create.qid.type); + _ = try x.h.ok(.{ .clunk = .{ .fid = 4 } }); + try x.h.walkTo(5, &.{ "scratch", "d" }); + _ = try x.h.ok(.{ .create = .{ .fid = 5, .name = "inner", .perm = 0o600, .mode = cloud9.owrite } }); + _ = try x.h.ok(.{ .write = .{ .fid = 5, .offset = 0, .data = "z" } }); + _ = try x.h.ok(.{ .clunk = .{ .fid = 5 } }); + try x.h.walkTo(6, &.{ "scratch", "d" }); + try x.h.expectFail(.{ .remove = .{ .fid = 6 } }, "directory not empty"); + try x.h.expectFail(.{ .clunk = .{ .fid = 6 } }, "unknown fid"); // remove always clunks + try x.h.walkTo(7, &.{ "scratch", "d", "inner" }); + _ = try x.h.ok(.{ .remove = .{ .fid = 7 } }); + try x.h.walkTo(8, &.{ "scratch", "d" }); + _ = try x.h.ok(.{ .remove = .{ .fid = 8 } }); + try x.h.walkTo(9, &.{ "scratch", "y" }); + _ = try x.h.ok(.{ .remove = .{ .fid = 9 } }); + try x.h.walkTo(10, &.{"scratch"}); + _ = try x.h.ok(.{ .open = .{ .fid = 10, .mode = cloud9.oread } }); + const names = try x.h.listDir(10, 1024); + defer testing.allocator.free(names); + try testing.expectEqual(@as(usize, 0), names.len); + // append-only files ignore the offset + try x.h.walkTo(11, &.{"scratch"}); + _ = try x.h.ok(.{ .create = .{ .fid = 11, .name = "log", .perm = cloud9.dmappend | 0o644, .mode = cloud9.ordwr } }); + _ = try x.h.ok(.{ .write = .{ .fid = 11, .offset = 100, .data = "a" } }); + _ = try x.h.ok(.{ .write = .{ .fid = 11, .offset = 0, .data = "b" } }); + const lr = try x.h.ok(.{ .read = .{ .fid = 11, .offset = 0, .count = 10 } }); + try testing.expectEqualStrings("ab", lr.read); + try testing.expect(lr.read.len == 2); + const ls = try x.h.ok(.{ .stat = .{ .fid = 11 } }); + try testing.expect(ls.stat.qid.type & cloud9.qtappend != 0); + // the scratch root cannot be removed + try x.h.walkTo(12, &.{"scratch"}); + try x.h.expectFail(.{ .remove = .{ .fid = 12 } }, "permission denied"); +} + +test "walk of a missing name and walking a file" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + try x.h.walkTo(1, &.{"scratch"}); + try x.h.expectFail(.{ .walk = .{ .fid = 1, .newfid = 2, .names = &.{"nope"} } }, "file does not exist"); + _ = try x.h.ok(.{ .create = .{ .fid = 1, .name = "f", .perm = 0o644, .mode = cloud9.oread } }); + _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); + try x.h.walkTo(3, &.{ "scratch", "f" }); + try x.h.expectFail(.{ .walk = .{ .fid = 3, .newfid = 4, .names = &.{"x"} } }, "not a directory"); + // a walk that fails past the first element is a partial Rwalk that leaves newfid unused + const r = try x.h.ok(.{ .walk = .{ .fid = 0, .newfid = 4, .names = &.{ "scratch", "nope", "x" } } }); + try testing.expectEqual(@as(u16, 1), r.walk.nwqid); + try x.h.expectFail(.{ .clunk = .{ .fid = 4 } }, "unknown fid"); + // .. from a file is not a directory; .. from the scratch root reaches the server root + try x.h.expectFail(.{ .walk = .{ .fid = 3, .newfid = 5, .names = &.{".."} } }, "not a directory"); + try x.h.walkTo(5, &.{"scratch"}); + const up = try x.h.ok(.{ .walk = .{ .fid = 5, .newfid = 6, .names = &.{ "..", "scratch", "..", "README" } } }); + try testing.expectEqual(@as(u16, 4), up.walk.nwqid); + try testing.expectEqual(@as(u64, 0), up.walk.wqid[1].path); // provider 0, root + try testing.expect(up.walk.wqid[0].type & cloud9.qtdir != 0); +} + +test "directory read across consecutive offsets returns every record exactly once" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + const n = 40; + for (0..n) |i| { + try x.h.walkTo(1, &.{"scratch"}); + var name_buf: [64]u8 = undefined; + const name = try std.fmt.bufPrint(&name_buf, "file-with-a-long-name-{d:0>3}", .{i}); + _ = try x.h.ok(.{ .create = .{ .fid = 1, .name = name, .perm = 0o644, .mode = cloud9.oread } }); + _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); + } + try x.h.walkTo(2, &.{"scratch"}); + _ = try x.h.ok(.{ .open = .{ .fid = 2, .mode = cloud9.oread } }); + // 200 bytes fits two records, so this takes many reads. + const names = try x.h.listDir(2, 200); + defer TS.Harness.freeNames(names); + try testing.expectEqual(@as(usize, n), names.len); + var seen: [n]bool = @splat(false); + for (names) |nm| { + const idx = try std.fmt.parseInt(usize, nm[nm.len - 3 ..], 10); + try testing.expect(!seen[idx]); + seen[idx] = true; + } + for (seen) |s| try testing.expect(s); + try x.h.expectFail(.{ .read = .{ .fid = 2, .offset = 7, .count = 200 } }, "bad offset"); + // a read that cannot fit even one record returns nothing rather than splitting it + const tiny = try x.h.ok(.{ .read = .{ .fid = 2, .offset = 0, .count = 30 } }); + try testing.expectEqual(@as(usize, 0), tiny.read.len); + _ = try x.h.ok(.{ .clunk = .{ .fid = 2 } }); + // Tversion resets every fid and every reference + try x.h.version(8192); + try testing.expectEqual(@as(usize, 0), x.h.conn.fidCount()); + for (x.scratch.root.children.items) |ch| try testing.expectEqual(@as(u32, 0), ch.refs); + _ = try x.h.ok(.{ .attach = .{ .fid = 0, .uname = "tester" } }); +} + +test "DMEXCL admits one open fid at a time" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + try x.h.walkTo(1, &.{"scratch"}); + const cr = try x.h.ok(.{ .create = .{ .fid = 1, .name = "lock", .perm = cloud9.dmexcl | 0o644, .mode = cloud9.owrite } }); + try testing.expect(cr.create.qid.type & cloud9.qtexcl != 0); + try x.h.walkTo(2, &.{ "scratch", "lock" }); + try x.h.expectFail(.{ .open = .{ .fid = 2, .mode = cloud9.oread } }, "exclusive use file already open"); + _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); + _ = try x.h.ok(.{ .open = .{ .fid = 2, .mode = cloud9.oread } }); + try x.h.walkTo(3, &.{ "scratch", "lock" }); + try x.h.expectFail(.{ .open = .{ .fid = 3, .mode = cloud9.oread } }, "exclusive use file already open"); + // a Tversion reset drops the open and frees the file for the next session + try x.h.version(8192); + _ = try x.h.ok(.{ .attach = .{ .fid = 0, .uname = "tester" } }); + try x.h.walkTo(4, &.{ "scratch", "lock" }); + _ = try x.h.ok(.{ .open = .{ .fid = 4, .mode = cloud9.oread } }); + try testing.expectEqual(@as(u32, 1), x.nodeOf(4).opens); + _ = try x.h.ok(.{ .remove = .{ .fid = 4 } }); +} + +test "scratch memory: zero-length writes, truncation frees, global budget" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + x.scratch.max_file = 4096; + try x.h.walkTo(1, &.{"scratch"}); + _ = try x.h.ok(.{ .create = .{ .fid = 1, .name = "f", .perm = 0o644, .mode = cloud9.ordwr } }); + // a zero-length write at a huge offset must not extend the file + const w0 = try x.h.ok(.{ .write = .{ .fid = 1, .offset = std.math.maxInt(u64), .data = "" } }); + try testing.expectEqual(@as(u32, 0), w0.write); + var st = try x.h.ok(.{ .stat = .{ .fid = 1 } }); + try testing.expectEqual(@as(u64, 0), st.stat.length); + // growth is charged to the budget; truncation releases it (memory too) + _ = try x.h.ok(.{ .write = .{ .fid = 1, .offset = 1000, .data = "x" } }); + try testing.expectEqual(@as(usize, 1001), x.scratch.bytes); + var ws = dontcare; + ws.length = 10; + _ = try x.h.ok(.{ .wstat = .{ .fid = 1, .stat = ws } }); + try testing.expectEqual(@as(usize, 10), x.scratch.bytes); + try testing.expectEqual(@as(usize, 10), x.nodeOf(1).data.capacity); + // per-file cap and the global budget both answer "no space" + try x.h.expectFail(.{ .write = .{ .fid = 1, .offset = 4096, .data = "x" } }, "no space left on device"); + try x.h.expectFail(.{ .write = .{ .fid = 1, .offset = std.math.maxInt(u64), .data = "x" } }, "no space left on device"); + x.scratch.bytes = budget - 10; // pretend other files hold the rest + try x.h.expectFail(.{ .write = .{ .fid = 1, .offset = 10, .data = "0123456789A" } }, "no space left on device"); + _ = try x.h.ok(.{ .write = .{ .fid = 1, .offset = 10, .data = "0123456789" } }); + try testing.expectEqual(budget, x.scratch.bytes); + ws.length = 4096; + try x.h.expectFail(.{ .wstat = .{ .fid = 1, .stat = ws } }, "no space left on device"); + x.scratch.bytes -= budget - 20; + // OTRUNC releases too + try x.h.walkTo(2, &.{ "scratch", "f" }); + _ = try x.h.ok(.{ .open = .{ .fid = 2, .mode = cloud9.owrite | cloud9.otrunc } }); + try testing.expectEqual(@as(usize, 0), x.scratch.bytes); + st = try x.h.ok(.{ .stat = .{ .fid = 2 } }); + try testing.expectEqual(@as(u64, 0), st.stat.length); + // removing a file with content returns its bytes once the last fid lets go + _ = try x.h.ok(.{ .write = .{ .fid = 2, .offset = 0, .data = "abc" } }); + try testing.expectEqual(@as(usize, 3), x.scratch.bytes); + _ = try x.h.ok(.{ .remove = .{ .fid = 2 } }); + try testing.expectEqual(@as(usize, 3), x.scratch.bytes); // fid 1 still holds it + const r = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = 10 } }); + try testing.expectEqualStrings("abc", r.read); + _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); + try testing.expectEqual(@as(usize, 0), x.scratch.bytes); +} + +test "wstat with every field equal to the current stat changes nothing" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + try x.h.walkTo(1, &.{"scratch"}); + _ = try x.h.ok(.{ .create = .{ .fid = 1, .name = "same", .perm = 0o640, .mode = cloud9.oread } }); + const before = (try x.h.ok(.{ .stat = .{ .fid = 1 } })).stat; + var copy = before; + var name_buf: [core.max_name]u8 = undefined; + @memcpy(name_buf[0..before.name.len], before.name); + copy.name = name_buf[0..before.name.len]; + copy.uid = "tester"; + copy.gid = "tester"; + copy.muid = "tester"; + _ = try x.h.ok(.{ .wstat = .{ .fid = 1, .stat = copy } }); + const after = (try x.h.ok(.{ .stat = .{ .fid = 1 } })).stat; + try testing.expectEqual(before.qid, after.qid); + try testing.expectEqual(before.mtime, after.mtime); + try testing.expectEqual(before.mode, after.mode); + try testing.expectEqualStrings("same", after.name); + // and a rename to the very same name is also a no-op + var st = dontcare; + st.name = "same"; + _ = try x.h.ok(.{ .wstat = .{ .fid = 1, .stat = st } }); + try testing.expectEqual(before.qid, (try x.h.ok(.{ .stat = .{ .fid = 1 } })).stat.qid); + // renaming onto an existing sibling is refused + _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); + try x.h.walkTo(2, &.{"scratch"}); + _ = try x.h.ok(.{ .create = .{ .fid = 2, .name = "other", .perm = 0o640, .mode = cloud9.oread } }); + st.name = "same"; + try x.h.expectFail(.{ .wstat = .{ .fid = 2, .stat = st } }, "file already exists"); + // the mode's directory bit is immutable, mtime is settable + st = dontcare; + st.mode = cloud9.dmdir | 0o640; + try x.h.expectFail(.{ .wstat = .{ .fid = 2, .stat = st } }, "permission denied"); + st = dontcare; + st.mtime = 12345; + _ = try x.h.ok(.{ .wstat = .{ .fid = 2, .stat = st } }); + try testing.expectEqual(@as(u32, 12345), (try x.h.ok(.{ .stat = .{ .fid = 2 } })).stat.mtime); + _ = try x.h.ok(.{ .remove = .{ .fid = 2 } }); +} + +test "ORCLOSE removes on clunk and removed files stay readable through open fids" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + try x.h.walkTo(1, &.{"scratch"}); + _ = try x.h.ok(.{ .create = .{ .fid = 1, .name = "tmp", .perm = 0o644, .mode = cloud9.ordwr | cloud9.orclose } }); + _ = try x.h.ok(.{ .write = .{ .fid = 1, .offset = 0, .data = "gone" } }); + try x.h.walkTo(2, &.{ "scratch", "tmp" }); + _ = try x.h.ok(.{ .open = .{ .fid = 2, .mode = cloud9.oread } }); + _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); + try x.h.walkTo(3, &.{"scratch"}); + try x.h.expectFail(.{ .walk = .{ .fid = 3, .newfid = 4, .names = &.{"tmp"} } }, "file does not exist"); + const r = try x.h.ok(.{ .read = .{ .fid = 2, .offset = 0, .count = 10 } }); + try testing.expectEqualStrings("gone", r.read); + try testing.expectEqual(@as(usize, 4), x.scratch.bytes); + _ = try x.h.ok(.{ .clunk = .{ .fid = 2 } }); + try testing.expectEqual(@as(usize, 0), x.scratch.bytes); + // create inside a removed directory fails + _ = try x.h.ok(.{ .create = .{ .fid = 3, .name = "d", .perm = cloud9.dmdir | 0o755, .mode = cloud9.oread } }); + try x.h.walkTo(5, &.{ "scratch", "d" }); + _ = try x.h.ok(.{ .remove = .{ .fid = 5 } }); + _ = try x.h.ok(.{ .clunk = .{ .fid = 3 } }); + try x.h.walkTo(6, &.{"scratch"}); + try x.h.expectFail(.{ .walk = .{ .fid = 6, .newfid = 7, .names = &.{"d"} } }, "file does not exist"); + try x.h.expectFail(.{ .create = .{ .fid = 5, .name = "x", .perm = 0o644, .mode = cloud9.oread } }, "unknown fid"); // remove clunked it +} + +test "qid paths are stable identities, not addresses: remove + recreate differ" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + try x.h.walkTo(1, &.{"scratch"}); + const a = try x.h.ok(.{ .create = .{ .fid = 1, .name = "f", .perm = 0o644, .mode = cloud9.oread } }); + const path_a = a.create.qid.path; + try testing.expectEqual(x.nodeOf(1).path, path_a & ((1 << 56) - 1)); + try testing.expect(path_a != 0); + // a rename keeps the identity + var st = dontcare; + st.name = "g"; + _ = try x.h.ok(.{ .wstat = .{ .fid = 1, .stat = st } }); + try testing.expectEqual(path_a, (try x.h.ok(.{ .stat = .{ .fid = 1 } })).stat.qid.path); + _ = try x.h.ok(.{ .remove = .{ .fid = 1 } }); + // the allocator very likely reuses the freed node's address here + try x.h.walkTo(2, &.{"scratch"}); + const b = try x.h.ok(.{ .create = .{ .fid = 2, .name = "f", .perm = 0o644, .mode = cloud9.oread } }); + try testing.expect(b.create.qid.path != path_a); + _ = try x.h.ok(.{ .remove = .{ .fid = 2 } }); + // the scratch root keeps path 0 (its handle), like every provider root + try x.h.walkTo(3, &.{"scratch"}); + try testing.expectEqual(@as(u64, 0), (try x.h.ok(.{ .stat = .{ .fid = 3 } })).stat.qid.path); + // directory listing reports the same identities as walking + _ = try x.h.ok(.{ .create = .{ .fid = 3, .name = "listed", .perm = 0o644, .mode = cloud9.oread } }); + const via_create = (try x.h.ok(.{ .stat = .{ .fid = 3 } })).stat.qid; + try x.h.walkTo(4, &.{"scratch"}); + _ = try x.h.ok(.{ .open = .{ .fid = 4, .mode = cloud9.oread } }); + const r = try x.h.ok(.{ .read = .{ .fid = 4, .offset = 0, .count = 1024 } }); + const listed = try cloud9.Stat.decode(r.read[0 .. std.mem.readInt(u16, r.read[0..2], .little) + 2]); + try testing.expectEqual(via_create, listed.qid); + _ = try x.h.ok(.{ .remove = .{ .fid = 3 } }); +} diff --git a/introspect/src/vars.zig b/introspect/src/vars.zig new file mode 100644 index 0000000..20cadfc --- /dev/null +++ b/introspect/src/vars.zig @@ -0,0 +1,824 @@ +//! Comptime value renderers for /vars. For a type T, `vtableFor(T)` builds (at +//! comptime) a flat table of the files and directories that describe a value of +//! that type: +//! +//! /value rendered text /type @typeName +//! /size @sizeOf /addr 0x… +//! /raw the bytes /f//... recursively (depth <= max_depth) +//! +//! The core serves a variable by walking this table; a node index is the whole +//! state it needs. Rendering writes into a `*std.Io.Writer` and never allocates. +//! Writes to scalar `value` files are plain stores (not atomic). +const std = @import("std"); +const Writer = std.Io.Writer; + +/// Deepest `f/` nesting: /vars/x/f/a/f/b/f/c/f/d/value is depth 4. +pub const max_depth: u8 = 4; +/// Longest string rendered from a `[]const u8` / `[*:0]const u8` before "…". +pub const max_string: usize = 256; +/// Most array/slice elements rendered before "…". +pub const max_elems: usize = 64; + +pub const Kind = enum(u8) { + /// The directory of a value: value, type, size, addr, raw, [f]. + dir, + /// Rendered text; writable when `set` is non-null. + value, + /// @typeName, static content. + type_name, + /// @sizeOf, static content. + size, + /// "0x…" of the value's address. + addr, + /// The bytes of the value, length = size. + raw, + /// The `f` directory: one `dir` per struct field. + fields, +}; + +pub const RenderFn = *const fn (base: [*]const u8, w: *Writer) Writer.Error!void; +pub const SetFn = *const fn (base: [*]u8, text: []const u8) SetError!void; +pub const SetError = error{ Invalid, Unsupported }; + +/// One node of a type's tree. Children occupy `first..first+count`. +pub const Node = struct { + name: []const u8, + kind: Kind, + parent: u32, + first: u32 = 0, + count: u32 = 0, + /// Byte offset of the described value from the variable's base address. + offset: usize, + /// @sizeOf the described value. + size: usize, + /// Static text for `type_name` and `size` leaves. + content: []const u8 = "", + render: ?RenderFn = null, + set: ?SetFn = null, + + pub fn isDir(n: Node) bool { + return n.kind == .dir or n.kind == .fields; + } + + pub fn writable(n: Node) bool { + return n.kind == .value and n.set != null; + } +}; + +pub const VTable = struct { + nodes: []const Node, + type_name: []const u8, + size: usize, + + /// The child of `dir` named `name`, if any. + pub fn child(vt: *const VTable, dir: u32, name: []const u8) ?u32 { + const d = vt.nodes[dir]; + for (d.first..d.first + d.count) |i| { + if (std.mem.eql(u8, vt.nodes[i].name, name)) return @intCast(i); + } + return null; + } +}; + +/// The comptime-generated table for `T`; the same pointer for the same `T`. +pub fn vtableFor(comptime T: type) *const VTable { + const S = struct { + const nodes = buildTable(T); + const vt: VTable = .{ .nodes = &nodes, .type_name = @typeName(T), .size = @sizeOf(T) }; + }; + return &S.vt; +} + +/// Whether a struct's fields get an `f/` directory at this depth. +fn hasFields(comptime T: type, depth: u8) bool { + if (depth >= max_depth) return false; + return switch (@typeInfo(T)) { + .@"struct" => |s| s.layout != .@"packed" and fieldCount(T) > 0, + else => false, + }; +} + +fn fieldCount(comptime T: type) usize { + var n: usize = 0; + for (@typeInfo(T).@"struct".fields) |f| { + if (!f.is_comptime and @sizeOf(f.type) != 0) n += 1; + } + return n; +} + +fn countNodes(comptime T: type, depth: u8) usize { + var n: usize = 6; // dir + value, type, size, addr, raw + if (hasFields(T, depth)) { + n += 1; // f + for (@typeInfo(T).@"struct".fields) |f| { + if (f.is_comptime or @sizeOf(f.type) == 0) continue; + n += countNodes(f.type, depth + 1); + } + } + return n; +} + +fn fill(nodes: []Node, next: *usize, idx: usize, comptime T: type, name: []const u8, offset: usize, depth: u8, parent: u32) void { + const with_fields = hasFields(T, depth); + const count: u32 = if (with_fields) 6 else 5; + const first = next.*; + next.* += count; + nodes[idx] = .{ .name = name, .kind = .dir, .parent = parent, .first = @intCast(first), .count = count, .offset = offset, .size = @sizeOf(T) }; + const me: u32 = @intCast(idx); + nodes[first + 0] = .{ .name = "value", .kind = .value, .parent = me, .offset = offset, .size = @sizeOf(T), .render = renderFor(T), .set = setFor(T) }; + nodes[first + 1] = .{ .name = "type", .kind = .type_name, .parent = me, .offset = offset, .size = @sizeOf(T), .content = @typeName(T) }; + nodes[first + 2] = .{ .name = "size", .kind = .size, .parent = me, .offset = offset, .size = @sizeOf(T), .content = std.fmt.comptimePrint("{d}", .{@sizeOf(T)}) }; + nodes[first + 3] = .{ .name = "addr", .kind = .addr, .parent = me, .offset = offset, .size = @sizeOf(T) }; + nodes[first + 4] = .{ .name = "raw", .kind = .raw, .parent = me, .offset = offset, .size = @sizeOf(T) }; + if (with_fields) { + const fdir: u32 = @intCast(first + 5); + const nf = fieldCount(T); + const ffirst = next.*; + next.* += nf; + nodes[fdir] = .{ .name = "f", .kind = .fields, .parent = me, .first = @intCast(ffirst), .count = @intCast(nf), .offset = offset, .size = @sizeOf(T) }; + var i: usize = 0; + for (@typeInfo(T).@"struct".fields) |f| { + if (f.is_comptime or @sizeOf(f.type) == 0) continue; + fill(nodes, next, ffirst + i, f.type, f.name, offset + @offsetOf(T, f.name), depth + 1, fdir); + i += 1; + } + } +} + +fn buildTable(comptime T: type) [countNodes(T, 0)]Node { + @setEvalBranchQuota(1_000_000); + var nodes: [countNodes(T, 0)]Node = undefined; + var next: usize = 1; + fill(&nodes, &next, 0, T, "", 0, 0, 0); + std.debug.assert(next == nodes.len); + return nodes; +} + +fn renderFor(comptime T: type) RenderFn { + return &struct { + fn f(base: [*]const u8, w: *Writer) Writer.Error!void { + const p: *const T = @ptrCast(@alignCast(base)); + try render(p, w, max_depth); + } + }.f; +} + +fn setFor(comptime T: type) ?SetFn { + if (!settable(T)) return null; + return &struct { + fn f(base: [*]u8, text: []const u8) SetError!void { + const p: *T = @ptrCast(@alignCast(base)); + try set(p, text); + } + }.f; +} + +fn settable(comptime T: type) bool { + return switch (@typeInfo(T)) { + .int, .float, .bool, .@"enum" => true, + else => false, + }; +} + +// --------------------------------------------------------------------------- +// Rendering +// --------------------------------------------------------------------------- + +/// Renders `ptr.*`. Structs become "field: value" lines (nested structs +/// indented); everything else is a single line without a trailing newline. +/// `depth` bounds struct/union/optional nesting; deeper values print as "…". +pub fn render(ptr: anytype, w: *Writer, depth: usize) Writer.Error!void { + const T = @TypeOf(ptr.*); + if (comptime isPlainStruct(T)) { + try renderStruct(T, ptr, w, depth, 0); + } else { + try renderValue(T, ptr, w, depth); + } +} + +fn isPlainStruct(comptime T: type) bool { + return switch (@typeInfo(T)) { + .@"struct" => |s| !s.is_tuple and s.fields.len > 0, + else => false, + }; +} + +/// The multi-line form: each field on its own line, nested structs indented. +fn renderStruct(comptime T: type, ptr: *const T, w: *Writer, depth: usize, indent: usize) Writer.Error!void { + if (depth == 0) { + try w.splatByteAll(' ', indent); + try w.writeAll("…\n"); + return; + } + const packed_layout = @typeInfo(T).@"struct".layout == .@"packed"; + inline for (@typeInfo(T).@"struct".fields) |f| { + try w.splatByteAll(' ', indent); + try w.writeAll(f.name); + try w.writeByte(':'); + if (comptime f.is_comptime) { + try w.writeAll(" (comptime)\n"); + } else if (comptime packed_layout) { + // Fields of a packed struct have no byte address: render a copy. + const v = @field(ptr.*, f.name); + try w.writeByte(' '); + try renderValue(f.type, &v, w, depth - 1); + try w.writeByte('\n'); + } else if (comptime isPlainStruct(f.type)) { + try w.writeByte('\n'); + try renderStruct(f.type, &@field(ptr.*, f.name), w, depth - 1, indent + 2); + } else { + try w.writeByte(' '); + try renderValue(f.type, &@field(ptr.*, f.name), w, depth - 1); + try w.writeByte('\n'); + } + } +} + +/// The single-line form of any value. +fn renderValue(comptime T: type, ptr: *const T, w: *Writer, depth: usize) Writer.Error!void { + switch (@typeInfo(T)) { + .int, .comptime_int => try w.print("{d}", .{ptr.*}), + .float, .comptime_float => try w.print("{d}", .{ptr.*}), + .bool => { + // The variable is live memory that anything (a debugger's /mem write, + // a torn update) may have corrupted: judge the byte, not the bool. + const b = @as(*const u8, @ptrCast(ptr)).*; + switch (b) { + 0 => try w.writeAll("false"), + 1 => try w.writeAll("true"), + else => try w.print("{d}", .{b}), + } + }, + .void => try w.writeAll("{}"), + .@"enum" => |e| { + // Read the storage bytes as one integer: @tagName/switch on a corrupt + // value is a safety panic, and the value is caller memory we do not + // control. A load through the tag type would truncate to its bit + // width (a u2 tag in a byte), so the full storage width is read. + if (@sizeOf(T) == 0) return w.writeAll(e.fields[0].name); + const Raw = std.meta.Int(.unsigned, @sizeOf(T) * 8); + const raw = @as(*align(@alignOf(T)) const Raw, @ptrCast(ptr)).*; + const TagU = std.meta.Int(.unsigned, @bitSizeOf(e.tag_type)); + const padding: Raw = if (@bitSizeOf(TagU) == @bitSizeOf(Raw)) 0 else ~@as(Raw, std.math.maxInt(TagU)); + if (raw & padding == 0) { + const low: TagU = @truncate(raw); + inline for (e.fields) |f| { + if (low == @as(TagU, @bitCast(@as(e.tag_type, f.value)))) return w.writeAll(f.name); + } + } + try w.print("{d}", .{raw}); + }, + .error_set => try w.print("error.{s}", .{@errorName(ptr.*)}), + .error_union => |eu| if (ptr.*) |v| { + try renderValue(eu.payload, &v, w, depth); + } else |e| { + try w.print("error.{s}", .{@errorName(e)}); + }, + .optional => |o| if (ptr.*) |v| { + try renderValue(o.child, &v, w, depth); + } else { + try w.writeAll("null"); + }, + .pointer => |p| switch (p.size) { + .slice => if (p.child == u8) { + try renderString(ptr.*, w); + } else { + try renderElems(p.child, ptr.*, w, depth); + }, + .many => if (p.child == u8 and p.sentinel() == 0) { + try renderCString(ptr.*, w); + } else { + try w.print("0x{x}", .{@intFromPtr(ptr.*)}); + }, + .one, .c => try w.print("0x{x}", .{@intFromPtr(ptr.*)}), + }, + .array => |a| if (a.child == u8) { + try renderString(ptr.*[0..], w); + } else { + try renderElems(a.child, ptr.*[0..], w, depth); + }, + .vector => |v| { + const arr: [v.len]v.child = ptr.*; + try renderElems(v.child, &arr, w, depth); + }, + .@"struct" => |s| { + if (depth == 0) { + try w.writeAll("…"); + return; + } + if (s.fields.len == 0) { + try w.writeAll("{}"); + return; + } + try w.writeAll("{ "); + inline for (s.fields, 0..) |f, i| { + if (i != 0) try w.writeAll(", "); + if (!s.is_tuple) { + try w.writeAll(f.name); + try w.writeAll(": "); + } + if (comptime f.is_comptime) { + try w.writeAll("(comptime)"); + } else if (comptime s.layout == .@"packed") { + const v = @field(ptr.*, f.name); + try renderValue(f.type, &v, w, depth - 1); + } else { + try renderValue(f.type, &@field(ptr.*, f.name), w, depth - 1); + } + } + try w.writeAll(" }"); + }, + .@"union" => |u| { + const Tag = u.tag_type orelse { + try w.print("(untagged union, {d} bytes)", .{@sizeOf(T)}); + return; + }; + if (depth == 0) { + try w.writeAll("…"); + return; + } + // A switch on a corrupt tag is a safety panic: match the integer first. + const raw = @intFromEnum(@as(Tag, ptr.*)); + inline for (u.fields) |f| { + if (raw == @intFromEnum(@field(Tag, f.name))) { + try w.writeAll(f.name); + if (f.type != void) { + try w.writeAll(": "); + try renderValue(f.type, &@field(ptr.*, f.name), w, depth - 1); + } + return; + } + } + try w.print("(invalid tag {d})", .{raw}); + }, + .@"fn" => try w.print("0x{x}", .{@intFromPtr(ptr)}), + else => try w.print("<{s}>", .{@typeName(T)}), + } +} + +fn renderElems(comptime E: type, items: []const E, w: *Writer, depth: usize) Writer.Error!void { + try w.writeByte('['); + for (items, 0..) |*item, i| { + if (i == max_elems) { + try w.writeAll(", …"); + break; + } + if (i != 0) try w.writeAll(", "); + try renderValue(E, item, w, depth); + } + try w.writeByte(']'); +} + +/// A NUL-terminated string, scanning at most `max_string` + 1 bytes for the +/// terminator so that a missing one cannot walk off the end of the mapping. +fn renderCString(s: [*:0]const u8, w: *Writer) Writer.Error!void { + var n: usize = 0; + while (n <= max_string and s[n] != 0) n += 1; + try renderString(s[0..n], w); +} + +/// A double-quoted string with C-style escapes, truncated to `max_string` bytes. +fn renderString(s: []const u8, w: *Writer) Writer.Error!void { + try w.writeByte('"'); + for (s[0..@min(s.len, max_string)]) |b| switch (b) { + '\n' => try w.writeAll("\\n"), + '\r' => try w.writeAll("\\r"), + '\t' => try w.writeAll("\\t"), + '\\' => try w.writeAll("\\\\"), + '"' => try w.writeAll("\\\""), + ' '...'!', '#'...'[', ']'...'~' => try w.writeByte(b), + else => { + const hex = "0123456789abcdef"; + try w.writeAll("\\x"); + try w.writeByte(hex[b >> 4]); + try w.writeByte(hex[b & 15]); + }, + }; + try w.writeByte('"'); + if (s.len > max_string) try w.writeAll("…"); +} + +// --------------------------------------------------------------------------- +// Setting +// --------------------------------------------------------------------------- + +/// Parses `text` and stores it into `ptr.*`: ints in decimal or 0x/0o/0b, +/// floats, bools (true/false/1/0), enums by tag name (or by integer value for +/// non-exhaustive enums). Other types are `error.Unsupported`. +pub fn set(ptr: anytype, text: []const u8) SetError!void { + const T = @TypeOf(ptr.*); + const s = std.mem.trim(u8, text, " \t\r\n\x00"); + switch (@typeInfo(T)) { + .int => ptr.* = std.fmt.parseInt(T, s, 0) catch return error.Invalid, + .float => ptr.* = std.fmt.parseFloat(T, s) catch return error.Invalid, + .bool => { + if (std.mem.eql(u8, s, "true") or std.mem.eql(u8, s, "1")) { + ptr.* = true; + } else if (std.mem.eql(u8, s, "false") or std.mem.eql(u8, s, "0")) { + ptr.* = false; + } else return error.Invalid; + }, + .@"enum" => |e| { + if (std.meta.stringToEnum(T, s)) |v| { + ptr.* = v; + } else if (!e.is_exhaustive) { + const raw = std.fmt.parseInt(e.tag_type, s, 0) catch return error.Invalid; + ptr.* = @enumFromInt(raw); + } else return error.Invalid; + }, + else => return error.Unsupported, + } +} + +// --------------------------------------------------------------------------- +// Tests +// --------------------------------------------------------------------------- + +const testing = std.testing; + +fn renderToBuf(buf: []u8, ptr: anytype) ![]const u8 { + var w: Writer = .fixed(buf); + try render(ptr, &w, max_depth); + return w.buffered(); +} + +test "render scalars, strings, pointers, optionals, enums, arrays" { + var buf: [512]u8 = undefined; + const i: i32 = -42; + try testing.expectEqualStrings("-42", try renderToBuf(&buf, &i)); + const f: f32 = 1.5; + try testing.expectEqualStrings("1.5", try renderToBuf(&buf, &f)); + const b: bool = true; + try testing.expectEqualStrings("true", try renderToBuf(&buf, &b)); + const s: []const u8 = "hi \"there\"\n"; + try testing.expectEqualStrings("\"hi \\\"there\\\"\\n\"", try renderToBuf(&buf, &s)); + const z: [*:0]const u8 = "zed"; + try testing.expectEqualStrings("\"zed\"", try renderToBuf(&buf, &z)); + const p: *const i32 = &i; + var expect_buf: [32]u8 = undefined; + const expect = try std.fmt.bufPrint(&expect_buf, "0x{x}", .{@intFromPtr(&i)}); + try testing.expectEqualStrings(expect, try renderToBuf(&buf, &p)); + const o: ?u8 = null; + try testing.expectEqualStrings("null", try renderToBuf(&buf, &o)); + const o2: ?u8 = 7; + try testing.expectEqualStrings("7", try renderToBuf(&buf, &o2)); + const E = enum { red, green }; + const e: E = .green; + try testing.expectEqualStrings("green", try renderToBuf(&buf, &e)); + const NE = enum(u8) { a, _ }; + const ne: NE = @enumFromInt(9); + try testing.expectEqualStrings("9", try renderToBuf(&buf, &ne)); + const arr = [_]u16{ 1, 2, 3 }; + try testing.expectEqualStrings("[1, 2, 3]", try renderToBuf(&buf, &arr)); + const bytes = [_]u8{ 0, 'a', 0xff }; + try testing.expectEqualStrings("\"\\x00a\\xff\"", try renderToBuf(&buf, &bytes)); + const U = union(enum) { none, some: u32 }; + const u: U = .{ .some = 5 }; + try testing.expectEqualStrings("some: 5", try renderToBuf(&buf, &u)); + const un: U = .none; + try testing.expectEqualStrings("none", try renderToBuf(&buf, &un)); +} + +test "render structs multi-line with nested indentation and depth limit" { + const Inner = struct { x: f32, flags: [2]bool }; + const Outer = struct { a: u32, b: bool, name: []const u8, inner: Inner, items: []const Inner }; + const v: Outer = .{ .a = 1, .b = false, .name = "n", .inner = .{ .x = 2.5, .flags = .{ true, false } }, .items = &.{.{ .x = 0, .flags = .{ false, false } }} }; + var buf: [512]u8 = undefined; + try testing.expectEqualStrings( + \\a: 1 + \\b: false + \\name: "n" + \\inner: + \\ x: 2.5 + \\ flags: [true, false] + \\items: [{ x: 0, flags: [false, false] }] + \\ + , try renderToBuf(&buf, &v)); + var w: Writer = .fixed(&buf); + try render(&v, &w, 1); + try testing.expectEqualStrings( + \\a: 1 + \\b: false + \\name: "n" + \\inner: + \\ … + \\items: […] + \\ + , w.buffered()); +} + +test "long strings and arrays are truncated" { + const long = [_]u8{'x'} ** 300; + var buf: [1024]u8 = undefined; + const s: []const u8 = &long; + const out = try renderToBuf(&buf, &s); + try testing.expectEqual(@as(usize, 1 + max_string + 1 + "…".len), out.len); + try testing.expect(std.mem.endsWith(u8, out, "\"…")); + const nums: [100]u32 = @splat(1); + const out2 = try renderToBuf(&buf, &nums); + try testing.expect(std.mem.endsWith(u8, out2, ", …]")); + try testing.expectEqual(@as(usize, max_elems), std.mem.count(u8, out2, "1")); +} + +test "set parses ints, floats, bools and enums" { + var i: u32 = 0; + try set(&i, "42\n"); + try testing.expectEqual(@as(u32, 42), i); + try set(&i, "0x10"); + try testing.expectEqual(@as(u32, 16), i); + try testing.expectError(error.Invalid, set(&i, "-1")); + try testing.expectError(error.Invalid, set(&i, "abc")); + var si: i8 = 0; + try set(&si, " -7 "); + try testing.expectEqual(@as(i8, -7), si); + try testing.expectError(error.Invalid, set(&si, "200")); + var f: f64 = 0; + try set(&f, "2.25"); + try testing.expectEqual(@as(f64, 2.25), f); + var b: bool = false; + try set(&b, "true"); + try testing.expect(b); + try set(&b, "0"); + try testing.expect(!b); + try testing.expectError(error.Invalid, set(&b, "maybe")); + const E = enum { off, on }; + var e: E = .off; + try set(&e, "on"); + try testing.expectEqual(E.on, e); + try testing.expectError(error.Invalid, set(&e, "blue")); + var s: []const u8 = "x"; + try testing.expectError(error.Unsupported, set(&s, "y")); +} + +test "vtable table layout for a nested struct" { + const Inner = struct { x: f32 }; + const T = struct { a: u32, b: bool, name: []const u8, inner: Inner }; + const vt = vtableFor(T); + try testing.expectEqual(vt, vtableFor(T)); + try testing.expectEqualStrings(@typeName(T), vt.type_name); + const root = vt.nodes[0]; + try testing.expect(root.isDir()); + try testing.expectEqual(@as(u32, 6), root.count); + const value = vt.child(0, "value").?; + try testing.expect(!vt.nodes[value].writable()); // a struct is not settable + try testing.expectEqualStrings(std.fmt.comptimePrint("{d}", .{@sizeOf(T)}), vt.nodes[vt.child(0, "size").?].content); + const f = vt.child(0, "f").?; + try testing.expectEqual(Kind.fields, vt.nodes[f].kind); + try testing.expectEqual(@as(u32, 4), vt.nodes[f].count); + const a = vt.child(f, "a").?; + try testing.expectEqual(@offsetOf(T, "a"), vt.nodes[a].offset); + const a_value = vt.child(a, "value").?; + try testing.expect(vt.nodes[a_value].writable()); + try testing.expectEqual(@as(usize, 4), vt.nodes[a_value].size); + try testing.expectEqual(a, vt.nodes[a_value].parent); + const inner = vt.child(f, "inner").?; + const inner_f = vt.child(inner, "f").?; + const x = vt.child(inner_f, "x").?; + try testing.expectEqual(@offsetOf(T, "inner") + @offsetOf(Inner, "x"), vt.nodes[x].offset); + try testing.expectEqualStrings("f32", vt.nodes[vt.child(x, "type").?].content); + try testing.expect(vt.child(f, "nope") == null); + // rendering and setting through the table + var v: T = .{ .a = 1, .b = true, .name = "n", .inner = .{ .x = 0.5 } }; + const base: [*]u8 = @ptrCast(&v); + var buf: [256]u8 = undefined; + var w: Writer = .fixed(&buf); + const x_value = vt.child(x, "value").?; + try vt.nodes[x_value].render.?(base + vt.nodes[x_value].offset, &w); + try testing.expectEqualStrings("0.5", w.buffered()); + try vt.nodes[a_value].set.?(base + vt.nodes[a_value].offset, "42"); + try testing.expectEqual(@as(u32, 42), v.a); + w = .fixed(&buf); + try vt.nodes[value].render.?(base, &w); + try testing.expect(std.mem.startsWith(u8, w.buffered(), "a: 42\nb: true\n")); +} + +test "depth limit stops the f/ tree at max_depth" { + const L4 = struct { v: u8 }; + const L3 = struct { l4: L4 }; + const L2 = struct { l3: L3 }; + const L1 = struct { l2: L2 }; + const L0 = struct { l1: L1 }; + const vt = vtableFor(L0); + var node: u32 = 0; + var depth: usize = 0; + while (vt.child(node, "f")) |f| : (depth += 1) { + node = vt.nodes[f].first; // the single field + } + try testing.expectEqual(@as(usize, max_depth), depth); + try testing.expect(vt.child(node, "value") != null); +} + +test "every @typeInfo category renders without dereferencing anything unbounded" { + var buf: [2048]u8 = undefined; + // packed and extern structs (packed fields have no address: rendered by copy) + const Packed = packed struct { a: u3, b: bool, c: u12, e: enum(u2) { p, q, r } }; + const pk: Packed = .{ .a = 5, .b = true, .c = 300, .e = .r }; + try testing.expectEqualStrings("a: 5\nb: true\nc: 300\ne: r\n", try renderToBuf(&buf, &pk)); + const Ext = extern struct { x: u16, y: f32, inner: extern struct { z: u8 } }; + const ex: Ext = .{ .x = 1, .y = 0.5, .inner = .{ .z = 9 } }; + try testing.expectEqualStrings("x: 1\ny: 0.5\ninner:\n z: 9\n", try renderToBuf(&buf, &ex)); + const Holder = struct { p: Packed, list: [2]Packed }; + const ho: Holder = .{ .p = pk, .list = .{ pk, pk } }; + try testing.expect(std.mem.startsWith(u8, try renderToBuf(&buf, &ho), "p:\n a: 5\n")); + // the f/ tree has no entries for a packed struct and works through a table + const vt = vtableFor(Packed); + try testing.expect(vt.child(0, "f") == null); + var w: Writer = .fixed(&buf); + try vt.nodes[vt.child(0, "value").?].render.?(@ptrCast(&pk), &w); + try testing.expect(std.mem.startsWith(u8, w.buffered(), "a: 5\n")); + // optionals of pointers are printed, never followed + var target: u32 = 7; + const op: ?*u32 = ⌖ + var expect_buf: [32]u8 = undefined; + try testing.expectEqualStrings(try std.fmt.bufPrint(&expect_buf, "0x{x}", .{@intFromPtr(&target)}), try renderToBuf(&buf, &op)); + const np: ?*u32 = null; + try testing.expectEqualStrings("null", try renderToBuf(&buf, &np)); + const dangling: *const u32 = @ptrFromInt(0x1000); + try testing.expectEqualStrings("0x1000", try renderToBuf(&buf, &dangling)); + const cptr: [*c]const u8 = @ptrFromInt(0x2000); + try testing.expectEqualStrings("0x2000", try renderToBuf(&buf, &cptr)); + const manyp: [*]const u32 = @ptrFromInt(0x3000); + try testing.expectEqualStrings("0x3000", try renderToBuf(&buf, &manyp)); + // untagged and tagged unions, error unions, error sets + const Untagged = union { a: u32, b: f32 }; + const un: Untagged = .{ .a = 1 }; + try testing.expectEqualStrings(std.fmt.comptimePrint("(untagged union, {d} bytes)", .{@sizeOf(Untagged)}), try renderToBuf(&buf, &un)); + const Tagged = union(enum(u8)) { none, some: u32, pair: struct { l: u8, r: u8 } }; + const tg: Tagged = .{ .pair = .{ .l = 1, .r = 2 } }; + try testing.expectEqualStrings("pair: { l: 1, r: 2 }", try renderToBuf(&buf, &tg)); + const eu: anyerror!u8 = error.Boom; + try testing.expectEqualStrings("error.Boom", try renderToBuf(&buf, &eu)); + const eu2: error{X}!u8 = 4; + try testing.expectEqualStrings("4", try renderToBuf(&buf, &eu2)); + const es: anyerror = error.Zap; + try testing.expectEqualStrings("error.Zap", try renderToBuf(&buf, &es)); + // wide ints and floats, vectors, sentinel arrays, slices of slices, void, comptime fields + const big: u128 = std.math.maxInt(u128); + try testing.expectEqualStrings("340282366920938463463374607431768211455", try renderToBuf(&buf, &big)); + const neg: i128 = std.math.minInt(i128); + try testing.expectEqualStrings("-170141183460469231731687303715884105728", try renderToBuf(&buf, &neg)); + const h: f16 = 1.5; + try testing.expectEqualStrings("1.5", try renderToBuf(&buf, &h)); + const ld: f80 = 2.25; + try testing.expectEqualStrings("2.25", try renderToBuf(&buf, &ld)); + const quad: f128 = 3.125; + try testing.expectEqualStrings("3.125", try renderToBuf(&buf, &quad)); + const vec: @Vector(4, i16) = .{ 1, -2, 3, -4 }; + try testing.expectEqualStrings("[1, -2, 3, -4]", try renderToBuf(&buf, &vec)); + const sarr: [3:0]u8 = .{ 'a', 'b', 'c' }; + try testing.expectEqualStrings("\"abc\"", try renderToBuf(&buf, &sarr)); + const rows: []const []const u8 = &.{ "ab", "cd" }; + try testing.expectEqualStrings("[\"ab\", \"cd\"]", try renderToBuf(&buf, &rows)); + const Odd = struct { v: void, comptime k: u8 = 3, n: u8 }; + const odd: Odd = .{ .v = {}, .n = 1 }; + try testing.expectEqualStrings("v: {}\nk: (comptime)\nn: 1\n", try renderToBuf(&buf, &odd)); + try testing.expectEqual(@as(u32, 1), vtableFor(Odd).nodes[vtableFor(Odd).child(0, "f").?].count); + // self-referential through a pointer: rendered as an address, table stays finite + const Link = struct { next: ?*const @This(), v: u8 }; + var a: Link = .{ .next = null, .v = 1 }; + const b: Link = .{ .next = &a, .v = 2 }; + a.next = &b; + try testing.expectEqualStrings(try std.fmt.bufPrint(&expect_buf, "next: 0x{x}\nv: 2\n", .{@intFromPtr(&a)}), try renderToBuf(&buf, &b)); + try testing.expect(vtableFor(Link).nodes.len < 32); + // tuples + const tup: struct { u8, []const u8 } = .{ 1, "x" }; + try testing.expectEqualStrings("{ 1, \"x\" }", try renderToBuf(&buf, &tup)); +} + +test "corrupt live memory renders instead of trapping: enums, unions, bools" { + var buf: [128]u8 = undefined; + const E = enum(u8) { a, b }; + var raw_e: u8 = 7; + try testing.expectEqualStrings("7", try renderToBuf(&buf, @as(*const E, @ptrCast(&raw_e)))); + raw_e = 1; + try testing.expectEqualStrings("b", try renderToBuf(&buf, @as(*const E, @ptrCast(&raw_e)))); + // a u2 tag in a byte: the whole byte is judged, not the truncated tag (ReleaseSafe would say "c") + const E3 = enum { a, b, c }; + var raw3: u8 = 0xEE; + try testing.expectEqualStrings("238", try renderToBuf(&buf, @as(*const E3, @ptrCast(&raw3)))); + raw3 = 3; + try testing.expectEqualStrings("3", try renderToBuf(&buf, @as(*const E3, @ptrCast(&raw3)))); + raw3 = 2; + try testing.expectEqualStrings("c", try renderToBuf(&buf, @as(*const E3, @ptrCast(&raw3)))); + const E12 = enum(u12) { p = 5, q = 4095 }; + var raw12: u16 = 0xF005; + try testing.expectEqualStrings("61445", try renderToBuf(&buf, @as(*const E12, @ptrCast(&raw12)))); + raw12 = 4095; + try testing.expectEqualStrings("q", try renderToBuf(&buf, @as(*const E12, @ptrCast(&raw12)))); + const ES = enum(i8) { neg = -3, pos = 7 }; + var raws: u8 = 0xFD; + try testing.expectEqualStrings("neg", try renderToBuf(&buf, @as(*const ES, @ptrCast(&raws)))); + raws = 0x80; + try testing.expectEqualStrings("128", try renderToBuf(&buf, @as(*const ES, @ptrCast(&raws)))); + const E1 = enum { only }; + const e1: E1 = .only; + try testing.expectEqualStrings("only", try renderToBuf(&buf, &e1)); + const NE = enum(u16) { x = 5, _ }; + var raw_ne: u16 = 5; + try testing.expectEqualStrings("x", try renderToBuf(&buf, @as(*const NE, @ptrCast(&raw_ne)))); + raw_ne = 6; + try testing.expectEqualStrings("6", try renderToBuf(&buf, @as(*const NE, @ptrCast(&raw_ne)))); + var raw_b: u8 = 2; + try testing.expectEqualStrings("2", try renderToBuf(&buf, @as(*const bool, @ptrCast(&raw_b)))); + const U = union(enum(u8)) { x: u32, y: bool }; + var raw_u: [@sizeOf(U)]u8 align(@alignOf(U)) = @splat(0x55); + const out = try renderToBuf(&buf, @as(*const U, @ptrCast(&raw_u))); + try testing.expectEqualStrings("(invalid tag 85)", out); + const S = struct { e: E, u: U, b: bool }; + var raw_s: [@sizeOf(S)]u8 align(@alignOf(S)) = @splat(0xEE); + const ps: *const S = @ptrCast(&raw_s); + _ = try renderToBuf(&buf, ps); // no trap + try testing.expect(std.mem.indexOf(u8, try renderToBuf(&buf, ps), "238") != null); +} + +test "a [*:0]const u8 without a terminator is read at most max_string + 1 bytes" { + // Only the first max_string + 1 bytes exist; anything beyond is the + // testing allocator's guard, which a wider scan would touch. + const mem = try testing.allocator.alloc(u8, max_string + 1); + defer testing.allocator.free(mem); + @memset(mem, 'x'); + const z: [*:0]const u8 = @ptrCast(mem.ptr); + var buf: [1024]u8 = undefined; + const out = try renderToBuf(&buf, &z); + try testing.expectEqual(@as(usize, 1 + max_string + 1 + "…".len), out.len); + try testing.expect(std.mem.endsWith(u8, out, "\"…")); + // exactly max_string bytes then NUL: no ellipsis + const mem2 = try testing.allocator.alloc(u8, max_string + 1); + defer testing.allocator.free(mem2); + @memset(mem2, 'y'); + mem2[max_string] = 0; + const z2: [*:0]const u8 = @ptrCast(mem2.ptr); + const out2 = try renderToBuf(&buf, &z2); + try testing.expectEqual(@as(usize, 1 + max_string + 1), out2.len); + // a garbage-length []const u8 still reads at most max_string bytes + const garbage: []const u8 = mem[0..max_string]; + _ = try renderToBuf(&buf, &garbage); +} + +test "set rejects hostile input without partial writes" { + var u: u8 = 200; + for ([_][]const u8{ "-1", "256", "1e3", "0x", "", " ", "1.5", "+", "0b2", "١", "12abc", "0x100", "\x00", "1 2" }) |bad| { + try testing.expectError(error.Invalid, set(&u, bad)); + try testing.expectEqual(@as(u8, 200), u); + } + try set(&u, "0b1111_1111"); + try testing.expectEqual(@as(u8, 255), u); + try set(&u, "+0o17"); + try testing.expectEqual(@as(u8, 15), u); + var i: i64 = 1; + try set(&i, "-9223372036854775808"); + try testing.expectEqual(std.math.minInt(i64), i); + try testing.expectError(error.Invalid, set(&i, "9223372036854775808")); + var w: u128 = 0; + try set(&w, "340282366920938463463374607431768211455"); + try testing.expectEqual(std.math.maxInt(u128), w); + try testing.expectError(error.Invalid, set(&w, "340282366920938463463374607431768211456")); + // floats: exponents, hex floats, inf/nan spellings, and junk + var f: f32 = 1; + try set(&f, "1.5e3"); + try testing.expectEqual(@as(f32, 1500), f); + try set(&f, "-0x1p-2"); + try testing.expectEqual(@as(f32, -0.25), f); + try set(&f, "1e999"); + try testing.expect(std.math.isInf(f)); + try testing.expectError(error.Invalid, set(&f, "1.5.5")); + try testing.expectError(error.Invalid, set(&f, "e5")); + try testing.expectError(error.Invalid, set(&f, "")); + var h: f16 = 0; + try set(&h, "65504"); + try testing.expectEqual(@as(f16, 65504), h); + var q: f128 = 0; + try set(&q, "2.5"); + try testing.expectEqual(@as(f128, 2.5), q); + // enums: NULs inside the tag, case, trailing junk; non-exhaustive by integer only when out of names + const E = enum(u8) { off, on }; + var e: E = .off; + for ([_][]const u8{ "on\x00x", "On", "on x", "1", "0x1", "" }) |bad| { + try testing.expectError(error.Invalid, set(&e, bad)); + try testing.expectEqual(E.off, e); + } + try set(&e, "\x00on\n"); + try testing.expectEqual(E.on, e); + const NE = enum(u8) { a, _ }; + var ne: NE = .a; + try set(&ne, "200"); + try testing.expectEqual(@as(u8, 200), @intFromEnum(ne)); + try testing.expectError(error.Invalid, set(&ne, "256")); + try testing.expectError(error.Invalid, set(&ne, "-1")); + try set(&ne, "a"); + try testing.expectEqual(NE.a, ne); + // bools + var b: bool = true; + for ([_][]const u8{ "yes", "TRUE", "2", "", "01" }) |bad| { + try testing.expectError(error.Invalid, set(&b, bad)); + try testing.expect(b); + } + // unsupported types are refused without touching memory + var opt: ?u8 = 3; + try testing.expectError(error.Unsupported, set(&opt, "4")); + try testing.expectEqual(@as(?u8, 3), opt); + var arr: [2]u8 = .{ 1, 2 }; + try testing.expectError(error.Unsupported, set(&arr, "x")); + var un: union(enum) { a: u8 } = .{ .a = 1 }; + try testing.expectError(error.Unsupported, set(&un, "a")); +} diff --git a/introspect/test/adv_core_hostile.py b/introspect/test/adv_core_hostile.py new file mode 100755 index 0000000..56ef5a3 --- /dev/null +++ b/introspect/test/adv_core_hostile.py @@ -0,0 +1,1018 @@ +#!/usr/bin/env python3 +"""Hostile raw-9P2000 client aimed at the introspect *core* (stdlib only). + +Complements adv_introspect_hostile.py in this directory (framing, tags, scratch, floods) +with attacks on the freestanding engine's own paths: the /vars tree and its +comptime renderers, snapshot slots, the static tree, the fid table at its +configured maximum, directory-read offsets, msize 24, the ctl staging rule, +and the demo's debug providers driven as black boxes. + +Usage: + adv_core_hostile.py --server zig-out/bin/introspect # spawns it on a temp unix socket + adv_core_hostile.py --socket PATH # attacks a running server + +Exit status is non-zero if any check fails or the server dies. +""" +import argparse +import os +import signal +import struct +import subprocess +import sys +import tempfile +import threading +import time + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +import adv_introspect_hostile as base # noqa: E402 +from adv_introspect_hostile import ( # noqa: E402 + NOTAG, Tversion, Tflush, Rflush, Twalk, Rwalk, Topen, Ropen, Rcreate, + Tread, Rread, Twrite, Rwrite, Tclunk, Rclunk, Tremove, Rremove, Tstat, Rstat, Twstat, Rwstat, + Rerror, OREAD, OWRITE, ORDWR, OEXEC, OTRUNC, ORCLOSE, DMDIR, + Nine, frame, s16, mkstat, parse_stat, ok, healthy, expect_dead, +) + +MAX_FIDS = 32768 # demo/main.zig cfg.max_fids +SNAPSHOT_SLOTS = 8 # demo/main.zig cfg.snapshot_slots (per connection) +SCRATCH_BUDGET = 512 << 20 +SCRATCH_MAX_FILE = 64 << 20 + + +def records(d): + """Splits a directory read into (name, raw-record) pairs.""" + out = [] + while d: + n, = struct.unpack_from("/name.""" + c.walk_ok(0, 40, [b"threads"]) + c.open(40, OREAD) + d = c.read_all(40) + c.clunk(40) + for name, _ in records(d): + if c.path_read([b"threads", name, b"name"], fid=41) == b"worker": + return name + return None + + +# --------------------------------------------------------------------------- /vars + + +def attack_vars(path): + print("# /vars: deep walks, hostile names, renderer edge cases, hostile writes") + c = Nine(path) + c.session(1 << 20) + deep = [b"vars", b"state", b"f", b"last_job", b"f", b"id", b".", b"..", b"id", b".", b"..", b"id", b".", b"..", b"id", b"value"] + assert len(deep) == 16 + ok("16-element walk deep into /vars/state/f/... succeeds", c.walk_ok(0, 1, deep) == 16) + rt, _, _ = c.open(1, OREAD) + ok("deep walk lands on a readable value file", rt == Ropen, rt) + c.clunk(1) + up = [b"vars", b"state", b"f", b"inner"] if False else [b"vars", b"state", b"f", b"last_job"] + [b".."] * 12 + n = c.walk_ok(0, 1, up) + ok("12 x '..' from inside /vars climbs to the root and stays there", n == 16, n) + rt, st = c.stat(1) + ok("fid after the climb is the root directory", rt == Rstat and st["qid"][2] == 0xFF << 56, st) + c.clunk(1) + # names that are hex/decimal edge cases or otherwise hostile: never anything but Rerror/partial walk + for nm in (b"0", b"-1", b"0x", b"0x0", b"state\x00", b"State", b" state", b"state ", b"a" * 255, b"a" * 65535, b"\xff\xfe", b"..\x00", b"f", b"value"): + n = c.walk_ok(0, 1, [b"vars", nm]) + ok(f"walk /vars/{nm[:12]!r}{'...' if len(nm) > 12 else ''} is a partial walk (1)", n == 1, n) + ok(" and newfid stays unbound", c.err(Tclunk, struct.pack(" state -> /vars -> /", len(q) == 4 and q[3][2] == 0xFF << 56 and (q[1][0] & 0x80), q) + c.clunk(1) + c.clunk(2) + # every file under /vars/state reads; raw reads beyond @sizeOf are empty + size = int(c.path_read([b"vars", b"state", b"size"])) + ok("/vars/state/size is a number", size > 0, size) + raw = c.path_read([b"vars", b"state", b"raw"]) + ok("/vars/state/raw has exactly @sizeOf bytes", raw is not None and len(raw) == size, (len(raw) if raw else raw, size)) + c.walk_ok(0, 1, [b"vars", b"state", b"raw"]) + c.open(1, OREAD) + rt, d = c.read(1, size, 100) + ok("raw read at offset @sizeOf is empty", rt == Rread and d == b"", (rt, d)) + rt, d = c.read(1, size - 1, 100) + ok("raw read at @sizeOf-1 returns one byte", rt == Rread and len(d) == 1, (rt, d)) + rt, d = c.read(1, (1 << 64) - 1, 100) + ok("raw read at 2^64-1 is empty", rt == Rread and d == b"") + rt, d = c.read(1, 0, 0xFFFFFFFF) + ok("raw read with count 2^32-1 is clamped", rt == Rread and len(d) == size, (rt, len(d) if d else d)) + rt, st = c.stat(1) + ok("raw stat length is @sizeOf and mode 0444", rt == Rstat and st["length"] == size and st["mode"] == 0o444, st) + ok("raw is read-only", c.err(Twrite, struct.pack(" the refused one now opens; clunk via Tremove (denied) also frees the slot + c.clunk(100) + rt, _, _ = c.open(100 + opened, OREAD) + ok("after one clunk the refused open succeeds", rt == Ropen, rt) + ok("remove of an open dynamic file is denied", c.err(Tremove, struct.pack("/stack/x is 'not a directory'", c.walk_ok(0, 1, [b"threads", tid, b"stack"]) == 3 and c.err(Twalk, struct.pack("/stack is 'not a directory'", c.err(Twalk, struct.pack("/../..//name walks", c.walk_ok(0, 1, [b"threads", tid, b"..", b"..", b"threads", tid, b"name"]) == 7) + c.clunk(1) + c.walk_ok(0, 1, [b"threads", tid]) + ok("create under /threads/ is denied", c.err(base.Tcreate, struct.pack(" is denied", c.err(Twstat, struct.pack(" is denied", c.err(Tremove, struct.pack(" reads @sizeOf bytes", rt == Rread and len(d) == size, (rt, len(d) if d else d)) + rt, d = c.read(1, 0, 0xFFFFFFFF) + ok("/mem read with count 2^32-1 is clamped and answered", rt in (Rread, Rerror), rt) + c.clunk(1) + hexd = c.path_read([b"hex", hx]) + ok("/hex/ is a hexdump", hexd is not None and len(hexd) > 64, hexd[:40] if hexd else hexd) + # a value written through /mem must render, not trap: corrupt the phase enum and read /vars/state/value + phase_addr = int(c.path_read([b"vars", b"state", b"f", b"phase", b"addr"]), 16) + c.walk_ok(0, 1, [b"mem", b"%x" % phase_addr]) + c.open(1, OWRITE) + rt, _, _ = c.write(1, 0, b"\xee") + ok("write a corrupt enum byte through /mem", rt == Rwrite, rt) + c.clunk(1) + v = c.path_read([b"vars", b"state", b"value"]) + ok("/vars/state/value renders the corrupt enum as a number instead of trapping", v is not None and b"phase: 238" in v, v) + pv = c.path_read([b"vars", b"state", b"f", b"phase", b"value"]) + ok("/vars/state/f/phase/value renders 238", pv == b"238", pv) + c.walk_ok(0, 1, [b"vars", b"state", b"f", b"phase", b"value"]) + c.open(1, OWRITE) + rt, _, _ = c.write(1, 0, b"idle") + ok("the enum can be repaired through /vars", rt == Rwrite, rt) + c.clunk(1) + # /panic: ctl refuses reads and garbage; message/stack read + ok("/panic/message reads (empty, no panic)", c.path_read([b"panic", b"message"]) == b"") + ok("/panic/stack reads", c.path_read([b"panic", b"stack"]) is not None) + c.walk_ok(0, 1, [b"panic", b"ctl"]) + ok("/panic/ctl refuses OREAD", c.err(Topen, struct.pack("= 1, (len(held), err)) + for f in held: + c.clunk(f) + ok("after clunking, /addr opens again", c.path_read([b"addr", b"1000"]) is not None) + # Tversion with debug files open (hexdumps of the exposed state: mapped memory) + base_addr = int(addr, 16) if addr else 0 + opened = 0 + for i in range(4): + c.walk_ok(0, 300 + i, [b"hex", b"%x" % (base_addr + i)]) + opened += c.open(300 + i, OREAD)[0] == Ropen + ok("four /hex snapshots open", opened == 4, opened) + rt, _, _ = c.version(65536) + ok("Tversion with debug snapshots open", rt == base.Rversion) + c.attach() + ok("/hex still opens after the reset", c.path_read([b"hex", b"%x" % base_addr]) is not None) + ok("/hex of unmapped memory is an Rerror at open, not a crash", c.walk_ok(0, 1, [b"hex", b"3000"]) == 2 and c.open(1, OREAD)[0] == Rerror) + c.clunk(1) + c.close() + ok("server healthy after debug provider attacks", healthy(path)) + + +# --------------------------------------------------------------------------- fids at the maximum + + +def flood(c, ids, names): + """Pipelines one Twalk per id and counts the Rwalk replies; returns (ok_count, error_count, seconds).""" + got = [0, 0] + dead = [False] + + def reader(): + try: + for _ in ids: + rt, _, _ = c.recv_frame() + if rt == Rwalk: + got[0] += 1 + else: + got[1] += 1 + except (EOFError, OSError): + dead[0] = True + + t = threading.Thread(target=reader) + t.start() + t0 = time.time() + body = b"".join(frame(Twalk, i & 0xFFFE, struct.pack(" the handler's writer fails + rt2, d = c.read(1, 0, 100) + rt3, st5 = c.stat(1) + ok("an over-long result is an Rerror and the previous result survives", rt == Rerror and d == b"y" * 100 and st5["length"] == 60000, (rt, d[:10] if d else d, st5["length"])) + c.clunk(1) + c.close() + ok("server healthy after ctl attacks", healthy(path)) + + +# --------------------------------------------------------------------------- fid state machine on scratch + + +def attack_fid_states(path): + print("# fid state machine on /scratch") + c = Nine(path) + c.session() + tag = os.urandom(3).hex().encode() + root = b"fs-" + tag + c.walk_ok(0, 1, [b"scratch"]) + c.create(1, root, DMDIR | 0o755, OREAD) + c.clunk(1) + S = [b"scratch", root] + c.walk_ok(0, 1, S) + rt, _, _ = c.create(1, b"f", 0o644, ORDWR) + ok("create f", rt == Rcreate) + ok("open of an open fid is 'file already open'", c.err(Topen, struct.pack("> 20} MiB fill the {SCRATCH_BUDGET >> 20} MiB budget exactly", made == count, made) + print(f" filled the budget in {time.time() - t0:.1f}s") + c.walk_ok(0, 1, S) + c.create(1, b"one-more", 0o644, OWRITE) + ok("one more byte is 'no space left on device'", c.err(Twrite, struct.pack(" [--fast] (part of zig build introspect-adv) +# Spawns the server on a temporary unix socket and attacks it with +# introspect/test/adv_core_hostile.py (Python 3 stdlib). Exit 1 on any failure. +set -u +INTROSPECT=$(realpath "${1:?path to introspect}") +shift +command -v python3 >/dev/null || { echo "SKIP: python3 missing"; exit 0; } +exec python3 "$(dirname "$0")/adv_core_hostile.py" --server "$INTROSPECT" "$@" diff --git a/introspect/test/adv_introspect_hostile.py b/introspect/test/adv_introspect_hostile.py new file mode 100755 index 0000000..7dbb5c0 --- /dev/null +++ b/introspect/test/adv_introspect_hostile.py @@ -0,0 +1,999 @@ +#!/usr/bin/env python3 +"""Hostile raw-9P2000 client for the introspect server (stdlib only). + +Usage: + adv_introspect_hostile.py --server zig-out/bin/introspect # spawns it on a temp unix socket + adv_introspect_hostile.py --socket PATH # attacks a running server + +Every attack is followed by a "server still healthy" probe on a fresh connection. +Exit status is non-zero if any check fails, the server dies, or a probe hangs. +""" +import argparse +import os +import signal +import socket +import struct +import subprocess +import sys +import tempfile +import threading +import time + +NOTAG = 0xFFFF +NOFID = 0xFFFFFFFF +Tversion, Rversion, Tauth, Rauth, Tattach, Rattach, Rerror = 100, 101, 102, 103, 104, 105, 107 +Tflush, Rflush, Twalk, Rwalk, Topen, Ropen, Tcreate, Rcreate = 108, 109, 110, 111, 112, 113, 114, 115 +Tread, Rread, Twrite, Rwrite, Tclunk, Rclunk, Tremove, Rremove = 116, 117, 118, 119, 120, 121, 122, 123 +Tstat, Rstat, Twstat, Rwstat = 124, 125, 126, 127 +OREAD, OWRITE, ORDWR, OEXEC, OTRUNC, ORCLOSE = 0, 1, 2, 3, 0x10, 0x40 +DMDIR, DMAPPEND, DMEXCL = 0x80000000, 0x40000000, 0x20000000 +NAMES = {v: k for k, v in globals().items() if k[:1] in "TR" and isinstance(v, int) and 100 <= v <= 127} + +FAILS = [] +PASSES = 0 + + +def ok(name, cond, detail=""): + global PASSES + if cond: + PASSES += 1 + print(f"ok - {name}") + else: + FAILS.append(name) + print(f"FAIL - {name} {detail}") + + +def s16(b): + return struct.pack(" connection closed + c.raw(frame(Twalk, 1, struct.pack("0 + ok("walk to file", c.walk_ok(0, 1, [b"build", b"target"]) == 2) + ok("walk from file fails 'not a directory'", c.err(Twalk, struct.pack(" 0) + ok("read dir at bad offset", c.err(Tread, struct.pack(" 6: + return + if c.walk_ok(0, 9, names) != len(names): + ok("walk " + b"/".join(names).decode(), False) + return + rt, st = c.stat(9) + if st["mode"] & DMDIR: + c.open(9, OREAD) + d = c.read_all(9, 512) + c.clunk(9) + while d: + n, = struct.unpack_from(" 0: + ok("server process still running", proc.poll() is None, proc.poll()) + finally: + if proc is not None: + proc.send_signal(signal.SIGTERM) + try: + _, err = proc.communicate(timeout=5) + except subprocess.TimeoutExpired: + proc.kill() + _, err = proc.communicate() + lines = [ln for ln in err.decode("utf-8", "replace").splitlines() if "connection ended" not in ln and "read: " not in ln] + if lines: + print("# server stderr (filtered):") + for ln in lines[:40]: + print(" " + ln) + if tmp: + try: + os.unlink(path) + os.rmdir(tmp) + except OSError: + pass + print(f"# {PASSES} passed, {len(FAILS)} failed") + for f in FAILS: + print("# FAIL " + f) + sys.exit(1 if FAILS else 0) + + +if __name__ == "__main__": + main() diff --git a/introspect/test/adv_introspect_hostile.sh b/introspect/test/adv_introspect_hostile.sh new file mode 100755 index 0000000..3ee2ecb --- /dev/null +++ b/introspect/test/adv_introspect_hostile.sh @@ -0,0 +1,10 @@ +#!/usr/bin/env bash +# Adversarial raw-9P2000 client tests for the introspect server. +# Usage: bash introspect/test/adv_introspect_hostile.sh [--fast] (part of zig build introspect-adv) +# Spawns the server on a temporary unix socket and attacks it with +# introspect/test/adv_introspect_hostile.py (Python 3 stdlib). Exit 1 on any failure. +set -u +INTROSPECT=$(realpath "${1:?path to introspect}") +shift +command -v python3 >/dev/null || { echo "SKIP: python3 missing"; exit 0; } +exec python3 "$(dirname "$0")/adv_introspect_hostile.py" --server "$INTROSPECT" "$@" diff --git a/introspect/test/adv_linux_probe.py b/introspect/test/adv_linux_probe.py new file mode 100755 index 0000000..83de771 --- /dev/null +++ b/introspect/test/adv_linux_probe.py @@ -0,0 +1,421 @@ +#!/usr/bin/env python3 +"""Adversarial tests of the introspect Linux layer: probe loop, debug provider, +signal machinery. Raw 9P2000 over a unix socket, plus one 9player mount. +Usage: adv_linux_probe.py --player <9player> --server +Reuses the client of adv_introspect_hostile.py. Exit 1 on any failure. +""" +import argparse +import ctypes +import os +import signal +import socket +import struct +import subprocess +import sys +import tempfile +import threading +import time + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +from adv_introspect_hostile import ( # noqa: E402 + NOFID, NOTAG, OREAD, OWRITE, Nine, Rerror, Ropen, Rread, Rversion, Rwalk, Rwrite, + Tattach, Tread, Tversion, Twrite, frame, healthy, ok, parse_stat, s16) +import adv_introspect_hostile as hostile # noqa: E402 + +libc = ctypes.CDLL(None, use_errno=True) +SYS_tgkill = 234 if os.uname().machine == "x86_64" else 131 # aarch64: 131 + + +def tgkill(pid, tid, sig): + return libc.syscall(SYS_tgkill, pid, tid, sig) + + +class Srv: + def __init__(self, server, extra=()): + self.tmp = tempfile.mkdtemp(prefix="advlin.") + self.path = os.path.join(self.tmp, "sock") + self.proc = subprocess.Popen([server, "--unix", self.path, *extra], stderr=subprocess.PIPE) + for _ in range(200): + if os.path.exists(self.path): + break + time.sleep(0.02) + self.pid = self.proc.pid + + def alive(self): + return self.proc.poll() is None + + def stop(self): + if self.proc.poll() is None: + self.proc.send_signal(signal.SIGTERM) + try: + self.proc.wait(timeout=5) + except subprocess.TimeoutExpired: + self.proc.kill() + self.proc.wait() + err = self.proc.stderr.read().decode("utf-8", "replace") + try: + os.unlink(self.path) + except OSError: + pass + try: + os.rmdir(self.tmp) + except OSError: + pass + return err + + +def client(path, timeout=10): + c = Nine(path, timeout=timeout) + c.session() + return c + + +def rd(c, names, fid=50, offset=0, count=8192): + """walk+open+one read; returns (rtype-or-tag, data-or-error-string).""" + if c.walk_ok(0, fid, names) != len(names): + c.clunk(fid) + return "walkfail", None + rt, _, rb = c.open(fid, OREAD) + if rt != Ropen: + c.clunk(fid) + return "openfail", rb + rt, d = c.read(fid, offset, count) + c.clunk(fid) + if rt == Rerror: + n, = struct.unpack_from("/maps (read in 4 KiB pieces)", via == real, f"{len(via)} vs {len(real)}") + maps = real.decode() + for tag in ("[stack]", "[vdso]", "[heap]"): + m = [ln for ln in maps.splitlines() if ln.endswith(tag)] + if not m: + continue + lo = int(m[0].split("-")[0], 16) + 0x100 + ok(f"/addr of {tag} renders ? (never handed to std)", rd(c, [b"addr", b"%x" % lo])[1] == b"?\n?\n?\n") + ok(f"/hex of {tag} dumps", rd(c, [b"hex", b"%x" % lo])[0] == Rread) + m = [ln for ln in maps.splitlines() if ln.endswith("[stack]")][0] + hi = int(m.split()[0].split("-")[1], 16) + rt, d = rd(c, [b"mem", b"%x" % (hi - 16)], count=4096) + ok("read across the end of a mapping is a short read of 16 bytes", rt == Rread and len(d) == 16, (rt, d and len(d))) + rt, d = rd(c, [b"hex", b"%x" % (hi - 16)]) + ok("hexdump across the end of a mapping stops at the boundary", rt == Rread and d.count(b"\n") == 1, (rt, d)) + code = [ln for ln in maps.splitlines() if "r-xp" in ln and "introspect" in ln][0] + clo = int(code.split("-")[0], 16) + ok("write into read-only code is an error, not a fault", wr(c, [b"mem", b"%x" % (clo + 0x100)], b"\xcc") == ("err", "i/o error")) + ok("server alive after memory attacks", s.alive() and healthy(s.path)) + # a big write over the demo's own globals (state onwards) must not fault the server path + addr = rd(c, [b"vars", b"state", b"addr"])[1].decode().strip()[2:] + # (it zeroes the demo's own globals, this connection's state included, so + # the reply may never come; the server as a whole must keep working) + wr(c, [b"mem", addr.encode()], b"\x00" * 65536) + time.sleep(0.3) + ok("server alive and serving after a 64 KiB overwrite of its own globals", s.alive() and healthy(s.path)) + s.stop() + + +def attack_signals(server): + print("# signal machinery") + s = Srv(server) + c = client(s.path, timeout=8) + w = thread_by_name(c, s.pid, b"worker") + probe = thread_by_name(c, s.pid, b"introspect") + ok("worker and probe threads found by name", bool(w and probe), (w, probe)) + names = sorted(open(f"/proc/{s.pid}/task/{t}/comm").read().strip() for t in os.listdir(f"/proc/{s.pid}/task")) + ok("thread names are introspect, introspect, worker", names == ["introspect", "introspect", "worker"], names) + + # 4 clients hammer stacks of every thread while trap/continue interleave + errs, count = [], [0] + stop = threading.Event() + + def hammer(k): + try: + cc = client(s.path, timeout=8) + while not stop.is_set(): + for t in (w, probe, str(s.pid).encode()): + rt, d = rd(cc, [b"threads", t, b"stack"], fid=10 + k) + count[0] += 1 + if rt in ("walkfail", "openfail") or (rt == "err" and "i/o" not in d): + errs.append((t, rt, d)) + except Exception as e: # noqa: BLE001 + errs.append(repr(e)) + + ths = [threading.Thread(target=hammer, args=(k,)) for k in range(4)] + for t in ths: + t.start() + rounds_ok = True + for _ in range(5): + wr(c, [b"runtime", b"ctl"], b"trap") + time.sleep(0.15) + rounds_ok &= ls(c, [b"breakpoints"]) == [w] + rounds_ok &= b"workerLoop" in (rd(c, [b"breakpoints", w, b"stack"])[1] or b"") + rounds_ok &= rd(c, [b"threads", w, b"stack"])[0] == Rread # capture of a paused thread + rounds_ok &= wr(c, [b"breakpoints", w, b"ctl"], b"continue")[0] == Rwrite + rounds_ok &= wr(c, [b"breakpoints", w, b"ctl"], b"continue")[0] == "walkfail" # twice: gone + stop.set() + for t in ths: + t.join() + ok("trap/inspect/continue rounds while 4 clients capture stacks", rounds_ok) + ok(f"{count[0]} concurrent captures without a wrong answer", count[0] > 50 and not errs, errs[:3]) + + # SIGTRAP from outside (tgkill, not int3): parks without corrupting the thread + ok("tgkill SIGTRAP to the worker", tgkill(s.pid, int(w), signal.SIGTRAP) == 0) + time.sleep(0.3) + ok("worker listed under /breakpoints after tgkill", ls(c, [b"breakpoints"]) == [w]) + ok("its stack names workerLoop", b"workerLoop" in (rd(c, [b"breakpoints", w, b"stack"])[1] or b"")) + t1 = ticks(c) + time.sleep(0.3) + ok("ticks frozen while parked", ticks(c) == t1) + ok("continue after tgkill", wr(c, [b"breakpoints", w, b"ctl"], b"continue")[0] == Rwrite) + time.sleep(0.4) + ok("ticks advance after continue (no instruction skipped)", ticks(c) > t1) + + # SIGTRAP on the probe thread itself: stepped over, the server keeps serving + ok("tgkill SIGTRAP to the probe thread", tgkill(s.pid, int(probe), signal.SIGTRAP) == 0) + time.sleep(0.2) + ok("server serves after a SIGTRAP on its own thread", healthy(s.path)) + ok("probe thread not parked", ls(c, [b"breakpoints"]) == []) + # process-directed SIGTRAP lands on some thread; whichever it is, it is resumable + os.kill(s.pid, signal.SIGTRAP) + time.sleep(0.3) + ok("alive after kill -TRAP ", s.alive() and healthy(s.path)) + for t in ls(c, [b"breakpoints"]): + ok(f"thread {t.decode()} parked by kill -TRAP resumes", wr(c, [b"breakpoints", t, b"ctl"], b"continue")[0] == Rwrite) + ok("continue on a never-paused tid does not walk", wr(c, [b"breakpoints", w, b"ctl"], b"continue")[0] == "walkfail") + ok("a bogus tid does not walk", c.walk_ok(0, 31, [b"threads", b"999999"]) == 1) + + # panic: held, inspectable, capture of the held thread works, trap meanwhile harmless, continue aborts + wr(c, [b"runtime", b"ctl"], b"panic") + time.sleep(0.3) + ok("panic message published", rd(c, [b"panic", b"message"])[1] == b"demo panic requested over 9p") + ok("panic stack names workerLoop", b"workerLoop" in rd(c, [b"panic", b"stack"])[1]) + ok("capture of the held panicking thread answers", rd(c, [b"threads", w, b"stack"])[0] == Rread) + ok("trap request while a panic is held is harmless", wr(c, [b"runtime", b"ctl"], b"trap")[0] == Rwrite and s.alive()) + ok("panic continue", wr(c, [b"panic", b"ctl"], b"continue")[0] == Rwrite) + ok("second panic continue is an error", wr(c, [b"panic", b"ctl"], b"continue") == ("err", "file does not exist")) + time.sleep(1.0) + ok("process aborted after continue", not s.alive() and s.proc.poll() not in (0, None), s.proc.poll()) + s.stop() + + s = Srv(server, ["--no-hold"]) + c = client(s.path, timeout=5) + wr(c, [b"runtime", b"ctl"], b"panic") + time.sleep(1.0) + ok("--no-hold: panic aborts at once", not s.alive() and s.proc.poll() not in (0, None), s.proc.poll()) + s.stop() + + s = Srv(server) + c = client(s.path, timeout=5) + wr(c, [b"runtime", b"ctl"], b"trap") + time.sleep(0.3) + ok("worker parked", ls(c, [b"breakpoints"]) != []) + path = s.path + s.proc.send_signal(signal.SIGTERM) + try: + rc = s.proc.wait(timeout=5) + except subprocess.TimeoutExpired: + rc = None + ok("SIGTERM with a parked thread exits promptly", rc is not None, rc) + ok("SIGTERM unlinks the unix socket (clean stop path)", not os.path.exists(path)) + s.stop() + + +def attack_probe(player, server): + print("# probe loop and admission") + s = Srv(server) + c0 = cpu_ticks(s.pid) + time.sleep(5.0) + ok("0 CPU ticks over 5 s idle", cpu_ticks(s.pid) - c0 == 0, cpu_ticks(s.pid) - c0) + f0 = fds(s.pid) + for i in range(1000): + so = socket.socket(socket.AF_UNIX, socket.SOCK_STREAM) + so.connect(s.path) + if i % 3 == 0: + so.sendall(frame(Tversion, NOTAG, struct.pack(" (part of zig build introspect-adv) +# Exit 1 on any failure; SKIP (exit 0) without python3. +set -u +PLAYER=$(realpath "${1:?path to 9player}") +INTROSPECT=$(realpath "${2:?path to introspect}") +command -v python3 >/dev/null || { echo "SKIP: python3 missing"; exit 0; } +exec python3 "$(dirname "$0")/adv_linux_probe.py" --player "$PLAYER" --server "$INTROSPECT" diff --git a/introspect/test/adversarial.sh b/introspect/test/adversarial.sh new file mode 100755 index 0000000..918bb9a --- /dev/null +++ b/introspect/test/adversarial.sh @@ -0,0 +1,19 @@ +#!/usr/bin/env bash +# Runs every introspect adversarial suite in sequence: hostile raw-9P clients +# against the demo (framing, tags, floods; the core's /vars, snapshots, fids) +# and the Linux layer through one 9player mount (memory, signals, poll loop). +# Usage: bash introspect/test/adversarial.sh <9player> [--fast] +# `zig build introspect-adv` runs the same suites as separate steps. +set -u +PLAYER=${1:?path to 9player} +INTROSPECT=${2:?path to introspect} +shift 2 +HERE=$(cd "$(dirname "$0")" && pwd) +status=0 +for suite in adv_introspect_hostile adv_core_hostile; do + echo "### $suite" + if bash "$HERE/$suite.sh" "$INTROSPECT" "$@"; then echo "### $suite: ok"; else echo "### $suite: FAILED"; status=1; fi +done +echo "### adv_linux_probe" +if bash "$HERE/adv_linux_probe.sh" "$PLAYER" "$INTROSPECT"; then echo "### adv_linux_probe: ok"; else echo "### adv_linux_probe: FAILED"; status=1; fi +exit $status diff --git a/introspect/test/debug.sh b/introspect/test/debug.sh new file mode 100755 index 0000000..c3be406 --- /dev/null +++ b/introspect/test/debug.sh @@ -0,0 +1,84 @@ +#!/usr/bin/env bash +# End-to-end test of the introspect debug facilities through 9player: +# threads and stacks, address resolution, memory, exposed values, breakpoints, panics. +# Usage: bash introspect/test/debug.sh <9player> (zig build introspect-debug-itest) +set -u +PLAYER=$(realpath "${1:?path to 9player}") +INTROSPECT=$(realpath "${2:?path to introspect}") +TMP=$(mktemp -d /tmp/9pdbg.XXXXXX) +M=/mnt/9p +FAILED=0; PASSED=0 +SRV= + +cleanup() { [ -n "$SRV" ] && kill "$SRV" 2>/dev/null; rm -rf "$TMP"; } +trap cleanup EXIT +if ! unshare -Urm true 2>/dev/null || [ ! -c /dev/fuse ]; then echo "SKIP: namespaces or /dev/fuse unavailable"; exit 0; fi + +pass() { PASSED=$((PASSED + 1)); echo "ok - $1"; } +fail() { FAILED=$((FAILED + 1)); echo "FAIL - $1"; shift; [ $# -gt 0 ] && printf ' %s\n' "$@"; } +expect_eq() { if [ "$2" = "$3" ]; then pass "$1"; else fail "$1" "expected: $(printf %q "$2")" "actual: $(printf %q "$3")"; fi; } +expect_contains() { case "$3" in *"$2"*) pass "$1" ;; *) fail "$1" "missing: $(printf %q "$2")" "in: $(printf %q "$3")" ;; esac; } +run_in() { timeout 60 "$PLAYER" --unix "$SOCK" -- sh -c "$1" 2>"$TMP/stderr"; } + +SOCK=$TMP/dbg.sock +"$INTROSPECT" --unix "$SOCK" >"$TMP/server.log" 2>&1 & +SRV=$! +for _ in $(seq 1 100); do [ -S "$SOCK" ] && break; sleep 0.05; done +[ -S "$SOCK" ] || { echo "server did not start"; cat "$TMP/server.log"; exit 1; } + +echo "# threads" +expect_eq "threads listed" "yes" "$(run_in "[ \$(ls $M/threads | wc -l) -ge 2 ] && echo yes")" +WORKER=$(run_in "for t in $M/threads/*; do if grep -q '^worker' \$t/name 2>/dev/null; then basename \$t; fi; done | head -1") +expect_eq "worker thread found by name" "yes" "$([ -n "$WORKER" ] && echo yes)" +STACK=$(run_in "cat $M/threads/$WORKER/stack") +expect_contains "worker stack names workerLoop" "workerLoop" "$STACK" +expect_contains "worker stack has source locations" "demo/main.zig:" "$STACK" +expect_contains "worker regs" "0x" "$(run_in "cat $M/threads/$WORKER/regs | head -3")" +expect_eq "own (server) thread stack works" "yes" "$(run_in "for t in $M/threads/*; do cat \$t/stack >/dev/null 2>&1 || echo bad; done; echo yes")" + +echo "# addresses and memory" +FRAME=$(printf '%s\n' "$STACK" | grep -oE '0x[0-9a-f]+' | head -1) +expect_contains "addr resolves a stack frame to the demo source" "demo/main.zig" "$(run_in "cat $M/addr/${FRAME#0x}")" +expect_contains "addr of garbage is an error, not a crash" "No such file" "$(run_in "cat $M/addr/zzz 2>&1")" +expect_contains "mem/maps readable" "r-xp" "$(run_in "head -c 4000 $M/mem/maps")" +STATE_ADDR=$(run_in "cat $M/vars/state/addr") +expect_contains "hexdump of the exposed state" " " "$(run_in "head -2 $M/hex/${STATE_ADDR#0x}")" +expect_eq "raw bytes of the state match its size" "$(run_in "cat $M/vars/state/size")" "$(run_in "cat $M/mem/${STATE_ADDR#0x} | head -c \$(cat $M/vars/state/size) | wc -c")" +expect_eq "reading unmapped memory is an error, not a crash" "no" "$(run_in "cat $M/mem/8 >/dev/null 2>&1 && echo yes || echo no")" + +echo "# exposed values" +T1=$(run_in "cat $M/vars/state/f/ticks/value"); sleep 0.4; T2=$(run_in "cat $M/vars/state/f/ticks/value") +expect_eq "ticks is numeric" "num" "$(printf '%s' "$T1" | grep -Eq '^[0-9]+$' && echo num)" +expect_eq "ticks advance" "yes" "$([ "$T2" -gt "$T1" ] 2>/dev/null && echo yes)" +expect_contains "rendered struct value" "ticks" "$(run_in "cat $M/vars/state/value")" +expect_contains "type name" "State" "$(run_in "cat $M/vars/state/type")" +T3=$(run_in "echo 5 > $M/vars/state/f/ticks/value && cat $M/vars/state/f/ticks/value") +expect_eq "writing a scalar changes the live variable" "yes" "$([ "$T3" -lt "$T2" ] 2>/dev/null && echo yes)" + +echo "# breakpoints" +expect_eq "no breakpoints initially" "" "$(run_in "ls $M/breakpoints")" +run_in "echo trap > $M/runtime/ctl" >/dev/null; sleep 0.6 +PAUSED=$(run_in "ls $M/breakpoints | head -1") +expect_eq "worker paused at @breakpoint()" "$WORKER" "$PAUSED" +expect_contains "paused stack names workerLoop" "workerLoop" "$(run_in "cat $M/breakpoints/$WORKER/stack 2>&1")" +P1=$(run_in "cat $M/vars/state/f/ticks/value"); sleep 0.4; P2=$(run_in "cat $M/vars/state/f/ticks/value") +expect_eq "ticks frozen while paused" "$P1" "$P2" +run_in "echo continue > $M/breakpoints/$WORKER/ctl" >/dev/null; sleep 0.4 +expect_eq "breakpoint list empty after continue" "" "$(run_in "ls $M/breakpoints")" +P3=$(run_in "cat $M/vars/state/f/ticks/value") +expect_eq "ticks advance after continue" "yes" "$([ "$P3" -gt "$P2" ] 2>/dev/null && echo yes)" + +echo "# panic" +expect_eq "no panic recorded" "" "$(run_in "cat $M/panic/message")" +run_in "echo panic > $M/runtime/ctl" >/dev/null; sleep 0.6 +expect_contains "panic message published" "demo panic" "$(run_in "cat $M/panic/message")" +expect_contains "panic stack names the worker" "workerLoop" "$(run_in "cat $M/panic/stack")" +expect_eq "server still alive while holding the panic" "yes" "$(kill -0 $SRV 2>/dev/null && echo yes)" +run_in "echo continue > $M/panic/ctl" >/dev/null 2>&1 +for _ in $(seq 1 50); do kill -0 $SRV 2>/dev/null || break; sleep 0.1; done +if kill -0 $SRV 2>/dev/null; then fail "server exits after panic continue"; else wait $SRV; RC=$?; SRV=; expect_eq "server exit status is non-zero after the panic" "yes" "$([ $RC -ne 0 ] && echo yes)"; fi +expect_contains "default panic output reached stderr" "demo panic" "$(cat "$TMP/server.log")" + +echo +echo "passed=$PASSED failed=$FAILED" +[ "$FAILED" -eq 0 ] -- cgit v1.3