From ba996acfcad1698adbf4a1834fe50e73b1c6cab9 Mon Sep 17 00:00:00 2001 From: Gabriel Schneider Date: Sat, 19 Sep 2026 23:28:22 -0300 Subject: Rename programs: 9player -> 9ns, introspect -> 9proc, app -> web (9web) Directories, binaries, build options (-D9ns, -D9proc), step names, module name (9proc), thread and fs names, env var NINEPLAYER_MOUNT -> NINE_MOUNT, docs and test scripts. Browser assets move to web/static. Co-Authored-By: Claude Fable 5.1 --- 9ns/README.md | 156 ++ 9ns/build.zig | 71 + 9ns/docs/DESIGN.md | 405 +++++ 9ns/src/bridge.zig | 971 +++++++++++ 9ns/src/fuse.zig | 653 +++++++ 9ns/src/main.zig | 444 +++++ 9ns/src/nine.zig | 756 ++++++++ 9ns/src/ns.zig | 1082 ++++++++++++ 9ns/test/adv_bridge_hostile.py | 487 ++++++ 9ns/test/adv_bridge_hostile.sh | 181 ++ 9ns/test/adv_bridge_semantics.sh | 205 +++ 9ns/test/adv_bridge_stress.sh | 64 + 9ns/test/adv_ns_process.sh | 202 +++ 9ns/test/adversarial.sh | 15 + 9ns/test/integration.sh | 160 ++ 9player/README.md | 156 -- 9player/build.zig | 71 - 9player/docs/DESIGN.md | 405 ----- 9player/src/bridge.zig | 971 ----------- 9player/src/fuse.zig | 653 ------- 9player/src/main.zig | 444 ----- 9player/src/nine.zig | 756 -------- 9player/src/ns.zig | 1082 ------------ 9player/test/adv_bridge_hostile.py | 487 ------ 9player/test/adv_bridge_hostile.sh | 181 -- 9player/test/adv_bridge_semantics.sh | 205 --- 9player/test/adv_bridge_stress.sh | 64 - 9player/test/adv_ns_process.sh | 202 --- 9player/test/adversarial.sh | 15 - 9player/test/integration.sh | 160 -- 9proc/README.md | 130 ++ 9proc/build.zig | 169 ++ 9proc/demo/main.zig | 423 +++++ 9proc/docs/LIBRARY.md | 287 ++++ 9proc/src/core.zig | 2656 +++++++++++++++++++++++++++++ 9proc/src/freestanding_check.zig | 71 + 9proc/src/linux/debug.zig | 1458 ++++++++++++++++ 9proc/src/linux/probe.zig | 829 +++++++++ 9proc/src/linux/provider.zig | 604 +++++++ 9proc/src/linux/runtime.zig | 90 + 9proc/src/root.zig | 24 + 9proc/src/scratch.zig | 689 ++++++++ 9proc/src/vars.zig | 824 +++++++++ 9proc/test/adv_9proc_hostile.py | 999 +++++++++++ 9proc/test/adv_9proc_hostile.sh | 10 + 9proc/test/adv_core_hostile.py | 1018 +++++++++++ 9proc/test/adv_core_hostile.sh | 10 + 9proc/test/adv_linux_probe.py | 421 +++++ 9proc/test/adv_linux_probe.sh | 10 + 9proc/test/adversarial.sh | 19 + 9proc/test/debug.sh | 84 + README.md | 41 +- app/client.zig | 176 -- app/main.zig | 204 --- app/web/app.mjs | 183 -- app/web/client.mjs | 138 -- app/web/index.html | 53 - app/web/style.css | 87 - build.zig | 48 +- build.zig.zon | 2 +- docs/design.md | 9 +- docs/http.md | 8 +- docs/validation.md | 2 +- introspect/README.md | 130 -- introspect/build.zig | 169 -- introspect/demo/main.zig | 423 ----- introspect/docs/LIBRARY.md | 287 ---- introspect/src/core.zig | 2656 ----------------------------- introspect/src/freestanding_check.zig | 71 - introspect/src/linux/debug.zig | 1458 ---------------- introspect/src/linux/probe.zig | 829 --------- introspect/src/linux/provider.zig | 604 ------- introspect/src/linux/runtime.zig | 90 - introspect/src/root.zig | 24 - introspect/src/scratch.zig | 689 -------- introspect/src/vars.zig | 824 --------- introspect/test/adv_core_hostile.py | 1018 ----------- introspect/test/adv_core_hostile.sh | 10 - introspect/test/adv_introspect_hostile.py | 999 ----------- introspect/test/adv_introspect_hostile.sh | 10 - introspect/test/adv_linux_probe.py | 421 ----- introspect/test/adv_linux_probe.sh | 10 - introspect/test/adversarial.sh | 19 - introspect/test/debug.sh | 84 - test/web/e2e.mjs | 4 +- web/client.zig | 176 ++ web/main.zig | 204 +++ web/static/app.mjs | 183 ++ web/static/client.mjs | 138 ++ web/static/index.html | 53 + web/static/style.css | 87 + 91 files changed, 17578 insertions(+), 17572 deletions(-) create mode 100644 9ns/README.md create mode 100644 9ns/build.zig create mode 100644 9ns/docs/DESIGN.md create mode 100644 9ns/src/bridge.zig create mode 100644 9ns/src/fuse.zig create mode 100644 9ns/src/main.zig create mode 100644 9ns/src/nine.zig create mode 100644 9ns/src/ns.zig create mode 100755 9ns/test/adv_bridge_hostile.py create mode 100755 9ns/test/adv_bridge_hostile.sh create mode 100755 9ns/test/adv_bridge_semantics.sh create mode 100755 9ns/test/adv_bridge_stress.sh create mode 100755 9ns/test/adv_ns_process.sh create mode 100755 9ns/test/adversarial.sh create mode 100755 9ns/test/integration.sh delete mode 100644 9player/README.md delete mode 100644 9player/build.zig delete mode 100644 9player/docs/DESIGN.md delete mode 100644 9player/src/bridge.zig delete mode 100644 9player/src/fuse.zig delete mode 100644 9player/src/main.zig delete mode 100644 9player/src/nine.zig delete mode 100644 9player/src/ns.zig delete mode 100755 9player/test/adv_bridge_hostile.py delete mode 100755 9player/test/adv_bridge_hostile.sh delete mode 100755 9player/test/adv_bridge_semantics.sh delete mode 100755 9player/test/adv_bridge_stress.sh delete mode 100755 9player/test/adv_ns_process.sh delete mode 100755 9player/test/adversarial.sh delete mode 100755 9player/test/integration.sh create mode 100644 9proc/README.md create mode 100644 9proc/build.zig create mode 100644 9proc/demo/main.zig create mode 100644 9proc/docs/LIBRARY.md create mode 100644 9proc/src/core.zig create mode 100644 9proc/src/freestanding_check.zig create mode 100644 9proc/src/linux/debug.zig create mode 100644 9proc/src/linux/probe.zig create mode 100644 9proc/src/linux/provider.zig create mode 100644 9proc/src/linux/runtime.zig create mode 100644 9proc/src/root.zig create mode 100644 9proc/src/scratch.zig create mode 100644 9proc/src/vars.zig create mode 100755 9proc/test/adv_9proc_hostile.py create mode 100755 9proc/test/adv_9proc_hostile.sh create mode 100755 9proc/test/adv_core_hostile.py create mode 100755 9proc/test/adv_core_hostile.sh create mode 100755 9proc/test/adv_linux_probe.py create mode 100755 9proc/test/adv_linux_probe.sh create mode 100755 9proc/test/adversarial.sh create mode 100755 9proc/test/debug.sh delete mode 100644 app/client.zig delete mode 100644 app/main.zig delete mode 100644 app/web/app.mjs delete mode 100644 app/web/client.mjs delete mode 100644 app/web/index.html delete mode 100644 app/web/style.css delete mode 100644 introspect/README.md delete mode 100644 introspect/build.zig delete mode 100644 introspect/demo/main.zig delete mode 100644 introspect/docs/LIBRARY.md delete mode 100644 introspect/src/core.zig delete mode 100644 introspect/src/freestanding_check.zig delete mode 100644 introspect/src/linux/debug.zig delete mode 100644 introspect/src/linux/probe.zig delete mode 100644 introspect/src/linux/provider.zig delete mode 100644 introspect/src/linux/runtime.zig delete mode 100644 introspect/src/root.zig delete mode 100644 introspect/src/scratch.zig delete mode 100644 introspect/src/vars.zig delete mode 100755 introspect/test/adv_core_hostile.py delete mode 100755 introspect/test/adv_core_hostile.sh delete mode 100755 introspect/test/adv_introspect_hostile.py delete mode 100755 introspect/test/adv_introspect_hostile.sh delete mode 100755 introspect/test/adv_linux_probe.py delete mode 100755 introspect/test/adv_linux_probe.sh delete mode 100755 introspect/test/adversarial.sh delete mode 100755 introspect/test/debug.sh create mode 100644 web/client.zig create mode 100644 web/main.zig create mode 100644 web/static/app.mjs create mode 100644 web/static/client.mjs create mode 100644 web/static/index.html create mode 100644 web/static/style.css diff --git a/9ns/README.md b/9ns/README.md new file mode 100644 index 0000000..04edffb --- /dev/null +++ b/9ns/README.md @@ -0,0 +1,156 @@ +# 9ns + +Mount a 9P2000 file tree into a fresh mount namespace and run a program in it, +as a plain user, without touching the host's mount table. + +```sh +9ns --unix /run/user/1000/acme -- fish # a shell that sees the tree at /mnt/9p +9ns --tcp 127.0.0.1:564 -- claude # an agent that sees it too +9ns --spawn '9proc-demo --stdio' -- bash # start the server yourself, talk over a socketpair +``` + +Inside, the tree is ordinary files: `ls`, `cat`, `echo x > ctl`, editors, +`find`, `rsync`, whatever. `$NINE_MOUNT` tells programs where it is +(default `/mnt/9p`). When the program exits, 9ns exits with its status +and the namespace, mount and connection disappear. + +## How it works + +The kernel's own `9p` filesystem cannot be mounted inside an unprivileged user +namespace, so 9ns is a small FUSE server that speaks 9P2000 to the real +server. There is no libfuse and no libc: `src/fuse.zig` implements the subset +of the kernel FUSE protocol needed, straight from `linux/fuse.h`. + +``` + program (fish/bash/claude) 9ns (parent) 9P server + in a new user+mount namespace │ + /mnt/9p ── FUSE ──▶ kernel ────▶│ fuse.zig ─▶ bridge.zig ─▶ nine.zig ──▶ unix / tcp / socketpair + │ (framing) (translation) (cloud9 Client) +``` + +1. The parent connects to the 9P server (version + attach) so failures are + reported before anything is forked. +2. The child does `unshare(CLONE_NEWUSER|CLONE_NEWNS)`, maps its own uid/gid, + makes every mount private, opens `/dev/fuse` (it must be opened inside the + new user namespace), mounts it on the mountpoint and passes the descriptor + back to the parent over `SCM_RIGHTS`, then execs the program. +3. The parent serves FUSE requests by translating them into 9P transactions + (`Twalk`, `Topen`, `Tread`, `Twrite`, `Tcreate`, `Tremove`, `Tstat`, + `Twstat`) until the child exits or the namespace disappears. + +Files are opened with `FOPEN_DIRECT_IO`, so synthetic files that report length +0 (the 9P convention for control files) still read correctly, and `O_TRUNC` +travels inside the 9P open mode (`OTRUNC`) rather than as a separate +truncate. Repeated lookups of the same qid map to the same inode. + +If `/mnt/9p` does not exist and cannot be created (the normal case), 9ns +mounts a tmpfs over `/mnt` *inside the namespace only* and bind-mounts every +existing entry of `/mnt` back into it, so nothing is hidden. Pass `--mount DIR` +to use any other directory. + +## Building and testing + +9ns lives in the [cloud9](../) repository as `cloud9/9ns/`, beside +the 9P2000 protocol library it is built on, and is wired into cloud9's +`build.zig` through the fragment `9ns/build.zig`. Everything is run from +the cloud9 root with Zig 0.16: + +```sh +zig build # zig-out/bin/{9ns,9proc-demo,9web,cloud9-probe} +zig build 9ns # build and install only zig-out/bin/9ns +zig build 9ns-test # unit tests (protocol structs, session, bridge, namespace helpers) +zig build 9ns-itest # integration tests: real namespaces, real FUSE, + # 9proc-demo over unix/tcp/socketpair, and plan9port's + # ramfs when /usr/lib/plan9/bin/ramfs is installed +zig build 9ns-adv # adversarial suites: hostile 9P servers, FUSE semantics, + # namespace/signal edge cases, stress (several minutes) +zig build -Doptimize=ReleaseSafe +zig build -D9ns=false # leave 9ns out (the default on non-Linux targets) +``` + +The integration suites mount the `9proc-demo` server (`../9proc`), +so they need `-D9proc=true` (the default on Linux), unprivileged user +namespaces (`kernel.unprivileged_userns_clone=1` on distributions that have +the knob), `/dev/fuse` and Python 3; they skip themselves otherwise. +`zig build programs-test` and `programs-itest` run the unit and integration +steps of every program in the repository. + +## Usage + +``` +9ns [options] -- PROGRAM [ARGS...] + +Transport (exactly one): + --unix PATH Unix stream socket + --tcp IP:PORT TCP (IPv4/IPv6 literal) + --fd N an already-connected inherited descriptor + --spawn CMD run CMD via /bin/sh -c with a socketpair on its stdin/stdout + +Options: + --mount PATH mountpoint inside the new namespace (default /mnt/9p) + --uname NAME 9P user name (default $USER) + --aname NAME 9P tree to attach (default "") + --msize BYTES maximum 9P message size to request (default 131072, max 16 MiB) + --cache SECONDS attr/entry cache validity, fractional allowed (default 1) + --no-direct-io let the kernel cache file pages (trusts stat length) + --debug trace FUSE and 9P operations on stderr + --help, --version +``` + +PROGRAM defaults to `$SHELL`. Exit status is the program's (`128+signal` if it +was killed); 125 means 9ns itself failed (usage, connect, namespace, +mount); 126/127 are exec failures as usual. + +## 9proc-demo: a demo 9P server + +`9proc-demo` is a single-binary 9P2000 server whose file tree is the binary +itself: build-time facts, `comptime` reflection and live runtime state. It is +the demo of the [9proc library](../9proc) (`../9proc/demo/main.zig`), +built and installed by `zig build 9proc`. + +``` +/README +/build/{zig_version,target,optimize,time,change} captured by build.zig (jj change id, UTC time) +/comptime/types//{name,size,align,fields} @sizeOf/@alignOf/@typeInfo, generated at comptime +/comptime/decls pub declarations of the server module +/runtime/{pid,ppid,uptime,argv,cwd,env,clients} +/runtime/fn/ reading calls a Zig function (hostname, now, random, uname, fib30); + the directory is generated from @typeInfo of the Fns struct +/runtime/ctl write "fib N" | "add A B" | "echo TEXT" | "sleep-ms N", read the result +/scratch/ in-memory read/write tree +``` + +`/runtime/env` exposes the server's whole environment, so serve 9proc-demo on +a Unix socket or loopback only. + +```sh +zig-out/bin/9proc-demo --unix /tmp/intro.sock & +zig-out/bin/9ns --unix /tmp/intro.sock -- sh -c ' + cat $NINE_MOUNT/build/zig_version; echo + cat $NINE_MOUNT/comptime/types/Qid/fields + echo "fib 20" > $NINE_MOUNT/runtime/ctl; cat $NINE_MOUNT/runtime/ctl' +``` + +The same command with `claude -p "explore /mnt/9p ..."` as the program gives an +agent a live, file-shaped view into a running process; that is the intended +use. + +## Limitations + +* One 9P request is in flight at a time; a server read that blocks (event + files) stalls the mount until it returns. +* Base 9P2000 only: no symlinks, ownership, xattrs or locks. Every file is + reported as owned by the invoking user. Cross-directory rename is `EXDEV`. +* No PID namespace and no `/proc` remount. `--tcp` takes IP literals only + (no libc, no resolver). +* Linux only. + +## Relation to cloud9 + +9ns consumes cloud9 as the module `cloud9` and keeps all mounting, +namespace and process policy on its side, which is what cloud9's design asks +of applications. It ships from the cloud9 repository as the sibling directory +`9ns/` (sources in `src/`, suites in `test/`, this README and +`docs/DESIGN.md`) with a build fragment that the root `build.zig` enables with +`-D9ns` on Linux targets; nothing in the code depends on that layout. +`docs/DESIGN.md` has the full module contracts. diff --git a/9ns/build.zig b/9ns/build.zig new file mode 100644 index 0000000..7f2155f --- /dev/null +++ b/9ns/build.zig @@ -0,0 +1,71 @@ +//! Build fragment for 9ns: mount a 9P2000 tree into a fresh mount +//! namespace via FUSE and run a program in it (Linux only, no libc). It is +//! `@import`ed by the root build.zig and called with the root builder, so +//! every `b.path(...)` here is relative to the cloud9 root (hence the +//! `9ns/` prefix), every option is defined by the root (no +//! `standardTargetOptions` here) and every step it registers lands in the +//! root's step list under the `9ns` prefix. +//! +//! Steps: 9ns, 9ns-test, 9ns-itest, 9ns-adv. +const std = @import("std"); + +/// What the root passes in. The root owns target/optimize resolution and the +/// cloud9 module; the integration suites also need a 9P server to mount, +/// which is the 9proc demo built by the sibling fragment. +pub const Context = struct { + target: std.Build.ResolvedTarget, + optimize: std.builtin.OptimizeMode, + cloud9: *std.Build.Module, + /// The `9proc-demo` server; null when 9proc is disabled, in + /// which case the end-to-end steps exist but fail with a notice. + proc_demo: ?*std.Build.Step.Compile, +}; + +pub const Artifacts = struct { + /// The 9ns executable (also installed by the plain `zig build`). + exe: *std.Build.Step.Compile, + /// `9ns-test`, `9ns-itest`, `9ns-adv`. + test_step: *std.Build.Step, + itest_step: *std.Build.Step, + adv_step: *std.Build.Step, +}; + +pub fn add(b: *std.Build, ctx: Context) Artifacts { + const target = ctx.target; + const optimize = ctx.optimize; + std.debug.assert(target.result.os.tag == .linux); // the root only enables 9ns on Linux + + const ns_mod = b.createModule(.{ + .root_source_file = b.path("9ns/src/main.zig"), + .target = target, + .optimize = optimize, + .imports = &.{.{ .name = "cloud9", .module = ctx.cloud9 }}, + }); + const ns = b.addExecutable(.{ .name = "9ns", .root_module = ns_mod }); + const install = b.addInstallArtifact(ns, .{}); + b.getInstallStep().dependOn(&install.step); + b.step("9ns", "Build and install only the 9ns binary").dependOn(&install.step); + + // Unit tests: protocol structs, session, bridge, namespace helpers (main.zig + // reaches every module). + const test_step = b.step("9ns-test", "Run 9ns's unit tests"); + test_step.dependOn(&b.addRunArtifact(b.addTest(.{ .root_module = ns_mod })).step); + + // End-to-end suites: real namespaces, real FUSE, a real 9P server. + const itest = b.step("9ns-itest", "Run 9ns/test/integration.sh (needs unprivileged user namespaces and /dev/fuse)"); + const adv = b.step("9ns-adv", "Run 9ns/test/adversarial.sh (hostile servers, namespaces, stress; several minutes)"); + const demo = ctx.proc_demo orelse { + const fail = b.addFail("9ns-itest and 9ns-adv need the 9proc-demo server (build with -D9proc=true)"); + itest.dependOn(&fail.step); + adv.dependOn(&fail.step); + return .{ .exe = ns, .test_step = test_step, .itest_step = itest, .adv_step = adv }; + }; + inline for (.{ .{ itest, "integration" }, .{ adv, "adversarial" } }) |pair| { + const run = b.addSystemCommand(&.{"bash"}); + run.addFileArg(b.path("9ns/test/" ++ pair[1] ++ ".sh")); + run.addArtifactArg(ns); + run.addArtifactArg(demo); + pair[0].dependOn(&run.step); + } + return .{ .exe = ns, .test_step = test_step, .itest_step = itest, .adv_step = adv }; +} diff --git a/9ns/docs/DESIGN.md b/9ns/docs/DESIGN.md new file mode 100644 index 0000000..7f943a8 --- /dev/null +++ b/9ns/docs/DESIGN.md @@ -0,0 +1,405 @@ +# 9ns design + +`9ns` mounts a 9P2000 file tree served over a Unix or TCP stream socket +into a **fresh mount namespace** and runs a program inside it. The program +(fish, bash, `claude`, anything) sees the 9P tree as ordinary files, without +root and without touching the host's mount table. + +## Why FUSE + +The kernel's own `9p` filesystem is not mountable inside an unprivileged user +namespace (it lacks `FS_USERNS_MOUNT`) and loading it needs root. FUSE has been +user-namespace mountable since Linux 4.18, and `/dev/fuse` is world read/write. +So 9ns is a tiny FUSE server that speaks 9P2000 to the real server: + +``` + program (fish/bash/claude) 9ns (parent) 9P server + in new user+mount namespace │ (9proc-demo, + /mnt/9p ─── FUSE ───▶ kernel ──▶│ fuse.zig ──▶ bridge.zig ──▶ nine.zig ──▶ ramfs, ...) + │ (framing) (translation) (cloud9 Client) +``` + +No libfuse: `src/fuse.zig` implements the small subset of the kernel FUSE +protocol we need directly against `/usr/include/linux/fuse.h`. + +## Toolchain facts (Zig 0.16) + +* Zig 0.16.0 at `/usr/bin/zig`, std at `/usr/lib/zig/std`. **Grep the std + tree before assuming an API exists**; 0.16 moved a lot of process/fs code + behind `std.Io`. Raw Linux syscalls in `std.os.linux` (`fork`, `execve`, + `mount`, `unshare`, `waitpid`, `pipe2`, `socketpair`, `poll`, `read`, `write`, + `open`, `openat`, `getdents64`, `sigaction`, `kill`, `readlinkat`, `mkdirat`, + `symlinkat`) are the intended low-level path. They return `usize`; decode + with `std.os.linux.errno(rc)` (an `E` enum, `.SUCCESS` when ok). +* `std.posix.poll`, `std.posix.sigaction`, `std.posix.read`, `std.posix.kill` + exist. `std.posix.fork/execve/waitpid/pipe2/socketpair` do **not**. +* `pub fn main() !void` and `pub fn main(init: std.process.Init) !void` are + both supported. Prefer `main(init: std.process.Init)`; `init.gpa` is a + general purpose allocator, `init.arena` an arena, `init.minimal.args` the + argv (`toSlice(allocator)`), `init.minimal.environ.block` the envp block. +* No libc is linked. Do not use `std.c.*`. Hostname lookups are therefore out + of scope: `--tcp` takes IP literals only. +* 9ns lives in the cloud9 repository as `cloud9/9ns/` and is built by + the root `build.zig` through the fragment `9ns/build.zig` (steps + `9ns`, `9ns-test`, `9ns-itest`, `9ns-adv`; toggle + `-D9ns`). cloud9 itself is imported as module `cloud9` + (`@import("cloud9")`). Read `../src/client.zig`, `Server.zig`, `wire.zig` + and `../docs/design.md`. Its core is allocation-free and caller-driven: you + push bytes in, take results out. The demo 9P server the tests mount is the + sibling program `../9proc` (`zig build 9proc`). +* Standalone module tests while other files are missing (from the cloud9 + root): + `zig test --dep cloud9 -Mroot=9ns/src/.zig -Mcloud9=src/root.zig`. +* Format everything with `zig fmt`. + +## Process model + +``` +9ns [options] -- PROGRAM [ARGS...] +``` + +1. Parent parses args, probes that `/dev/fuse` exists, connects to the 9P + server, negotiates `version` and `attach`es (fid 0 = root). Connection + failures are reported before anything is forked. +2. Parent forks with a `socketpair` status channel. **Child**: + 1. `unshare(CLONE_NEWUSER | CLONE_NEWNS)`. + 2. Writes `/proc/self/setgroups` = `deny`, `/proc/self/uid_map` = + `" 1"`, `/proc/self/gid_map` = `" 1"` (same ids + inside as outside; the child creating the namespace holds full + capabilities in it until exec). + 3. `mount(NULL, "/", NULL, MS_REC|MS_PRIVATE, NULL)` so nothing propagates. + 4. Ensures the mountpoint exists (see below). + 5. Opens `/dev/fuse` (`O_RDWR|O_CLOEXEC`). The kernel refuses to mount a + fuse descriptor opened from a different user namespace than the mount + ("wrong user namespace for fuse device"), so this must happen here, not + in the parent. + 6. `mount("9ns", mountpoint, "fuse", MS_NOSUID|MS_NODEV, + "fd=,rootmode=40000,user_id=,group_id=,max_read=")`. + 7. Sends the fuse fd to the parent over the status socket (`SCM_RIGHTS`). + 8. `statx` of the mountpoint: this forces one GETATTR, which the parent + serves. Without it the kernel keeps the root inode's initial uid 0 + (unmapped in the namespace) and every create in the root gets `EACCES`. + 9. Sets `NINE_MOUNT=` in the environment. + 10. `execve` of PROGRAM with PATH search (implemented by hand; no libc). + Exec failures are reported through the `CLOEXEC` status socket + (errno + message); the parent prints them after the serve loop ends. +3. **Parent** receives the fuse fd, then runs the FUSE loop (`bridge.serve`) + until either the child exits (SIGCHLD via self-pipe) or the FUSE fd reports + `ENODEV` (last process in the namespace gone, mount destroyed). It then + closes the fuse fd and exits with the child's status (`128+sig` if + signalled). The self-pipe is also watched by the 9P session while a reply + is outstanding (`Session.stop_fd` → `error.Stopped`), so a server that + never answers cannot keep 9ns alive after the child is gone; a 3 s + watchdog armed from the SIGCHLD handler is the last resort. +4. Signals in the parent: `SIGINT`/`SIGQUIT` ignored (the child owns the tty + and gets them itself); `SIGTERM`/`SIGHUP` forwarded to the child; + `SIGPIPE` ignored; `SIGCHLD` → self-pipe. + +The FUSE fd is shared with the child only until exec (CLOEXEC); the parent's +copy keeps the connection alive. + +### Mountpoint policy + +Default mountpoint: `/mnt/9p`. A relative `--mount` is resolved against cwd. + +* If the path is a directory: use it. +* Else try `mkdir`. If that fails with `EACCES`/`EPERM`/`EROFS` (the normal + case for `/mnt/9p` as a plain user), **shadow the parent directory**: + open an fd to the parent, mount a `tmpfs` over it, then recreate every + existing entry inside the tmpfs: directories → `mkdir` + bind mount from + `/proc/self/fd//`; symlinks → `readlinkat` + `symlink`; anything + else → empty regular file + bind mount. Then `mkdir` the target inside. + Refuse (with a clear message) if the parent has more than 4096 entries or + is `/`. This only affects the new namespace. +* Else fail with the errno and a hint to pass `--mount` an existing dir. + +## Module contracts + +### `src/fuse.zig` — kernel FUSE protocol (no policy) + +Extern structs mirroring `linux/fuse.h`, with `comptime` size asserts: +`InHeader` (40), `OutHeader` (16), `Attr` (88), `EntryOut` (128), +`AttrOut` (104), `GetattrIn` (16), `SetattrIn` (88), `OpenIn` (8), +`OpenOut` (16), `ReleaseIn` (24), `FlushIn` (24), `ReadIn` (40), +`WriteIn` (40), `WriteOut` (8), `CreateIn` (16), `MkdirIn` (8), +`RenameIn` (8), `Rename2In` (16), `ForgetIn` (8), `BatchForgetIn` (8), +`ForgetOne` (16), `FsyncIn` (16), `AccessIn` (8), `InterruptIn` (8), +`Kstatfs` (80), `StatfsOut` (80), `InitIn` (64), `InitOut` (64), +`Dirent` (24 header, name padded to 8), `LseekIn` (24). + +`pub const Opcode = enum(u32) { lookup = 1, forget = 2, getattr = 3, setattr = 4, +readlink = 5, symlink = 6, mknod = 8, mkdir = 9, unlink = 10, rmdir = 11, +rename = 12, link = 13, open = 14, read = 15, write = 16, statfs = 17, +release = 18, fsync = 20, setxattr = 21, getxattr = 22, listxattr = 23, +removexattr = 24, flush = 25, init = 26, opendir = 27, readdir = 28, +releasedir = 29, fsyncdir = 30, getlk = 31, setlk = 32, setlkw = 33, +access = 34, create = 35, interrupt = 36, bmap = 37, destroy = 38, +ioctl = 39, poll = 40, notify_reply = 41, batch_forget = 42, fallocate = 43, +readdirplus = 44, rename2 = 45, lseek = 46, copy_file_range = 47, +setupmapping = 48, removemapping = 49, syncfs = 50, tmpfile = 51, statx = 52, _ }` + +Constants: `kernel_version = 7`, `kernel_minor = 31` (what we answer; the +kernel adapts to the lower minor), `FOPEN_DIRECT_IO = 1`, `FOPEN_KEEP_CACHE = 2`, +`FOPEN_NONSEEKABLE = 4`, `FUSE_ASYNC_READ = 1`, `FUSE_MAX_PAGES = 1<<22`, +`FATTR_MODE=1, FATTR_UID=2, FATTR_GID=4, FATTR_SIZE=8, FATTR_ATIME=16, +FATTR_MTIME=32, FATTR_FH=64, FATTR_ATIME_NOW=128, FATTR_MTIME_NOW=256, +FATTR_LOCKOWNER=512, FATTR_CTIME=1024`. `root_id = 1`. + +I/O helpers (blocking fd, no allocation beyond the caller's buffer): + +```zig +pub const Request = struct { header: InHeader, body: []const u8 }; +/// One kernel request. Returns null on ENODEV (unmounted). Retries EINTR/EAGAIN/ENOENT. +pub fn readRequest(fd: i32, buf: []u8) !?Request; +/// Success reply: header + concatenated payload slices, single writev. +pub fn reply(fd: i32, unique: u64, payloads: []const []const u8) !void; +/// Error reply: negative errno. +pub fn replyError(fd: i32, unique: u64, err: std.os.linux.E) !void; +/// Append a fuse_dirent (8-byte padded) to `buf`; returns false if it doesn't fit. +pub fn addDirent(buf: []u8, used: *usize, ino: u64, off: u64, dtype: u32, name: []const u8) bool; +pub fn body(comptime T: type, req: Request) !*const T; // aligned copy-free view, checks size +pub fn nameAfter(comptime T: type, req: Request) ![]const u8; // NUL-terminated name after a struct +``` + +The request buffer must be ≥ `max_write + 4096`; 9ns uses 1 MiB + 4 KiB. +Requests with an unknown/unsupported opcode get `ENOSYS`. + +### `src/nine.zig` — synchronous 9P2000 session on a blocking fd + +Thin, synchronous RPC layer over `cloud9.Client` (which is push/take, +non-blocking-agnostic). One outstanding request at a time (the FUSE loop is +single-threaded). Fids are allocated from a free list. + +```zig +pub const Address = union(enum) { unix: []const u8, tcp: struct { host: []const u8, port: u16 }, fd: i32 }; +pub const Session = struct { + pub const Error = error{ Nine, Protocol, Io, Closed, TooLarge, OutOfMemory }; + /// After error.Nine, `ename` holds the server's Rerror text (copied, bounded). + ename: [256]u8, ename_len: usize, + msize: u32, + + pub fn connect(gpa: std.mem.Allocator, address: Address, msize: u32) !Session; // socket+connect, version + pub fn deinit(s: *Session) void; + pub fn attach(s: *Session, fid: u32, uname: []const u8, aname: []const u8) Error!cloud9.Qid; + pub fn allocFid(s: *Session) u32; + pub fn freeFid(s: *Session, fid: u32) void; + /// Generic RPC. Result slices borrow the input buffer until the next call. + pub fn rpc(s: *Session, req: cloud9.Client.Request) Error!cloud9.Client.Result; + // Conveniences (all built on rpc): + pub fn walk(s, fid: u32, newfid: u32, names: []const []const u8) Error!Walk; // Walk = { nwqid, wqid[16] }; partial walk → error.Nine with ename "not found"-ish + pub fn clone(s, fid: u32) Error!u32; // allocFid + walk with no names + pub fn open(s, fid: u32, mode: u8) Error!Open; // Open = { qid, iounit } + pub fn create(s, fid: u32, name: []const u8, perm: u32, mode: u8) Error!Open; + pub fn read(s, fid: u32, offset: u64, buf: []u8) Error!usize; // chunks by maxRead/iounit; stops at short read + pub fn write(s, fid: u32, offset: u64, data: []const u8) Error!usize; // chunks; stops at short write + pub fn stat(s, fid: u32) Error!cloud9.Stat; // strings borrow the input buffer + pub fn wstat(s, fid: u32, st: cloud9.Stat) Error!void; + pub fn clunk(s, fid: u32) Error!void; // frees the fid even on error + pub fn remove(s, fid: u32) Error!void; // frees the fid even on error + pub fn errno(s: *const Session) std.os.linux.E; // map ename → errno (see below) +}; +pub const dontcare = cloud9.Stat{ .type = 0xFFFF, .dev = 0xFFFF_FFFF, .qid = .{ .type = 0xFF, .version = 0xFFFF_FFFF, .path = 0xFFFF_FFFF_FFFF_FFFF }, .mode = 0xFFFF_FFFF, .atime = 0xFFFF_FFFF, .mtime = 0xFFFF_FFFF, .length = 0xFFFF_FFFF_FFFF_FFFF, .name = "", .uid = "", .gid = "", .muid = "" }; +``` + +`connect`: for `.unix` and `.tcp` create a blocking `SOCK_STREAM|SOCK_CLOEXEC` +socket and connect (`TCP_NODELAY` on TCP); for `.fd` adopt it. Buffers of +`msize` bytes for in/out are heap allocated. Then submit `.version`, drain +output to the socket, read until `take()` yields the version result. The +negotiated msize is `result.version.msize`; if the server answered +`"unknown"`, fail with `error.Protocol`. + +`rpc`: submit, write all of `client.output()` (calling `wrote`), then loop: +`take()`; if null, `read` from the fd into a temp buffer and `push` (push +returns how much fit; the frame is at most msize so it always fits after a +`take`). If the fd returns 0 → `error.Closed`. If the client dies → +`error.Protocol`. A `.fail` result copies the ename and returns `error.Nine`. + +Rerror text → errno mapping (case-insensitive substring, in this order): +`"not exist"`, `"not found"`, `"no such"` → `ENOENT`; `"exists"` → `EEXIST`; +`"not empty"` → `ENOTEMPTY`; `"not a dir"` → `ENOTDIR`; +`"is a dir"` → `EISDIR`; `"permission"`, `"denied"` → `EACCES`; +`"read-only"`, `"read only"`, `"readonly"` → `EROFS`; `"no space"` → `ENOSPC`; +`"not allowed"`, `"not permitted"`, `"cannot"` → `EPERM`; +`"fid"` → `EBADF`; `"bad offset"`, `"invalid"`, `"bad "` → `EINVAL`; +`"busy"`, `"in use"` → `EBUSY`; `"too long"` → `ENAMETOOLONG`; +`"not supported"`, `"unsupported"` → `ENOTSUP`; otherwise `EIO`. + +### `src/bridge.zig` — FUSE ↔ 9P translation + +```zig +pub const Options = struct { + uid: u32, gid: u32, // reported owner of every file + attr_timeout_ns: u64 = 1e9, // attr/entry cache validity (0 = none) + direct_io: bool = true, // FOPEN_DIRECT_IO on every regular file + debug: bool = false, // trace to stderr +}; +/// Runs until the FUSE fd reports ENODEV or `stop_fd` becomes readable. +pub fn serve(gpa: std.mem.Allocator, fuse_fd: i32, nine: *nine.Session, root_fid: u32, stop_fd: i32, opts: Options) !void; +``` + +State: + +* `inodes: AutoHashMap(u64 /*nodeid*/, Inode{ fid: u32, qid: Qid, nlookup: u64 })`. + Node 1 is the root (`root_fid`, never forgotten). +* `by_qid: AutoHashMap(u64 /*qid.path*/, u64 /*nodeid*/)` so that repeated + lookups of the same file map to the same inode (the old fid is clunked and + the fresh one kept). Dedupe only merges when the qid type (dir bit) also + matches, so a server reusing a path across a file and a directory cannot + poison an inode. `ino` in attrs is `qid.path` (root, or anything carrying + the root's path: 1). +* `handles: AutoHashMap(u64 /*fh*/, Handle{ fid: u32, dir: ?DirList })`. + `DirList` is the entire directory read at first `READDIR` offset 0: + `[]Entry{ name: []u8, ino: u64, dtype: u32 }` with synthetic `.` and `..` + first. `READDIR` offsets are indices into that list; a `READDIR` at offset 0 + re-reads the directory (rewinddir). + +Op mapping (9P2000 has no symlinks, links, xattrs, locks, mknod): + +| FUSE | 9P | +|---|---| +| INIT | reply `InitOut{ major=7, minor=31, max_readahead=in.max_readahead, flags = FUSE_ASYNC_READ \| FUSE_ATOMIC_O_TRUNC \| FUSE_AUTO_INVAL_DATA \| FUSE_BIG_WRITES (plus FUSE_MAX_PAGES with max_pages=256 if offered), max_background=16, congestion_threshold=12, max_write=1 MiB, time_gran=1 }`. Atomic O_TRUNC matters: without it the kernel truncates via a separate SETATTR(size=0) that synthetic control files reject; with it `O_TRUNC` becomes 9P `OTRUNC` inside the open | +| LOOKUP(parent,name) | `walk(parent.fid → newfid, [name])`; `stat(newfid)`; dedupe by qid; `EntryOut` | +| FORGET / BATCH_FORGET | `nlookup -= n`; at 0 `clunk` and drop (no reply) | +| GETATTR | `stat(inode.fid)` → `AttrOut` | +| SETATTR | `stat` then `wstat` with a *dontcare* Stat: `FATTR_SIZE`→length; `FATTR_MODE`→`(old.mode & ~0o777) \| (mode & 0o777)`; `FATTR_MTIME`→mtime (`FATTR_MTIME_NOW` → now); `FATTR_ATIME` ignored; `FATTR_UID/GID` → `EPERM` unless unchanged; then `stat` again for the reply | +| OPEN | `clone(inode.fid)` then `open(newfid, mode)`; mode from `O_ACCMODE` (`oread/owrite/ordwr`), `O_TRUNC` → `otrunc`; reply `OpenOut{ fh, open_flags = FOPEN_DIRECT_IO }`; on failure clunk | +| OPENDIR | same with `oread`; `fh` with `dir = null` | +| READ | `read(fh.fid, offset, buf[0..min(size, 1 MiB)])`; reply data | +| WRITE | `write(fh.fid, offset, data)`; `WriteOut{ size = n }` | +| READDIR | fill `Dirent`s from the `DirList` starting at `offset`, up to `size` bytes | +| RELEASE / RELEASEDIR | `clunk(fh.fid)`; free DirList | +| FLUSH / FSYNC / FSYNCDIR | ok (no-op) | +| CREATE(parent,name,flags,mode) | `clone(parent)`; `create(fid, name, mode & 0o777, openmode)` → this fid is the **open** file; then `walk(parent → fid2, [name])` + `stat(fid2)` for the inode; reply `EntryOut ++ OpenOut` | +| MKDIR | `clone(parent)`; `create(fid, name, DMDIR \| (mode & 0o777), oread)`; `clunk`; then lookup as above | +| UNLINK / RMDIR | `walk(parent → tmp, [name])`; `remove(tmp)` | +| RENAME / RENAME2 | if `newdir != parent` → `EXDEV`; else `walk(parent → tmp, [oldname])`, `wstat(tmp, dontcare with .name = newname)`, `clunk`. 9P rename never replaces, POSIX does: when the target exists (and `RENAME_NOREPLACE` is not set) a directory target is removed first; a file target is parked under a temporary name, the rename retried, and the parked file removed only after success (restored on failure) | +| STATFS | constant `Kstatfs{ bsize = 4096, namelen = 255, frsize = 4096 }` | +| ACCESS | `ENOSYS` (kernel stops asking; the server enforces permissions on open) | +| READLINK, SYMLINK, LINK, MKNOD, *XATTR, *LK, IOCTL, POLL, BMAP, FALLOCATE, LSEEK, COPY_FILE_RANGE, TMPFILE, STATX | `ENOSYS` | +| INTERRUPT | ignored (reply nothing) | +| DESTROY | return from `serve` | + +Attr mapping from `cloud9.Stat`: `mode = (S_IFDIR if DMDIR else S_IFREG) | +(st.mode & 0o777)`; `nlink = 1`; `size = length`; `blocks = (length+511)/512`; +`blksize = 4096`; `atime/mtime/ctime = st.atime/st.mtime/st.mtime`; +`uid/gid = opts.uid/gid`. `Dirent.type` = `DT_DIR` (4) / `DT_REG` (8). + +Errors: `nine.Session.Error.Nine` → `nine.errno()`; `Closed`/`Protocol`/`Io` +→ `EIO` and, since the session is dead, `serve` returns `error.Closed` after +replying so 9ns can report "9P server went away". + +With `direct_io` the kernel never trusts `length` for reads: synthetic files +that report length 0 (very common in 9P) still `cat` correctly, and reads run +until the server returns a short read. With `--no-direct-io` the bridge forces +`attr_timeout_ns = 0`, because a cached stale size truncates reads (observed +data loss on a 4 MiB copy otherwise). + +Hostile-server rules: directory listings are capped at 64 MiB (a server that +ignores read offsets otherwise loops forever); directory records with names +containing `/`, NUL, empty, `.`/`..` or longer than `FUSE_NAME_MAX` are dropped +rather than poisoning the whole READDIR reply; `length` near 2^64 is clamped +to `i64` max; the errno of a failing 9P call is latched before any cleanup +clunk overwrites the session's ename. + +### `src/ns.zig` — namespace and process plumbing + +```zig +pub const Spawn = struct { + argv: []const []const u8, // argv[0] is PATH-searched unless it contains '/' + envp: [*:null]const ?[*:0]const u8, // inherited environment + mountpoint: []const u8, // absolute + fuse_fd: i32, + uid: u32, gid: u32, + max_read: u32, +}; +pub const Child = struct { pid: i32 }; +/// fork; the child sets up the namespace, mounts, and execs. Returns once exec succeeded +/// (status pipe closed) or fails with the child's error (message on stderr). +pub fn spawn(gpa: std.mem.Allocator, s: Spawn) !Child; +pub fn ensureMountpoint(path: [:0]const u8) !void; // the shadowing logic, testable alone +pub fn resolveMountpoint(gpa, path: []const u8) ![:0]u8; // absolute, no trailing slash +pub fn findInPath(gpa, envp, name) ![:0]u8; +``` + +Also exports the signal plumbing used by `main.zig`: +`installSignals(child_pid_ptr: *i32) !i32` returning the SIGCHLD self-pipe +read end (used as `stop_fd` for `bridge.serve`), and +`waitChild(pid) !u8` → exit status (`128+sig` on signal death). + +### `src/main.zig` — CLI + +``` +Usage: 9ns [options] -- PROGRAM [ARGS...] +Transport (exactly one): + --unix PATH Unix stream socket + --tcp IP:PORT TCP (IPv4/IPv6 literal) + --fd N already-connected inherited descriptor + --spawn CMD run CMD (via /bin/sh -c) with a socketpair on its stdin/stdout +Options: + --mount PATH mountpoint inside the new namespace (default /mnt/9p) + --uname NAME 9P user name (default $USER, else "none") + --aname NAME 9P tree to attach (default "") + --msize BYTES maximum 9P message size to request (default 131072, max 16 MiB) + --cache SECONDS attr/entry cache validity, may be fractional (default 1) + --no-direct-io let the kernel cache file pages (trusts stat length) + --debug trace FUSE and 9P operations on stderr + --help, --version +PROGRAM defaults to $SHELL (else /bin/sh). The mountpoint is exported as $NINE_MOUNT. +``` + +Exit codes: child's status; 125 for 9ns's own failures (usage, connect, +mount); 126/127 as usual for exec failures. + +### `../9proc/demo/main.zig` — demo 9P2000 server (binary `9proc-demo`) + +The demo server is a separate program in this repository, built on the +9proc library; see `../9proc/docs/LIBRARY.md` for the library +contract (freestanding core, value renderers, Linux debug probe). The tree it serves keeps the paths the integration tests read +(`/build/*`, `/comptime/types//*`, `/comptime/decls`, `/runtime/fn/*`, +`/runtime/ctl`, `/runtime/{pid,ppid,uptime,argv,cwd,env,clients}`, +`/scratch/`) and adds `/vars`, `/threads`, `/addr`, `/mem`, `/hex`, +`/breakpoints` and `/panic`. + +## Integration test plan (`test/integration.sh`) + +Run by `zig build 9ns-itest`; args: path to `9ns`, path to `9proc-demo`. +Everything under a temp dir. Skips (exit 0 with a notice) when +`unshare -Urm true` fails or `/dev/fuse` is missing. + +1. 9proc-demo on a Unix socket; `9ns --unix … -- sh -c` scripts: + `cat /mnt/9p/build/zig_version` == `zig version`; `ls` listings; `stat` + sizes; `/runtime/fn/now` is numeric; `ctl` round trip; `/scratch`: create, + append (`>>`), overwrite, truncate, `mkdir -p a/b/c`, rename within dir, + `mv` across dirs fails with `EXDEV`-ish message, `rm`, `rmdir`, 1 MiB + random file round trip compared with `sha256sum`, `dd` with odd block + sizes, many small files, `find`, exit-status propagation (`exit 7` → 7), + `$NINE_MOUNT` set, nested `9ns` inside `9ns`. +2. `--spawn "<9proc-demo> --stdio"` variant. +3. `--tcp 127.0.0.1:` variant. +4. If `/usr/lib/plan9/bin/ramfs` exists: `NAMESPACE=$tmp ramfs -s ramfs` + creates `$tmp/ramfs`; run the scratch battery against it. +5. `--mount` with an existing dir, with a relative path, and the default + `/mnt/9p` (exercises parent shadowing; verify `/mnt`'s other entries are + still visible inside). +6. Kill tests: 9ns exits when the child exits; server death during use + yields `EIO`, not a hang. + +## Verification + +`zig build 9ns-test` (unit), `zig build 9ns-itest` (74 end-to-end +checks against 9proc-demo over unix/tcp/socketpair and against plan9port's +`ramfs`) and `zig build 9ns-adv` (adversarial suites: a scriptable +hostile 9P server with ~30 misbehaviour modes, FUSE semantics through the +bridge, process/namespace/signal edge cases with 51 checks, and stress). The +suites that attack the 9proc server itself (a hostile raw-9P client with +181 checks, the core, the Linux layer) moved with it to +`../9proc/test` (`zig build 9proc-adv`). All pass in Debug and +ReleaseSafe. + +## Out of scope for v1 (documented, not hidden) + +* One 9P request in flight at a time: a 9P read that blocks (event files) + stalls the whole mount until it returns (but not past the child's exit). +* No 9P2000.u/.L: no symlinks, ownership, or extended attributes. +* No PID namespace, no `/proc` remount. `--tcp` needs an IP literal. +* Cross-directory rename returns `EXDEV` (9P2000 cannot move files). diff --git a/9ns/src/bridge.zig b/9ns/src/bridge.zig new file mode 100644 index 0000000..3a072ec --- /dev/null +++ b/9ns/src/bridge.zig @@ -0,0 +1,971 @@ +//! FUSE ↔ 9P2000 translation: the request loop that turns kernel FUSE requests +//! into synchronous 9P calls on a `nine.Session` and sends the replies back. +//! +//! Everything here is single-threaded and one request at a time. State is three +//! tables: inodes (nodeid → fid/qid, deduplicated by qid.path), open handles +//! (fh → fid plus a cached directory listing), and the reverse qid map. +const std = @import("std"); +const cloud9 = @import("cloud9"); +const fuse = @import("fuse.zig"); +const nine = @import("nine.zig"); +const linux = std.os.linux; + +pub const Options = struct { + /// Reported owner of every file. + uid: u32, + gid: u32, + /// attr/entry cache validity (0 = none). + attr_timeout_ns: u64 = 1_000_000_000, + /// FOPEN_DIRECT_IO on every regular file. + direct_io: bool = true, + /// Trace every request, reply and 9P call to stderr. + debug: bool = false, +}; + +/// Largest single READ/WRITE payload we accept from the kernel. +pub const max_write: u32 = 1 << 20; +/// Upper bound on the raw bytes of one directory listing (about a million entries); +/// past it the listing fails with EIO instead of eating memory. +pub const max_dir_bytes: u64 = 64 << 20; +/// The kernel refuses dirents longer than this (FUSE_NAME_MAX) with EIO. +pub const max_name_len: usize = 1024; +/// Request buffer: `max_write` plus room for the header and the largest in-struct. +pub const request_buf_len: usize = max_write + 4096; + +const Inode = struct { + fid: u32, + qid: cloud9.Qid, + nlookup: u64, + /// nodeid of the directory this inode was looked up in (root: itself). Used for "..". + parent: u64, +}; + +pub const Entry = struct { name: []u8, ino: u64, dtype: u32 }; + +pub const DirList = struct { + entries: std.ArrayList(Entry) = .empty, + + pub fn deinit(d: *DirList, gpa: std.mem.Allocator) void { + for (d.entries.items) |e| gpa.free(e.name); + d.entries.deinit(gpa); + } +}; + +const Handle = struct { + fid: u32, + nodeid: u64, + dir: ?DirList, +}; + +/// Errors a request handler may surface. Policy failures are ordinary errors +/// that the dispatcher maps to an errno; `FuseIo` means the kernel side is broken. +const HandlerError = nine.Session.Error || error{ + BadRequest, + NoEntry, + BadHandle, + Exdev, + Perm, + NotSup, + /// A directory listing the server sent could not be parsed (EIO, not fatal). + BadDir, + FuseIo, +}; + +/// Runs until the FUSE fd reports ENODEV, a DESTROY arrives, or `stop_fd` +/// becomes readable (also while a 9P reply is outstanding). Returns +/// `error.Closed` if the 9P server went away. +pub fn serve(gpa: std.mem.Allocator, fuse_fd: i32, session: *nine.Session, root_fid: u32, stop_fd: i32, opts: Options) !void { + var effective = opts; + // With page caching on, a nonzero attr cache lets the kernel trust a stale + // (often zero) size and truncate reads: 9P sizes are authoritative and change + // under us. direct_io ignores the cached size, so the cache is safe only there. + if (!effective.direct_io) effective.attr_timeout_ns = 0; + var b: Bridge = .{ + .gpa = gpa, + .fuse_fd = fuse_fd, + .nine = session, + .opts = effective, + }; + defer b.deinit(); + + b.req_buf = try gpa.alignedAlloc(u8, .@"8", request_buf_len); + b.data_buf = try gpa.alloc(u8, max_write); + + // Abandon any pending 9P reply once the child is gone (stop_fd readable), + // including the initial root stat below: a silent server must not pin us. + session.stop_fd = stop_fd; + defer session.stop_fd = -1; + + // Node 1 is the root; its qid comes from a stat so lookups resolving back to + // it (e.g. via a walk) dedupe onto node 1. + var root_qid: cloud9.Qid = .{ .type = cloud9.qtdir, .version = 0, .path = 0 }; + if (b.stat(root_fid)) |st| { + root_qid = st.qid; + b.root_path = st.qid.path; + try b.by_qid.put(gpa, st.qid.path, fuse.root_id); + } else |e| switch (e) { + error.Nine => {}, + error.Stopped => return, + else => return error.Closed, + } + try b.inodes.put(gpa, fuse.root_id, .{ .fid = root_fid, .qid = root_qid, .nlookup = 1, .parent = fuse.root_id }); + + var pfds = [_]linux.pollfd{ + .{ .fd = fuse_fd, .events = linux.POLL.IN, .revents = 0 }, + .{ .fd = stop_fd, .events = linux.POLL.IN, .revents = 0 }, + }; + while (true) { + pfds[0].revents = 0; + pfds[1].revents = 0; + const rc = linux.poll(&pfds, pfds.len, -1); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .INTR, .AGAIN => continue, + else => return error.Io, + } + if (pfds[1].revents != 0) { + b.trace("stop_fd readable; leaving serve loop", .{}); + return; + } + if (pfds[0].revents == 0) continue; + const req = (fuse.readRequest(fuse_fd, b.req_buf) catch |e| switch (e) { + error.Protocol => return error.FuseProtocol, + else => return error.FuseIo, + }) orelse { + b.trace("fuse fd reports ENODEV; unmounted", .{}); + return; + }; + if (!try b.dispatch(req)) return; + } +} + +const Bridge = struct { + gpa: std.mem.Allocator, + fuse_fd: i32, + nine: *nine.Session, + opts: Options, + req_buf: []align(8) u8 = &.{}, + data_buf: []u8 = &.{}, + inodes: std.AutoHashMapUnmanaged(u64, Inode) = .empty, + by_qid: std.AutoHashMapUnmanaged(u64, u64) = .empty, + handles: std.AutoHashMapUnmanaged(u64, Handle) = .empty, + next_node: u64 = 2, + next_fh: u64 = 1, + /// qid.path of the root, reported as ino 1 wherever it shows up. + root_path: u64 = 0, + /// errno of the most recent Rerror that a handler did not swallow. Kept here + /// because `Session.rpc` clears its ename on every call, and error paths + /// clunk (an rpc) before the dispatcher maps the failure to an errno. + last_err: linux.E = .IO, + + fn deinit(b: *Bridge) void { + var it = b.handles.valueIterator(); + while (it.next()) |h| if (h.dir) |*d| d.deinit(b.gpa); + b.handles.deinit(b.gpa); + b.inodes.deinit(b.gpa); + b.by_qid.deinit(b.gpa); + if (b.req_buf.len != 0) b.gpa.free(b.req_buf); + if (b.data_buf.len != 0) b.gpa.free(b.data_buf); + } + + fn trace(b: *const Bridge, comptime fmt: []const u8, args: anytype) void { + if (b.opts.debug) std.debug.print("9ns: " ++ fmt ++ "\n", args); + } + + // -- dispatch -------------------------------------------------------------------- + + /// Handles one request. Returns false when the loop should stop (DESTROY). + /// Fatal errors (dead 9P session, broken FUSE fd) propagate. + fn dispatch(b: *Bridge, req: fuse.Request) !bool { + const h = req.header; + const op = h.op(); + b.trace("<- {s} unique={d} nodeid={d} len={d} (fids={d} inodes={d} handles={d})", .{ opName(op), h.unique, h.nodeid, h.len, b.nine.fidsInUse(), b.inodes.count(), b.handles.count() }); + const wants_reply = switch (op) { + .forget, .batch_forget, .interrupt => false, + else => true, + }; + if (op == .destroy) { + b.reply(h.unique, &.{}) catch {}; + return false; + } + b.handle(req) catch |e| { + const code: linux.E = switch (e) { + error.Nine => b.last_err, + error.BadRequest => .INVAL, + error.NoEntry => .NOENT, + error.BadHandle => .BADF, + error.Exdev => .XDEV, + error.Perm => .PERM, + error.NotSup => .NOSYS, + error.OutOfMemory => .NOMEM, + error.TooLarge => .NAMETOOLONG, + error.BadDir => .IO, + error.Closed, error.Protocol, error.Io, error.Stopped => .IO, + error.FuseIo => return error.FuseIo, + }; + if (wants_reply) try b.replyError(h.unique, code); + switch (e) { + error.Closed, error.Protocol, error.Io => return error.Closed, + error.Stopped => return false, // the child is gone; the mount is being torn down + else => {}, + } + }; + return true; + } + + fn handle(b: *Bridge, req: fuse.Request) HandlerError!void { + const u = req.header.unique; + switch (req.header.op()) { + .init => { + const in = try body(fuse.InitIn, req); + const out = fuse.initReply(in, max_write); + try b.reply(u, &.{std.mem.asBytes(&out)}); + }, + .lookup => { + const name = try nameAfter(void, req); + const entry = try b.lookupEntry(req.header.nodeid, name); + try b.reply(u, &.{std.mem.asBytes(&entry)}); + }, + .forget => { + const in = try body(fuse.ForgetIn, req); + try b.forget(req.header.nodeid, in.nlookup); + }, + .batch_forget => { + const in = try body(fuse.BatchForgetIn, req); + const rest = req.body[@sizeOf(fuse.BatchForgetIn)..]; + const count: usize = in.count; + if (rest.len < count * @sizeOf(fuse.ForgetOne)) return error.BadRequest; + for (0..count) |i| { + const one = std.mem.bytesToValue(fuse.ForgetOne, rest[i * @sizeOf(fuse.ForgetOne) ..][0..@sizeOf(fuse.ForgetOne)]); + try b.forget(one.nodeid, one.nlookup); + } + }, + .getattr => { + const ino = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; + const st = try b.stat(ino.fid); + const out = b.attrOut(st, b.inoOf(req.header.nodeid, ino.qid)); + try b.reply(u, &.{std.mem.asBytes(&out)}); + }, + .setattr => try b.setattr(req), + .open => try b.openFile(req, false), + .opendir => try b.openFile(req, true), + .read => { + const in = try body(fuse.ReadIn, req); + const h = b.handles.get(in.fh) orelse return error.BadHandle; + const want: usize = @min(in.size, max_write); + const n = try b.read(h.fid, in.offset, b.data_buf[0..want]); + try b.reply(u, &.{b.data_buf[0..n]}); + }, + .write => { + const in = try body(fuse.WriteIn, req); + const h = b.handles.get(in.fh) orelse return error.BadHandle; + const rest = req.body[@sizeOf(fuse.WriteIn)..]; + if (rest.len < in.size) return error.BadRequest; + const n = try b.write(h.fid, in.offset, rest[0..in.size]); + const out = fuse.WriteOut{ .size = @intCast(n) }; + try b.reply(u, &.{std.mem.asBytes(&out)}); + }, + .readdir => try b.readdir(req), + .release, .releasedir => { + const in = try body(fuse.ReleaseIn, req); + const kv = b.handles.fetchRemove(in.fh) orelse return error.BadHandle; + var h = kv.value; + if (h.dir) |*d| d.deinit(b.gpa); + try b.clunk(h.fid); + try b.reply(u, &.{}); + }, + .flush, .fsync, .fsyncdir => try b.reply(u, &.{}), + .create => try b.create(req), + .mkdir => { + const in = try body(fuse.MkdirIn, req); + const name = try nameAfter(fuse.MkdirIn, req); + const parent = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; + const fid = try b.clone(parent.fid); + _ = b.create9(fid, name, cloud9.dmdir | (in.mode & 0o777), cloud9.oread) catch |e| { + b.clunkQuiet(fid); + return e; + }; + try b.clunk(fid); + const entry = try b.lookupEntry(req.header.nodeid, name); + try b.reply(u, &.{std.mem.asBytes(&entry)}); + }, + .unlink, .rmdir => { + const name = try nameAfter(void, req); + const parent = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; + const tmp = try b.walkName(parent.fid, name); + try b.remove(tmp); + try b.reply(u, &.{}); + }, + .rename => { + const in = try body(fuse.RenameIn, req); + const old = try nameAfter(fuse.RenameIn, req); + const new = try secondName(req, old, @sizeOf(fuse.RenameIn)); + try b.rename(req.header.nodeid, in.newdir, old, new, 0); + try b.reply(u, &.{}); + }, + .rename2 => { + const in = try body(fuse.Rename2In, req); + const old = try nameAfter(fuse.Rename2In, req); + const new = try secondName(req, old, @sizeOf(fuse.Rename2In)); + try b.rename(req.header.nodeid, in.newdir, old, new, in.flags); + try b.reply(u, &.{}); + }, + .statfs => { + const out = fuse.StatfsOut{ .st = .{ .bsize = 4096, .namelen = 255, .frsize = 4096 } }; + try b.reply(u, &.{std.mem.asBytes(&out)}); + }, + .interrupt => {}, + .destroy => unreachable, // handled in dispatch + .access => return error.NotSup, + else => return error.NotSup, + } + } + + // -- handlers ---------------------------------------------------------------------- + + /// walk(parent → new fid, [name]) + stat, deduplicated by qid.path. Bumps nlookup. + fn lookupEntry(b: *Bridge, parent_id: u64, name: []const u8) HandlerError!fuse.EntryOut { + const parent = b.inodes.get(parent_id) orelse return error.NoEntry; + const newfid = try b.walkName(parent.fid, name); + const st = b.stat(newfid) catch |e| { + b.clunkQuiet(newfid); + return e; + }; + const qid = st.qid; + var nodeid: u64 = undefined; + if (b.by_qid.get(qid.path)) |existing| { + // A directory and a file sharing a qid.path (a server bug) must not + // share a node: the kernel would mark the inode bad, and for the + // root that is fatal for the whole mount. + const merge = if (b.inodes.getPtr(existing)) |ino| (ino.qid.type & cloud9.qtdir) == (qid.type & cloud9.qtdir) else false; + if (merge) { + const ino = b.inodes.getPtr(existing).?; + ino.nlookup += 1; + ino.qid = qid; + nodeid = existing; + if (existing == fuse.root_id) { + b.clunkQuiet(newfid); + } else { + // Keep the fresh fid (it is bound to the current file at this + // name) and retire the older one. + const stale = ino.fid; + ino.fid = newfid; + b.clunkQuiet(stale); + } + } else { + // Stale reverse entry, or a type clash: bind a fresh node to it. + nodeid = try b.newInode(newfid, qid, parent_id); + } + } else { + nodeid = try b.newInode(newfid, qid, parent_id); + } + var out = fuse.EntryOut{ + .nodeid = nodeid, + .generation = 0, + .attr = b.attrFrom(st, b.inoOf(nodeid, qid)), + }; + out.entry_valid = b.opts.attr_timeout_ns / 1_000_000_000; + out.entry_valid_nsec = @intCast(b.opts.attr_timeout_ns % 1_000_000_000); + out.attr_valid = out.entry_valid; + out.attr_valid_nsec = out.entry_valid_nsec; + return out; + } + + fn newInode(b: *Bridge, fid: u32, qid: cloud9.Qid, parent: u64) HandlerError!u64 { + const nodeid = b.next_node; + b.inodes.put(b.gpa, nodeid, .{ .fid = fid, .qid = qid, .nlookup = 1, .parent = parent }) catch |e| { + b.clunkQuiet(fid); + return e; + }; + b.by_qid.put(b.gpa, qid.path, nodeid) catch |e| { + _ = b.inodes.remove(nodeid); + b.clunkQuiet(fid); + return e; + }; + b.next_node += 1; + return nodeid; + } + + fn forget(b: *Bridge, nodeid: u64, n: u64) HandlerError!void { + if (nodeid == fuse.root_id) return; + const ino = b.inodes.getPtr(nodeid) orelse return; + if (ino.nlookup > n) { + ino.nlookup -= n; + return; + } + const fid = ino.fid; + const path = ino.qid.path; + _ = b.inodes.remove(nodeid); + if (b.by_qid.get(path)) |mapped| { + if (mapped == nodeid) _ = b.by_qid.remove(path); + } + b.clunk(fid) catch |e| switch (e) { + error.Nine => {}, + else => return e, + }; + } + + fn setattr(b: *Bridge, req: fuse.Request) HandlerError!void { + const in = try body(fuse.SetattrIn, req); + const ino = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; + const old = try b.stat(ino.fid); + const old_mode = old.mode; + + var st = nine.dontcare; + var changed = false; + if (in.valid & fuse.FATTR_UID != 0 and in.uid != b.opts.uid) return error.Perm; + if (in.valid & fuse.FATTR_GID != 0 and in.gid != b.opts.gid) return error.Perm; + if (in.valid & fuse.FATTR_SIZE != 0) { + st.length = in.size; + changed = true; + } + if (in.valid & fuse.FATTR_MODE != 0) { + st.mode = (old_mode & ~@as(u32, 0o777)) | (in.mode & 0o777); + changed = true; + } + if (in.valid & fuse.FATTR_MTIME_NOW != 0) { + st.mtime = nowSeconds(); + changed = true; + } else if (in.valid & fuse.FATTR_MTIME != 0) { + st.mtime = @truncate(in.mtime); + changed = true; + } + if (changed) try b.wstat(ino.fid, st); + const fresh = try b.stat(ino.fid); + const out = b.attrOut(fresh, b.inoOf(req.header.nodeid, ino.qid)); + try b.reply(req.header.unique, &.{std.mem.asBytes(&out)}); + } + + fn openFile(b: *Bridge, req: fuse.Request, is_dir: bool) HandlerError!void { + const in = try body(fuse.OpenIn, req); + const ino = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; + const mode: u8 = if (is_dir) cloud9.oread else openMode(in.flags); + const fid = try b.clone(ino.fid); + _ = b.open9(fid, mode) catch |e| { + b.clunkQuiet(fid); + return e; + }; + const fh = try b.newHandle(fid, req.header.nodeid); + const out = fuse.OpenOut{ + .fh = fh, + .open_flags = if (!is_dir and b.opts.direct_io) fuse.FOPEN_DIRECT_IO else 0, + }; + try b.reply(req.header.unique, &.{std.mem.asBytes(&out)}); + } + + fn newHandle(b: *Bridge, fid: u32, nodeid: u64) HandlerError!u64 { + const fh = b.next_fh; + b.handles.put(b.gpa, fh, .{ .fid = fid, .nodeid = nodeid, .dir = null }) catch |e| { + b.clunkQuiet(fid); + return e; + }; + b.next_fh += 1; + return fh; + } + + fn create(b: *Bridge, req: fuse.Request) HandlerError!void { + const in = try body(fuse.CreateIn, req); + const name = try nameAfter(fuse.CreateIn, req); + const parent = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; + // The created fid becomes the open file. + const fid = try b.clone(parent.fid); + _ = b.create9(fid, name, in.mode & 0o777, openMode(in.flags)) catch |e| { + b.clunkQuiet(fid); + return e; + }; + const entry = b.lookupEntry(req.header.nodeid, name) catch |e| { + b.clunkQuiet(fid); + return e; + }; + const fh = try b.newHandle(fid, entry.nodeid); + const oo = fuse.OpenOut{ + .fh = fh, + .open_flags = if (b.opts.direct_io) fuse.FOPEN_DIRECT_IO else 0, + }; + try b.reply(req.header.unique, &.{ std.mem.asBytes(&entry), std.mem.asBytes(&oo) }); + } + + fn rename(b: *Bridge, parent_id: u64, newdir: u64, old: []const u8, new: []const u8, flags: u32) HandlerError!void { + if (newdir != parent_id) return error.Exdev; + const rf: linux.RENAME = @bitCast(flags); + if (rf.EXCHANGE or rf.WHITEOUT) return error.BadRequest; + const parent = b.inodes.get(parent_id) orelse return error.NoEntry; + const tmp = try b.walkName(parent.fid, old); + defer b.clunkQuiet(tmp); + var st = nine.dontcare; + st.name = new; + b.wstat(tmp, st) catch |e| { + // 9P2000 rename never replaces an existing name; POSIX rename does. + if (e != error.Nine or rf.NOREPLACE or b.nine.errno() != .EXIST) return e; + try b.renameOver(parent.fid, tmp, new); + }; + } + + /// Replace `new` with the file behind `src`. An (empty) directory target is + /// removed first: it holds no data and the VFS already ruled out mismatched + /// types. A file target is parked under a temporary name so that a failing + /// second rename can put it back instead of having destroyed it. + fn renameOver(b: *Bridge, parent_fid: u32, src: u32, new: []const u8) HandlerError!void { + const victim = try b.walkName(parent_fid, new); + const vst = b.stat(victim) catch |e| { + b.clunkQuiet(victim); + return e; + }; + var st = nine.dontcare; + st.name = new; + if (vst.mode & cloud9.dmdir != 0) { + b.trace(" rename target is a directory; removing it and retrying", .{}); + try b.remove(victim); + return b.wstat(src, st); + } + var park_buf: [48]u8 = undefined; + const park = std.fmt.bufPrint(&park_buf, ".9ns-rename-{x}", .{randomU64()}) catch unreachable; + b.trace(" rename target exists; parking it as {s} and retrying", .{park}); + var pst = nine.dontcare; + pst.name = park; + b.wstat(victim, pst) catch |e| { + b.clunkQuiet(victim); + return e; + }; + b.wstat(src, st) catch |e| { + b.trace(" rename still failed; restoring the target", .{}); + const saved = b.last_err; + b.wstat(victim, st) catch {}; + b.last_err = saved; + b.clunkQuiet(victim); + return e; + }; + b.remove(victim) catch b.trace(" could not remove the parked target {s}", .{park}); + } + + fn readdir(b: *Bridge, req: fuse.Request) HandlerError!void { + const in = try body(fuse.ReadIn, req); + const h = b.handles.getPtr(in.fh) orelse return error.BadHandle; + if (in.offset == 0 or h.dir == null) { + if (h.dir) |*d| d.deinit(b.gpa); + h.dir = null; + h.dir = try b.loadDir(h.fid, h.nodeid); + } + const dir = &h.dir.?; + const size: usize = @min(in.size, max_write); + const used = packDirents(dir.entries.items, in.offset, b.data_buf[0..size]); + try b.reply(req.header.unique, &.{b.data_buf[0..used]}); + } + + /// Reads the whole directory and builds its listing, "." and ".." first. + fn loadDir(b: *Bridge, fid: u32, nodeid: u64) HandlerError!DirList { + var list: DirList = .{}; + errdefer list.deinit(b.gpa); + const self_ino = b.inoOfNode(nodeid); + const parent_ino = if (b.inodes.get(nodeid)) |ino| b.inoOfNode(ino.parent) else self_ino; + try list.entries.append(b.gpa, .{ .name = try b.gpa.dupe(u8, "."), .ino = self_ino, .dtype = fuse.DT_DIR }); + try list.entries.append(b.gpa, .{ .name = try b.gpa.dupe(u8, ".."), .ino = parent_ino, .dtype = fuse.DT_DIR }); + + var offset: u64 = 0; + while (true) { + // A server that ignores the offset would otherwise feed us forever. + if (offset >= max_dir_bytes) return error.BadDir; + const n = try b.read(fid, offset, b.data_buf); + if (n == 0) break; + try parseDirRecords(b.gpa, b.data_buf[0..n], &list); + offset += n; + } + // Entries carrying the root's own qid.path get the root's ino (1), as GETATTR would report it. + for (list.entries.items[2..]) |*e| if (e.ino == b.root_path) { + e.ino = fuse.root_id; + }; + return list; + } + + // -- 9P wrappers (tracing) ----------------------------------------------------------- + + fn stat(b: *Bridge, fid: u32) nine.Session.Error!cloud9.Stat { + const st = b.nine.stat(fid) catch |e| return b.nineErr("stat", fid, e); + b.trace(" 9p stat fid={d} -> name={s} mode={o} len={d} qid={x}", .{ fid, st.name, st.mode, st.length, st.qid.path }); + return st; + } + + fn walkName(b: *Bridge, fid: u32, name: []const u8) nine.Session.Error!u32 { + const newfid = b.nine.allocFid(); + _ = b.nine.walk(fid, newfid, &.{name}) catch |e| { + b.nine.freeFid(newfid); + return b.nineErr("walk", fid, e); + }; + b.trace(" 9p walk fid={d} newfid={d} name={s} -> ok", .{ fid, newfid, name }); + return newfid; + } + + fn clone(b: *Bridge, fid: u32) nine.Session.Error!u32 { + const newfid = b.nine.clone(fid) catch |e| return b.nineErr("clone", fid, e); + b.trace(" 9p walk fid={d} newfid={d} (clone) -> ok", .{ fid, newfid }); + return newfid; + } + + fn open9(b: *Bridge, fid: u32, mode: u8) nine.Session.Error!nine.Session.Open { + const o = b.nine.open(fid, mode) catch |e| return b.nineErr("open", fid, e); + b.trace(" 9p open fid={d} mode={d} -> iounit={d}", .{ fid, mode, o.iounit }); + return o; + } + + fn create9(b: *Bridge, fid: u32, name: []const u8, perm: u32, mode: u8) nine.Session.Error!nine.Session.Open { + const o = b.nine.create(fid, name, perm, mode) catch |e| return b.nineErr("create", fid, e); + b.trace(" 9p create fid={d} name={s} perm={o} mode={d} -> iounit={d}", .{ fid, name, perm, mode, o.iounit }); + return o; + } + + fn read(b: *Bridge, fid: u32, offset: u64, buf: []u8) nine.Session.Error!usize { + const n = b.nine.read(fid, offset, buf) catch |e| return b.nineErr("read", fid, e); + b.trace(" 9p read fid={d} offset={d} count={d} -> {d}", .{ fid, offset, buf.len, n }); + return n; + } + + fn write(b: *Bridge, fid: u32, offset: u64, data: []const u8) nine.Session.Error!usize { + const n = b.nine.write(fid, offset, data) catch |e| return b.nineErr("write", fid, e); + b.trace(" 9p write fid={d} offset={d} count={d} -> {d}", .{ fid, offset, data.len, n }); + return n; + } + + fn wstat(b: *Bridge, fid: u32, st: cloud9.Stat) nine.Session.Error!void { + b.nine.wstat(fid, st) catch |e| return b.nineErr("wstat", fid, e); + b.trace(" 9p wstat fid={d} name={s} mode={x} len={x} mtime={x} -> ok", .{ fid, st.name, st.mode, st.length, st.mtime }); + } + + fn clunk(b: *Bridge, fid: u32) nine.Session.Error!void { + b.nine.clunk(fid) catch |e| return b.nineErr("clunk", fid, e); + b.trace(" 9p clunk fid={d} -> ok", .{fid}); + } + + /// Best-effort clunk during error unwinding; a dead session surfaces on the + /// next call. Does not disturb the errno of the failure being unwound. + fn clunkQuiet(b: *Bridge, fid: u32) void { + const saved = b.last_err; + defer b.last_err = saved; + b.clunk(fid) catch {}; + } + + fn remove(b: *Bridge, fid: u32) nine.Session.Error!void { + b.nine.remove(fid) catch |e| return b.nineErr("remove", fid, e); + b.trace(" 9p remove fid={d} -> ok", .{fid}); + } + + fn nineErr(b: *Bridge, what: []const u8, fid: u32, e: nine.Session.Error) nine.Session.Error { + if (e == error.Nine) { + b.last_err = b.nine.errno(); + b.trace(" 9p {s} fid={d} -> Rerror \"{s}\" ({s})", .{ what, fid, b.nine.ename[0..b.nine.ename_len], @tagName(b.nine.errno()) }); + } else { + b.trace(" 9p {s} fid={d} -> {s}", .{ what, fid, @errorName(e) }); + } + return e; + } + + // -- FUSE wrappers (tracing) -------------------------------------------------------- + + fn reply(b: *Bridge, unique: u64, payloads: []const []const u8) error{FuseIo}!void { + var total: usize = 0; + for (payloads) |p| total += p.len; + b.trace("-> unique={d} ok ({d} bytes)", .{ unique, total }); + fuse.reply(b.fuse_fd, unique, payloads) catch return error.FuseIo; + } + + fn replyError(b: *Bridge, unique: u64, code: linux.E) error{FuseIo}!void { + b.trace("-> unique={d} error E{s}", .{ unique, @tagName(code) }); + fuse.replyError(b.fuse_fd, unique, code) catch return error.FuseIo; + } + + // -- attrs --------------------------------------------------------------------------- + + fn inoOf(b: *const Bridge, nodeid: u64, qid: cloud9.Qid) u64 { + return if (nodeid == fuse.root_id or qid.path == b.root_path) fuse.root_id else qid.path; + } + + fn inoOfNode(b: *const Bridge, nodeid: u64) u64 { + if (nodeid == fuse.root_id) return fuse.root_id; + const ino = b.inodes.get(nodeid) orelse return nodeid; + return b.inoOf(nodeid, ino.qid); + } + + fn attrFrom(b: *const Bridge, st: cloud9.Stat, ino: u64) fuse.Attr { + return attrFromStat(st, ino, b.opts.uid, b.opts.gid); + } + + fn attrOut(b: *const Bridge, st: cloud9.Stat, ino: u64) fuse.AttrOut { + return .{ + .attr_valid = b.opts.attr_timeout_ns / 1_000_000_000, + .attr_valid_nsec = @intCast(b.opts.attr_timeout_ns % 1_000_000_000), + .attr = b.attrFrom(st, ino), + }; + } +}; + +// -- pure helpers (unit-tested) ------------------------------------------------------------ + +/// Attr from a 9P Stat: DMDIR → S_IFDIR else S_IFREG, low 9 permission bits kept. +pub fn attrFromStat(st: cloud9.Stat, ino: u64, uid: u32, gid: u32) fuse.Attr { + const ftype: u32 = if (st.mode & cloud9.dmdir != 0) fuse.S_IFDIR else fuse.S_IFREG; + return .{ + .ino = ino, + // The kernel marks an inode bad when size > LLONG_MAX; clamp hostile lengths. + .size = @min(st.length, std.math.maxInt(i64)), + // Saturating: a hostile length of 2^64-1 must not overflow. + .blocks = st.length / 512 + @intFromBool(st.length % 512 != 0), + .atime = st.atime, + .mtime = st.mtime, + .ctime = st.mtime, + .mode = ftype | (st.mode & 0o777), + .nlink = 1, + .uid = uid, + .gid = gid, + .blksize = 4096, + }; +} + +/// Kernel open(2) flags → 9P open mode. O_APPEND has no 9P equivalent and is ignored. +pub fn openMode(flags: u32) u8 { + const o: linux.O = @bitCast(flags); + var mode: u8 = switch (o.ACCMODE) { + .RDONLY => cloud9.oread, + .WRONLY => cloud9.owrite, + .RDWR => cloud9.ordwr, + }; + if (o.TRUNC) mode |= cloud9.otrunc; + return mode; +} + +/// Parses consecutive 9P directory records (2-byte size + Stat) and appends entries. +pub fn parseDirRecords(gpa: std.mem.Allocator, bytes: []const u8, list: *DirList) error{ OutOfMemory, BadDir }!void { + var pos: usize = 0; + while (pos < bytes.len) { + if (bytes.len - pos < 2) return error.BadDir; + const size: usize = std.mem.readInt(u16, bytes[pos..][0..2], .little); + if (bytes.len - pos < 2 + size) return error.BadDir; + const st = cloud9.Stat.decode(bytes[pos..][0 .. 2 + size]) catch return error.BadDir; + pos += 2 + size; + // The kernel rejects a whole READDIR reply (EIO) over one bad name, and + // "." and ".." are synthesised by loadDir: drop such records instead. + if (!validDirentName(st.name)) continue; + const name = try gpa.dupe(u8, st.name); + errdefer gpa.free(name); + try list.entries.append(gpa, .{ + .name = name, + .ino = st.qid.path, + .dtype = if (st.mode & cloud9.dmdir != 0) fuse.DT_DIR else fuse.DT_REG, + }); + } +} + +/// A name the kernel will accept in a dirent and that does not duplicate the synthetic "." / "..". +pub fn validDirentName(name: []const u8) bool { + if (name.len == 0 or name.len > max_name_len) return false; + if (std.mem.indexOfAny(u8, name, "/\x00") != null) return false; + if (std.mem.eql(u8, name, ".") or std.mem.eql(u8, name, "..")) return false; + return true; +} + +/// Packs dirents from `entries[offset..]` into `buf`; each record's `off` is its index + 1. +/// Returns the number of bytes used. +pub fn packDirents(entries: []const Entry, offset: u64, buf: []u8) usize { + var used: usize = 0; + var i: usize = @intCast(@min(offset, entries.len)); + while (i < entries.len) : (i += 1) { + const e = entries[i]; + if (!fuse.addDirent(buf, &used, e.ino, @as(u64, i) + 1, e.dtype, e.name)) break; + } + return used; +} + +fn randomU64() u64 { + var bytes: [8]u8 = undefined; + if (linux.errno(linux.getrandom(&bytes, bytes.len, 0)) == .SUCCESS) return std.mem.readInt(u64, &bytes, .little); + var ts: linux.timespec = undefined; + _ = linux.clock_gettime(.MONOTONIC, &ts); + return @as(u64, @bitCast(ts.nsec)) ^ (@as(u64, @bitCast(ts.sec)) << 32); +} + +fn nowSeconds() u32 { + var ts: linux.timespec = undefined; + if (linux.errno(linux.clock_gettime(.REALTIME, &ts)) != .SUCCESS) return 0; + return @intCast(@as(u64, @intCast(ts.sec)) & 0xFFFF_FFFF); +} + +fn opName(op: fuse.Opcode) []const u8 { + return switch (op) { + _ => "unknown", + else => @tagName(op), + }; +} + +// Thin adapters so fuse.zig's parse errors become HandlerError.BadRequest. +fn body(comptime T: type, req: fuse.Request) error{BadRequest}!*const T { + return fuse.body(T, req) catch error.BadRequest; +} + +fn nameAfter(comptime T: type, req: fuse.Request) error{BadRequest}![]const u8 { + return fuse.nameAfter(T, req) catch error.BadRequest; +} + +fn secondName(req: fuse.Request, first: []const u8, offset: usize) error{BadRequest}![]const u8 { + return fuse.secondName(req, first, offset) catch error.BadRequest; +} + +// -- tests ------------------------------------------------------------------------------ + +const testing = std.testing; + +test { + // Force semantic analysis of `serve` and the whole dispatch path, which no + // unit test can exercise without a FUSE mount. + testing.refAllDecls(@This()); +} + +fn testStat(name: []const u8, mode: u32, length: u64, path: u64) cloud9.Stat { + return .{ + .type = 0, + .dev = 0, + .qid = .{ .type = if (mode & cloud9.dmdir != 0) cloud9.qtdir else 0, .version = 0, .path = path }, + .mode = mode, + .atime = 100, + .mtime = 200, + .length = length, + .name = name, + .uid = "u", + .gid = "g", + .muid = "u", + }; +} + +test "attr mapping: DMDIR → S_IFDIR|perm, length → size/blocks" { + const d = attrFromStat(testStat("d", cloud9.dmdir | 0o755, 0, 9), 9, 1000, 1001); + try testing.expectEqual(fuse.S_IFDIR | 0o755, d.mode); + try testing.expectEqual(@as(u64, 9), d.ino); + try testing.expectEqual(@as(u64, 0), d.size); + try testing.expectEqual(@as(u64, 0), d.blocks); + try testing.expectEqual(@as(u32, 1000), d.uid); + try testing.expectEqual(@as(u32, 1001), d.gid); + try testing.expectEqual(@as(u32, 1), d.nlink); + + const f = attrFromStat(testStat("f", 0o640 | cloud9.dmappend, 1025, 4), 4, 0, 0); + try testing.expectEqual(fuse.S_IFREG | 0o640, f.mode); // dmappend bit not leaked + try testing.expectEqual(@as(u64, 1025), f.size); + try testing.expectEqual(@as(u64, 3), f.blocks); + try testing.expectEqual(@as(u32, 4096), f.blksize); + try testing.expectEqual(@as(u64, 100), f.atime); + try testing.expectEqual(@as(u64, 200), f.mtime); + try testing.expectEqual(@as(u64, 200), f.ctime); + + try testing.expectEqual(@as(u64, 1), attrFromStat(testStat("f", 0o600, 512, 4), 4, 0, 0).blocks); + try testing.expectEqual(@as(u64, 2), attrFromStat(testStat("f", 0o600, 513, 4), 4, 0, 0).blocks); +} + +test "open flag → 9P mode mapping" { + const rdonly: u32 = @bitCast(linux.O{ .ACCMODE = .RDONLY }); + const wronly: u32 = @bitCast(linux.O{ .ACCMODE = .WRONLY }); + const rdwr: u32 = @bitCast(linux.O{ .ACCMODE = .RDWR }); + const trunc: u32 = @bitCast(linux.O{ .TRUNC = true }); + const append: u32 = @bitCast(linux.O{ .APPEND = true }); + const creat: u32 = @bitCast(linux.O{ .CREAT = true }); + try testing.expectEqual(cloud9.oread, openMode(rdonly)); + try testing.expectEqual(cloud9.owrite, openMode(wronly)); + try testing.expectEqual(cloud9.ordwr, openMode(rdwr)); + try testing.expectEqual(cloud9.owrite | cloud9.otrunc, openMode(wronly | trunc)); + try testing.expectEqual(cloud9.ordwr | cloud9.otrunc, openMode(rdwr | trunc | creat)); + try testing.expectEqual(cloud9.owrite, openMode(wronly | append)); // O_APPEND ignored +} + +test "dirlist parsing from two hand-encoded Stat records" { + var buf: [512]u8 = undefined; + const a = try cloud9.Stat.encode(testStat("alpha", 0o644, 10, 0x11), &buf); + const bb = try cloud9.Stat.encode(testStat("beta", cloud9.dmdir | 0o755, 0, 0x22), buf[a.len..]); + const bytes = buf[0 .. a.len + bb.len]; + // Sanity: the record is prefixed by its own 2-byte size. + try testing.expectEqual(a.len - 2, std.mem.readInt(u16, bytes[0..2], .little)); + + var list: DirList = .{}; + defer list.deinit(testing.allocator); + try parseDirRecords(testing.allocator, bytes, &list); + try testing.expectEqual(@as(usize, 2), list.entries.items.len); + try testing.expectEqualStrings("alpha", list.entries.items[0].name); + try testing.expectEqual(@as(u64, 0x11), list.entries.items[0].ino); + try testing.expectEqual(fuse.DT_REG, list.entries.items[0].dtype); + try testing.expectEqualStrings("beta", list.entries.items[1].name); + try testing.expectEqual(@as(u64, 0x22), list.entries.items[1].ino); + try testing.expectEqual(fuse.DT_DIR, list.entries.items[1].dtype); + + // Truncated input is a protocol error and leaves earlier entries intact. + try testing.expectError(error.BadDir, parseDirRecords(testing.allocator, bytes[0 .. bytes.len - 1], &list)); + try testing.expectEqual(@as(usize, 3), list.entries.items.len); +} + +test "readdir packing and offset resumption" { + const names = [_][]const u8{ ".", "..", "one", "two", "three" }; + var entries: [names.len]Entry = undefined; + for (&entries, names, 0..) |*e, n, i| e.* = .{ .name = @constCast(n), .ino = 100 + i, .dtype = if (i < 2) fuse.DT_DIR else fuse.DT_REG }; + + // Everything fits: five records, off = index + 1. + var big: [1024]u8 = undefined; + const used = packDirents(&entries, 0, &big); + var pos: usize = 0; + var idx: usize = 0; + while (pos < used) : (idx += 1) { + const d = std.mem.bytesToValue(fuse.Dirent, big[pos..][0..@sizeOf(fuse.Dirent)]); + try testing.expectEqual(@as(u64, 100 + idx), d.ino); + try testing.expectEqual(@as(u64, idx + 1), d.off); + try testing.expectEqualStrings(names[idx], big[pos + @sizeOf(fuse.Dirent) ..][0..d.namelen]); + pos += (@sizeOf(fuse.Dirent) + d.namelen + 7) & ~@as(usize, 7); + } + try testing.expectEqual(names.len, idx); + + // A buffer that fits exactly two records ("." = 32, ".." = 32) stops there… + var small: [64]u8 = undefined; + const first_used = packDirents(&entries, 0, &small); + try testing.expectEqual(@as(usize, 64), first_used); + const last = std.mem.bytesToValue(fuse.Dirent, small[32..][0..@sizeOf(fuse.Dirent)]); + try testing.expectEqual(@as(u64, 2), last.off); + // …and resuming at the last `off` yields "one" next. + const second_used = packDirents(&entries, last.off, &small); + const next = std.mem.bytesToValue(fuse.Dirent, small[0..@sizeOf(fuse.Dirent)]); + try testing.expectEqualStrings("one", small[@sizeOf(fuse.Dirent)..][0..next.namelen]); + try testing.expectEqual(@as(u64, 3), next.off); + try testing.expect(second_used > 0); + + // Past the end: nothing (EOF for the kernel). + try testing.expectEqual(@as(usize, 0), packDirents(&entries, names.len, &big)); + try testing.expectEqual(@as(usize, 0), packDirents(&entries, 1000, &big)); +} + +test "attr mapping saturates hostile lengths instead of overflowing" { + const a = attrFromStat(testStat("f", 0o600, std.math.maxInt(u64), 4), 4, 0, 0); + try testing.expectEqual(@as(u64, std.math.maxInt(i64)), a.size); + try testing.expectEqual(@as(u64, std.math.maxInt(u64) / 512 + 1), a.blocks); + const b = attrFromStat(testStat("f", 0o600, 1024, 4), 4, 0, 0); + try testing.expectEqual(@as(u64, 2), b.blocks); + try testing.expectEqual(@as(u64, 1024), b.size); +} + +test "dirent names the kernel would reject are dropped from listings" { + try testing.expect(validDirentName("a")); + try testing.expect(validDirentName("x" ** 1024)); + try testing.expect(!validDirentName("")); + try testing.expect(!validDirentName("a/b")); + try testing.expect(!validDirentName("a\x00b")); + try testing.expect(!validDirentName(".")); + try testing.expect(!validDirentName("..")); + try testing.expect(!validDirentName("x" ** 1025)); + + var buf: [4096]u8 = undefined; + var n: usize = 0; + for ([_][]const u8{ ".", "..", "", "a/b", "keep", "x" ** 1025, "also" }) |name| { + n += (try cloud9.Stat.encode(testStat(name, 0o644, 1, 0x30), buf[n..])).len; + } + var list: DirList = .{}; + defer list.deinit(testing.allocator); + try parseDirRecords(testing.allocator, buf[0..n], &list); + try testing.expectEqual(@as(usize, 2), list.entries.items.len); + try testing.expectEqualStrings("keep", list.entries.items[0].name); + try testing.expectEqualStrings("also", list.entries.items[1].name); +} + +test "DirList frees its names" { + var list: DirList = .{}; + try list.entries.append(testing.allocator, .{ .name = try testing.allocator.dupe(u8, "x"), .ino = 1, .dtype = fuse.DT_REG }); + list.deinit(testing.allocator); +} diff --git a/9ns/src/fuse.zig b/9ns/src/fuse.zig new file mode 100644 index 0000000..682b3d5 --- /dev/null +++ b/9ns/src/fuse.zig @@ -0,0 +1,653 @@ +//! Kernel FUSE protocol subset (no libfuse, no libc, no policy). +//! +//! Extern structs mirror `/usr/include/linux/fuse.h` (kernel header 7.45); +//! every layout is checked against the header's size at comptime. Only the +//! opcodes and structs 9ns needs are here. The I/O helpers are blocking +//! and allocation-free: the caller owns a single request buffer. +//! +//! Wire rules worth remembering: +//! * The kernel delivers exactly one request per `read(2)` on `/dev/fuse`, +//! and a reply must be exactly one `write(2)`/`writev(2)`. +//! * Request bodies start right after the 40-byte `InHeader`; since every +//! in-struct is 8-byte aligned in the header, `body()` requires the caller's +//! buffer to be 8-byte aligned (`std.heap` page allocations and +//! `align(8)` arrays both qualify). +//! * A write that fails with `ENOENT` means the request was interrupted and +//! the kernel already forgot it: the reply is silently dropped. +//! * `ENODEV` on read means the filesystem was unmounted. + +const std = @import("std"); +const linux = std.os.linux; + +pub const kernel_version: u32 = 7; +/// The minor we answer; the kernel adapts to the lower of the two. +pub const kernel_minor: u32 = 31; +pub const root_id: u64 = 1; + +pub const FOPEN_DIRECT_IO: u32 = 1 << 0; +pub const FOPEN_KEEP_CACHE: u32 = 1 << 1; +pub const FOPEN_NONSEEKABLE: u32 = 1 << 2; + +pub const FUSE_ASYNC_READ: u32 = 1 << 0; +/// The kernel passes O_TRUNC in OPEN instead of a separate SETATTR(size=0); 9P has OTRUNC for exactly this. +pub const FUSE_ATOMIC_O_TRUNC: u32 = 1 << 3; +/// Without this the kernel's cached-write path (`--no-direct-io`) sends one 4 KiB WRITE per page. +pub const FUSE_BIG_WRITES: u32 = 1 << 5; +/// Re-fetch a cached inode's size/mtime and drop stale pages when they change. +/// Required for `--no-direct-io` correctness: 9P sizes change under us, and +/// without this the kernel trusts a stale cached size and truncates reads. +pub const FUSE_AUTO_INVAL_DATA: u32 = 1 << 12; +pub const FUSE_MAX_PAGES: u32 = 1 << 22; + +pub const FATTR_MODE: u32 = 1 << 0; +pub const FATTR_UID: u32 = 1 << 1; +pub const FATTR_GID: u32 = 1 << 2; +pub const FATTR_SIZE: u32 = 1 << 3; +pub const FATTR_ATIME: u32 = 1 << 4; +pub const FATTR_MTIME: u32 = 1 << 5; +pub const FATTR_FH: u32 = 1 << 6; +pub const FATTR_ATIME_NOW: u32 = 1 << 7; +pub const FATTR_MTIME_NOW: u32 = 1 << 8; +pub const FATTR_LOCKOWNER: u32 = 1 << 9; +pub const FATTR_CTIME: u32 = 1 << 10; + +/// File type bits for `Attr.mode` and `Dirent.type` (used by the bridge). +pub const S_IFDIR: u32 = linux.S.IFDIR; +pub const S_IFREG: u32 = linux.S.IFREG; +pub const DT_DIR: u32 = linux.DT.DIR; +pub const DT_REG: u32 = linux.DT.REG; + +pub const Opcode = enum(u32) { + lookup = 1, + forget = 2, + getattr = 3, + setattr = 4, + readlink = 5, + symlink = 6, + mknod = 8, + mkdir = 9, + unlink = 10, + rmdir = 11, + rename = 12, + link = 13, + open = 14, + read = 15, + write = 16, + statfs = 17, + release = 18, + fsync = 20, + setxattr = 21, + getxattr = 22, + listxattr = 23, + removexattr = 24, + flush = 25, + init = 26, + opendir = 27, + readdir = 28, + releasedir = 29, + fsyncdir = 30, + getlk = 31, + setlk = 32, + setlkw = 33, + access = 34, + create = 35, + interrupt = 36, + bmap = 37, + destroy = 38, + ioctl = 39, + poll = 40, + notify_reply = 41, + batch_forget = 42, + fallocate = 43, + readdirplus = 44, + rename2 = 45, + lseek = 46, + copy_file_range = 47, + setupmapping = 48, + removemapping = 49, + syncfs = 50, + tmpfile = 51, + statx = 52, + _, +}; + +// --------------------------------------------------------------------------- +// Structs (field order and widths follow linux/fuse.h exactly) +// --------------------------------------------------------------------------- + +pub const InHeader = extern struct { + len: u32, + opcode: u32, + unique: u64, + nodeid: u64, + uid: u32, + gid: u32, + pid: u32, + total_extlen: u16, + padding: u16, + + pub fn op(h: InHeader) Opcode { + return @enumFromInt(h.opcode); + } +}; + +pub const OutHeader = extern struct { + len: u32, + @"error": i32, + unique: u64, +}; + +pub const Attr = extern struct { + ino: u64 = 0, + size: u64 = 0, + blocks: u64 = 0, + atime: u64 = 0, + mtime: u64 = 0, + ctime: u64 = 0, + atimensec: u32 = 0, + mtimensec: u32 = 0, + ctimensec: u32 = 0, + mode: u32 = 0, + nlink: u32 = 0, + uid: u32 = 0, + gid: u32 = 0, + rdev: u32 = 0, + blksize: u32 = 0, + flags: u32 = 0, +}; + +pub const EntryOut = extern struct { + nodeid: u64 = 0, + generation: u64 = 0, + entry_valid: u64 = 0, + attr_valid: u64 = 0, + entry_valid_nsec: u32 = 0, + attr_valid_nsec: u32 = 0, + attr: Attr = .{}, +}; + +pub const AttrOut = extern struct { + attr_valid: u64 = 0, + attr_valid_nsec: u32 = 0, + dummy: u32 = 0, + attr: Attr = .{}, +}; + +pub const GetattrIn = extern struct { getattr_flags: u32, dummy: u32, fh: u64 }; + +pub const SetattrIn = extern struct { + valid: u32, + padding: u32, + fh: u64, + size: u64, + lock_owner: u64, + atime: u64, + mtime: u64, + ctime: u64, + atimensec: u32, + mtimensec: u32, + ctimensec: u32, + mode: u32, + unused4: u32, + uid: u32, + gid: u32, + unused5: u32, +}; + +pub const OpenIn = extern struct { flags: u32, open_flags: u32 }; +pub const OpenOut = extern struct { fh: u64 = 0, open_flags: u32 = 0, backing_id: i32 = 0 }; +pub const ReleaseIn = extern struct { fh: u64, flags: u32, release_flags: u32, lock_owner: u64 }; +pub const FlushIn = extern struct { fh: u64, unused: u32, padding: u32, lock_owner: u64 }; + +pub const ReadIn = extern struct { + fh: u64, + offset: u64, + size: u32, + read_flags: u32, + lock_owner: u64, + flags: u32, + padding: u32, +}; + +pub const WriteIn = extern struct { + fh: u64, + offset: u64, + size: u32, + write_flags: u32, + lock_owner: u64, + flags: u32, + padding: u32, +}; + +pub const WriteOut = extern struct { size: u32, padding: u32 = 0 }; +pub const CreateIn = extern struct { flags: u32, mode: u32, umask: u32, open_flags: u32 }; +pub const MkdirIn = extern struct { mode: u32, umask: u32 }; +pub const RenameIn = extern struct { newdir: u64 }; +pub const Rename2In = extern struct { newdir: u64, flags: u32, padding: u32 }; +pub const ForgetIn = extern struct { nlookup: u64 }; +pub const BatchForgetIn = extern struct { count: u32, dummy: u32 }; +pub const ForgetOne = extern struct { nodeid: u64, nlookup: u64 }; +pub const FsyncIn = extern struct { fh: u64, fsync_flags: u32, padding: u32 }; +pub const AccessIn = extern struct { mask: u32, padding: u32 }; +pub const InterruptIn = extern struct { unique: u64 }; +pub const LseekIn = extern struct { fh: u64, offset: u64, whence: u32, padding: u32 }; + +pub const Kstatfs = extern struct { + blocks: u64 = 0, + bfree: u64 = 0, + bavail: u64 = 0, + files: u64 = 0, + ffree: u64 = 0, + bsize: u32 = 0, + namelen: u32 = 0, + frsize: u32 = 0, + padding: u32 = 0, + spare: [6]u32 = [_]u32{0} ** 6, +}; + +pub const StatfsOut = extern struct { st: Kstatfs = .{} }; + +pub const InitIn = extern struct { + major: u32, + minor: u32, + max_readahead: u32, + flags: u32, + flags2: u32, + unused: [11]u32, +}; + +/// 64 bytes; the kernel accepts this size whenever the answered minor >= 23. +pub const InitOut = extern struct { + major: u32 = kernel_version, + minor: u32 = kernel_minor, + max_readahead: u32 = 0, + flags: u32 = 0, + max_background: u16 = 0, + congestion_threshold: u16 = 0, + max_write: u32 = 0, + time_gran: u32 = 0, + max_pages: u16 = 0, + map_alignment: u16 = 0, + flags2: u32 = 0, + max_stack_depth: u32 = 0, + request_timeout: u16 = 0, + unused: [11]u16 = [_]u16{0} ** 11, +}; + +/// Fixed 24-byte head of `fuse_dirent`; the name follows, padded to 8 bytes. +pub const Dirent = extern struct { ino: u64, off: u64, namelen: u32, type: u32 }; + +comptime { + std.debug.assert(@sizeOf(InHeader) == 40); + std.debug.assert(@sizeOf(OutHeader) == 16); + std.debug.assert(@sizeOf(Attr) == 88); + std.debug.assert(@sizeOf(EntryOut) == 128); + std.debug.assert(@sizeOf(AttrOut) == 104); + std.debug.assert(@sizeOf(GetattrIn) == 16); + std.debug.assert(@sizeOf(SetattrIn) == 88); + std.debug.assert(@sizeOf(OpenIn) == 8); + std.debug.assert(@sizeOf(OpenOut) == 16); + std.debug.assert(@sizeOf(ReleaseIn) == 24); + std.debug.assert(@sizeOf(FlushIn) == 24); + std.debug.assert(@sizeOf(ReadIn) == 40); + std.debug.assert(@sizeOf(WriteIn) == 40); + std.debug.assert(@sizeOf(WriteOut) == 8); + std.debug.assert(@sizeOf(CreateIn) == 16); + std.debug.assert(@sizeOf(MkdirIn) == 8); + std.debug.assert(@sizeOf(RenameIn) == 8); + std.debug.assert(@sizeOf(Rename2In) == 16); + std.debug.assert(@sizeOf(ForgetIn) == 8); + std.debug.assert(@sizeOf(BatchForgetIn) == 8); + std.debug.assert(@sizeOf(ForgetOne) == 16); + std.debug.assert(@sizeOf(FsyncIn) == 16); + std.debug.assert(@sizeOf(AccessIn) == 8); + std.debug.assert(@sizeOf(InterruptIn) == 8); + std.debug.assert(@sizeOf(Kstatfs) == 80); + std.debug.assert(@sizeOf(StatfsOut) == 80); + std.debug.assert(@sizeOf(InitIn) == 64); + std.debug.assert(@sizeOf(InitOut) == 64); + std.debug.assert(@sizeOf(Dirent) == 24); + std.debug.assert(@sizeOf(LseekIn) == 24); +} + +// --------------------------------------------------------------------------- +// Request / reply helpers +// --------------------------------------------------------------------------- + +pub const Error = error{ Protocol, Io, TooManyPayloads }; + +pub const Request = struct { + header: InHeader, + /// Bytes after the header; a slice into the caller's buffer. + body: []const u8, +}; + +/// Reads one kernel request with a single `read(2)`. Returns null on ENODEV +/// (unmounted). Retries EINTR/EAGAIN/ENOENT. `buf` should be at least +/// `max_write + 4096` bytes and 8-byte aligned so `body()` can view it. +pub fn readRequest(fd: i32, buf: []u8) Error!?Request { + while (true) { + const rc = linux.read(fd, buf.ptr, buf.len); + switch (linux.errno(rc)) { + .SUCCESS => { + const n: usize = rc; + if (n < @sizeOf(InHeader)) return error.Protocol; + const header = std.mem.bytesToValue(InHeader, buf[0..@sizeOf(InHeader)]); + if (header.len != n) return error.Protocol; + return .{ .header = header, .body = buf[@sizeOf(InHeader)..n] }; + }, + .INTR, .AGAIN, .NOENT => continue, + .NODEV => return null, + else => return error.Io, + } + } +} + +/// Maximum number of payload slices a single `reply` can carry. +pub const max_payloads = 7; + +/// Success reply: `OutHeader` followed by the concatenated `payloads`, sent in +/// one `writev(2)`. An ENOENT from the kernel means the request was +/// interrupted; the reply is dropped and this returns normally. +pub fn reply(fd: i32, unique: u64, payloads: []const []const u8) Error!void { + if (payloads.len > max_payloads) return error.TooManyPayloads; + var total: usize = @sizeOf(OutHeader); + for (payloads) |p| total += p.len; + if (total > std.math.maxInt(u32)) return error.Protocol; + const header = OutHeader{ .len = @intCast(total), .@"error" = 0, .unique = unique }; + var iov: [max_payloads + 1]std.posix.iovec_const = undefined; + iov[0] = .{ .base = @ptrCast(&header), .len = @sizeOf(OutHeader) }; + for (payloads, 1..) |p, i| iov[i] = .{ .base = p.ptr, .len = p.len }; + return writeAll(fd, &iov, payloads.len + 1, total); +} + +/// Error reply: an `OutHeader` carrying `-errno` and no payload. +pub fn replyError(fd: i32, unique: u64, err: linux.E) Error!void { + const code: i32 = @intCast(@intFromEnum(err)); + const header = OutHeader{ .len = @sizeOf(OutHeader), .@"error" = -code, .unique = unique }; + var iov = [_]std.posix.iovec_const{.{ .base = @ptrCast(&header), .len = @sizeOf(OutHeader) }}; + return writeAll(fd, &iov, 1, @sizeOf(OutHeader)); +} + +fn writeAll(fd: i32, iov: [*]const std.posix.iovec_const, count: usize, total: usize) Error!void { + while (true) { + const rc = linux.writev(fd, iov, count); + switch (linux.errno(rc)) { + .SUCCESS => return if (rc == total) {} else error.Protocol, + .INTR => continue, + .NOENT => return, // request was interrupted; reply dropped + else => return error.Io, + } + } +} + +/// Appends a `fuse_dirent` (head + name, padded to a multiple of 8) at +/// `buf[used.*..]`. Returns false and leaves `buf`/`used` unchanged if the +/// record does not fit. +pub fn addDirent(buf: []u8, used: *usize, ino: u64, off: u64, dtype: u32, name: []const u8) bool { + const raw = @sizeOf(Dirent) + name.len; + const rec = (raw + 7) & ~@as(usize, 7); + if (used.* > buf.len or buf.len - used.* < rec) return false; + const dst = buf[used.*..][0..rec]; + const head = Dirent{ .ino = ino, .off = off, .namelen = @intCast(name.len), .type = dtype }; + @memcpy(dst[0..@sizeOf(Dirent)], std.mem.asBytes(&head)); + @memcpy(dst[@sizeOf(Dirent)..raw], name); + @memset(dst[raw..rec], 0); + used.* += rec; + return true; +} + +/// Views the first `@sizeOf(T)` bytes of `req.body` as `T` (copy-free). +/// Fails with `error.Protocol` if the body is too short or misaligned. +pub fn body(comptime T: type, req: Request) Error!*const T { + if (req.body.len < @sizeOf(T)) return error.Protocol; + if (@intFromPtr(req.body.ptr) % @alignOf(T) != 0) return error.Protocol; + return @ptrCast(@alignCast(req.body.ptr)); +} + +/// The NUL-terminated string at `req.body[offset..]`, without the NUL. +pub fn nameAt(req: Request, offset: usize) Error![]const u8 { + if (offset > req.body.len) return error.Protocol; + const rest = req.body[offset..]; + const end = std.mem.indexOfScalar(u8, rest, 0) orelse return error.Protocol; + return rest[0..end]; +} + +/// The NUL-terminated string following a `T` body (or at offset 0 when +/// `T == void`), e.g. LOOKUP's name (`void`) or MKDIR's name (`MkdirIn`). +pub fn nameAfter(comptime T: type, req: Request) Error![]const u8 { + const offset = if (T == void) 0 else @sizeOf(T); + return nameAt(req, offset); +} + +/// The string that follows `first` (obtained via `nameAt(req, offset)`), +/// for "old\0new\0" pairs such as RENAME's. +pub fn secondName(req: Request, first: []const u8, offset: usize) Error![]const u8 { + return nameAt(req, offset + first.len + 1); +} + +/// Builds the INIT reply per docs/DESIGN.md. +pub fn initReply(in: *const InitIn, max_write: u32) InitOut { + var out = InitOut{ + .major = kernel_version, + .minor = @min(kernel_minor, in.minor), + .max_readahead = in.max_readahead, + .flags = FUSE_ASYNC_READ | FUSE_ATOMIC_O_TRUNC | FUSE_BIG_WRITES | FUSE_AUTO_INVAL_DATA, + .max_background = 16, + .congestion_threshold = 12, + .max_write = max_write, + .time_gran = 1, + }; + if (in.flags & FUSE_MAX_PAGES != 0) { + out.flags |= FUSE_MAX_PAGES; + out.max_pages = 256; + } + return out; +} + +// --------------------------------------------------------------------------- +// Tests +// --------------------------------------------------------------------------- + +const testing = std.testing; + +test "struct sizes match linux/fuse.h" { + // The comptime block above is the real check; this makes it run under + // `zig test` even if the module is otherwise unreferenced. + try testing.expectEqual(@as(usize, 40), @sizeOf(InHeader)); + try testing.expectEqual(@as(usize, 64), @sizeOf(InitOut)); + try testing.expectEqual(@as(usize, 24), @sizeOf(Dirent)); + try testing.expectEqual(@as(u32, 26), @intFromEnum(Opcode.init)); + try testing.expectEqual(Opcode.statx, @as(Opcode, @enumFromInt(52))); +} + +test "addDirent pads records to 8 bytes and refuses when full" { + var buf: [1024]u8 = undefined; + var used: usize = 0; + const name = "abcdefghijklmnopq"; // 17 chars + var expect_total: usize = 0; + var n: usize = 1; + while (n <= 17) : (n += 1) { + const before = used; + try testing.expect(addDirent(&buf, &used, n, n, DT_REG, name[0..n])); + const rec = used - before; + try testing.expectEqual(@as(usize, 0), rec % 8); + try testing.expectEqual((24 + n + 7) & ~@as(usize, 7), rec); + // check head fields and NUL padding + const head = std.mem.bytesToValue(Dirent, buf[before..][0..24]); + try testing.expectEqual(n, head.ino); + try testing.expectEqual(@as(u32, @intCast(n)), head.namelen); + try testing.expectEqualStrings(name[0..n], buf[before + 24 ..][0..n]); + for (buf[before + 24 + n .. used]) |b| try testing.expectEqual(@as(u8, 0), b); + expect_total += rec; + } + try testing.expectEqual(expect_total, used); + + // A record that does not fit leaves everything untouched. + var small: [40]u8 = undefined; + var used2: usize = 0; + try testing.expect(addDirent(&small, &used2, 1, 1, DT_DIR, "0123456789abcdef")); // 24+16 = 40 + try testing.expectEqual(@as(usize, 40), used2); + try testing.expect(!addDirent(&small, &used2, 2, 2, DT_DIR, "x")); + try testing.expectEqual(@as(usize, 40), used2); + var tight: [31]u8 = undefined; + var used3: usize = 0; + try testing.expect(!addDirent(&tight, &used3, 1, 1, DT_REG, "a")); // needs 32 + try testing.expectEqual(@as(usize, 0), used3); +} + +test "body/nameAfter/secondName on hand-built requests" { + var buf: [128]u8 align(8) = undefined; + // LOOKUP(parent=1, "hello") + const name = "hello"; + const hdr = InHeader{ + .len = @intCast(@sizeOf(InHeader) + name.len + 1), + .opcode = @intFromEnum(Opcode.lookup), + .unique = 7, + .nodeid = root_id, + .uid = 1000, + .gid = 1000, + .pid = 42, + .total_extlen = 0, + .padding = 0, + }; + @memcpy(buf[0..40], std.mem.asBytes(&hdr)); + @memcpy(buf[40..45], name); + buf[45] = 0; + const req = Request{ .header = hdr, .body = buf[40..hdr.len] }; + try testing.expectEqual(Opcode.lookup, req.header.op()); + try testing.expectEqualStrings("hello", try nameAfter(void, req)); + try testing.expectError(error.Protocol, body(MkdirIn, Request{ .header = hdr, .body = buf[40..44] })); + + // MKDIR(mode=0o755) + "dir" + const mk = MkdirIn{ .mode = 0o755, .umask = 0o22 }; + @memcpy(buf[40..48], std.mem.asBytes(&mk)); + @memcpy(buf[48..51], "dir"); + buf[51] = 0; + const mreq = Request{ .header = hdr, .body = buf[40..52] }; + const got = try body(MkdirIn, mreq); + try testing.expectEqual(@as(u32, 0o755), got.mode); + try testing.expectEqualStrings("dir", try nameAfter(MkdirIn, mreq)); + + // RENAME(newdir) + "old\0new\0" + const rn = RenameIn{ .newdir = 9 }; + @memcpy(buf[40..48], std.mem.asBytes(&rn)); + @memcpy(buf[48..56], "old\x00new\x00"); + const rreq = Request{ .header = hdr, .body = buf[40..56] }; + try testing.expectEqual(@as(u64, 9), (try body(RenameIn, rreq)).newdir); + const old = try nameAfter(RenameIn, rreq); + try testing.expectEqualStrings("old", old); + try testing.expectEqualStrings("new", try secondName(rreq, old, @sizeOf(RenameIn))); + try testing.expectError(error.Protocol, secondName(rreq, "new", @sizeOf(RenameIn) + 4)); + + // Missing NUL and misalignment are protocol errors. + try testing.expectError(error.Protocol, nameAt(Request{ .header = hdr, .body = buf[48..51] }, 0)); + try testing.expectError(error.Protocol, body(MkdirIn, Request{ .header = hdr, .body = buf[41..57] })); +} + +test "initReply fields" { + var in = InitIn{ .major = 7, .minor = 45, .max_readahead = 131072, .flags = 0, .flags2 = 0, .unused = [_]u32{0} ** 11 }; + const a = initReply(&in, 1 << 20); + try testing.expectEqual(@as(u32, 7), a.major); + try testing.expectEqual(@as(u32, 31), a.minor); + try testing.expectEqual(@as(u32, 131072), a.max_readahead); + try testing.expectEqual(FUSE_ASYNC_READ | FUSE_ATOMIC_O_TRUNC | FUSE_BIG_WRITES | FUSE_AUTO_INVAL_DATA, a.flags); + try testing.expectEqual(@as(u16, 0), a.max_pages); + try testing.expectEqual(@as(u16, 16), a.max_background); + try testing.expectEqual(@as(u16, 12), a.congestion_threshold); + try testing.expectEqual(@as(u32, 1 << 20), a.max_write); + try testing.expectEqual(@as(u32, 1), a.time_gran); + + in.flags = FUSE_MAX_PAGES | FUSE_ASYNC_READ; + in.minor = 27; + const b = initReply(&in, 4096); + try testing.expectEqual(@as(u32, 27), b.minor); + try testing.expectEqual(FUSE_ASYNC_READ | FUSE_ATOMIC_O_TRUNC | FUSE_BIG_WRITES | FUSE_AUTO_INVAL_DATA | FUSE_MAX_PAGES, b.flags); + try testing.expectEqual(@as(u16, 256), b.max_pages); + try testing.expectEqual(@as(u32, 4096), b.max_write); +} + +fn makePipe() ![2]i32 { + var fds: [2]i32 = undefined; + if (linux.errno(linux.pipe2(&fds, .{ .CLOEXEC = true })) != .SUCCESS) return error.Io; + return fds; +} + +fn readExact(fd: i32, out: []u8) !void { + var got: usize = 0; + while (got < out.len) { + const rc = linux.read(fd, out[got..].ptr, out.len - got); + if (linux.errno(rc) != .SUCCESS or rc == 0) return error.Io; + got += rc; + } +} + +test "reply writes header + payloads through a pipe" { + const fds = try makePipe(); + defer _ = linux.close(fds[0]); + defer _ = linux.close(fds[1]); + + const oo = OpenOut{ .fh = 0x1234, .open_flags = FOPEN_DIRECT_IO }; + try reply(fds[1], 99, &.{ std.mem.asBytes(&oo), "tail" }); + + var out: [16 + 16 + 4]u8 = undefined; + try readExact(fds[0], &out); + const h = std.mem.bytesToValue(OutHeader, out[0..16]); + try testing.expectEqual(@as(u32, 36), h.len); + try testing.expectEqual(@as(i32, 0), h.@"error"); + try testing.expectEqual(@as(u64, 99), h.unique); + try testing.expectEqualSlices(u8, std.mem.asBytes(&oo), out[16..32]); + try testing.expectEqualStrings("tail", out[32..36]); + + // Empty payload list: header only. + try reply(fds[1], 5, &.{}); + var only: [16]u8 = undefined; + try readExact(fds[0], &only); + try testing.expectEqual(@as(u32, 16), std.mem.bytesToValue(OutHeader, &only).len); + + var too_many: [max_payloads + 1][]const u8 = undefined; + for (&too_many) |*p| p.* = "x"; + try testing.expectError(error.TooManyPayloads, reply(fds[1], 1, &too_many)); +} + +test "replyError writes a negative errno" { + const fds = try makePipe(); + defer _ = linux.close(fds[0]); + defer _ = linux.close(fds[1]); + + try replyError(fds[1], 0xdead_beef, .NOENT); + var out: [16]u8 = undefined; + try readExact(fds[0], &out); + const h = std.mem.bytesToValue(OutHeader, &out); + try testing.expectEqual(@as(u32, 16), h.len); + try testing.expectEqual(@as(i32, -2), h.@"error"); + try testing.expectEqual(@as(u64, 0xdead_beef), h.unique); + + try replyError(fds[1], 1, .NOSYS); + try readExact(fds[0], &out); + try testing.expectEqual(-@as(i32, @intCast(@intFromEnum(linux.E.NOSYS))), std.mem.bytesToValue(OutHeader, &out).@"error"); +} + +test "readRequest parses one request from a pipe and rejects bad lengths" { + const fds = try makePipe(); + defer _ = linux.close(fds[0]); + defer _ = linux.close(fds[1]); + + var wire: [48]u8 align(8) = undefined; + const hdr = InHeader{ .len = 48, .opcode = @intFromEnum(Opcode.forget), .unique = 3, .nodeid = 2, .uid = 0, .gid = 0, .pid = 0, .total_extlen = 0, .padding = 0 }; + @memcpy(wire[0..40], std.mem.asBytes(&hdr)); + @memcpy(wire[40..48], std.mem.asBytes(&ForgetIn{ .nlookup = 11 })); + try testing.expectEqual(@as(usize, 48), linux.write(fds[1], &wire, wire.len)); + + var buf: [4096]u8 align(8) = undefined; + const req = (try readRequest(fds[0], &buf)) orelse return error.Io; + try testing.expectEqual(Opcode.forget, req.header.op()); + try testing.expectEqual(@as(u64, 2), req.header.nodeid); + try testing.expectEqual(@as(u64, 11), (try body(ForgetIn, req)).nlookup); + + // Header length disagreeing with what was read is a protocol error. + var bad = wire; + std.mem.bytesAsValue(InHeader, bad[0..40]).len = 40; + try testing.expectEqual(@as(usize, 48), linux.write(fds[1], &bad, bad.len)); + try testing.expectError(error.Protocol, readRequest(fds[0], &buf)); +} diff --git a/9ns/src/main.zig b/9ns/src/main.zig new file mode 100644 index 0000000..26bd699 --- /dev/null +++ b/9ns/src/main.zig @@ -0,0 +1,444 @@ +//! 9ns: mount a 9P2000 tree into a fresh user+mount namespace via FUSE +//! and run a program inside it. +//! +//! Exit codes: the child's status (128+sig if signalled); 125 for 9ns's +//! own failures (usage, connect, attach, namespace/mount); 126/127 for exec +//! failures. + +const std = @import("std"); +const linux = std.os.linux; +const ns = @import("ns.zig"); +const nine = @import("nine.zig"); +const bridge = @import("bridge.zig"); + +const version_string = "9ns 0.1.0"; + +const usage_text = + \\Usage: 9ns [options] -- PROGRAM [ARGS...] + \\Transport (exactly one): + \\ --unix PATH Unix stream socket + \\ --tcp IP:PORT TCP (IPv4/IPv6 literal) + \\ --fd N already-connected inherited descriptor + \\ --spawn CMD run CMD (via /bin/sh -c) with a socketpair on its stdin/stdout + \\Options: + \\ --mount PATH mountpoint inside the new namespace (default /mnt/9p) + \\ --uname NAME 9P user name (default $USER, else "none") + \\ --aname NAME 9P tree to attach (default "") + \\ --msize BYTES maximum 9P message size to request (default 131072) + \\ --cache SECONDS attr/entry cache validity, may be fractional (default 1) + \\ --no-direct-io let the kernel cache file pages (trusts stat length) + \\ --debug trace FUSE and 9P operations on stderr + \\ --help, --version + \\PROGRAM defaults to $SHELL (else /bin/sh). The mountpoint is exported as $NINE_MOUNT. + \\ +; + +const own_failure: u8 = 125; +/// Largest 9P message size we agree to request: the session allocates two +/// buffers of this size up front, before the server negotiates it down. +const max_msize: u32 = 16 * 1024 * 1024; + +/// Write `text` to stdout (informational output such as --help); errors are +/// ignored, there is nowhere better to report them. +fn printStdout(text: []const u8) void { + var off: usize = 0; + while (off < text.len) { + const rc = linux.write(1, text[off..].ptr, text.len - off); + switch (linux.errno(rc)) { + .SUCCESS => off += rc, + .INTR => continue, + else => return, + } + } +} + +const Config = struct { + address: ?nine.Address = null, + spawn_cmd: ?[]const u8 = null, + mount: []const u8 = "/mnt/9p", + uname: ?[]const u8 = null, + aname: []const u8 = "", + msize: u32 = 131072, + cache_ns: u64 = 1_000_000_000, + direct_io: bool = true, + debug: bool = false, + /// Empty means "default program". + program: []const []const u8 = &.{}, +}; + +const ParseResult = union(enum) { + run: Config, + /// Usage error, already reported on stderr; exit with this status. + exit: u8, + /// --help/--version: text for stdout, then exit 0. Printing is left to + /// `main` so that no test path writes to fd 1 (under `zig build test` + /// that is the test runner's protocol pipe). + info: []const u8, +}; + +fn usageError(comptime fmt: []const u8, args: anytype) ParseResult { + std.debug.print("9ns: " ++ fmt ++ "\n(try 9ns --help)\n", args); + return .{ .exit = own_failure }; +} + +fn parseArgs(arena: std.mem.Allocator, args: []const [:0]const u8) !ParseResult { + var cfg = Config{}; + var transports: usize = 0; + var i: usize = 1; + var program_start: ?usize = null; + while (i < args.len) : (i += 1) { + const arg: []const u8 = args[i]; + if (std.mem.eql(u8, arg, "--")) { + program_start = i + 1; + break; + } + if (!std.mem.startsWith(u8, arg, "--")) { + // A single-dash word is a typo for an option, not a program. + if (arg.len > 1 and arg[0] == '-') return usageError("unknown option {s} (options start with --)", .{arg}); + // A bare word starts PROGRAM, as if "--" were given. + program_start = i; + break; + } + // Split "--opt=value". + var name = arg; + var inline_value: ?[]const u8 = null; + if (std.mem.indexOfScalar(u8, arg, '=')) |eq| { + name = arg[0..eq]; + inline_value = arg[eq + 1 ..]; + } + const Opt = enum { unix, tcp, fd, spawn, mount, uname, aname, msize, cache, @"no-direct-io", debug, help, version, unknown }; + const opt = std.meta.stringToEnum(Opt, name[2..]) orelse .unknown; + switch (opt) { + .@"no-direct-io", .debug, .help, .version => if (inline_value != null) return usageError("{s} takes no value", .{name}), + .unknown => return usageError("unknown option {s}", .{name}), + else => {}, + } + const value: []const u8 = switch (opt) { + .@"no-direct-io", .debug, .help, .version, .unknown => "", + else => inline_value orelse blk: { + i += 1; + if (i >= args.len) return usageError("{s} needs a value", .{name}); + break :blk args[i]; + }, + }; + switch (opt) { + .unix => { + if (value.len == 0) return usageError("--unix wants a socket path", .{}); + cfg.address = .{ .unix = value }; + transports += 1; + }, + .tcp => { + cfg.address = parseTcp(value) orelse return usageError("--tcp wants IP:PORT (IPv6 as [ADDR]:PORT), got '{s}'", .{value}); + transports += 1; + }, + .fd => { + const n = std.fmt.parseInt(i32, value, 10) catch return usageError("--fd wants a number, got '{s}'", .{value}); + if (n < 0) return usageError("--fd wants a non-negative number", .{}); + cfg.address = .{ .fd = n }; + transports += 1; + }, + .spawn => { + if (value.len == 0) return usageError("--spawn wants a command", .{}); + cfg.spawn_cmd = value; + transports += 1; + }, + .mount => { + if (value.len == 0) return usageError("--mount wants a path", .{}); + cfg.mount = value; + }, + .uname => cfg.uname = value, + .aname => cfg.aname = value, + .msize => { + cfg.msize = std.fmt.parseInt(u32, value, 10) catch return usageError("--msize wants a number, got '{s}'", .{value}); + if (cfg.msize < 4096 or cfg.msize > max_msize) return usageError("--msize must be between 4096 and {d}", .{max_msize}); + }, + .cache => { + const secs = std.fmt.parseFloat(f64, value) catch return usageError("--cache wants seconds, got '{s}'", .{value}); + if (!(secs >= 0) or secs > 1e9) return usageError("--cache out of range", .{}); + cfg.cache_ns = @intFromFloat(secs * 1e9); + }, + .@"no-direct-io" => cfg.direct_io = false, + .debug => cfg.debug = true, + .help => return .{ .info = usage_text }, + .version => return .{ .info = version_string ++ "\n" }, + .unknown => unreachable, + } + } + if (transports == 0) return usageError("one transport is required (--unix, --tcp, --fd or --spawn)", .{}); + if (transports > 1) return usageError("exactly one transport is allowed", .{}); + if (program_start) |start| { + const prog = try arena.alloc([]const u8, args.len - start); + for (args[start..], 0..) |a, j| prog[j] = a; + cfg.program = prog; + } + return .{ .run = cfg }; +} + +fn parseTcp(spec: []const u8) ?nine.Address { + const colon = std.mem.lastIndexOfScalar(u8, spec, ':') orelse return null; + var host = spec[0..colon]; + if (host.len >= 2 and host[0] == '[' and host[host.len - 1] == ']') host = host[1 .. host.len - 1]; + if (host.len == 0) return null; + const port = std.fmt.parseInt(u16, spec[colon + 1 ..], 10) catch return null; + return .{ .tcp = .{ .host = host, .port = port } }; +} + +/// `--spawn`: run CMD under /bin/sh with one end of a socketpair as its +/// stdin/stdout; the other end is the 9P transport. +const Server = struct { pid: i32, fd: i32 }; + +fn spawnServer(cmd: [:0]const u8, envp: [*:null]const ?[*:0]const u8) !Server { + var sv: [2]i32 = undefined; + switch (linux.errno(linux.socketpair(linux.AF.UNIX, linux.SOCK.STREAM | linux.SOCK.CLOEXEC, 0, &sv))) { + .SUCCESS => {}, + else => |e| { + std.debug.print("9ns: socketpair: E{t}\n", .{e}); + return error.SystemResources; + }, + } + const rc = linux.fork(); + switch (linux.errno(rc)) { + .SUCCESS => {}, + else => |e| { + _ = linux.close(sv[0]); + _ = linux.close(sv[1]); + std.debug.print("9ns: fork: E{t}\n", .{e}); + return error.SystemResources; + }, + } + if (rc == 0) { + // Child: dup2 clears CLOEXEC on 0 and 1; everything else is CLOEXEC. + if (linux.errno(linux.dup2(sv[1], 0)) != .SUCCESS or linux.errno(linux.dup2(sv[1], 1)) != .SUCCESS) linux.exit_group(125); + // The server shares our process group, so a Ctrl-C meant for the + // program would kill it and take the mount down with it: ignore the + // tty signals (inherited across exec). SIGPIPE goes back to its + // default, we only ignore it for ourselves. + ignoreSignal(.INT); + ignoreSignal(.QUIT); + defaultSignal(.PIPE); + const argv = [_:null]?[*:0]const u8{ "sh", "-c", cmd.ptr }; + const e = linux.errno(linux.execve("/bin/sh", &argv, envp)); + std.debug.print("9ns: --spawn: exec /bin/sh: E{t}\n", .{e}); + linux.exit_group(127); + } + _ = linux.close(sv[1]); + return .{ .pid = @intCast(rc), .fd = sv[0] }; +} + +fn stopServer(server: ?Server) void { + const s = server orelse return; + _ = linux.kill(s.pid, .TERM); + ns.reapAny(s.pid); +} + +/// Fail early (before spawning servers or forking) if /dev/fuse is unusable. +fn probeFuseDevice() bool { + const rc = linux.open("/dev/fuse", .{ .ACCMODE = .RDWR, .CLOEXEC = true }, 0); + switch (linux.errno(rc)) { + .SUCCESS => { + _ = linux.close(@intCast(rc)); + return true; + }, + .NOENT => std.debug.print("9ns: /dev/fuse: ENOENT (is the fuse module loaded? try: modprobe fuse)\n", .{}), + else => |e| std.debug.print("9ns: open /dev/fuse: E{t}\n", .{e}), + } + return false; +} + +fn ignoreSignal(sig: linux.SIG) void { + const ign = linux.Sigaction{ .handler = .{ .handler = linux.SIG.IGN }, .mask = linux.sigemptyset(), .flags = 0 }; + std.posix.sigaction(sig, &ign, null); +} + +fn defaultSignal(sig: linux.SIG) void { + const dfl = linux.Sigaction{ .handler = .{ .handler = linux.SIG.DFL }, .mask = linux.sigemptyset(), .flags = 0 }; + std.posix.sigaction(sig, &dfl, null); +} + +/// `--fd N`: the descriptor is ours from now on; it must not leak into the +/// program (which could otherwise read 9P replies meant for us). Fails on a +/// bad descriptor, which is the earliest place to report it. +fn adoptFd(fd: i32) bool { + switch (linux.errno(linux.fcntl(fd, linux.F.SETFD, linux.FD_CLOEXEC))) { + .SUCCESS => return true, + else => |e| { + std.debug.print("9ns: --fd {d}: E{t}\n", .{ fd, e }); + return false; + }, + } +} + +fn describeAddress(a: nine.Address, buf: []u8) []const u8 { + return switch (a) { + .unix => |p| std.fmt.bufPrint(buf, "unix socket {s}", .{p}) catch "unix socket", + .tcp => |t| std.fmt.bufPrint(buf, "tcp {s}:{d}", .{ t.host, t.port }) catch "tcp", + .fd => |fd| std.fmt.bufPrint(buf, "fd {d}", .{fd}) catch "fd", + }; +} + +pub fn main(init: std.process.Init) !u8 { + const gpa = init.gpa; + const arena = init.arena.allocator(); + const args = try init.minimal.args.toSlice(arena); + const envp: [*:null]const ?[*:0]const u8 = init.minimal.environ.block.slice.ptr; + + var cfg = switch (try parseArgs(arena, args)) { + .exit => |code| return code, + .info => |text| { + printStdout(text); + return 0; + }, + .run => |c| c, + }; + + // Defaults that come from the environment. + if (cfg.program.len == 0) { + const env_shell = ns.getenv(envp, "SHELL") orelse ""; + const shell = if (env_shell.len == 0) "/bin/sh" else env_shell; + cfg.program = try arena.dupe([]const u8, &.{shell}); + } + const uname = cfg.uname orelse ns.getenv(envp, "USER") orelse "none"; + const mountpoint = ns.resolveMountpoint(gpa, cfg.mount) catch |err| { + std.debug.print("9ns: --mount {s}: {t}\n", .{ cfg.mount, err }); + return own_failure; + }; + defer gpa.free(mountpoint); + + if (!probeFuseDevice()) return own_failure; + + // Writes to a dead server socket must not kill us. + ignoreSignal(.PIPE); + + var server: ?Server = null; + var address: nine.Address = undefined; + if (cfg.spawn_cmd) |cmd| { + const cmd_z = try arena.dupeZ(u8, cmd); + server = spawnServer(cmd_z, envp) catch return own_failure; + ns.watchServer(server.?.pid); + address = .{ .fd = server.?.fd }; + } else { + address = cfg.address.?; + if (address == .fd and !adoptFd(address.fd)) return own_failure; + } + + var addr_buf: [256]u8 = undefined; + var session = nine.Session.connect(gpa, address, cfg.msize) catch |err| { + std.debug.print("9ns: connect to {s}: {t}\n", .{ describeAddress(address, &addr_buf), err }); + stopServer(server); + return own_failure; + }; + defer session.deinit(); + defer stopServer(server); + + _ = session.attach(0, uname, cfg.aname) catch |err| { + switch (err) { + error.Nine => std.debug.print("9ns: attach (uname={s}, aname='{s}'): {s}\n", .{ uname, cfg.aname, session.ename[0..session.ename_len] }), + else => std.debug.print("9ns: attach: {t}\n", .{err}), + } + return own_failure; + }; + if (cfg.debug) std.debug.print("9ns: attached to {s} (msize {d}), mounting on {s}\n", .{ describeAddress(address, &addr_buf), session.msize, mountpoint }); + + var child_pid: i32 = 0; + const stop_fd = ns.installSignals(&child_pid) catch return own_failure; + + const uid = linux.getuid(); + const gid = linux.getgid(); + const child = ns.spawn(gpa, .{ + .argv = cfg.program, + .envp = envp, + .mountpoint = mountpoint, + .uid = uid, + .gid = gid, + .max_read = bridge.max_write, + }) catch return own_failure; + + bridge.serve(gpa, child.fuse_fd, &session, 0, stop_fd, .{ + .uid = uid, + .gid = gid, + .attr_timeout_ns = cfg.cache_ns, + .direct_io = cfg.direct_io, + .debug = cfg.debug, + }) catch |err| switch (err) { + error.Closed => std.debug.print("9ns: 9P server connection closed\n", .{}), + else => std.debug.print("9ns: fuse: {t}\n", .{err}), + }; + + // Closing the device aborts the FUSE connection: anything still using + // the mount gets ENOTCONN instead of hanging on an unserved request. + _ = linux.close(child.fuse_fd); + + const status = ns.reapIfExited(child.pid) orelse ns.waitChild(child.pid) catch own_failure; + // An exec failure (126/127) is already in `status`; this prints its message. + _ = ns.reportExecFailure(child); + return status; +} + +test "parseTcp" { + const a = parseTcp("127.0.0.1:564").?; + try std.testing.expectEqualStrings("127.0.0.1", a.tcp.host); + try std.testing.expectEqual(@as(u16, 564), a.tcp.port); + const b = parseTcp("[::1]:9999").?; + try std.testing.expectEqualStrings("::1", b.tcp.host); + try std.testing.expectEqual(@as(u16, 9999), b.tcp.port); + try std.testing.expect(parseTcp("nohost") == null); + try std.testing.expect(parseTcp(":564") == null); + try std.testing.expect(parseTcp("1.2.3.4:") == null); + try std.testing.expect(parseTcp("1.2.3.4:70000") == null); +} + +test "parseArgs" { + const arena = std.testing.allocator; + { + const args = [_][:0]const u8{ "9ns", "--unix", "/s", "--cache", "0.5", "--msize=8192", "--no-direct-io", "--", "sh", "-c", "x" }; + const r = try parseArgs(arena, &args); + defer arena.free(r.run.program); + try std.testing.expectEqualStrings("/s", r.run.address.?.unix); + try std.testing.expectEqual(@as(u64, 500_000_000), r.run.cache_ns); + try std.testing.expectEqual(@as(u32, 8192), r.run.msize); + try std.testing.expect(!r.run.direct_io); + try std.testing.expectEqual(@as(usize, 3), r.run.program.len); + try std.testing.expectEqualStrings("x", r.run.program[2]); + } + { + const args = [_][:0]const u8{ "9ns", "--fd", "3" }; + const r = try parseArgs(arena, &args); + try std.testing.expectEqual(@as(i32, 3), r.run.address.?.fd); + try std.testing.expectEqual(@as(usize, 0), r.run.program.len); + try std.testing.expectEqualStrings("/mnt/9p", r.run.mount); + } + { + // Two transports, no transport, unknown option, missing value: all 125. + const two = [_][:0]const u8{ "9ns", "--fd", "3", "--unix", "/s" }; + try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &two)).exit); + const none = [_][:0]const u8{ "9ns", "--", "sh" }; + try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &none)).exit); + const unknown = [_][:0]const u8{ "9ns", "--bogus" }; + try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &unknown)).exit); + const missing = [_][:0]const u8{ "9ns", "--unix" }; + try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &missing)).exit); + const badcache = [_][:0]const u8{ "9ns", "--fd", "3", "--cache", "abc" }; + try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &badcache)).exit); + // Empty values, a single-dash typo, and an msize that would allocate gigabytes. + const emptyunix = [_][:0]const u8{ "9ns", "--unix=", "--", "sh" }; + try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &emptyunix)).exit); + const emptymount = [_][:0]const u8{ "9ns", "--fd", "3", "--mount", "" }; + try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &emptymount)).exit); + const singledash = [_][:0]const u8{ "9ns", "--fd", "3", "-mount", "/x" }; + try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &singledash)).exit); + const hugemsize = [_][:0]const u8{ "9ns", "--fd", "3", "--msize", "4294967295" }; + try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &hugemsize)).exit); + const okmsize = [_][:0]const u8{ "9ns", "--fd", "3", "--msize", "16777216" }; + try std.testing.expectEqual(@as(u32, 16777216), (try parseArgs(arena, &okmsize)).run.msize); + } + { + const ver = [_][:0]const u8{ "9ns", "--version" }; + try std.testing.expectEqualStrings(version_string ++ "\n", (try parseArgs(arena, &ver)).info); + const help = [_][:0]const u8{ "9ns", "--help" }; + try std.testing.expect(std.mem.startsWith(u8, (try parseArgs(arena, &help)).info, "Usage: 9ns")); + } +} + +test { + _ = ns; +} diff --git a/9ns/src/nine.zig b/9ns/src/nine.zig new file mode 100644 index 0000000..70633e6 --- /dev/null +++ b/9ns/src/nine.zig @@ -0,0 +1,756 @@ +//! Synchronous 9P2000 session over a blocking file descriptor. +//! +//! A thin RPC layer over `cloud9.Client` (push/take, allocation-free). One request +//! is outstanding at a time: the FUSE loop that drives this is single-threaded, so +//! every call here blocks until its reply (or the connection's death) arrives. +//! Fids are handed out from a free list; fid 0 is reserved for the root. +const std = @import("std"); +const cloud9 = @import("cloud9"); +const linux = std.os.linux; + +pub const Address = union(enum) { + unix: []const u8, + tcp: struct { host: []const u8, port: u16 }, + fd: i32, +}; + +/// A Stat whose every field means "leave unchanged" in a Twstat. +pub const dontcare = cloud9.Stat{ + .type = 0xFFFF, + .dev = 0xFFFF_FFFF, + .qid = .{ .type = 0xFF, .version = 0xFFFF_FFFF, .path = 0xFFFF_FFFF_FFFF_FFFF }, + .mode = 0xFFFF_FFFF, + .atime = 0xFFFF_FFFF, + .mtime = 0xFFFF_FFFF, + .length = 0xFFFF_FFFF_FFFF_FFFF, + .name = "", + .uid = "", + .gid = "", + .muid = "", +}; + +pub const Session = struct { + pub const Error = error{ Nine, Protocol, Io, Closed, Stopped, TooLarge, OutOfMemory }; + + pub const Walk = struct { nwqid: u16, wqid: [cloud9.max_welem]cloud9.Qid }; + pub const Open = struct { qid: cloud9.Qid, iounit: u32 }; + + gpa: std.mem.Allocator, + fd: i32, + client: cloud9.Client, + in_buf: []u8, + out_buf: []u8, + /// After `error.Nine`, the server's Rerror text (copied, bounded). + ename: [256]u8 = undefined, + ename_len: usize = 0, + /// Negotiated maximum message size. + msize: u32, + next_fid: u32 = 1, + free_fids: std.ArrayList(u32) = .empty, + /// Per-fid iounit learned from open/create (0 = none); used to chunk read/write. + iounits: std.AutoHashMapUnmanaged(u32, u32) = .empty, + /// Optional descriptor watched while waiting for a reply: when it becomes + /// readable (the bridge's "child exited" pipe) the pending rpc fails with + /// `error.Stopped` instead of blocking on a server that never answers. + stop_fd: i32 = -1, + + /// Connect to `address`, then negotiate the protocol version. + /// `msize` is the maximum message size to ask for (0 = the buffers' size). + pub fn connect(gpa: std.mem.Allocator, address: Address, msize: u32) !Session { + const want: u32 = if (msize == 0) 8192 else @max(msize, 24); + const fd = try openTransport(address); + errdefer if (address != .fd) { + _ = linux.close(fd); + }; + + const in_buf = try gpa.alloc(u8, want); + errdefer gpa.free(in_buf); + const out_buf = try gpa.alloc(u8, want); + errdefer gpa.free(out_buf); + + var s: Session = .{ + .gpa = gpa, + .fd = fd, + .client = .init(.{ .in = in_buf, .out = out_buf }), + .in_buf = in_buf, + .out_buf = out_buf, + .msize = want, + }; + const r = try s.rpc(.{ .version = .{ .msize = want } }); + if (!std.mem.eql(u8, r.version.version, "9P2000")) return error.Protocol; + s.msize = r.version.msize; + return s; + } + + /// Closes the descriptor and frees the buffers. Fids are not clunked. + pub fn deinit(s: *Session) void { + _ = linux.close(s.fd); + s.free_fids.deinit(s.gpa); + s.iounits.deinit(s.gpa); + s.gpa.free(s.in_buf); + s.gpa.free(s.out_buf); + s.* = undefined; + } + + pub fn attach(s: *Session, fid: u32, uname: []const u8, aname: []const u8) Error!cloud9.Qid { + const r = try s.rpc(.{ .attach = .{ .fid = fid, .uname = uname, .aname = aname } }); + return r.attach; + } + + /// Fid 0 is never handed out: it belongs to the root attach. + pub fn allocFid(s: *Session) u32 { + if (s.free_fids.pop()) |fid| return fid; + const fid = s.next_fid; + s.next_fid += 1; + return fid; + } + + /// Fids currently bound (excluding fid 0); a debugging aid for leak hunting. + pub fn fidsInUse(s: *const Session) usize { + return (s.next_fid - 1) - s.free_fids.items.len; + } + + pub fn freeFid(s: *Session, fid: u32) void { + _ = s.iounits.remove(fid); + // If the free list cannot grow the fid is simply leaked; the counter keeps going. + s.free_fids.append(s.gpa, fid) catch {}; + } + + /// Generic RPC. Result slices borrow the input buffer until the next call. + pub fn rpc(s: *Session, req: cloud9.Client.Request) Error!cloud9.Client.Result { + s.ename_len = 0; + _ = s.client.submit(req) catch |e| switch (e) { + error.NoTags, error.Handshake, error.Dead => return error.Protocol, + error.NoSpace, error.TooLarge => return error.TooLarge, + error.BadRequest => { + s.setEname("bad request"); + return error.Nine; + }, + }; + try s.flush(); + var tmp: [64 * 1024]u8 = undefined; + while (true) { + if (s.client.take()) |done| { + switch (done.result) { + .fail => |ename| { + s.setEname(ename); + return error.Nine; + }, + else => return done.result, + } + } + if (s.client.dead) return error.Protocol; + // After take() returned null the previous frame is gone, so the free + // space is at least what the pending frame still needs. + const room = s.client.in.len - s.client.in_len; + if (room == 0) return error.Protocol; + const n = try readSome(s.fd, s.stop_fd, tmp[0..@min(room, tmp.len)]); + if (n == 0) return error.Closed; + const pushed = s.client.push(tmp[0..n]); + if (pushed != n) return error.Protocol; + } + } + + /// Walk `names` from `fid` to `newfid`. A partial walk leaves `newfid` unbound + /// (9P semantics) and reports `error.Nine` with ename "file does not exist". + pub fn walk(s: *Session, fid: u32, newfid: u32, names: []const []const u8) Error!Walk { + const r = try s.rpc(.{ .walk = .{ .fid = fid, .newfid = newfid, .names = names } }); + if (r.walk.nwqid < names.len) { + s.setEname("file does not exist"); + return error.Nine; + } + return .{ .nwqid = r.walk.nwqid, .wqid = r.walk.wqid }; + } + + /// allocFid + zero-element walk. The fid is released again on failure. + pub fn clone(s: *Session, fid: u32) Error!u32 { + const newfid = s.allocFid(); + errdefer s.freeFid(newfid); + _ = try s.walk(fid, newfid, &.{}); + return newfid; + } + + pub fn open(s: *Session, fid: u32, mode: u8) Error!Open { + const r = try s.rpc(.{ .open = .{ .fid = fid, .mode = mode } }); + s.noteIounit(fid, r.open.iounit); + return .{ .qid = r.open.qid, .iounit = r.open.iounit }; + } + + pub fn create(s: *Session, fid: u32, name: []const u8, perm: u32, mode: u8) Error!Open { + const r = try s.rpc(.{ .create = .{ .fid = fid, .name = name, .perm = perm, .mode = mode } }); + s.noteIounit(fid, r.create.iounit); + return .{ .qid = r.create.qid, .iounit = r.create.iounit }; + } + + /// Reads into `buf`, chunking by min(maxRead, iounit) and stopping at the first + /// short read. Returns the number of bytes read (0 at end of file). + pub fn read(s: *Session, fid: u32, offset: u64, buf: []u8) Error!usize { + return readWith(s, rpc, fid, offset, buf, s.chunk(fid)); + } + + /// Writes `data`, chunking like `read` and stopping at the first short write. + pub fn write(s: *Session, fid: u32, offset: u64, data: []const u8) Error!usize { + return writeWith(s, rpc, fid, offset, data, s.chunkWrite(fid)); + } + + /// The returned Stat's strings (name/uid/gid/muid) borrow the session's input + /// buffer: they are valid only until the next rpc. Copy what must outlive it. + pub fn stat(s: *Session, fid: u32) Error!cloud9.Stat { + const r = try s.rpc(.{ .stat = .{ .fid = fid } }); + return r.stat; + } + + pub fn wstat(s: *Session, fid: u32, st: cloud9.Stat) Error!void { + _ = try s.rpc(.{ .wstat = .{ .fid = fid, .stat = st } }); + } + + /// Frees the fid locally even when the server reports an error. + pub fn clunk(s: *Session, fid: u32) Error!void { + defer s.freeFid(fid); + _ = try s.rpc(.{ .clunk = .{ .fid = fid } }); + } + + /// Frees the fid locally even when the server reports an error. + pub fn remove(s: *Session, fid: u32) Error!void { + defer s.freeFid(fid); + _ = try s.rpc(.{ .remove = .{ .fid = fid } }); + } + + /// Maps the last Rerror text to an errno (case-insensitive substring match). + pub fn errno(s: *const Session) linux.E { + return enameToErrno(s.ename[0..s.ename_len]); + } + + // -- internals -------------------------------------------------------------- + + fn setEname(s: *Session, text: []const u8) void { + const n = @min(text.len, 255); + @memcpy(s.ename[0..n], text[0..n]); + s.ename_len = n; + } + + fn noteIounit(s: *Session, fid: u32, iounit: u32) void { + if (iounit == 0) { + _ = s.iounits.remove(fid); + } else { + s.iounits.put(s.gpa, fid, iounit) catch {}; + } + } + + fn chunk(s: *Session, fid: u32) u32 { + return chunkSize(s.client.maxRead(), s.iounits.get(fid) orelse 0); + } + + fn chunkWrite(s: *Session, fid: u32) u32 { + return chunkSize(s.client.maxWrite(), s.iounits.get(fid) orelse 0); + } + + /// Writes everything in the client's output buffer to the socket. + fn flush(s: *Session) Error!void { + while (s.client.output().len != 0) { + const out = s.client.output(); + const rc = linux.write(s.fd, out.ptr, out.len); + switch (linux.errno(rc)) { + .SUCCESS => { + if (rc == 0) return error.Closed; + s.client.wrote(rc); + }, + .INTR, .AGAIN => continue, + .PIPE, .CONNRESET => return error.Closed, + else => return error.Io, + } + } + } +}; + +fn chunkSize(max: u32, iounit: u32) u32 { + if (iounit != 0 and iounit < max) return iounit; + return max; +} + +/// Chunked read over any rpc-shaped function (injected so the loop is testable). +fn readWith( + s: anytype, + comptime rpcFn: anytype, + fid: u32, + offset: u64, + buf: []u8, + max_chunk: u32, +) Session.Error!usize { + if (max_chunk == 0) return error.Protocol; + var done: usize = 0; + while (done < buf.len) { + const want: u32 = @intCast(@min(buf.len - done, max_chunk)); + const r = try rpcFn(s, .{ .read = .{ .fid = fid, .offset = offset + done, .count = want } }); + const data = r.read; + @memcpy(buf[done..][0..data.len], data); + done += data.len; + if (data.len < want) break; + } + return done; +} + +/// Chunked write over any rpc-shaped function. +fn writeWith( + s: anytype, + comptime rpcFn: anytype, + fid: u32, + offset: u64, + data: []const u8, + max_chunk: u32, +) Session.Error!usize { + if (max_chunk == 0) return error.Protocol; + var done: usize = 0; + while (done < data.len) { + const want: usize = @min(data.len - done, max_chunk); + const r = try rpcFn(s, .{ .write = .{ .fid = fid, .offset = offset + done, .data = data[done..][0..want] } }); + done += r.write; + if (r.write < want) break; + } + return done; +} + +fn readSome(fd: i32, stop_fd: i32, buf: []u8) Session.Error!usize { + while (true) { + if (stop_fd >= 0) { + var pfds = [_]linux.pollfd{ + .{ .fd = fd, .events = linux.POLL.IN, .revents = 0 }, + .{ .fd = stop_fd, .events = linux.POLL.IN, .revents = 0 }, + }; + const prc = linux.poll(&pfds, pfds.len, -1); + switch (linux.errno(prc)) { + .SUCCESS => {}, + .INTR, .AGAIN => continue, + else => return error.Io, + } + if (pfds[1].revents != 0 and pfds[0].revents == 0) return error.Stopped; + } + const rc = linux.read(fd, buf.ptr, buf.len); + switch (linux.errno(rc)) { + .SUCCESS => return rc, + .INTR, .AGAIN => continue, + .CONNRESET => return error.Closed, + else => return error.Io, + } + } +} + +/// Rerror text → errno, per docs/DESIGN.md (first match wins). +pub fn enameToErrno(ename: []const u8) linux.E { + const Rule = struct { needle: []const u8, err: linux.E }; + const rules = [_]Rule{ + .{ .needle = "not exist", .err = .NOENT }, + .{ .needle = "not found", .err = .NOENT }, + .{ .needle = "no such", .err = .NOENT }, + .{ .needle = "exists", .err = .EXIST }, + .{ .needle = "not empty", .err = .NOTEMPTY }, + .{ .needle = "not a dir", .err = .NOTDIR }, + .{ .needle = "is a dir", .err = .ISDIR }, + .{ .needle = "permission", .err = .ACCES }, + .{ .needle = "denied", .err = .ACCES }, + .{ .needle = "read-only", .err = .ROFS }, + .{ .needle = "read only", .err = .ROFS }, + .{ .needle = "readonly", .err = .ROFS }, + .{ .needle = "no space", .err = .NOSPC }, + .{ .needle = "not allowed", .err = .PERM }, + .{ .needle = "not permitted", .err = .PERM }, + .{ .needle = "cannot", .err = .PERM }, + .{ .needle = "fid", .err = .BADF }, + .{ .needle = "bad offset", .err = .INVAL }, + .{ .needle = "invalid", .err = .INVAL }, + .{ .needle = "bad ", .err = .INVAL }, + .{ .needle = "busy", .err = .BUSY }, + .{ .needle = "in use", .err = .BUSY }, + .{ .needle = "too long", .err = .NAMETOOLONG }, + .{ .needle = "not supported", .err = .OPNOTSUPP }, + .{ .needle = "unsupported", .err = .OPNOTSUPP }, + }; + for (rules) |rule| { + if (std.ascii.findIgnoreCase(ename, rule.needle) != null) return rule.err; + } + return .IO; +} + +// -- transport ------------------------------------------------------------------ + +fn openTransport(address: Address) !i32 { + switch (address) { + .fd => |fd| return fd, + .unix => |path| { + if (path.len == 0 or path.len >= 108) return error.NameTooLong; + var sa: linux.sockaddr.un = .{ .path = @splat(0) }; + @memcpy(sa.path[0..path.len], path); + const fd = try newSocket(linux.AF.UNIX, 0); + errdefer _ = linux.close(fd); + try doConnect(fd, @ptrCast(&sa), @sizeOf(linux.sockaddr.un)); + return fd; + }, + .tcp => |t| { + const ip = std.Io.net.IpAddress.parse(t.host, t.port) catch return error.InvalidAddress; + switch (ip) { + .ip4 => |a| { + const sa: linux.sockaddr.in = .{ + .port = std.mem.nativeToBig(u16, t.port), + .addr = @bitCast(a.bytes), + }; + const fd = try newSocket(linux.AF.INET, linux.IPPROTO.TCP); + errdefer _ = linux.close(fd); + setNodelay(fd); + try doConnect(fd, @ptrCast(&sa), @sizeOf(linux.sockaddr.in)); + return fd; + }, + .ip6 => |a| { + const sa: linux.sockaddr.in6 = .{ + .port = std.mem.nativeToBig(u16, t.port), + .flowinfo = 0, + .addr = a.bytes, + .scope_id = 0, + }; + const fd = try newSocket(linux.AF.INET6, linux.IPPROTO.TCP); + errdefer _ = linux.close(fd); + setNodelay(fd); + try doConnect(fd, @ptrCast(&sa), @sizeOf(linux.sockaddr.in6)); + return fd; + }, + } + }, + } +} + +fn newSocket(domain: u32, protocol: u32) !i32 { + const rc = linux.socket(domain, linux.SOCK.STREAM | linux.SOCK.CLOEXEC, protocol); + switch (linux.errno(rc)) { + .SUCCESS => return @intCast(rc), + .MFILE, .NFILE => return error.ProcessFdQuotaExceeded, + .AFNOSUPPORT, .PROTONOSUPPORT => return error.AddressFamilyNotSupported, + .ACCES => return error.AccessDenied, + .NOMEM, .NOBUFS => return error.SystemResources, + else => return error.Unexpected, + } +} + +fn setNodelay(fd: i32) void { + const one: u32 = 1; + _ = linux.setsockopt(fd, linux.IPPROTO.TCP, linux.TCP.NODELAY, @ptrCast(&one), @sizeOf(u32)); +} + +fn doConnect(fd: i32, addr: *const linux.sockaddr, len: linux.socklen_t) !void { + while (true) { + const rc = linux.connect(fd, addr, len); + switch (linux.errno(rc)) { + .SUCCESS => return, + .INTR => continue, + .CONNREFUSED => return error.ConnectionRefused, + .NOENT, .NOTDIR => return error.FileNotFound, + .ACCES, .PERM => return error.AccessDenied, + .TIMEDOUT => return error.ConnectionTimedOut, + .NETUNREACH, .HOSTUNREACH => return error.NetworkUnreachable, + .ADDRNOTAVAIL => return error.AddressNotAvailable, + .AGAIN, .INPROGRESS => return error.WouldBlock, + else => return error.Unexpected, + } + } +} + +// -- tests ---------------------------------------------------------------------- + +const testing = std.testing; + +test { + testing.refAllDecls(@This()); +} + +test "ename → errno mapping" { + try testing.expectEqual(linux.E.NOENT, enameToErrno("file does not exist")); + try testing.expectEqual(linux.E.NOENT, enameToErrno("No Such File")); + try testing.expectEqual(linux.E.NOENT, enameToErrno("directory entry not found")); + try testing.expectEqual(linux.E.EXIST, enameToErrno("file already exists")); + try testing.expectEqual(linux.E.NOTEMPTY, enameToErrno("directory not empty")); + try testing.expectEqual(linux.E.NOTDIR, enameToErrno("not a directory")); + try testing.expectEqual(linux.E.ISDIR, enameToErrno("is a directory")); + try testing.expectEqual(linux.E.ACCES, enameToErrno("permission denied")); + try testing.expectEqual(linux.E.ACCES, enameToErrno("access denied")); + try testing.expectEqual(linux.E.ROFS, enameToErrno("read-only file system")); + try testing.expectEqual(linux.E.NOSPC, enameToErrno("no space left")); + try testing.expectEqual(linux.E.PERM, enameToErrno("operation not permitted")); + try testing.expectEqual(linux.E.PERM, enameToErrno("cannot remove root")); + try testing.expectEqual(linux.E.BADF, enameToErrno("unknown fid")); + try testing.expectEqual(linux.E.BADF, enameToErrno("fid in use")); // "fid" precedes "in use" + try testing.expectEqual(linux.E.INVAL, enameToErrno("bad offset")); + try testing.expectEqual(linux.E.INVAL, enameToErrno("invalid argument")); + try testing.expectEqual(linux.E.INVAL, enameToErrno("bad request")); + try testing.expectEqual(linux.E.BUSY, enameToErrno("device busy")); + try testing.expectEqual(linux.E.NAMETOOLONG, enameToErrno("name too long")); + try testing.expectEqual(linux.E.OPNOTSUPP, enameToErrno("operation not supported")); + try testing.expectEqual(linux.E.IO, enameToErrno("something odd happened")); + try testing.expectEqual(linux.E.IO, enameToErrno("")); +} + +test "fid allocator recycles and never hands out 0" { + var s: Session = undefined; + s.gpa = testing.allocator; + s.next_fid = 1; + s.free_fids = .empty; + s.iounits = .empty; + defer s.free_fids.deinit(s.gpa); + defer s.iounits.deinit(s.gpa); + + const a = s.allocFid(); + const b = s.allocFid(); + const c = s.allocFid(); + try testing.expectEqual(@as(u32, 1), a); + try testing.expectEqual(@as(u32, 2), b); + try testing.expectEqual(@as(u32, 3), c); + s.freeFid(b); + try testing.expectEqual(b, s.allocFid()); + s.freeFid(a); + s.freeFid(c); + const x = s.allocFid(); + const y = s.allocFid(); + try testing.expect((x == a and y == c) or (x == c and y == a)); + try testing.expectEqual(@as(u32, 4), s.allocFid()); + try testing.expect(a != 0 and b != 0 and c != 0); +} + +test "chunkSize honours iounit only when smaller" { + try testing.expectEqual(@as(u32, 100), chunkSize(100, 0)); + try testing.expectEqual(@as(u32, 40), chunkSize(100, 40)); + try testing.expectEqual(@as(u32, 100), chunkSize(100, 400)); +} + +/// Fake rpc for the chunked read/write loops: a file of `len` bytes where byte i == i & 0xff. +const FakeFile = struct { + len: usize, + calls: usize = 0, + max_count: u32 = 0, + short_write_at: ?usize = null, + scratch: [4096]u8 = undefined, + + fn rpc(f: *FakeFile, req: cloud9.Client.Request) Session.Error!cloud9.Client.Result { + f.calls += 1; + switch (req) { + .read => |r| { + f.max_count = @max(f.max_count, r.count); + if (r.offset >= f.len) return .{ .read = "" }; + const n: usize = @min(@as(usize, r.count), f.len - @as(usize, @intCast(r.offset))); + for (f.scratch[0..n], 0..) |*b, i| b.* = @truncate(r.offset + i); + return .{ .read = f.scratch[0..n] }; + }, + .write => |w| { + f.max_count = @max(f.max_count, @as(u32, @intCast(w.data.len))); + if (f.short_write_at) |at| { + if (w.offset + w.data.len > at) { + const n: usize = if (w.offset >= at) 0 else @intCast(at - w.offset); + return .{ .write = @intCast(n) }; + } + } + return .{ .write = @intCast(w.data.len) }; + }, + else => unreachable, + } + } +}; + +test "read chunks by max_chunk and stops at a short read" { + var f: FakeFile = .{ .len = 2500 }; + var buf: [4000]u8 = undefined; + const n = try readWith(&f, FakeFile.rpc, 7, 0, &buf, 1000); + try testing.expectEqual(@as(usize, 2500), n); + try testing.expectEqual(@as(usize, 3), f.calls); // 1000, 1000, 500 (short → stop) + try testing.expectEqual(@as(u32, 1000), f.max_count); + for (buf[0..n], 0..) |b, i| try testing.expectEqual(@as(u8, @truncate(i)), b); + + // Reading exactly up to a chunk boundary uses one call per chunk and no more. + f = .{ .len = 2000 }; + try testing.expectEqual(@as(usize, 2000), try readWith(&f, FakeFile.rpc, 7, 0, buf[0..2000], 1000)); + try testing.expectEqual(@as(usize, 2), f.calls); + + // Offset past EOF → 0. + f = .{ .len = 10 }; + try testing.expectEqual(@as(usize, 0), try readWith(&f, FakeFile.rpc, 7, 50, &buf, 1000)); +} + +test "write chunks and stops at a short write" { + var f: FakeFile = .{ .len = 0 }; + var data: [2500]u8 = undefined; + for (&data, 0..) |*b, i| b.* = @truncate(i); + try testing.expectEqual(@as(usize, 2500), try writeWith(&f, FakeFile.rpc, 7, 0, &data, 1000)); + try testing.expectEqual(@as(usize, 3), f.calls); + try testing.expectEqual(@as(u32, 1000), f.max_count); + + f = .{ .len = 0, .short_write_at = 1500 }; + try testing.expectEqual(@as(usize, 1500), try writeWith(&f, FakeFile.rpc, 7, 0, &data, 1000)); + try testing.expectEqual(@as(usize, 2), f.calls); +} + +// -- in-process server test --------------------------------------------------------- + +/// A tiny 9P2000 backend on a cloud9.Server: answers version/attach/walk/stat/open/ +/// read/clunk/remove with canned data. Runs in its own thread over a socketpair. +const FakeServer = struct { + fd: i32, + msize: u32, + max_read_count: u32 = 0, + file_len: usize, + + const file_qid: cloud9.Qid = .{ .type = 0, .version = 3, .path = 0x1234 }; + const dir_qid: cloud9.Qid = .{ .type = cloud9.qtdir, .version = 1, .path = 0x1 }; + + fn run(fs: *FakeServer) void { + fs.loop() catch |e| std.debug.print("fake server: {s}\n", .{@errorName(e)}); + _ = linux.close(fs.fd); + } + + fn loop(fs: *FakeServer) !void { + const gpa = testing.allocator; + const in = try gpa.alloc(u8, fs.msize); + defer gpa.free(in); + const out = try gpa.alloc(u8, fs.msize * 2); + defer gpa.free(out); + var srv: cloud9.Server = .init(.{ .in = in, .out = out }); + var tmp: [4096]u8 = undefined; + var data: [8192]u8 = undefined; + while (true) { + while (try srv.receive()) |req| { + const tag = req.tag; + switch (req.msg) { + .tversion => |m| try srv.negotiate(m.msize, m.version), + .tattach => try srv.reply(tag, .{ .rattach = .{ .qid = dir_qid } }), + .twalk => |m| { + var wq: [cloud9.max_welem]cloud9.Qid = @splat(dir_qid); + var n: u16 = 0; + for (m.wname[0..m.nwname]) |name| { + if (std.mem.eql(u8, name, "file")) { + wq[n] = file_qid; + } else if (std.mem.eql(u8, name, "dir")) { + wq[n] = dir_qid; + } else break; + n += 1; + } + if (n == 0 and m.nwname != 0) { + try srv.reply(tag, .{ .rerror = .{ .ename = "file does not exist" } }); + } else { + try srv.reply(tag, .{ .rwalk = .{ .nwqid = n, .wqid = wq } }); + } + }, + .tstat => try srv.reply(tag, .{ .rstat = .{ .stat = .{ + .type = 0, + .dev = 0, + .qid = file_qid, + .mode = 0o644, + .atime = 1, + .mtime = 2, + .length = fs.file_len, + .name = "file", + .uid = "u", + .gid = "g", + .muid = "u", + } } }), + .topen => |m| try srv.reply(tag, .{ .ropen = .{ .qid = file_qid, .iounit = if (m.mode == cloud9.owrite) 700 else 0 } }), + .tread => |m| { + fs.max_read_count = @max(fs.max_read_count, m.count); + var n: usize = 0; + if (m.offset < fs.file_len) n = @min(@as(usize, m.count), fs.file_len - @as(usize, @intCast(m.offset))); + n = @min(n, data.len); + for (data[0..n], 0..) |*b, i| b.* = @truncate(m.offset + i); + try srv.reply(tag, .{ .rread = .{ .data = data[0..n] } }); + }, + .twrite => |m| try srv.reply(tag, .{ .rwrite = .{ .count = @intCast(m.data.len) } }), + .tclunk => try srv.reply(tag, .rclunk), + .tremove => try srv.reply(tag, .{ .rerror = .{ .ename = "permission denied" } }), + .twstat => try srv.reply(tag, .rwstat), + // A flush is the test's "hang up now" signal. + .tflush => return, + else => try srv.reply(tag, .{ .rerror = .{ .ename = "not supported" } }), + } + srv.release(); + } + while (srv.output().len != 0) { + const o = srv.output(); + const rc = linux.write(fs.fd, o.ptr, o.len); + if (linux.errno(rc) != .SUCCESS) return error.Write; + srv.wrote(rc); + } + const rc = linux.read(fs.fd, &tmp, tmp.len); + if (linux.errno(rc) != .SUCCESS) return error.Read; + if (rc == 0) return; + if (srv.push(tmp[0..rc]) != rc) return error.Overflow; + } + } +}; + +test "session against an in-process cloud9.Server" { + var fds: [2]i32 = undefined; + try testing.expectEqual(linux.E.SUCCESS, linux.errno(linux.socketpair(linux.AF.UNIX, linux.SOCK.STREAM | linux.SOCK.CLOEXEC, 0, &fds))); + + var fs: FakeServer = .{ .fd = fds[1], .msize = 8192, .file_len = 20_000 }; + const th = try std.Thread.spawn(.{}, FakeServer.run, .{&fs}); + + var s = try Session.connect(testing.allocator, .{ .fd = fds[0] }, 8192); + defer { + s.deinit(); + th.join(); + } + try testing.expectEqual(@as(u32, 8192), s.msize); + + const root = try s.attach(0, "me", ""); + try testing.expectEqual(FakeServer.dir_qid.path, root.path); + + // Plain rpc + stat borrowing the input buffer. + const fid = s.allocFid(); + const w = try s.walk(0, fid, &.{"file"}); + try testing.expectEqual(@as(u16, 1), w.nwqid); + try testing.expectEqual(FakeServer.file_qid.path, w.wqid[0].path); + const st = try s.stat(fid); + try testing.expectEqualStrings("file", st.name); + try testing.expectEqual(@as(u64, 20_000), st.length); + + // Chunked read: 20000 bytes at maxRead = msize - 11 = 8181 per chunk. + _ = try s.open(fid, cloud9.oread); + const buf = try testing.allocator.alloc(u8, 30_000); + defer testing.allocator.free(buf); + const n = try s.read(fid, 0, buf); + try testing.expectEqual(@as(usize, 20_000), n); + for (buf[0..n], 0..) |b, i| try testing.expectEqual(@as(u8, @truncate(i)), b); + try testing.expectEqual(@as(u32, 8181), fs.max_read_count); + try testing.expectEqual(@as(usize, 0), try s.read(fid, 20_000, buf)); + + // iounit from open bounds the chunk. + const wfid = try s.clone(fid); + _ = try s.open(wfid, cloud9.owrite); + fs.max_read_count = 0; + _ = try s.read(wfid, 0, buf[0..3000]); + try testing.expectEqual(@as(u32, 700), fs.max_read_count); + try testing.expectEqual(@as(usize, 3000), try s.write(wfid, 0, buf[0..3000])); + + // Partial walk → error.Nine with a "not exist" ename → ENOENT. + const pfid = s.allocFid(); + try testing.expectError(error.Nine, s.walk(0, pfid, &.{ "dir", "nope" })); + try testing.expectEqual(linux.E.NOENT, s.errno()); + try testing.expectEqualStrings("file does not exist", s.ename[0..s.ename_len]); + s.freeFid(pfid); + + // Server Rerror → error.Nine, ename copied, fid freed by remove even on error. + try testing.expectError(error.Nine, s.remove(wfid)); + try testing.expectEqual(linux.E.ACCES, s.errno()); + try testing.expectEqual(wfid, s.allocFid()); // recycled + s.freeFid(wfid); + + // Unsupported op → "not supported" → ENOTSUP; a plain wstat succeeds. + try testing.expectError(error.Nine, s.rpc(.{ .auth = .{ .afid = 5, .uname = "me" } })); + try testing.expectEqual(linux.E.OPNOTSUPP, s.errno()); + try s.wstat(fid, dontcare); + try s.clunk(fid); + try testing.expectEqual(fid, s.allocFid()); + s.freeFid(fid); + + // A clone bound to a fid that then fails to walk must release the fid. + const before = s.next_fid; + const cfid = s.allocFid(); + s.freeFid(cfid); + try testing.expectError(error.Nine, s.walk(0, cfid, &.{"nope"})); + try testing.expectEqual(before, s.next_fid); + + // The server hanging up makes the pending rpc fail with error.Closed. + try testing.expectError(error.Closed, s.rpc(.{ .flush = .{ .oldtag = 0 } })); +} diff --git a/9ns/src/ns.zig b/9ns/src/ns.zig new file mode 100644 index 0000000..6da2c7f --- /dev/null +++ b/9ns/src/ns.zig @@ -0,0 +1,1082 @@ +//! Namespace and process plumbing for 9ns. +//! +//! Everything here is raw `std.os.linux` syscalls (no libc). The child side +//! of `spawn` runs between `fork` and `execve`; it does not allocate except +//! inside `ensureMountpoint` (the process is single-threaded by then, so the +//! inherited allocator is safe to use). +//! +//! Exit codes produced by the child before exec: 125 for namespace/mount +//! setup failures, 126 when the program was found but is not executable, +//! 127 when it was not found. + +const std = @import("std"); +const builtin = @import("builtin"); +const linux = std.os.linux; +const Allocator = std.mem.Allocator; +const E = linux.E; + +pub const Spawn = struct { + /// argv[0] is PATH-searched unless it contains '/'. + argv: []const []const u8, + /// Inherited environment; `NINE_MOUNT` is added or replaced. + envp: [*:null]const ?[*:0]const u8, + /// Absolute mountpoint (see `resolveMountpoint`). + mountpoint: []const u8, + uid: u32, + gid: u32, + max_read: u32, + /// When false the namespace is set up (including mountpoint shadowing) + /// but `/dev/fuse` is not opened and nothing is mounted; `Child.fuse_fd` + /// is then -1. Only for smoke tests. + mount_fuse: bool = true, +}; + +pub const Child = struct { + pid: i32, + /// The `/dev/fuse` connection backing the mount, opened by the child + /// inside its user namespace (the kernel refuses to mount a fuse fd that + /// was opened from another user namespace) and handed back over the + /// status socket with SCM_RIGHTS. Owned by the caller; CLOEXEC. + fuse_fd: i32, + /// Parent end of the status socket. The child reports an exec failure + /// on it (see `reportExecFailure`); it reads EOF once exec succeeded. + status_fd: i32, +}; + +/// Exit status used by the child for setup failures (matches 9ns's own). +pub const setup_failure_status: u8 = 125; +/// Refuse to shadow a directory with more entries than this. +pub const max_shadow_entries: usize = 4096; + +const default_path = "/usr/local/bin:/bin:/usr/bin"; +const path_max = 4096; + +// --------------------------------------------------------------------------- +// Mountpoint resolution +// --------------------------------------------------------------------------- + +/// Absolute path (relative paths resolved against cwd), duplicate slashes +/// collapsed, `.` and `..` components resolved lexically, no trailing slash. +/// `/` itself is rejected. +pub fn resolveMountpoint(gpa: Allocator, path: []const u8) ![:0]u8 { + var cwd_buf: [path_max]u8 = undefined; + var cwd: []const u8 = "/"; + if (path.len == 0 or path[0] != '/') { + const rc = linux.getcwd(&cwd_buf, cwd_buf.len); + switch (linux.errno(rc)) { + .SUCCESS => {}, + else => |e| { + std.debug.print("9ns: getcwd: E{t}\n", .{e}); + return error.Cwd; + }, + } + // rc counts the terminating NUL. + cwd = cwd_buf[0 .. rc - 1]; + } + return normalizePath(gpa, cwd, path); +} + +/// Pure part of `resolveMountpoint`: `cwd` is only used when `path` is relative. +fn normalizePath(gpa: Allocator, cwd: []const u8, path: []const u8) ![:0]u8 { + if (path.len == 0) return error.InvalidMountpoint; + var out: std.ArrayList(u8) = .empty; + defer out.deinit(gpa); + if (path[0] != '/') try appendComponents(gpa, &out, cwd); + try appendComponents(gpa, &out, path); + if (out.items.len == 0) return error.InvalidMountpoint; // "/" or equivalent + return out.toOwnedSliceSentinel(gpa, 0); +} + +fn appendComponents(gpa: Allocator, out: *std.ArrayList(u8), path: []const u8) !void { + var it = std.mem.tokenizeScalar(u8, path, '/'); + while (it.next()) |comp| { + if (std.mem.eql(u8, comp, ".")) continue; + if (std.mem.eql(u8, comp, "..")) { + // Pop the last component (lexically; "/.." stays "/"). + const idx = std.mem.lastIndexOfScalar(u8, out.items, '/') orelse 0; + out.shrinkRetainingCapacity(idx); + continue; + } + try out.append(gpa, '/'); + try out.appendSlice(gpa, comp); + } +} + +// --------------------------------------------------------------------------- +// Environment helpers +// --------------------------------------------------------------------------- + +/// Look a variable up in a raw envp block. +pub fn getenv(envp: [*:null]const ?[*:0]const u8, name: []const u8) ?[]const u8 { + var i: usize = 0; + while (envp[i]) |entry| : (i += 1) { + const kv = std.mem.span(entry); + if (kv.len > name.len and kv[name.len] == '=' and std.mem.eql(u8, kv[0..name.len], name)) { + return kv[name.len + 1 ..]; + } + } + return null; +} + +/// Every path `execve` should try for `name`, in order: just `name` if it +/// contains a '/', else `/` for each `$PATH` element (an empty +/// element means the current directory; `$PATH` unset falls back to +/// `/usr/local/bin:/bin:/usr/bin`). +pub fn pathCandidates(gpa: Allocator, envp: [*:null]const ?[*:0]const u8, name: []const u8) ![]const [:0]const u8 { + if (name.len == 0) return error.EmptyProgramName; + var list: std.ArrayList([:0]const u8) = .empty; + errdefer { + for (list.items) |c| gpa.free(c); + list.deinit(gpa); + } + if (std.mem.indexOfScalar(u8, name, '/') != null) { + try list.append(gpa, try gpa.dupeZ(u8, name)); + return list.toOwnedSlice(gpa); + } + const path = getenv(envp, "PATH") orelse default_path; + var it = std.mem.splitScalar(u8, path, ':'); + while (it.next()) |dir| { + const d = if (dir.len == 0) "." else dir; + try list.append(gpa, try std.fmt.allocPrintSentinel(gpa, "{s}/{s}", .{ d, name }, 0)); + } + return list.toOwnedSlice(gpa); +} + +/// First PATH candidate that is an executable regular file, or the name +/// itself when it contains a '/'. Provided for completeness; `spawn` simply +/// tries `execve` on every candidate instead. +pub fn findInPath(gpa: Allocator, envp: [*:null]const ?[*:0]const u8, name: []const u8) ![:0]u8 { + const cands = try pathCandidates(gpa, envp, name); + defer { + for (cands) |c| gpa.free(c); + gpa.free(cands); + } + for (cands) |c| { + var stx: linux.Statx = undefined; + const rc = linux.statx(linux.AT.FDCWD, c.ptr, 0, .{ .TYPE = true, .MODE = true }, &stx); + if (linux.errno(rc) != .SUCCESS) continue; + if (stx.mode & linux.S.IFMT != linux.S.IFREG) continue; + if (stx.mode & 0o111 == 0) continue; + return gpa.dupeZ(u8, c); + } + return error.FileNotFound; +} + +/// New envp block: every entry of `envp` except `NINE_MOUNT=...`, +/// followed by `NINE_MOUNT=`. +fn buildEnvp(gpa: Allocator, envp: [*:null]const ?[*:0]const u8, mountpoint: []const u8) ![:null]?[*:0]const u8 { + const key = "NINE_MOUNT="; + var keep: usize = 0; + var i: usize = 0; + while (envp[i]) |entry| : (i += 1) { + if (!std.mem.startsWith(u8, std.mem.span(entry), key)) keep += 1; + } + const out = try gpa.allocSentinel(?[*:0]const u8, keep + 1, null); + errdefer gpa.free(out); + var j: usize = 0; + i = 0; + while (envp[i]) |entry| : (i += 1) { + if (std.mem.startsWith(u8, std.mem.span(entry), key)) continue; + out[j] = entry; + j += 1; + } + const mount_entry = try std.fmt.allocPrintSentinel(gpa, key ++ "{s}", .{mountpoint}, 0); + out[j] = mount_entry.ptr; + return out; +} + +fn buildArgv(gpa: Allocator, argv: []const []const u8) ![:null]?[*:0]const u8 { + const out = try gpa.allocSentinel(?[*:0]const u8, argv.len, null); + for (argv, 0..) |a, i| out[i] = (try gpa.dupeZ(u8, a)).ptr; + return out; +} + +// --------------------------------------------------------------------------- +// Mountpoint policy +// --------------------------------------------------------------------------- + +/// Make sure `path` is a directory, inside the *current* mount namespace: +/// +/// * already a directory → done; +/// * else `mkdir`; on `EACCES`/`EPERM`/`EROFS` shadow the parent directory +/// with a tmpfs that re-exposes every existing entry (bind mounts for +/// directories and files, recreated symlinks) and `mkdir` inside it; +/// * anything else fails with the errno and a hint. +/// +/// Every failure prints `9ns: : E` to stderr before +/// returning. Meant to be called in the child of `spawn` (or from a +/// throwaway namespace: `unshare -Urm`). +pub fn ensureMountpoint(gpa: Allocator, path: [:0]const u8) !void { + if (fileType(linux.AT.FDCWD, path, false)) |ft| { + if (ft == .dir) return; + std.debug.print("9ns: mountpoint {s}: exists but is not a directory\n", .{path}); + return error.Mountpoint; + } + if (fileType(linux.AT.FDCWD, path, true) == .symlink) { + std.debug.print("9ns: mountpoint {s}: dangling symlink\n", .{path}); + return error.Mountpoint; + } + const mk = linux.errno(linux.mkdirat(linux.AT.FDCWD, path, 0o755)); + switch (mk) { + .SUCCESS => return, + .ACCES, .PERM, .ROFS => {}, + else => |e| { + std.debug.print("9ns: mkdir {s}: E{t} (pass --mount an existing directory)\n", .{ path, e }); + return error.Mountpoint; + }, + } + const parent = std.fs.path.dirname(path) orelse "/"; + if (std.mem.eql(u8, parent, "/") or isSameDirectory(parent, "/")) { + std.debug.print("9ns: mkdir {s}: E{t}; refusing to shadow / (pass --mount an existing directory)\n", .{ path, mk }); + return error.Mountpoint; + } + // The shadow rebuilds entries from /proc/self/fd//; a tmpfs + // over /proc (or a subtree of it) would take that away from itself. + if (std.mem.eql(u8, parent, "/proc") or std.mem.startsWith(u8, parent, "/proc/")) { + std.debug.print("9ns: mkdir {s}: E{t}; refusing to shadow {s} (pass --mount an existing directory)\n", .{ path, mk, parent }); + return error.Mountpoint; + } + const parent_z = try gpa.dupeZ(u8, parent); + defer gpa.free(parent_z); + try shadowDirectory(gpa, parent_z); + switch (linux.errno(linux.mkdirat(linux.AT.FDCWD, path, 0o755))) { + .SUCCESS => {}, + else => |e| { + std.debug.print("9ns: mkdir {s} (in shadow tmpfs): E{t}\n", .{ path, e }); + return error.Mountpoint; + }, + } +} + +const FileType = enum { dir, symlink, other }; + +/// True when both paths resolve (following symlinks, including magic ones +/// such as /proc/self/root) to the same inode. +fn isSameDirectory(a: []const u8, b: [*:0]const u8) bool { + var a_buf: [path_max]u8 = undefined; + const a_z = std.fmt.bufPrintZ(&a_buf, "{s}", .{a}) catch return false; + var sa: linux.Statx = undefined; + var sb: linux.Statx = undefined; + if (linux.errno(linux.statx(linux.AT.FDCWD, a_z, 0, .{ .INO = true }, &sa)) != .SUCCESS) return false; + if (linux.errno(linux.statx(linux.AT.FDCWD, b, 0, .{ .INO = true }, &sb)) != .SUCCESS) return false; + return sa.ino == sb.ino and sa.dev_major == sb.dev_major and sa.dev_minor == sb.dev_minor; +} + +fn fileType(dirfd: i32, name: [*:0]const u8, nofollow: bool) ?FileType { + var stx: linux.Statx = undefined; + const flags: u32 = if (nofollow) linux.AT.SYMLINK_NOFOLLOW else 0; + const rc = linux.statx(dirfd, name, flags, .{ .TYPE = true }, &stx); + if (linux.errno(rc) != .SUCCESS) return null; + return switch (stx.mode & linux.S.IFMT) { + linux.S.IFDIR => .dir, + linux.S.IFLNK => .symlink, + else => .other, + }; +} + +const Entry = struct { name: [:0]u8, kind: FileType }; + +/// Read every entry of the directory open at `fd` (excluding `.` and `..`). +fn listDir(gpa: Allocator, fd: i32, dirpath: []const u8) ![]Entry { + var list: std.ArrayList(Entry) = .empty; + errdefer { + for (list.items) |e| gpa.free(e.name); + list.deinit(gpa); + } + var buf: [32 * 1024]u8 align(@alignOf(linux.dirent64)) = undefined; + while (true) { + const rc = linux.getdents64(fd, &buf, buf.len); + switch (linux.errno(rc)) { + .SUCCESS => {}, + else => |e| { + std.debug.print("9ns: getdents64 {s}: E{t}\n", .{ dirpath, e }); + return error.Mountpoint; + }, + } + if (rc == 0) break; + var off: usize = 0; + while (off < rc) { + const d: *align(1) const linux.dirent64 = @ptrCast(&buf[off]); + const name_ptr: [*:0]const u8 = @ptrCast(&buf[off + @offsetOf(linux.dirent64, "name")]); + const name = std.mem.span(name_ptr); + const dtype = d.type; + off += d.reclen; + if (std.mem.eql(u8, name, ".") or std.mem.eql(u8, name, "..")) continue; + if (list.items.len >= max_shadow_entries) { + std.debug.print("9ns: refusing to shadow {s}: more than {d} entries\n", .{ dirpath, max_shadow_entries }); + return error.TooManyEntries; + } + const kind: FileType = switch (dtype) { + linux.DT.DIR => .dir, + linux.DT.LNK => .symlink, + linux.DT.UNKNOWN => fileType(fd, name_ptr, true) orelse .other, + else => .other, + }; + try list.append(gpa, .{ .name = try gpa.dupeZ(u8, name), .kind = kind }); + } + } + return list.toOwnedSlice(gpa); +} + +fn shadowDirectory(gpa: Allocator, parent: [:0]const u8) !void { + const open_rc = linux.open(parent, .{ .ACCMODE = .RDONLY, .DIRECTORY = true, .CLOEXEC = true }, 0); + switch (linux.errno(open_rc)) { + .SUCCESS => {}, + else => |e| { + std.debug.print("9ns: open {s}: E{t}\n", .{ parent, e }); + return error.Mountpoint; + }, + } + const pfd: i32 = @intCast(open_rc); + defer _ = linux.close(pfd); + + const entries = try listDir(gpa, pfd, parent); + defer { + for (entries) |e| gpa.free(e.name); + gpa.free(entries); + } + + const tmpfs_opts: [*:0]const u8 = "mode=755"; + switch (linux.errno(linux.mount("tmpfs", parent, "tmpfs", linux.MS.NOSUID | linux.MS.NODEV, @intFromPtr(tmpfs_opts)))) { + .SUCCESS => {}, + else => |e| { + std.debug.print("9ns: mount tmpfs on {s}: E{t}\n", .{ parent, e }); + return error.Mountpoint; + }, + } + + // `pfd` still refers to the original directory underneath the tmpfs, so + // `/proc/self/fd//` reaches the hidden entries. + var src_buf: [path_max]u8 = undefined; + var dst_buf: [path_max]u8 = undefined; + var link_buf: [path_max]u8 = undefined; + for (entries) |e| { + const src = std.fmt.bufPrintZ(&src_buf, "/proc/self/fd/{d}/{s}", .{ pfd, e.name }) catch { + std.debug.print("9ns: shadow {s}/{s}: name too long (skipped)\n", .{ parent, e.name }); + continue; + }; + const dst = std.fmt.bufPrintZ(&dst_buf, "{s}/{s}", .{ parent, e.name }) catch { + std.debug.print("9ns: shadow {s}/{s}: name too long (skipped)\n", .{ parent, e.name }); + continue; + }; + switch (e.kind) { + .dir => { + if (!check("mkdir", dst, linux.mkdirat(linux.AT.FDCWD, dst, 0o755))) continue; + _ = check("bind", dst, linux.mount(src, dst, null, linux.MS.BIND | linux.MS.REC, 0)); + }, + .symlink => { + const rc = linux.readlinkat(pfd, e.name, &link_buf, link_buf.len - 1); + if (!check("readlink", dst, rc)) continue; + link_buf[rc] = 0; + const target: [*:0]const u8 = @ptrCast(&link_buf); + _ = check("symlink", dst, linux.symlinkat(target, linux.AT.FDCWD, dst)); + }, + .other => { + const rc = linux.openat(linux.AT.FDCWD, dst, .{ .ACCMODE = .WRONLY, .CREAT = true, .CLOEXEC = true }, 0o644); + if (!check("create", dst, rc)) continue; + _ = linux.close(@intCast(rc)); + _ = check("bind", dst, linux.mount(src, dst, null, linux.MS.BIND | linux.MS.REC, 0)); + }, + } + } +} + +/// Report a failed per-entry step as a warning (the entry is skipped; the +/// rest of the shadow is still useful). Returns true on success. +fn check(step: []const u8, path: [*:0]const u8, rc: usize) bool { + switch (linux.errno(rc)) { + .SUCCESS => return true, + else => |e| { + std.debug.print("9ns: shadow: {s} {s}: E{t} (skipped)\n", .{ step, std.mem.span(path), e }); + return false; + }, + } +} + +// --------------------------------------------------------------------------- +// spawn +// --------------------------------------------------------------------------- + +const ChildArgs = struct { + gpa: Allocator, + status_sock: i32, + mountpoint: [:0]const u8, + fuse_opts_prefix: [:0]const u8, // everything after "fd=," + mount_fuse: bool, + uid_map: []const u8, + gid_map: []const u8, + argv: [:null]?[*:0]const u8, + envp: [:null]?[*:0]const u8, + candidates: []const [:0]const u8, + name: []const u8, +}; + +/// Status channel protocol (child → parent, over a CLOEXEC socketpair): +/// a 0 byte means "namespace and mount are up" and carries the fuse fd as +/// SCM_RIGHTS; a non-zero byte is an exit status followed by a message. +/// EOF ends the conversation (exec succeeded, or the child died). +const ok_byte: u8 = 0; + +/// fork; the child unshares user+mount namespaces, maps its uid/gid, +/// makes `/` private, ensures the mountpoint, opens `/dev/fuse`, mounts it +/// on the mountpoint and sends the fd back. `spawn` returns at that point +/// (with `error.ChildFailed` and a message on stderr if any step failed). +/// The child then stats the mountpoint, which makes the kernel fetch the +/// root's attributes once the parent serves (the kernel seeds the fuse root +/// with uid 0, unmapped in the new user namespace, so nothing could be +/// created in the root until then), sets `NINE_MOUNT` and execs +/// `argv`. An exec failure is reported on `Child.status_fd` and ends the +/// child with 126/127; collect it with `reportExecFailure` after +/// `bridge.serve` returns. +/// +/// If `installSignals` was called, the pid is stored into the registered +/// variable as soon as fork returns so no SIGCHLD can be missed. +pub fn spawn(gpa: Allocator, s: Spawn) !Child { + if (s.argv.len == 0 or s.argv[0].len == 0) { + std.debug.print("9ns: empty program name\n", .{}); + return error.EmptyProgramName; + } + + const mountpoint = try gpa.dupeZ(u8, s.mountpoint); + defer gpa.free(mountpoint); + const fuse_opts_prefix = try std.fmt.allocPrintSentinel(gpa, "rootmode=40000,user_id={d},group_id={d},max_read={d}", .{ s.uid, s.gid, s.max_read }, 0); + defer gpa.free(fuse_opts_prefix); + var uid_buf: [64]u8 = undefined; + var gid_buf: [64]u8 = undefined; + const uid_map = try std.fmt.bufPrint(&uid_buf, "{d} {d} 1\n", .{ s.uid, s.uid }); + const gid_map = try std.fmt.bufPrint(&gid_buf, "{d} {d} 1\n", .{ s.gid, s.gid }); + const argv = try buildArgv(gpa, s.argv); + defer { + for (argv) |a| gpa.free(std.mem.span(a.?)); + gpa.free(argv); + } + const envp = try buildEnvp(gpa, s.envp, s.mountpoint); + defer { + gpa.free(std.mem.span(envp[envp.len - 1].?)); // the NINE_MOUNT entry we created + gpa.free(envp); + } + const candidates = try pathCandidates(gpa, s.envp, s.argv[0]); + defer { + for (candidates) |c| gpa.free(c); + gpa.free(candidates); + } + + var sv: [2]i32 = undefined; + switch (linux.errno(linux.socketpair(linux.AF.UNIX, linux.SOCK.STREAM | linux.SOCK.CLOEXEC, 0, &sv))) { + .SUCCESS => {}, + else => |e| { + std.debug.print("9ns: socketpair: E{t}\n", .{e}); + return error.SystemResources; + }, + } + + const child_args = ChildArgs{ + .gpa = gpa, + .status_sock = sv[1], + .mountpoint = mountpoint, + .fuse_opts_prefix = fuse_opts_prefix, + .mount_fuse = s.mount_fuse, + .uid_map = uid_map, + .gid_map = gid_map, + .argv = argv, + .envp = envp, + .candidates = candidates, + .name = s.argv[0], + }; + + const fork_rc = linux.fork(); + switch (linux.errno(fork_rc)) { + .SUCCESS => {}, + else => |e| { + _ = linux.close(sv[0]); + _ = linux.close(sv[1]); + std.debug.print("9ns: fork: E{t}\n", .{e}); + return error.SystemResources; + }, + } + if (fork_rc == 0) childMain(&child_args); + + const pid: i32 = @intCast(fork_rc); + if (child_pid_ptr) |p| @atomicStore(i32, p, pid, .seq_cst); + _ = linux.close(sv[1]); + + // First byte: ok (with the fuse fd attached) or a failure status. + var first: [1]u8 = .{ok_byte}; // defined even if recvmsg stores nothing + var fuse_fd: i32 = -1; + var n: usize = 0; + while (true) { + const rc = recvWithFd(sv[0], &first, &fuse_fd, 0); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .INTR => continue, + else => break, + } + n = rc; + break; + } + if (n == 1 and first[0] == ok_byte and (fuse_fd >= 0 or !s.mount_fuse)) { + return .{ .pid = pid, .fuse_fd = fuse_fd, .status_fd = sv[0] }; + } + + // Failure. A status byte means the child is exiting on its own and a + // message follows. Anything else (EOF: the child died before reporting; + // an ok byte without the fd: the SCM_RIGHTS transfer was truncated, e.g. + // EMFILE) is a protocol violation: the child may be about to exec with a + // dead mount, so kill it before waiting rather than reading the status + // socket until an exec'd program eventually exits. + const reported = n == 1 and first[0] != ok_byte; + if (!reported) _ = linux.kill(pid, .KILL); + if (fuse_fd >= 0) _ = linux.close(fuse_fd); + var msg: [512]u8 = undefined; + var len: usize = 0; + while (reported and len < msg.len) { + const rc = linux.read(sv[0], msg[len..].ptr, msg.len - len); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .INTR => continue, + else => break, + } + if (rc == 0) break; + len += rc; + } + _ = linux.close(sv[0]); + if (reported) { + std.debug.print("9ns: {s}\n", .{msg[0..len]}); + } else if (n == 1) { + std.debug.print("9ns: child handshake failed: no fuse fd received (out of file descriptors?)\n", .{}); + } else { + std.debug.print("9ns: child exited before reporting\n", .{}); + } + _ = waitChild(pid) catch {}; + if (child_pid_ptr) |p| @atomicStore(i32, p, 0, .seq_cst); + const status: u8 = if (reported) first[0] else setup_failure_status; + return switch (status) { + 126 => error.ExecPermission, + 127 => error.ExecNotFound, + else => error.ChildFailed, + }; +} + +/// After the child is gone (or the mount is dead): print the exec failure +/// the child reported on `status_fd`, if any, and close it. Returns the +/// status byte the child announced, or null when exec succeeded / nothing +/// was reported. Never blocks. +pub fn reportExecFailure(child: Child) ?u8 { + defer _ = linux.close(child.status_fd); + var msg: [512]u8 = undefined; + var len: usize = 0; + while (len < msg.len) { + var iov = [_]std.posix.iovec{.{ .base = msg[len..].ptr, .len = msg.len - len }}; + var hdr = linux.msghdr{ + .name = null, + .namelen = 0, + .iov = &iov, + .iovlen = 1, + .control = null, + .controllen = 0, + .flags = 0, + }; + const rc = linux.recvmsg(child.status_fd, &hdr, linux.MSG.DONTWAIT); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .INTR => continue, + else => break, + } + if (rc == 0) break; + len += rc; + } + if (len == 0) return null; + std.debug.print("9ns: {s}\n", .{msg[1..len]}); + return msg[0]; +} + +const cmsg_fd_len = @sizeOf(linux.cmsghdr) + @sizeOf(i32); +const cmsg_fd_space = std.mem.alignForward(usize, cmsg_fd_len, @sizeOf(usize)); + +/// sendmsg one data byte, optionally with `fd` attached as SCM_RIGHTS. +fn sendWithFd(sock: i32, byte: u8, fd: ?i32) usize { + const data = [_]u8{byte}; + const iov = [_]std.posix.iovec_const{.{ .base = &data, .len = 1 }}; + var cbuf: [cmsg_fd_space]u8 align(@alignOf(linux.cmsghdr)) = @splat(0); + var msg = linux.msghdr_const{ + .name = null, + .namelen = 0, + .iov = &iov, + .iovlen = 1, + .control = null, + .controllen = 0, + .flags = 0, + }; + if (fd) |f| { + const hdr: *linux.cmsghdr = @ptrCast(&cbuf); + hdr.* = .{ .len = cmsg_fd_len, .level = linux.SOL.SOCKET, .type = linux.SCM.RIGHTS }; + @memcpy(cbuf[@sizeOf(linux.cmsghdr)..][0..@sizeOf(i32)], std.mem.asBytes(&f)); + msg.control = &cbuf; + msg.controllen = cmsg_fd_space; + } + return linux.sendmsg(sock, &msg, linux.MSG.NOSIGNAL); +} + +/// recvmsg into `buf`; an SCM_RIGHTS fd, if any, is stored in `fd_out`. +fn recvWithFd(sock: i32, buf: []u8, fd_out: *i32, flags: u32) usize { + var iov = [_]std.posix.iovec{.{ .base = buf.ptr, .len = buf.len }}; + var cbuf: [cmsg_fd_space]u8 align(@alignOf(linux.cmsghdr)) = @splat(0); + var msg = linux.msghdr{ + .name = null, + .namelen = 0, + .iov = &iov, + .iovlen = 1, + .control = &cbuf, + .controllen = cbuf.len, + .flags = 0, + }; + const rc = linux.recvmsg(sock, &msg, linux.MSG.CMSG_CLOEXEC | flags); + if (linux.errno(rc) != .SUCCESS) return rc; + if (msg.controllen >= cmsg_fd_len) { + const hdr: *const linux.cmsghdr = @ptrCast(&cbuf); + if (hdr.level == linux.SOL.SOCKET and hdr.type == linux.SCM.RIGHTS and hdr.len >= cmsg_fd_len) { + var fd: i32 = undefined; + @memcpy(std.mem.asBytes(&fd), cbuf[@sizeOf(linux.cmsghdr)..][0..@sizeOf(i32)]); + fd_out.* = fd; + } + } + return rc; +} + +/// Child side of `spawn`. Never returns. +fn childMain(c: *const ChildArgs) noreturn { + resetSignals(); + + const rc_unshare = linux.errno(linux.unshare(linux.CLONE.NEWUSER | linux.CLONE.NEWNS)); + if (rc_unshare != .SUCCESS) childFail(c, setup_failure_status, "unshare(CLONE_NEWUSER|CLONE_NEWNS)", rc_unshare, true); + writeProcFile(c, "/proc/self/setgroups", "deny", true); + writeProcFile(c, "/proc/self/uid_map", c.uid_map, false); + writeProcFile(c, "/proc/self/gid_map", c.gid_map, false); + + const root: [*:0]const u8 = "/"; + const rc_priv = linux.mount(null, root, null, linux.MS.REC | linux.MS.PRIVATE, 0); + if (linux.errno(rc_priv) != .SUCCESS) childFail(c, setup_failure_status, "mount(/, MS_REC|MS_PRIVATE)", linux.errno(rc_priv), true); + + ensureMountpoint(c.gpa, c.mountpoint) catch { + childFail(c, setup_failure_status, "mountpoint setup failed (pass --mount an existing directory)", .SUCCESS, false); + }; + + var fuse_fd: ?i32 = null; + if (c.mount_fuse) { + // Must be opened here, after unshare: the kernel only mounts a fuse + // device opened from the mount's own user namespace. + const rc_open = linux.open("/dev/fuse", .{ .ACCMODE = .RDWR, .CLOEXEC = true }, 0); + switch (linux.errno(rc_open)) { + .SUCCESS => {}, + .NOENT => childFail(c, setup_failure_status, "open /dev/fuse: ENOENT (is the fuse module loaded? try: modprobe fuse)", .SUCCESS, false), + else => |e| childFail(c, setup_failure_status, "open /dev/fuse", e, true), + } + const fd: i32 = @intCast(rc_open); + var opts_buf: [256]u8 = undefined; + const opts = std.fmt.bufPrintZ(&opts_buf, "fd={d},{s}", .{ fd, c.fuse_opts_prefix }) catch unreachable; + const rc = linux.mount("9ns", c.mountpoint, "fuse", linux.MS.NOSUID | linux.MS.NODEV, @intFromPtr(opts.ptr)); + if (linux.errno(rc) != .SUCCESS) childFail(c, setup_failure_status, "mount fuse", linux.errno(rc), true); + fuse_fd = fd; + } + const sent = sendWithFd(c.status_sock, ok_byte, fuse_fd); + if (linux.errno(sent) != .SUCCESS) linux.exit_group(setup_failure_status); + if (fuse_fd) |fd| { + _ = linux.close(fd); // the parent holds the connection now + // Force one GETATTR of the root (served by the parent, which is + // entering its serve loop now); see `spawn`. Errors don't matter. + var stx: linux.Statx = undefined; + _ = linux.statx(linux.AT.FDCWD, c.mountpoint, 0, .{ .TYPE = true }, &stx); + } + + var last: E = .NOENT; + var saw_acces = false; + for (c.candidates) |cand| { + const rc = linux.execve(cand.ptr, c.argv.ptr, c.envp.ptr); + last = linux.errno(rc); + switch (last) { + .NOENT, .NOTDIR, .LOOP, .NAMETOOLONG => continue, + .ACCES => { + saw_acces = true; + continue; + }, + else => break, + } + } + var buf: [512]u8 = undefined; + // "Not found" covers every candidate that could not even be resolved + // (a PATH element that is a file gives ENOTDIR, a symlink loop ELOOP); + // a candidate that existed but was not executable wins over those. + const not_found = switch (last) { + .NOENT, .NOTDIR, .LOOP, .NAMETOOLONG => true, + else => false, + }; + if (not_found and saw_acces) last = .ACCES; + const status: u8 = if (not_found and !saw_acces) 127 else 126; + const text = std.fmt.bufPrint(&buf, "exec {s}", .{c.name}) catch "exec"; + childFail(c, status, text, last, true); +} + +fn writeProcFile(c: *const ChildArgs, path: [*:0]const u8, data: []const u8, ignore_missing: bool) void { + const rc = linux.open(path, .{ .ACCMODE = .WRONLY, .CLOEXEC = true }, 0); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .NOENT => if (ignore_missing) return else childFail(c, setup_failure_status, std.mem.span(path), .NOENT, true), + else => |e| childFail(c, setup_failure_status, std.mem.span(path), e, true), + } + const fd: i32 = @intCast(rc); + const w = linux.write(fd, data.ptr, data.len); + const we = linux.errno(w); + _ = linux.close(fd); + if (we != .SUCCESS) childFail(c, setup_failure_status, std.mem.span(path), we, true); + if (w != data.len) childFail(c, setup_failure_status, std.mem.span(path), .IO, true); +} + +/// Write `[: E]` to the status socket and exit. +fn childFail(c: *const ChildArgs, status: u8, step: []const u8, e: E, with_errno: bool) noreturn { + var buf: [600]u8 = undefined; + buf[0] = status; + const rest = if (with_errno) + std.fmt.bufPrint(buf[1..], "{s}: E{t}", .{ step, e }) catch buf[1..1] + else + std.fmt.bufPrint(buf[1..], "{s}", .{step}) catch buf[1..1]; + const msg = buf[0 .. 1 + rest.len]; + var off: usize = 0; + while (off < msg.len) { + const rc = linux.write(c.status_sock, msg[off..].ptr, msg.len - off); + if (linux.errno(rc) == .INTR) continue; + if (linux.errno(rc) != .SUCCESS) break; + off += rc; + } + linux.exit_group(status); +} + +// --------------------------------------------------------------------------- +// Signals +// --------------------------------------------------------------------------- + +var child_pid_ptr: ?*i32 = null; +var chld_pipe_w: i32 = -1; +var reaped = std.atomic.Value(bool).init(false); +var reaped_status = std.atomic.Value(u32).init(0); +/// A second child (the `--spawn` server) that the SIGCHLD handler reaps so +/// it does not linger as a zombie when it dies mid-session. Its exit does +/// not stop the serve loop. 0 = none. +var server_pid = std.atomic.Value(i32).init(0); + +/// Register the `--spawn` server for reaping by the SIGCHLD handler. +pub fn watchServer(pid: i32) void { + server_pid.store(pid, .seq_cst); +} + +/// Seconds the serve loop gets to come back after the child died before +/// the watchdog ends the process anyway. +pub const exit_grace_seconds: isize = 3; + +/// The watched child is already dead but the serve loop has not come back +/// (it is stuck in a 9P request the server never answers): a terminal +/// signal, or the watchdog armed by `onChld`, then ends 9ns with the +/// child's status instead of hanging. Nothing is lost: the mount is torn +/// down when the process exits. +fn bailIfChildGone() void { + if (!reaped.load(.acquire)) return; + const srv = server_pid.load(.seq_cst); + if (srv > 0) _ = linux.kill(srv, .TERM); + linux.exit_group(decodeStatus(reaped_status.load(.acquire))); +} + +fn armWatchdog() void { + // setitimer takes an itimerval; std declares it with itimerspec, which + // has the same layout on 64-bit targets (the sub-second field is 0). + const t = linux.itimerspec{ + .it_interval = .{ .sec = 0, .nsec = 0 }, + .it_value = .{ .sec = exit_grace_seconds, .nsec = 0 }, + }; + _ = linux.setitimer(@intFromEnum(linux.ITIMER.REAL), &t, null); +} + +fn onAlarm(_: linux.SIG) callconv(.c) void { + bailIfChildGone(); +} + +fn onForward(sig: linux.SIG) callconv(.c) void { + const p = child_pid_ptr orelse return; + const pid = @atomicLoad(i32, p, .seq_cst); + if (pid > 0) _ = linux.kill(pid, sig); + bailIfChildGone(); +} + +/// SIGINT/SIGQUIT: the child owns the tty and gets them itself; we only +/// react when the child is already gone (see `bailIfChildGone`). +fn onTerminal(_: linux.SIG) callconv(.c) void { + bailIfChildGone(); +} + +/// Only the watched child counts: reap it here (WNOHANG), remember its +/// status, forget its pid (so a later SIGTERM cannot hit a recycled pid) +/// and poke the self-pipe. The `--spawn` server is reaped too but does not +/// interrupt `bridge.serve`; SIGCHLD from anything else is ignored. +fn onChld(_: linux.SIG) callconv(.c) void { + const srv = server_pid.load(.seq_cst); + if (srv > 0) { + var sst: u32 = 0; + const src = linux.waitpid(srv, &sst, linux.W.NOHANG); + if (linux.errno(src) == .SUCCESS and src != 0) server_pid.store(0, .seq_cst); + } + const p = child_pid_ptr orelse return; + const pid = @atomicLoad(i32, p, .seq_cst); + if (pid <= 0) return; + var st: u32 = 0; + const rc = linux.waitpid(pid, &st, linux.W.NOHANG); + if (linux.errno(rc) != .SUCCESS or rc == 0) return; + reaped_status.store(st, .release); + reaped.store(true, .release); + @atomicStore(i32, p, 0, .seq_cst); + const b = [_]u8{'c'}; + _ = linux.write(chld_pipe_w, &b, 1); + armWatchdog(); +} + +/// SIGPIPE ignored; SIGINT/SIGQUIT effectively ignored (the child owns the +/// tty) unless the child is already dead; SIGTERM/SIGHUP forwarded to +/// `*child_pid`; SIGCHLD for `*child_pid` reaps it, writes a byte to a +/// nonblocking self-pipe whose read end is returned (use it as `stop_fd`) +/// and arms a watchdog (`exit_grace_seconds`, SIGALRM) that ends the +/// process with the child's status should the serve loop stay blocked. +/// `*child_pid` is filled in by `spawn`. +pub fn installSignals(child_pid: *i32) !i32 { + child_pid_ptr = child_pid; + var fds: [2]i32 = undefined; + switch (linux.errno(linux.pipe2(&fds, .{ .CLOEXEC = true, .NONBLOCK = true }))) { + .SUCCESS => {}, + else => |e| { + std.debug.print("9ns: pipe2: E{t}\n", .{e}); + return error.SystemResources; + }, + } + chld_pipe_w = fds[1]; + + const ign = linux.Sigaction{ .handler = .{ .handler = linux.SIG.IGN }, .mask = linux.sigemptyset(), .flags = 0 }; + const term = linux.Sigaction{ .handler = .{ .handler = &onTerminal }, .mask = linux.sigemptyset(), .flags = linux.SA.RESTART }; + const fwd = linux.Sigaction{ .handler = .{ .handler = &onForward }, .mask = linux.sigemptyset(), .flags = linux.SA.RESTART }; + const chld = linux.Sigaction{ .handler = .{ .handler = &onChld }, .mask = linux.sigemptyset(), .flags = linux.SA.RESTART | linux.SA.NOCLDSTOP }; + const alrm = linux.Sigaction{ .handler = .{ .handler = &onAlarm }, .mask = linux.sigemptyset(), .flags = linux.SA.RESTART }; + std.posix.sigaction(.INT, &term, null); + std.posix.sigaction(.QUIT, &term, null); + std.posix.sigaction(.ALRM, &alrm, null); + std.posix.sigaction(.PIPE, &ign, null); + std.posix.sigaction(.TERM, &fwd, null); + std.posix.sigaction(.HUP, &fwd, null); + std.posix.sigaction(.CHLD, &chld, null); + return fds[0]; +} + +/// Restore default dispositions in the child before exec (ignored signals +/// would otherwise survive execve). +fn resetSignals() void { + const dfl = linux.Sigaction{ .handler = .{ .handler = linux.SIG.DFL }, .mask = linux.sigemptyset(), .flags = 0 }; + inline for (.{ linux.SIG.INT, linux.SIG.QUIT, linux.SIG.PIPE, linux.SIG.TERM, linux.SIG.HUP, linux.SIG.CHLD, linux.SIG.ALRM }) |sig| { + _ = linux.sigaction(sig, &dfl, null); + } +} + +// --------------------------------------------------------------------------- +// Waiting +// --------------------------------------------------------------------------- + +fn takeReaped() ?u32 { + if (!reaped.load(.acquire)) return null; + return reaped_status.load(.acquire); +} + +/// waitpid status → exit code (`128+sig` when killed by a signal). +pub fn decodeStatus(st: u32) u8 { + if (linux.W.IFEXITED(st)) return linux.W.EXITSTATUS(st); + if (linux.W.IFSIGNALED(st)) return 128 +% @as(u8, @truncate(@intFromEnum(linux.W.TERMSIG(st)))); + return 1; +} + +/// Block until `pid` exits (the SIGCHLD handler may have reaped it already). +pub fn waitChild(pid: i32) !u8 { + while (true) { + if (takeReaped()) |st| return decodeStatus(st); + var st: u32 = 0; + const rc = linux.waitpid(pid, &st, 0); + switch (linux.errno(rc)) { + .SUCCESS => return decodeStatus(st), + .INTR => continue, + .CHILD => { + if (takeReaped()) |s| return decodeStatus(s); + return error.NoChild; + }, + else => |e| { + std.debug.print("9ns: waitpid: E{t}\n", .{e}); + return error.Wait; + }, + } + } +} + +/// Non-blocking: the exit status of `pid` if it has exited, else null. +pub fn reapIfExited(pid: i32) ?u8 { + if (takeReaped()) |st| return decodeStatus(st); + var st: u32 = 0; + const rc = linux.waitpid(pid, &st, linux.W.NOHANG); + switch (linux.errno(rc)) { + .SUCCESS => return if (rc == 0) null else decodeStatus(st), + .CHILD => return if (takeReaped()) |s| decodeStatus(s) else null, + else => return null, + } +} + +/// Reap any child (used for the `--spawn` server at exit). Non-blocking. +pub fn reapAny(pid: i32) void { + var st: u32 = 0; + _ = linux.waitpid(pid, &st, linux.W.NOHANG); +} + +// --------------------------------------------------------------------------- +// Tests (no namespaces needed; `ensureMountpoint` is exercised by +// test/integration.sh through the 9ns binary) +// --------------------------------------------------------------------------- + +const testing = std.testing; + +test "normalizePath: absolute paths" { + const gpa = testing.allocator; + const cases = [_]struct { in: []const u8, out: []const u8 }{ + .{ .in = "/mnt/9p", .out = "/mnt/9p" }, + .{ .in = "/mnt/9p/", .out = "/mnt/9p" }, + .{ .in = "//mnt///9p//", .out = "/mnt/9p" }, + .{ .in = "/mnt/./9p/.", .out = "/mnt/9p" }, + .{ .in = "/mnt/x/../9p", .out = "/mnt/9p" }, + .{ .in = "/../mnt/9p", .out = "/mnt/9p" }, + .{ .in = "/a/b/c/../..", .out = "/a" }, + }; + for (cases) |c| { + const got = try normalizePath(gpa, "/cwd", c.in); + defer gpa.free(got); + try testing.expectEqualStrings(c.out, got); + try testing.expectEqual(@as(u8, 0), got[got.len]); + } +} + +test "normalizePath: relative paths use cwd" { + const gpa = testing.allocator; + const cases = [_]struct { cwd: []const u8, in: []const u8, out: []const u8 }{ + .{ .cwd = "/home/me", .in = "mnt", .out = "/home/me/mnt" }, + .{ .cwd = "/home/me", .in = "./mnt/", .out = "/home/me/mnt" }, + .{ .cwd = "/home/me", .in = "../mnt", .out = "/home/mnt" }, + .{ .cwd = "/home/me/", .in = ".", .out = "/home/me" }, + .{ .cwd = "/", .in = "x", .out = "/x" }, + }; + for (cases) |c| { + const got = try normalizePath(gpa, c.cwd, c.in); + defer gpa.free(got); + try testing.expectEqualStrings(c.out, got); + } +} + +test "normalizePath: rejects root and empty" { + const gpa = testing.allocator; + try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/cwd", "/")); + try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/cwd", "///")); + try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/cwd", "/mnt/..")); + try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/cwd", "")); + try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/", "..")); +} + +test "resolveMountpoint: relative resolves against the real cwd" { + const gpa = testing.allocator; + const got = try resolveMountpoint(gpa, "sub/dir"); + defer gpa.free(got); + try testing.expect(got[0] == '/'); + try testing.expect(std.mem.endsWith(u8, got, "/sub/dir")); +} + +test "getenv" { + const env = [_:null]?[*:0]const u8{ "PATH=/a:/b", "X=", "PATHX=no", "NINE_MOUNT=/m" }; + const envp: [*:null]const ?[*:0]const u8 = &env; + try testing.expectEqualStrings("/a:/b", getenv(envp, "PATH").?); + try testing.expectEqualStrings("", getenv(envp, "X").?); + try testing.expectEqualStrings("/m", getenv(envp, "NINE_MOUNT").?); + try testing.expect(getenv(envp, "NOPE") == null); + try testing.expect(getenv(envp, "PAT") == null); +} + +test "pathCandidates: PATH search" { + const gpa = testing.allocator; + const env = [_:null]?[*:0]const u8{ "PATH=/usr/local/bin::/usr/bin", "HOME=/h" }; + const cands = try pathCandidates(gpa, &env, "fish"); + defer { + for (cands) |c| gpa.free(c); + gpa.free(cands); + } + try testing.expectEqual(@as(usize, 3), cands.len); + try testing.expectEqualStrings("/usr/local/bin/fish", cands[0]); + try testing.expectEqualStrings("./fish", cands[1]); + try testing.expectEqualStrings("/usr/bin/fish", cands[2]); +} + +test "pathCandidates: slash means no search; default PATH" { + const gpa = testing.allocator; + const env = [_:null]?[*:0]const u8{"HOME=/h"}; + { + const cands = try pathCandidates(gpa, &env, "./bin/x"); + defer { + for (cands) |c| gpa.free(c); + gpa.free(cands); + } + try testing.expectEqual(@as(usize, 1), cands.len); + try testing.expectEqualStrings("./bin/x", cands[0]); + } + { + const cands = try pathCandidates(gpa, &env, "sh"); + defer { + for (cands) |c| gpa.free(c); + gpa.free(cands); + } + try testing.expectEqual(@as(usize, 3), cands.len); + try testing.expectEqualStrings("/usr/local/bin/sh", cands[0]); + try testing.expectEqualStrings("/bin/sh", cands[1]); + } + try testing.expectError(error.EmptyProgramName, pathCandidates(gpa, &env, "")); +} + +test "findInPath finds sh" { + const gpa = testing.allocator; + const env = [_:null]?[*:0]const u8{"PATH=/nonexistent:/bin:/usr/bin"}; + const p = try findInPath(gpa, &env, "sh"); + defer gpa.free(p); + try testing.expect(std.mem.endsWith(u8, p, "/sh")); + try testing.expectError(error.FileNotFound, findInPath(gpa, &env, "definitely-not-a-program-9ns")); +} + +test "buildEnvp replaces NINE_MOUNT" { + const gpa = testing.allocator; + const env = [_:null]?[*:0]const u8{ "A=1", "NINE_MOUNT=/old", "B=2" }; + const out = try buildEnvp(gpa, &env, "/mnt/9p"); + defer { + gpa.free(std.mem.span(out[out.len - 1].?)); + gpa.free(out); + } + try testing.expectEqual(@as(usize, 3), out.len); + try testing.expectEqualStrings("A=1", std.mem.span(out[0].?)); + try testing.expectEqualStrings("B=2", std.mem.span(out[1].?)); + try testing.expectEqualStrings("NINE_MOUNT=/mnt/9p", std.mem.span(out[2].?)); + try testing.expect(out[3] == null); + try testing.expectEqualStrings("/mnt/9p", getenv(out.ptr, "NINE_MOUNT").?); +} + +test "decodeStatus" { + try testing.expectEqual(@as(u8, 0), decodeStatus(0)); + try testing.expectEqual(@as(u8, 7), decodeStatus(7 << 8)); + try testing.expectEqual(@as(u8, 255), decodeStatus(255 << 8)); + try testing.expectEqual(@as(u8, 128 + 9), decodeStatus(9)); // SIGKILL + try testing.expectEqual(@as(u8, 128 + 15), decodeStatus(15)); // SIGTERM +} + +test "ensureMountpoint: existing directory is accepted, plain file rejected" { + const gpa = testing.allocator; + try ensureMountpoint(gpa, "/tmp"); + try testing.expectError(error.Mountpoint, ensureMountpoint(gpa, "/proc/self/status")); +} diff --git a/9ns/test/adv_bridge_hostile.py b/9ns/test/adv_bridge_hostile.py new file mode 100755 index 0000000..d353541 --- /dev/null +++ b/9ns/test/adv_bridge_hostile.py @@ -0,0 +1,487 @@ +#!/usr/bin/env python3 +"""A scriptable, hostile 9P2000 server on a Unix socket (stdlib only). + +Usage: adv_bridge_hostile.py SOCKET MODE + +Serves a tiny in-memory tree: + /f "hello world\\n" + /d/g "in d\\n" + /fids reading it returns the number of fids currently bound + /big 1 MiB of pseudo-random bytes +plus create/write/remove/wstat so the scratch battery can run in `ok` mode. + +MODE selects one misbehaviour (see MODES below). Everything not covered by +the mode behaves normally, so 9ns gets through version/attach/stat(root). +""" +import os +import random +import socket +import struct +import sys +import time + +NOTAG = 0xFFFF +NOFID = 0xFFFFFFFF +QTDIR = 0x80 +DMDIR = 0x80000000 + +Tversion, Rversion = 100, 101 +Tauth, Rauth = 102, 103 +Tattach, Rattach = 104, 105 +Rerror = 107 +Tflush, Rflush = 108, 109 +Twalk, Rwalk = 110, 111 +Topen, Ropen = 112, 113 +Tcreate, Rcreate = 114, 115 +Tread, Rread = 116, 117 +Twrite, Rwrite = 118, 119 +Tclunk, Rclunk = 120, 121 +Tremove, Rremove = 122, 123 +Tstat, Rstat = 124, 125 +Twstat, Rwstat = 126, 127 + +MODES = """ +ok behave (qid paths are recycled LIFO after remove, like many servers) +trunc Rread on /f: send half the frame, then close +short_frame Rread on /f: frame whose size field is 3 +huge_frame Rread on /f: frame whose size field is msize+1 +wrong_tag Rread on /f: reply carries tag+1 +wrong_type Tstat on /f: answer with an Rwalk +rread_big Rread on /f: count = requested+1 +rwalk_many Twalk to f: nwqid = nwname+1 +rwalk_zero Twalk to nope: Rwalk nwqid=0 instead of Rerror +rstat_garbage Tstat on /f: random bytes as the stat +rstat_overlong Tstat on /f: inner stat size disagrees with outer +dir_split Tread on /: a stat record split across two Rreads +dir_forever Tread on /: ignore offset, always return the same records +qid_collide every file and dir shares qid.path 7 (root keeps its own) +qid_zero every qid.path is 0, including the root +name_slash / has an entry "a/b" +name_empty / has an entry "" +name_huge / has an entry with a 60000-byte name +name_dots / lists "." and ".." too +rerror_big Twalk to nope: Rerror with 65535 bytes of text +extra_reply Rread on /f: an unsolicited Rclunk (tag 9) precedes the real reply +never Tread on /f: never reply (hang) +close_mid Tread on /f: close the socket without replying +renegotiate Tread on /f: an unsolicited Rversion precedes the real reply +length_max Tstat on /f: length = 2**64-1 +iounit_one Ropen: iounit = 1 +rwrite_big Rwrite: count = requested+1 +msize_tiny Rversion msize = 64 +version_unknown Rversion "unknown" +rename_fail Twstat with a new name always fails "file already exists" +slow every reply delayed 20 ms (for interrupt tests) +""" + + +def s8(x): return struct.pack('= 4: + n = struct.unpack('= n: + msg, buf = buf[:n], buf[n:] + out = self.handle(msg) + if out is None: + return # hang up / hang + if self.mode == 'slow': + time.sleep(0.02) + conn.sendall(out) + continue + data = conn.recv(65536) + if not data: + return + buf += data + + def handle(self, msg): + typ = msg[4] + tag = struct.unpack(' len(node.content): + node.content.extend(b'\0' * (off - len(node.content))) + node.content[off:off + len(data)] = data + node.mtime = int(time.time()) + n = len(data) + 1 if self.mode == 'rwrite_big' else len(data) + return self.frame(Rwrite, tag, s32(n)) + if typ == Tclunk: + fid = r.u32() + if fid not in self.fids: + return self.err(tag, 'unknown fid') + del self.fids[fid] + return self.frame(Rclunk, tag, b'') + if typ == Tremove: + fid = r.u32() + if fid not in self.fids: + return self.err(tag, 'unknown fid') + node = self.fids[fid][0] + del self.fids[fid] + if node is self.root: + return self.err(tag, 'cannot remove root') + if node.isdir and node.children: + return self.err(tag, 'directory not empty') + parent = self.find_parent(self.root, node) + if parent is not None: + del parent.children[node.name] + self.free_paths.append(node.path) + node.removed = True + return self.frame(Rremove, tag, b'') + if typ == Tstat: + fid = r.u32() + if fid not in self.fids: + return self.err(tag, 'unknown fid') + node = self.fids[fid][0] + if node is self.root.children.get('f'): + m = self.mode + if m == 'wrong_type': + return self.frame(Rwalk, tag, s16(0)) + if m == 'rstat_garbage': + junk = bytes([0xAB] * 60) + return self.frame(Rstat, tag, s16(len(junk)) + junk) + if m == 'rstat_overlong': + st = node.stat_bytes(self) + inner = st[2:] + return self.frame(Rstat, tag, s16(len(inner) + 5) + inner) + if m == 'length_max': + st = node.stat_bytes(self, length=2 ** 64 - 1) + return self.frame(Rstat, tag, s16(len(st)) + st) + st = node.stat_bytes(self) + return self.frame(Rstat, tag, s16(len(st)) + st) + if typ == Twstat: + fid = r.u32() + r.u16() + st = r.bytes(r.u16()) + if fid not in self.fids: + return self.err(tag, 'unknown fid') + node = self.fids[fid][0] + sr = Reader(st) + sr.u16(); sr.u32(); sr.bytes(13) + mode = sr.u32(); sr.u32(); mtime = sr.u32(); length = sr.u64() + name = sr.str() + if name and name != node.name: + if self.mode == 'rename_fail': + return self.err(tag, 'file already exists') + parent = self.find_parent(self.root, node) + if name in parent.children: + return self.err(tag, 'file already exists') + del parent.children[node.name] + node.name = name + parent.children[name] = node + if mode != 0xFFFFFFFF: + node.mode = mode & 0o777 + if mtime != 0xFFFFFFFF: + node.mtime = mtime + if length != 0xFFFFFFFFFFFFFFFF and not node.isdir: + if length < len(node.content): + del node.content[length:] + else: + node.content.extend(b'\0' * (length - len(node.content))) + return self.frame(Rwstat, tag, b'') + return self.err(tag, 'unsupported message') + + def find_parent(self, cur, node): + for c in cur.children.values(): + if c is node: + return cur + if c.isdir: + p = self.find_parent(c, node) + if p is not None: + return p + return None + + def readdir(self, tag, node, off, count): + recs = [] + if node is self.root: + m = self.mode + if m == 'name_slash': + recs.append(node.stat_bytes(self, name='a/b')) + if m == 'name_empty': + recs.append(node.stat_bytes(self, name='')) + if m == 'name_huge': + recs.append(node.stat_bytes(self, name='h' * 60000)) + if m == 'name_dots': + recs.append(node.stat_bytes(self, name='.')) + recs.append(node.stat_bytes(self, name='..')) + for c in node.children.values(): + recs.append(c.stat_bytes(self)) + blob = b''.join(recs) + if node is self.root and self.mode == 'dir_forever': + return self.frame(Rread, tag, s32(len(blob)) + blob) + if node is self.root and self.mode == 'dir_split': + # first read: up to the middle of the second record; second read: the rest + cut = len(recs[0]) + len(recs[1]) // 2 + if off == 0: + data = blob[:cut] + elif off == cut: + data = blob[cut:] + else: + data = b'' + return self.frame(Rread, tag, s32(len(data)) + data) + # 9P rule: offset 0 or previous offset+count; never split a record. + out = b'' + pos = 0 + for rec in recs: + if pos >= off and len(out) + len(rec) <= count: + out += rec + elif pos >= off: + break + pos += len(rec) + return self.frame(Rread, tag, s32(len(out)) + out) + + +class Reader: + def __init__(self, b): + self.b = b + self.i = 0 + + def bytes(self, n): + v = self.b[self.i:self.i + n] + self.i += n + return v + + def u8(self): return struct.unpack(' <9proc-demo> (9proc unused; part of zig build 9ns-adv) +set -u +NS=$(realpath "${1:?path to 9ns}") +HERE=$(cd "$(dirname "$0")" && pwd) +SRV=$HERE/adv_bridge_hostile.py +TMP=$(mktemp -d "${TMPDIR:-/tmp}/9ns-adv.XXXXXX") +M=/mnt/9p +FAILED=0 +PASSED=0 +SRVPID= + +cleanup() { [ -n "$SRVPID" ] && kill "$SRVPID" 2>/dev/null; pkill -f "adv_bridge_hostile.py $TMP" 2>/dev/null; rm -rf "$TMP"; } +trap cleanup EXIT + +if ! unshare -Urm true 2>/dev/null || [ ! -c /dev/fuse ]; then echo "SKIP: no user namespaces or /dev/fuse"; exit 0; fi + +pass() { PASSED=$((PASSED + 1)); echo "ok - $1"; } +fail() { FAILED=$((FAILED + 1)); echo "FAIL - $1"; shift; [ $# -gt 0 ] && printf ' %s\n' "$@"; } +expect_eq() { if [ "$2" = "$3" ]; then pass "$1"; else fail "$1" "expected: $(printf %q "$2")" "actual: $(printf %q "$3")"; fi; } +expect_contains() { case "$3" in *"$2"*) pass "$1" ;; *) fail "$1" "missing: $(printf %q "$2")" "in: $(printf %q "$3")" ;; esac; } + +start_server() { # mode + [ -n "$SRVPID" ] && { kill "$SRVPID" 2>/dev/null; wait "$SRVPID" 2>/dev/null; } + SOCK=$TMP/$1.sock + rm -f "$SOCK" + python3 "$SRV" "$SOCK" "$1" >"$TMP/$1.srv.out" 2>&1 "$TMP/stderr") + RC=$? + STDERR=$(cat "$TMP/stderr") +} + +# 9ns must not die of a signal or panic. RC 124 = timeout(1) fired. +no_crash() { # name + if [ "$RC" -ge 128 ] || [ "$RC" -eq 124 ]; then fail "$1: 9ns exit $RC" "$STDERR"; return; fi + case "$STDERR" in *panic*|*"Segmentation"*|*"integer overflow"*|*"reached unreachable"*|*"index out of bounds"*) fail "$1: crash text in stderr" "$STDERR";; *) pass "$1: no crash (exit $RC)";; esac +} + +echo "# sanity: the hostile server behaves in 'ok' mode" +run ok "cat $M/f"; expect_eq "ok: cat f" "hello world" "$OUT" +run ok "cat $M/d/g"; expect_eq "ok: nested" "in d" "$OUT" +run ok "cat $M/nope 2>&1 | sed 's/.*: //'"; expect_eq "ok: ENOENT" "No such file or directory" "$OUT" +run ok "head -c 1048576 $M/big | wc -c | grep -q 1048576 && echo yes"; expect_eq "ok: 1 MiB read matches" "yes" "$OUT" + +echo "# unlink + recreate with a recycled qid.path" +run ok "echo 1 > $M/a; rm $M/a; echo 2 > $M/a; cat $M/a; rm $M/a"; expect_eq "qid reuse: new content, not stale" "2" "$OUT" +run ok "echo 1 > $M/x; rm $M/x; mkdir $M/x; stat -c %F $M/x; rmdir $M/x"; expect_eq "qid reuse: file→dir on the same path" "directory" "$OUT" + +echo "# fids do not grow with the number of operations" +loop='i=0; while [ $i -lt N ]; do echo hi > M/t; cat M/t >/dev/null; mkdir M/dd; rmdir M/dd; rm M/t; i=$((i+1)); done; cat M/fids' +run ok "$(echo "$loop" | sed "s|N|20|; s|M/|$M/|g")"; a=$OUT +run ok "$(echo "$loop" | sed "s|N|200|; s|M/|$M/|g")"; b=$OUT +expect_eq "fids after 20 == after 200 iterations ($a)" "$a" "$b" +floop='i=0; while [ $i -lt N ]; do cat M/nope 2>/dev/null; echo x > M/fids 2>/dev/null; mkdir M/f 2>/dev/null; rm M/d 2>/dev/null; mv M/f M/d 2>/dev/null; i=$((i+1)); done; cat M/fids' +run ok "$(echo "$floop" | sed "s|N|20|; s|M/|$M/|g")"; a=$OUT +run ok "$(echo "$floop" | sed "s|N|200|; s|M/|$M/|g")"; b=$OUT +expect_eq "fids after 20 == after 200 failing iterations ($a)" "$a" "$b" + +echo "# rename over an existing file must not lose the target when the rename fails" +run rename_fail "echo A > $M/a; echo B > $M/b; mv $M/a $M/b 2>/dev/null; echo mv=\$?; cat $M/b; cat $M/a" +expect_contains "rename_fail: mv reports failure" "mv=1" "$OUT" +expect_contains "rename_fail: target b still has its content" "B" "$OUT" +expect_contains "rename_fail: source a still has its content" "A" "$OUT" + +echo "# protocol violations on a data read must yield an error, not a crash" +for mode in trunc short_frame huge_frame wrong_tag rread_big extra_reply close_mid renegotiate rwrite_big; do + if [ "$mode" = rwrite_big ]; then script="dd if=/dev/zero of=$M/f bs=10 count=1 2>&1; echo status=\$?"; else script="cat $M/f 2>&1; echo status=\$?"; fi + run "$mode" "$script" + no_crash "$mode" + expect_contains "$mode: child sees an error" "status=1" "$OUT" + case "$OUT" in *"Input/output error"*|*"not connected"*) pass "$mode: EIO/ENOTCONN";; *) fail "$mode: errno text" "$OUT";; esac +done + +echo "# protocol violations on lookup/stat" +for mode in wrong_type rwalk_many rstat_garbage rstat_overlong; do + run "$mode" "stat -c %s $M/f 2>&1; echo status=\$?" + no_crash "$mode" + expect_contains "$mode: child sees an error" "status=1" "$OUT" +done +run rwalk_zero "cat $M/nope 2>&1; echo status=\$?; cat $M/f 2>&1" +no_crash "rwalk_zero" +expect_contains "rwalk_zero: child sees an error" "status=1" "$OUT" +run rerror_big "cat $M/nope 2>&1; echo status=\$?; cat $M/f" +no_crash "rerror_big" +expect_contains "rerror_big: child sees an error" "status=1" "$OUT" +expect_contains "rerror_big: session survives a 64 KiB Rerror" "hello world" "$OUT" + +echo "# hostile stat contents" +run length_max "stat -c '%s %b' $M/f 2>&1; echo status=\$?" +no_crash "length_max" +expect_contains "length_max: stat succeeds with a saturated block count" "status=0" "$OUT" +run iounit_one "cat $M/f; head -c 3000 $M/big | wc -c" +no_crash "iounit_one" +expect_contains "iounit_one: read still complete" "hello world" "$OUT" +expect_contains "iounit_one: 3000 bytes" "3000" "$OUT" + +echo "# hostile directory listings" +run dir_split "ls $M 2>&1; echo status=\$?; cat $M/f" +no_crash "dir_split" +expect_contains "dir_split: readdir fails with EIO" "Input/output error" "$OUT" +expect_contains "dir_split: session survives" "hello world" "$OUT" +run dir_forever "ls $M 2>&1 | tail -c 200; echo status=\$?; cat $M/f; grep VmRSS /proc/\$PPID/status" +no_crash "dir_forever" +expect_contains "dir_forever: infinite directory is cut off with EIO" "Input/output error" "$OUT" +expect_contains "dir_forever: session survives" "hello world" "$OUT" +for mode in name_slash name_empty name_huge name_dots; do + run "$mode" "ls -a $M | tr '\n' ' '; echo; cat $M/f" + no_crash "$mode" + expect_contains "$mode: listing still works" "big d f fids" "$OUT" + expect_contains "$mode: file readable" "hello world" "$OUT" + case "$mode" in + name_slash) expect_eq "$mode: slash entry dropped" "" "$(printf '%s' "$OUT" | grep -o 'a/b')";; + name_dots) expect_eq "$mode: exactly one . and one .." "1 1" "$(printf '%s %s' "$(printf '%s\n' "$OUT" | head -1 | tr ' ' '\n' | grep -c '^\.$')" "$(printf '%s\n' "$OUT" | head -1 | tr ' ' '\n' | grep -c '^\.\.$')")";; + esac +done + +echo "# qid collisions" +run qid_collide "cat $M/f; cat $M/d/g; ls $M/d; stat -c %i $M/f $M/d 2>&1; echo status=\$?" +no_crash "qid_collide" +expect_contains "qid_collide: reads work" "hello world" "$OUT" +run qid_zero "cat $M/f; ls $M | tr '\n' ' '; echo; cat $M/d/g; echo status=\$?" +no_crash "qid_zero" +expect_contains "qid_zero: file with root's qid.path is still a readable file" "hello world" "$OUT" +expect_contains "qid_zero: root still lists" "big d f fids" "$OUT" +expect_contains "qid_zero: nested file readable" "in d" "$OUT" + +echo "# version negotiation" +run version_unknown "echo ran" +expect_eq "version_unknown: 9ns refuses (125)" "125" "$RC" +run msize_tiny "cat $M/f 2>&1; echo status=\$?" +no_crash "msize_tiny" + +echo "# a server that never replies" +start_server never +# SIGTERM is forwarded to the child; once the child is gone 9ns must leave the +# pending 9P reply behind and exit even though the server stays silent. +timeout -s TERM 3 "$NS" --unix "$SOCK" -- sh -c "cat $M/f; echo unreachable" >"$TMP/never.out" 2>"$TMP/never.err" & +TPID=$! +sleep 4 +if kill -0 "$TPID" 2>/dev/null; then + fail "never: SIGTERM did not end 9ns while a reply was outstanding"; kill -9 "$TPID" +else + pass "never: SIGTERM ends 9ns even while the server is silent" +fi +wait "$TPID" 2>/dev/null +# Without a signal the mount hangs (documented v1 limitation) until the server dies. +timeout 30 "$NS" --unix "$SOCK" -- sh -c "cat $M/f; echo unreachable" >"$TMP/never.out" 2>"$TMP/never.err" & +TPID=$! +sleep 1.5 +if kill -0 "$TPID" 2>/dev/null; then + pass "never: mount hangs while the server is silent (documented v1 limitation)" + kill "$SRVPID"; wait "$SRVPID" 2>/dev/null; SRVPID= + for _ in $(seq 1 50); do kill -0 "$TPID" 2>/dev/null || break; sleep 0.1; done + if kill -0 "$TPID" 2>/dev/null; then fail "never: 9ns still alive after its server died"; kill -9 "$TPID"; else pass "never: killing the server unblocks 9ns"; fi +else + wait "$TPID"; fail "never: 9ns exited early ($?)" "$(cat "$TMP/never.err")" +fi + +echo "# interrupting a slow read (INTERRUPT must not confuse reply matching)" +run slow "cat $M/big > /dev/null; cat $M/f; cat $M/fids" --msize 8192 +base=$(printf '%s\n' "$OUT" | tail -1) +run slow "(cat $M/big > /dev/null & sleep 0.3; kill -INT \$!; wait \$!) 2>/dev/null; cat $M/f; cat $M/fids" --msize 8192 +no_crash "slow" +expect_contains "slow: read after interrupted read works" "hello world" "$OUT" +expect_eq "slow: fids after an interrupted read == after a complete one ($base)" "$base" "$(printf '%s\n' "$OUT" | tail -1)" + +echo +echo "passed=$PASSED failed=$FAILED" +[ "$FAILED" -eq 0 ] diff --git a/9ns/test/adv_bridge_semantics.sh b/9ns/test/adv_bridge_semantics.sh new file mode 100755 index 0000000..b38ce9c --- /dev/null +++ b/9ns/test/adv_bridge_semantics.sh @@ -0,0 +1,205 @@ +#!/usr/bin/env bash +# FUSE semantics through the bridge against 9proc-demo's /scratch tree. +# Usage: bash 9ns/test/adv_bridge_semantics.sh <9ns> <9proc-demo> (part of zig build 9ns-adv) +set -u +NS=$(realpath "${1:?path to 9ns}") +PROC=$(realpath "${2:?path to 9proc-demo}") +TMP=$(mktemp -d "${TMPDIR:-/tmp}/9ns-sem.XXXXXX") +M=/mnt/9p +S=$M/scratch +FAILED=0 +PASSED=0 +SRVPID= +cleanup() { [ -n "$SRVPID" ] && kill "$SRVPID" 2>/dev/null; rm -rf "$TMP"; } +trap cleanup EXIT +if ! unshare -Urm true 2>/dev/null || [ ! -c /dev/fuse ]; then echo "SKIP: no user namespaces or /dev/fuse"; exit 0; fi + +pass() { PASSED=$((PASSED + 1)); echo "ok - $1"; } +fail() { FAILED=$((FAILED + 1)); echo "FAIL - $1"; shift; [ $# -gt 0 ] && printf ' %s\n' "$@"; } +expect_eq() { if [ "$2" = "$3" ]; then pass "$1"; else fail "$1" "expected: $(printf %q "$2")" "actual: $(printf %q "$3")"; fi; } +expect_contains() { case "$3" in *"$2"*) pass "$1" ;; *) fail "$1" "missing: $(printf %q "$2")" "in: $(printf %q "$3")" ;; esac; } + +SOCK=$TMP/i.sock +"$PROC" --unix "$SOCK" >"$TMP/srv.out" 2>&1 & +SRVPID=$! +for _ in $(seq 1 100); do [ -S "$SOCK" ] && break; sleep 0.02; done +# Each run is a fresh session; state persists in the server, so tests clean up after themselves. +run() { OUT=$(timeout 120 "$NS" --unix "$SOCK" "${EXTRA[@]}" -- sh -c "$1" 2>"$TMP/stderr"); RC=$?; STDERR=$(cat "$TMP/stderr"); } +EXTRA=() +py() { run "python3 - <<'PYEOF' +$1 +PYEOF"; } + +echo "# open/create flags" +py " +import os, errno +p='$S/excl' +fd=os.open(p, os.O_CREAT|os.O_WRONLY, 0o644); os.write(fd, b'x'); os.close(fd) +try: + os.open(p, os.O_CREAT|os.O_EXCL|os.O_WRONLY, 0o644); print('no error') +except OSError as e: print(errno.errorcode[e.errno]) +os.unlink(p) +fd=os.open('$S/ro', os.O_CREAT|os.O_RDONLY, 0o644); print(os.read(fd, 10)); os.close(fd) +print(os.path.exists('$S/ro')); os.unlink('$S/ro') +" +expect_eq "O_EXCL on an existing file is EEXIST" "EEXIST" "$(printf '%s\n' "$OUT" | sed -n 1p)" +expect_eq "create with O_RDONLY works and reads empty" $'b\'\'\nTrue' "$(printf '%s\n' "$OUT" | sed -n 2,3p)" + +run "mkdir -m 700 $S/m7 && stat -c %a $S/m7; chmod 755 $S/m7 && stat -c %a $S/m7; rmdir $S/m7" +expect_eq "mkdir -m 700 then chmod 755" $'700\n755' "$OUT" +run "echo x > $S/c && chmod 600 $S/c && stat -c %a $S/c; chmod 444 $S/c && stat -c %a $S/c; rm -f $S/c" +expect_eq "chmod on a file" $'600\n444' "$OUT" +run "echo x > $S/t && touch -d @1000000000 $S/t && stat -c %Y $S/t; touch $S/t && [ \$(stat -c %Y $S/t) -gt 1000000000 ] && echo now; rm $S/t" +expect_eq "utimes (explicit) and touch (now)" $'1000000000\nnow' "$OUT" +run "echo abc > $S/tr && truncate -s 10 $S/tr && stat -c %s $S/tr && od -An -c $S/tr | tr -s ' ' | tr -d '\n'; echo; rm $S/tr" +expect_eq "truncate to larger zero-fills" $'10\n a b c \\n \\0 \\0 \\0 \\0 \\0 \\0' "$OUT" +run "echo a > $S/ap && echo b >> $S/ap && echo c >> $S/ap && cat $S/ap | tr '\n' ' '; rm $S/ap" +expect_eq "shell append" "a b c " "$OUT" +py " +import os +p='$S/ap2' +f1=os.open(p, os.O_CREAT|os.O_WRONLY|os.O_APPEND, 0o644) +f2=os.open(p, os.O_WRONLY|os.O_APPEND) +os.write(f1, b'one '); os.write(f2, b'two '); os.write(f1, b'three') +os.close(f1); os.close(f2) +print(open(p).read()); os.unlink(p) +" +expect_eq "O_APPEND from two descriptors interleaves in order" "one two three" "$OUT" +run "echo 0123456789 > $S/tt && (echo X > $S/tt) && cat $S/tt && stat -c %s $S/tt; rm $S/tt" +expect_eq "O_TRUNC (atomic_o_trunc) truncates before write" $'X\n2' "$OUT" + +echo "# reads at the edges" +run "printf hello > $S/e; dd if=$S/e bs=1 skip=100 count=5 2>/dev/null | wc -c; head -c 0 $S/e | wc -c; dd if=/dev/null of=$S/e bs=1 count=0 conv=notrunc 2>/dev/null; cat $S/e; echo; rm $S/e" +expect_eq "read past EOF is 0 bytes; 0-byte read/write are no-ops" $'0\n0\nhello' "$OUT" +head -c 4194304 /dev/urandom >"$TMP/four" +SUM=$(sha256sum <"$TMP/four" | cut -d' ' -f1) +run "dd if=$TMP/four of=$S/four bs=4M status=none && dd if=$S/four bs=4M status=none | sha256sum | cut -d' ' -f1; stat -c %s $S/four; rm $S/four" +expect_eq "4 MiB single-request dd round trip" "$SUM"$'\n4194304' "$OUT" +py " +import os +p='$S/lseek' +open(p,'w').write('0123456789') +f=open(p,'rb'); f.seek(0, 2); print(f.tell()); f.seek(-3, 2); print(f.read()); f.close() +os.unlink(p) +" +expect_eq "lseek SEEK_END on a direct_io file" $'10\nb\'789\'' "$OUT" + +echo "# unlink of an open file" +py " +import os +p='$S/unl' +fd=os.open(p, os.O_CREAT|os.O_RDWR, 0o644) +os.write(fd, b'before') +os.unlink(p) +print(os.path.exists(p)) +os.lseek(fd, 0, 0); print(os.read(fd, 100)) +os.write(fd, b'-after'); os.lseek(fd, 0, 0); print(os.read(fd, 100)) +os.close(fd) +" +expect_eq "read/write through the fd after unlink" $'False\nb\'before\'\nb\'before-after\'' "$OUT" + +echo "# rename" +run "echo A > $S/ra; echo B > $S/rb; mv $S/ra $S/rb && cat $S/rb; ls $S | tr '\n' ' '; echo; rm $S/rb" +expect_eq "rename over an existing file replaces it, no leftovers" $'A\nrb ' "$OUT" +run "mkdir $S/rd1 && echo x > $S/rd1/f && mv $S/rd1 $S/rd2 && cat $S/rd2/f && ls $S/rd2; rm -r $S/rd2; ls $S | wc -l" +expect_eq "rename of a directory" $'x\nf\n0' "$OUT" +run "mkdir $S/e1 $S/e2 && mv -T $S/e1 $S/e2 && ls $S | tr '\n' ' '; rmdir $S/e2" +expect_eq "rename dir over an empty dir" "e2 " "$OUT" +run "mkdir $S/n1 $S/n2 && echo x > $S/n2/f && mv -T $S/n1 $S/n2 2>&1 | sed 's/.*: //'; rm -r $S/n1 $S/n2" +expect_eq "rename dir over a non-empty dir is ENOTEMPTY" "Directory not empty" "$OUT" +py " +import os +p='$S/same'; open(p,'w').write('x'); os.rename(p, p); print(open(p).read()); os.unlink(p) +" +expect_eq "rename onto itself is a no-op" "x" "$OUT" +py " +import os, ctypes, errno +libc = ctypes.CDLL(None, use_errno=True) +a, b = b'$S/nra', b'$S/nrb' +open(a,'w').write('A'); open(b,'w').write('B') +r = libc.renameat2(-100, a, -100, b, 1) # RENAME_NOREPLACE +print('rc', r, errno.errorcode.get(ctypes.get_errno())) +print(open(b).read()) +os.unlink(a); os.unlink(b) +" +expect_eq "RENAME_NOREPLACE keeps the target" $'rc -1 EEXIST\nB' "$OUT" + +echo "# directories" +run "mkdir $S/many && cd $S/many && i=0; while [ \$i -lt 5000 ]; do : > f\$i; i=\$((i+1)); done; ls | wc -l; ls -l | wc -l; grep VmRSS /proc/\$PPID/status | awk '{print \$2}' > $TMP/rss1; rm -f $S/many/*; rmdir $S/many; ls | wc -l; grep VmRSS /proc/\$PPID/status | awk '{print \$2}' > $TMP/rss2" +expect_eq "5000 entries: ls and ls -l" $'5000\n5001\n0' "$OUT" +r1=$(cat "$TMP/rss1"); r2=$(cat "$TMP/rss2") +if [ "$r2" -le $((r1 + 2048)) ]; then pass "RSS after cleanup ($r2 KiB) <= after listing ($r1 KiB)+2 MiB"; else fail "RSS grew after cleanup: $r1 -> $r2 KiB"; fi +py " +import os +d='$S/chg'; os.mkdir(d) +for i in range(50): open(f'{d}/a{i}','w').close() +it = os.scandir(d); first = next(it).name +for i in range(3000): open(f'{d}/b{i}','w').close() +rest = [e.name for e in it] +print(first[0], len(rest) >= 49, len(set(rest)) == len(rest)) +for n in os.listdir(d): os.unlink(f'{d}/{n}') +os.rmdir(d) +" +expect_eq "readdir of a directory that changes mid-iteration" "a True True" "$OUT" +run "ls $M/.. > /dev/null && echo ok; stat -c %i $M $M/. $M/scratch/..; cd $M/scratch && ls .. | grep -c scratch" +expect_eq ".. of the root and of a subdir" $'ok\n1\n1\n1\n1' "$OUT" +run "cd $M && find . -type d | wc -l && find . -type f | head -1 && find $S -type f | wc -l" +expect_contains "find -type works" "./README" "$OUT" +run "stat -f -c '%T %S %l' $M; df -P $M | tail -1 | awk '{print \$1}'; sync -f $M && echo synced; sync && echo synced2" +expect_eq "statfs, df, syncfs, sync" $'fuse 4096 255\n9ns\nsynced\nsynced2' "$OUT" + +echo "# server refusals keep their errno through the error path" +run "echo x > $M/build/zig_version; a=\$?; mkdir $M/build/x 2>/dev/null; b=\$?; rm $M/README 2>/dev/null; c=\$?; rmdir $M/build 2>/dev/null; d=\$?; echo \$a\$b\$c\$d" 2>/dev/null +expect_eq "open-for-write / mkdir / rm / rmdir on read-only nodes all fail (errno preserved through error path)" "1111" "$(printf '%s\n' "$OUT" | tail -1)" + +echo "# unsupported operations fail cleanly" +run "echo x > $S/l1; ln $S/l1 $S/l2 2>&1 | sed 's/.*: //'; ln -s l1 $S/l3 2>&1 | sed 's/.*: //'; mkfifo $S/p 2>&1 | sed 's/.*: //'; ls $S | tr '\n' ' '; echo; rm $S/l1" +# The kernel turns ENOSYS from LINK into EPERM (fuse_link); symlink/mknod keep ENOSYS. +expect_eq "link/symlink/mknod fail cleanly" $'Operation not permitted\nFunction not implemented\nFunction not implemented\nl1 ' "$OUT" +run "echo x > $S/x1; setfattr -n user.a -v 1 $S/x1 2>&1 | sed 's/.*: //'; getfattr -n user.a $S/x1 2>&1 | sed 's/.*: //'; getfattr -d $S/x1 2>&1 | sed 's/.*: //'; rm $S/x1" +expect_eq "xattr ops are EOPNOTSUPP" $'Operation not supported\nOperation not supported\nOperation not supported' "$OUT" +py " +import os, fcntl, mmap, errno +p='$S/mm'; open(p,'w').write('mapme') +fd=os.open(p, os.O_RDWR) +fcntl.flock(fd, fcntl.LOCK_EX); fcntl.flock(fd, fcntl.LOCK_UN); fcntl.lockf(fd, fcntl.LOCK_EX); fcntl.lockf(fd, fcntl.LOCK_UN); print('locks ok') +try: + m = mmap.mmap(fd, 5); print('shared', bytes(m)); m.close() +except OSError as e: print('shared', errno.errorcode[e.errno]) +try: + m = mmap.mmap(fd, 5, flags=mmap.MAP_PRIVATE, prot=mmap.PROT_READ); print('private', bytes(m)); m.close() +except OSError as e: print('private', errno.errorcode[e.errno]) +os.close(fd); os.unlink(p) +" +expect_contains "flock/lockf work (local locks)" "locks ok" "$OUT" +case "$OUT" in *"shared ENODEV"*|*"shared b'mapme'"*) pass "shared mmap: clean result ($(printf '%s\n' "$OUT" | sed -n 2p))";; *) fail "shared mmap" "$OUT";; esac +expect_contains "private mmap reads the file" "private b'mapme'" "$OUT" + +echo "# tools" +mkdir -p "$TMP/tree/sub/deeper"; echo one > "$TMP/tree/a"; echo two > "$TMP/tree/sub/b"; head -c 70000 /dev/urandom > "$TMP/tree/sub/deeper/blob"; chmod 640 "$TMP/tree/a" +run "cp -a $TMP/tree $S/tree 2>&1; diff -r $TMP/tree $S/tree && echo same; stat -c %a $S/tree/a; cp -a $S/tree $TMP/back && diff -r $TMP/tree $TMP/back && echo back; rm -r $S/tree" +expect_eq "cp -a there and back" $'same\n640\nback' "$OUT" +run "cd $TMP && tar cf $S/t.tar tree && cd $S && mkdir tx && tar xf t.tar -C tx && diff -r $TMP/tree tx/tree && echo tar-ok; rm -r $S/tx $S/t.tar" +expect_eq "tar into and out of the mount" "tar-ok" "$OUT" +run "rsync -a $TMP/tree/ $S/rs/ && diff -r $TMP/tree $S/rs && echo rsync-ok; sleep 1.1; echo mod > $TMP/tree/a; rsync -a $TMP/tree/ $S/rs/ && cat $S/rs/a; rm -r $S/rs" +expect_eq "rsync -a twice" $'rsync-ok\nmod' "$OUT" +run "cd $S && mkdir repo && cd repo && git init -q . && git config user.email a@b && git config user.name n && echo hi > f && git add f && git commit -qm init && git log --oneline | wc -l && git status --porcelain | wc -l; cd $S && rm -rf repo; ls $S | wc -l" +expect_eq "git init/add/commit inside the mount" $'1\n0\n0' "$OUT" + +echo "# --no-direct-io" +EXTRA=(--no-direct-io) +run "cp $TMP/four $S/nd && cmp $TMP/four $S/nd && echo same; stat -c %s $S/nd; rm $S/nd" +expect_eq "no-direct-io: 4 MiB round trip" $'same\n4194304' "$OUT" +[ "$RC" -eq 0 ] || echo " stderr: $STDERR" +# Buffered writes are per-page without a writeback cache (kernel behaviour); the +# point here is only that a large buffered write is delivered intact. +EXTRA=(--no-direct-io) +head -c 262144 /dev/urandom > "$TMP/w" +WSUM=$(sha256sum <"$TMP/w" | cut -d' ' -f1) +run "cp $TMP/w $S/w && sha256sum < $S/w | cut -d' ' -f1; stat -c %s $S/w; rm $S/w" +expect_eq "no-direct-io: 256 KiB buffered write is intact" "$WSUM"$'\n262144' "$OUT" +EXTRA=() + +echo +echo "passed=$PASSED failed=$FAILED" +[ "$FAILED" -eq 0 ] diff --git a/9ns/test/adv_bridge_stress.sh b/9ns/test/adv_bridge_stress.sh new file mode 100755 index 0000000..b306b28 --- /dev/null +++ b/9ns/test/adv_bridge_stress.sh @@ -0,0 +1,64 @@ +#!/usr/bin/env bash +# Resource and concurrency stress through the bridge against 9proc-demo. +# Usage: bash 9ns/test/adv_bridge_stress.sh <9ns> <9proc-demo> (~1-2 min; part of zig build 9ns-adv) +set -u +NS=$(realpath "${1:?path to 9ns}") +PROC=$(realpath "${2:?path to 9proc-demo}") +TMP=$(mktemp -d "${TMPDIR:-/tmp}/9ns-stress.XXXXXX") +M=/mnt/9p +S=$M/scratch +FAILED=0 +PASSED=0 +SRVPID= +cleanup() { [ -n "$SRVPID" ] && kill "$SRVPID" 2>/dev/null; rm -rf "$TMP"; } +trap cleanup EXIT +if ! unshare -Urm true 2>/dev/null || [ ! -c /dev/fuse ]; then echo "SKIP: no user namespaces or /dev/fuse"; exit 0; fi +pass() { PASSED=$((PASSED + 1)); echo "ok - $1"; } +fail() { FAILED=$((FAILED + 1)); echo "FAIL - $1"; shift; [ $# -gt 0 ] && printf ' %s\n' "$@"; } +expect_eq() { if [ "$2" = "$3" ]; then pass "$1"; else fail "$1" "expected: $(printf %q "$2")" "actual: $(printf %q "$3")"; fi; } + +SOCK=$TMP/i.sock +"$PROC" --unix "$SOCK" >"$TMP/srv.out" 2>&1 & +SRVPID=$! +for _ in $(seq 1 100); do [ -S "$SOCK" ] && break; sleep 0.02; done +run() { OUT=$(timeout 600 "$NS" --unix "$SOCK" -- sh -c "$1" 2>"$TMP/stderr"); RC=$?; STDERR=$(cat "$TMP/stderr"); } + +echo "# 100k+ 9P operations in one session; RSS must plateau" +# Each iteration: create+write+close, open+read+close, unlink, plus a failing lookup: ~15 RPCs. +run "rss() { grep VmRSS /proc/\$PPID/status | awk '{print \$2}'; } +i=0; while [ \$i -lt 8000 ]; do echo \$i > $S/s; cat $S/s > /dev/null; rm $S/s; cat $S/none 2>/dev/null; i=\$((i+1)); if [ \$i -eq 2000 ]; then rss; fi; done; rss; ls $S | wc -l" +r1=$(printf '%s\n' "$OUT" | sed -n 1p); r2=$(printf '%s\n' "$OUT" | sed -n 2p); left=$(printf '%s\n' "$OUT" | sed -n 3p) +expect_eq "scratch left clean" "0" "$left" +if [ -n "$r1" ] && [ -n "$r2" ] && [ "$r2" -le $((r1 + 4096)) ]; then pass "RSS at 2000 iterations = $r1 KiB, at 8000 = $r2 KiB"; else fail "RSS grows: $r1 -> $r2 KiB" "$STDERR"; fi +case "$STDERR" in *leak*) fail "allocator reported leaks" "$STDERR";; *) pass "no leak report from the debug allocator";; esac + +echo "# eight processes hammering the mount concurrently" +run "mkdir $S/par; for p in 1 2 3 4 5 6 7 8; do ( + d=$S/par/p\$p; mkdir \$d; i=0; bad=0 + while [ \$i -lt 300 ]; do + printf '%s-%s' \$p \$i > \$d/f\$((i % 7)); v=\$(cat \$d/f\$((i % 7))); [ \"\$v\" = \"\$p-\$i\" ] || bad=\$((bad+1)) + mkdir \$d/dd; rmdir \$d/dd; ls \$d > /dev/null; i=\$((i+1)) + done; rm -r \$d; echo \$p:\$bad ) & done; wait; ls $S/par | wc -l; rmdir $S/par" +expect_eq "all workers verified their own data" "1:0 2:0 3:0 4:0 5:0 6:0 7:0 8:0" "$(printf '%s\n' "$OUT" | grep ':' | sort | tr '\n' ' ' | sed 's/ $//')" +expect_eq "parallel tree fully removed" "0" "$(printf '%s\n' "$OUT" | grep -v ':')" + +echo "# a process killed mid-read and mid-write" +run "head -c 8000000 /dev/urandom > $S/kb; (cat $S/kb > /dev/null & sleep 0.05; kill -9 \$!; wait \$!) 2>/dev/null; (cat /dev/zero > $S/kw & sleep 0.05; kill -9 \$!; wait \$!) 2>/dev/null; sha256sum < $S/kb | cut -c1-8 > /dev/null && echo readable; [ -f $S/kw ] && echo written; rm $S/kb $S/kw; ls $S | wc -l" +expect_eq "survives SIGKILL mid-read/mid-write" $'readable\nwritten\n0' "$OUT" + +echo "# many open handles at once (fh counter, fid table)" +run "python3 - <<'EOF' +import os +d='$S/fh'; os.mkdir(d) +fds=[] +for i in range(1500): + fd=os.open(f'{d}/h{i%50}', os.O_CREAT|os.O_RDWR, 0o644); os.write(fd, b'z'); fds.append(fd) +for fd in fds: os.close(fd) +for i in range(50): os.unlink(f'{d}/h{i}') +os.rmdir(d); print('ok') +EOF" +expect_eq "1500 simultaneous handles" "ok" "$OUT" + +echo +echo "passed=$PASSED failed=$FAILED" +[ "$FAILED" -eq 0 ] diff --git a/9ns/test/adv_ns_process.sh b/9ns/test/adv_ns_process.sh new file mode 100755 index 0000000..db759ca --- /dev/null +++ b/9ns/test/adv_ns_process.sh @@ -0,0 +1,202 @@ +#!/usr/bin/env bash +# Adversarial regression tests for 9ns/src/ns.zig and 9ns/src/main.zig: process, +# namespace, signal and CLI handling. Real namespaces, real FUSE. +# Usage: bash 9ns/test/adv_ns_process.sh <9ns> <9proc-demo> (part of zig build 9ns-adv) +# Exit 0 on success (or when the machine cannot run the tests), 1 on failure. +set -u + +NS=$(realpath "${1:?path to 9ns}") +PROC=$(realpath "${2:?path to 9proc-demo}") +# Unix socket paths are limited to ~107 bytes; keep the temp dir short. +TMP=$(mktemp -d "${TMPDIR:-/tmp}/9padv.XXXXXX") +PIDS=() +FAILED=0 +PASSED=0 + +cleanup() { + for p in "${PIDS[@]:-}"; do [ -n "$p" ] && kill "$p" 2>/dev/null; done + rm -rf "$TMP" +} +trap cleanup EXIT + +if ! unshare -Urm true 2>/dev/null; then echo "SKIP: unprivileged user namespaces unavailable"; exit 0; fi +if [ ! -c /dev/fuse ]; then echo "SKIP: /dev/fuse missing"; exit 0; fi + +pass() { PASSED=$((PASSED + 1)); echo "ok - $1"; } +fail() { FAILED=$((FAILED + 1)); echo "FAIL - $1"; shift; [ $# -gt 0 ] && printf ' %s\n' "$@"; } +expect_eq() { if [ "$2" = "$3" ]; then pass "$1"; else fail "$1" "expected: $(printf %q "$2")" "actual: $(printf %q "$3")"; fi; } +expect_contains() { case "$3" in *"$2"*) pass "$1" ;; *) fail "$1" "missing: $(printf %q "$2")" "in: $(printf %q "$3")" ;; esac; } + +SOCK=$TMP/s +"$PROC" --unix "$SOCK" & +PIDS+=($!) +for _ in $(seq 1 100); do [ -S "$SOCK" ] && break; sleep 0.05; done +[ -S "$SOCK" ] || { echo "9proc-demo did not create $SOCK"; exit 1; } +MI_BEFORE=$(grep -v " $TMP" /proc/self/mountinfo | sort) + +TIMEOUT=$(command -v timeout) +run() { "$TIMEOUT" 60 "$NS" --unix "$SOCK" "$@"; } + +echo "# CLI" +expect_eq "--help goes to stdout, exit 0" "Usage: 9ns" "$(run --help 2>/dev/null | head -1 | cut -c1-10; )" +expect_eq "--help exit code" "0" "$("$NS" --help >/dev/null 2>&1; echo $?)" +expect_eq "--version on stdout" "9ns" "$("$NS" --version 2>/dev/null | cut -d' ' -f1)" +expect_eq "single-dash typo is a usage error, not a program" "125" "$(run -mount /x -- true 2>/dev/null; echo $?)" +expect_contains "single-dash typo message" "unknown option -mount" "$(run -mount /x -- true 2>&1)" +expect_eq "--unix= empty is a usage error" "125" "$("$NS" --unix= -- true 2>/dev/null; echo $?)" +expect_contains "--unix= message" "socket path" "$("$NS" --unix= -- true 2>&1)" +expect_eq "--mount '' is a usage error" "125" "$(run --mount '' -- true 2>/dev/null; echo $?)" +expect_eq "--msize huge rejected" "125" "$(run --msize 4294967295 -- true 2>/dev/null; echo $?)" +expect_eq "--msize 16 MiB accepted" "ok" "$(run --msize 16777216 -- sh -c 'echo ok')" +expect_contains "empty program name is reported" "empty program name" "$(run -- '' 2>&1)" +expect_eq "empty program name exit" "125" "$(run -- '' 2>/dev/null; echo $?)" +expect_eq "empty \$SHELL falls back to /bin/sh" "0" "$(SHELL= run -- /dev/null 2>&1; echo $?)" +expect_eq "--fd with a closed descriptor fails early" "125" "$("$NS" --fd 987 -- true 2>/dev/null; echo $?)" +expect_contains "--fd bad descriptor message" "--fd 987: EBADF" "$("$NS" --fd 987 -- true 2>&1)" + +echo "# exec failures" +expect_eq "not found is 127" "127" "$(run -- no-such-program-9ns 2>/dev/null; echo $?)" +expect_eq "PATH element that is a file: still 127" "127" "$(PATH=/etc/passwd run -- true 2>/dev/null; echo $?)" +expect_contains "PATH element that is a file: message" "exec true: E" "$(PATH=/etc/passwd run -- true 2>&1)" +printf '#!/bin/sh\necho no\n' >"$TMP/nx"; chmod 644 "$TMP/nx" +expect_eq "non-executable is 126" "126" "$(run -- "$TMP/nx" 2>/dev/null; echo $?)" +mkdir -p "$TMP/p1" "$TMP/p2"; cp "$TMP/nx" "$TMP/p1/prog"; printf '#!/bin/sh\necho right\n' >"$TMP/p2/prog"; chmod 755 "$TMP/p2/prog" +expect_eq "non-executable first in PATH, executable later" "right" "$(PATH=$TMP/p1:$TMP/p2 run -- prog)" +expect_eq "non-executable only in PATH is 126" "126" "$(PATH=$TMP/p1 run -- prog 2>/dev/null; echo $?)" +expect_eq "argv[0] preserved" "sh" "$(run -- sh -c 'echo $0')" +expect_eq "PATH unset uses default" "ok" "$(env -u PATH "$NS" --unix "$SOCK" -- sh -c 'echo ok')" + +echo "# fd hygiene" +# 9ns passes inherited descriptors through untouched, so compare with what a +# plain child of this script sees (the runner may itself hold extra fds). +FD_LIST='ls /proc/self/fd | grep -v "^3$" | sort -n | tr "\n" " " | sed "s/ $//"' +FD_BASE=$(sh -c "$FD_LIST") +expect_eq "no extra fds in the program (unix)" "$FD_BASE" "$(run -- sh -c "$FD_LIST")" +expect_eq "no extra fds in the program (spawn)" "$FD_BASE" "$("$TIMEOUT" 60 "$NS" --spawn "$PROC --stdio" -- sh -c "$FD_LIST")" +expect_eq "--fd transport does not leak into the program" "0 1 2" "$(python3 - "$NS" "$SOCK" <<'EOF' +import socket, subprocess, sys, os +s = socket.socket(socket.AF_UNIX, socket.SOCK_STREAM); s.connect(sys.argv[2]) +r = subprocess.run([sys.argv[1], "--fd", str(s.fileno()), "--", "sh", "-c", + 'ls /proc/self/fd | grep -v "^3$" | sort -n | tr "\n" " " | sed "s/ $//"'], + pass_fds=[s.fileno()], capture_output=True, text=True) +print(r.stdout.strip()) +EOF +)" + +echo "# signals" +expect_eq "SIGTERM forwarded" "143" "$(run -- sh -c 'kill -TERM $PPID; sleep 5; echo alive' >/dev/null 2>&1; echo $?)" +expect_eq "SIGHUP forwarded" "129" "$(run -- sh -c 'kill -HUP $PPID; sleep 5; echo alive' >/dev/null 2>&1; echo $?)" +expect_eq "SIGINT to 9ns is ignored while the child lives" "still-here" "$(run -- sh -c 'kill -INT $PPID; sleep 0.3; echo still-here')" +# Ctrl-C from the tty must not kill the --spawn server (same process group). +expect_eq "Ctrl-C on the tty leaves the --spawn server alive" "ok" "$(timeout 30 python3 - "$NS" "$PROC" <<'EOF' +import os, pty, sys, time, select +P, I = sys.argv[1], sys.argv[2] +prog = ["python3", "-c", """ +import os, signal, sys, time +signal.signal(signal.SIGINT, lambda *a: None) +m = os.environ['NINE_MOUNT'] +open(m + '/build/optimize').read() +sys.stdin.readline() +try: + open(m + '/build/optimize').read(); print('ok', flush=True) +except Exception as e: + print('mount dead:', e, flush=True) +"""] +pid, fd = pty.fork() +if pid == 0: + os.execv(P, [P, "--spawn", I + " --stdio", "--"] + prog) +out = b"" +def rd(t): + global out + end = time.time() + t + while time.time() < end: + r, _, _ = select.select([fd], [], [], 0.1) + if r: + try: d = os.read(fd, 4096) + except OSError: return + if not d: return + out += d +rd(1.5); os.write(fd, b"\x03"); rd(0.7); os.write(fd, b"\n"); rd(3) +os.waitpid(pid, 0) +print(out.decode(errors="replace").replace("^C", "").strip().splitlines()[-1] if out.strip() else "no output") +EOF +)" +# The --spawn server dying mid-session is reaped (no zombie) and does not end the session. +OUT=$("$TIMEOUT" 60 "$NS" --spawn "$PROC --stdio" -- sh -c 'srv=$(cat $NINE_MOUNT/runtime/pid); kill -TERM $srv; sleep 0.5; st=$(ps -o stat= -p $srv 2>/dev/null); echo "${st:-gone}"; exit 5' 2>/dev/null); RC=$? +expect_eq "server death mid-session: exit status still the child's, server reaped (no zombie)" "5 gone" "$RC $OUT" +# A server that never answers: once the child is dead, SIGTERM must end 9ns. +cat >"$TMP/hang.py" <<'EOF' +import struct, os, sys, time +def rd(n): + b = b"" + while len(b) < n: + c = os.read(0, n - len(b)) + if not c: sys.exit(0) + b += c + return b +while True: + size, = struct.unpack("/dev/null & +HP=$! +sleep 1; kill -TERM $HP +START=$(date +%s) +for _ in $(seq 1 100); do kill -0 $HP 2>/dev/null || break; sleep 0.1; done +if kill -0 $HP 2>/dev/null; then kill -KILL $HP; RC=hung; else wait $HP; RC=$?; fi +expect_eq "hung server: one SIGTERM ends 9ns once the child is dead (watchdog)" "143" "$RC" +expect_eq "hung server: exit was prompt" "yes" "$([ $(( $(date +%s) - START )) -lt 8 ] && echo yes)" +pkill -f "$TMP/hang.py" 2>/dev/null + +echo "# child/parent protocol" +if command -v strace >/dev/null 2>&1 && strace -qq -e trace=none true 2>/dev/null; then + expect_eq "child killed before handoff" "125" "$(strace -f -qq -e trace=unshare -e inject=unshare:signal=KILL -o /dev/null timeout 20 "$NS" --unix "$SOCK" -- true 2>/dev/null; echo $?)" + expect_contains "child killed before handoff: message" "child exited before reporting" "$(strace -f -qq -e trace=unshare -e inject=unshare:signal=KILL -o /dev/null timeout 20 "$NS" --unix "$SOCK" -- true 2>&1)" + expect_eq "status handoff fails" "125" "$(strace -f -qq -e trace=sendmsg -e inject=sendmsg:error=EPIPE -o /dev/null timeout 20 "$NS" --unix "$SOCK" -- true 2>/dev/null; echo $?)" + # recvmsg skipped (returns 1 without the fd): the child must be killed, not exec'd onto a dead mount. + OUT=$(strace -f -qq -e trace=recvmsg -e inject=recvmsg:retval=1:when=1 -o /dev/null timeout 20 "$NS" --unix "$SOCK" -- sh -c 'echo child-ran' 2>&1; echo "rc=$?") + expect_contains "truncated fd handoff: child not exec'd" "rc=125" "$OUT" + expect_eq "truncated fd handoff: program never ran" "no" "$(case "$OUT" in *child-ran*) echo yes;; *) echo no;; esac)" + expect_contains "fuse mount failure is reported" "mount fuse: EPERM" "$(strace -f -qq -e trace=mount -e inject=mount:error=EPERM:when=2 -o /dev/null timeout 20 "$NS" --unix "$SOCK" --mount "$TMP/mp" -- true 2>&1)" +else + echo "skip - strace unavailable (child failure injection)" +fi +expect_contains "fork failure is reported" "fork: E" "$(python3 -c " +import resource, os +resource.setrlimit(resource.RLIMIT_NPROC, (1, 1)) +os.execv('$NS', ['$NS', '--unix', '$SOCK', '--', 'true'])" 2>&1)" + +echo "# mountpoint policy" +ln -s /nonexistent "$TMP/dangling" +expect_contains "dangling symlink mountpoint" "dangling symlink" "$(run --mount "$TMP/dangling" -- true 2>&1)" +expect_eq "refuse to shadow / via /proc/self/root" "125" "$(run --mount /proc/self/root/x9p -- true 2>/dev/null; echo $?)" +expect_contains "refuse to shadow / via /proc/self/root: message" "refusing to shadow /" "$(run --mount /proc/self/root/x9p -- true 2>&1)" +expect_eq "refuse to shadow under /proc" "125" "$(run --mount /proc/self/fd/x9p -- true 2>/dev/null; echo $?)" +if [ "$(ls -A /usr/lib | wc -l)" -gt 4096 ]; then + expect_contains "parent with >4096 entries refused" "more than 4096 entries" "$(run --mount /usr/lib/x9p -- true 2>&1)" +else + echo "skip - no root-owned directory with >4096 entries" +fi +expect_eq "shadowed /run keeps its entries" "$(ls -A /run | sort | tr '\n' ' ')" "$(run --mount /run/x9p -- sh -c 'ls -A /run | grep -v "^x9p$" | sort | tr "\n" " "')" +expect_eq "mountpoint with spaces" "ok" "$(mkdir -p "$TMP/with space" && run --mount "$TMP/with space" -- sh -c '[ -f "$NINE_MOUNT/README" ] && echo ok')" +expect_eq "mountpoint is a file" "125" "$(run --mount "$TMP/nx" -- true 2>/dev/null; echo $?)" + +echo "# leaks" +for i in $(seq 1 30); do run -- sh -c 'cat $NINE_MOUNT/build/optimize >/dev/null' 2>/dev/null; done +BG=(); for i in $(seq 1 10); do ( run -- sh -c 'cat $NINE_MOUNT/build/optimize >/dev/null' 2>/dev/null ) & BG+=($!); done; wait "${BG[@]}" # not a bare wait: that would also wait for the server +sleep 0.3 +expect_eq "no stray 9ns processes" "" "$(pgrep -f "^$NS " | tr '\n' ' ')" +expect_eq "no stray --stdio servers" "" "$(pgrep -f "$PROC --stdio" | tr '\n' ' ')" +expect_eq "host mount table untouched" "same" "$([ "$MI_BEFORE" = "$(grep -v " $TMP" /proc/self/mountinfo | sort)" ] && echo same || echo changed)" + +echo +echo "passed=$PASSED failed=$FAILED" +[ "$FAILED" -eq 0 ] diff --git a/9ns/test/adversarial.sh b/9ns/test/adversarial.sh new file mode 100755 index 0000000..71c6f44 --- /dev/null +++ b/9ns/test/adversarial.sh @@ -0,0 +1,15 @@ +#!/usr/bin/env bash +# Runs every 9ns adversarial suite in sequence (hostile servers, FUSE +# semantics, process/namespace edge cases, stress). The suites aimed at the +# 9proc server itself live in 9proc/test (zig build 9proc-adv). +# Usage: bash 9ns/test/adversarial.sh <9ns> <9proc-demo> (zig build 9ns-adv) +set -u +NS=${1:?path to 9ns} +PROC=${2:?path to 9proc-demo} +HERE=$(cd "$(dirname "$0")" && pwd) +status=0 +for suite in adv_ns_process adv_bridge_hostile adv_bridge_semantics adv_bridge_stress; do + echo "### $suite" + if bash "$HERE/$suite.sh" "$NS" "$PROC"; then echo "### $suite: ok"; else echo "### $suite: FAILED"; status=1; fi +done +exit $status diff --git a/9ns/test/integration.sh b/9ns/test/integration.sh new file mode 100755 index 0000000..5e0ad13 --- /dev/null +++ b/9ns/test/integration.sh @@ -0,0 +1,160 @@ +#!/usr/bin/env bash +# Integration tests for 9ns: real user+mount namespaces, real FUSE, real 9P servers. +# Usage: bash 9ns/test/integration.sh <9ns> <9proc-demo> (zig build 9ns-itest) +# Exit 0 on success (or when the machine cannot run the tests), 1 on failure. +set -u + +NS=$(realpath "${1:?path to 9ns}") +PROC=$(realpath "${2:?path to 9proc-demo}") +TMP=$(mktemp -d "${TMPDIR:-/tmp}/9ns-itest.XXXXXX") +PIDS=() +FAILED=0 +PASSED=0 +M=/mnt/9p + +cleanup() { + for p in "${PIDS[@]:-}"; do [ -n "$p" ] && kill "$p" 2>/dev/null; done + rm -rf "$TMP" +} +trap cleanup EXIT + +if ! unshare -Urm true 2>/dev/null; then + echo "SKIP: unprivileged user namespaces unavailable"; exit 0 +fi +if [ ! -c /dev/fuse ]; then + echo "SKIP: /dev/fuse missing"; exit 0 +fi + +pass() { PASSED=$((PASSED + 1)); echo "ok - $1"; } +fail() { FAILED=$((FAILED + 1)); echo "FAIL - $1"; shift; [ $# -gt 0 ] && printf ' %s\n' "$@"; } +expect_eq() { # name expected actual + if [ "$2" = "$3" ]; then pass "$1"; else fail "$1" "expected: $(printf %q "$2")" "actual: $(printf %q "$3")"; fi +} +expect_contains() { # name needle haystack + case "$3" in *"$2"*) pass "$1" ;; *) fail "$1" "missing: $(printf %q "$2")" "in: $(printf %q "$3")" ;; esac +} + +wait_socket() { # path + for _ in $(seq 1 100); do [ -S "$1" ] && return 0; sleep 0.05; done + return 1 +} + +# run_in "" — run inside a namespace with the current transport ($TRANSPORT array). +run_in() { timeout 60 "$NS" "${TRANSPORT[@]}" -- sh -c "$1" 2>"$TMP/stderr"; } + +# --- scratch battery: works against any writable 9P tree rooted at $1 (relative to mount) --- +scratch_battery() { # label scratchdir + local label=$1 S=$2 + + expect_eq "$label: create+append+read" $'hello\nworld' "$(run_in "echo hello > $M/$S/a && echo world >> $M/$S/a && cat $M/$S/a")" + expect_eq "$label: overwrite" "x" "$(run_in "echo x > $M/$S/a && cat $M/$S/a")" + expect_eq "$label: stat size after overwrite" "2" "$(run_in "stat -c %s $M/$S/a")" + expect_eq "$label: truncate" "0" "$(run_in "truncate -s 0 $M/$S/a && stat -c %s $M/$S/a")" + expect_eq "$label: truncate extend" "10" "$(run_in "truncate -s 10 $M/$S/a && stat -c %s $M/$S/a")" + expect_eq "$label: mkdir -p nested" "directory" "$(run_in "mkdir -p $M/$S/d1/d2/d3 && stat -c %F $M/$S/d1/d2/d3")" + expect_eq "$label: rename within dir" "moved" "$(run_in "echo moved > $M/$S/d1/d2/f && mv $M/$S/d1/d2/f $M/$S/d1/d2/g && cat $M/$S/d1/d2/g")" + # mv(1) silently falls back to copy+delete on EXDEV, so probe rename(2) directly. + expect_contains "$label: rename across dirs is EXDEV" "EXDEV" "$(run_in "python3 -c 'import os,errno +try: os.rename(\"$M/$S/d1/d2/g\", \"$M/$S/d1/g\") +except OSError as e: print(errno.errorcode[e.errno]) +'")" + expect_eq "$label: rm file" "gone" "$(run_in "rm $M/$S/d1/d2/g && [ ! -e $M/$S/d1/d2/g ] && echo gone")" + expect_eq "$label: rmdir non-empty fails" "1" "$(run_in "rmdir $M/$S/d1 2>/dev/null; echo \$?")" + expect_eq "$label: rmdir chain" "ok" "$(run_in "rmdir $M/$S/d1/d2/d3 $M/$S/d1/d2 $M/$S/d1 && echo ok")" + expect_eq "$label: ENOENT" "1" "$(run_in "cat $M/$S/nope 2>/dev/null; echo \$?")" + expect_eq "$label: ENOENT errno text" "No such file or directory" "$(run_in "cat $M/$S/nope 2>&1 | sed 's/.*: //'")" + + head -c 1048576 /dev/urandom >"$TMP/rand" + local sum; sum=$(sha256sum <"$TMP/rand" | cut -d' ' -f1) + expect_eq "$label: 1 MiB round trip (cp)" "$sum" "$(run_in "cp $TMP/rand $M/$S/big && sha256sum < $M/$S/big | cut -d' ' -f1")" + expect_eq "$label: 1 MiB size" "1048576" "$(run_in "stat -c %s $M/$S/big")" + expect_eq "$label: odd block sizes (dd bs=1000)" "$sum" "$(run_in "dd if=$M/$S/big of=$M/$S/big2 bs=1000 status=none && sha256sum < $M/$S/big2 | cut -d' ' -f1")" + expect_eq "$label: partial read at offset" "$(tail -c 12345 "$TMP/rand" | sha256sum | cut -d' ' -f1)" "$(run_in "tail -c 12345 $M/$S/big | sha256sum | cut -d' ' -f1")" + expect_eq "$label: many small files" "200" "$(run_in "mkdir $M/$S/many && for i in \$(seq 1 200); do echo \$i > $M/$S/many/f\$i; done; ls $M/$S/many | wc -l")" + expect_eq "$label: find count" "201" "$(run_in "find $M/$S/many | wc -l")" + expect_eq "$label: readdir contents" "f1 f100 f200" "$(run_in "cd $M/$S/many && ls f1 f100 f200 | tr '\n' ' ' | sed 's/ \$//'")" + expect_eq "$label: cleanup many" "0" "$(run_in "rm -r $M/$S/many $M/$S/big $M/$S/big2 $M/$S/a; ls $M/$S | wc -l")" +} + +# ============================================================================ +echo "# 9proc-demo over a Unix socket" +SOCK=$TMP/9proc.sock +"$PROC" --unix "$SOCK" & +PIDS+=($!) +wait_socket "$SOCK" || { echo "9proc-demo did not create $SOCK"; exit 1; } +TRANSPORT=(--unix "$SOCK") + +expect_eq "zig_version" "$(zig version)" "$(run_in "cat $M/build/zig_version")" +expect_eq "mount exported" "$M" "$(run_in 'echo $NINE_MOUNT')" +expect_contains "root listing" "build" "$(run_in "ls $M")" +expect_contains "root listing has scratch" "scratch" "$(run_in "ls $M")" +expect_eq "README size > 0" "yes" "$(run_in "[ \$(stat -c %s $M/README) -gt 0 ] && echo yes")" +expect_eq "README readable" "yes" "$(run_in "[ -s $M/README ] && head -c 1 $M/README >/dev/null && echo yes")" +expect_eq "fn/now numeric" "num" "$(run_in "cat $M/runtime/fn/now | grep -Eq '^[0-9]+\$' && echo num")" +expect_eq "fn listing from comptime" "yes" "$(run_in "ls $M/runtime/fn | grep -q hostname && echo yes")" +expect_eq "ctl round trip" "5" "$(run_in "echo 'add 2 3' > $M/runtime/ctl && cat $M/runtime/ctl")" +expect_eq "ctl echo" "hi there" "$(run_in "echo 'echo hi there' > $M/runtime/ctl && cat $M/runtime/ctl")" +expect_contains "comptime types" "Qid" "$(run_in "ls $M/comptime/types")" +expect_eq "comptime size of Qid" "16" "$(run_in "cat $M/comptime/types/Qid/size")" +expect_eq "runtime pid is server pid" "${PIDS[-1]}" "$(run_in "cat $M/runtime/pid")" +expect_eq "exit status propagates" "7" "$(run_in 'exit 7'; echo $?)" +expect_eq "mount is fuse" "yes" "$(run_in "grep -q \"^9ns $M fuse\" /proc/mounts && echo yes")" +expect_eq "host /mnt entries still visible" "$(ls -A /mnt | sort | tr '\n' ' ')" "$(run_in "ls -A /mnt | grep -v '^9p\$' | sort | tr '\n' ' '")" +expect_eq "host mount table untouched" "no" "$(grep -q " $M " /proc/self/mountinfo && echo yes || echo no)" +scratch_battery "9proc-demo" scratch + +echo "# nested 9ns" +expect_eq "nested mount" "$(zig version)" "$(run_in "$NS --unix $SOCK --mount $TMP/inner -- sh -c 'cat \$NINE_MOUNT/build/zig_version'")" + +echo "# --mount variants" +mkdir -p "$TMP/mnt" +expect_eq "--mount existing dir" "ok" "$(timeout 60 "$NS" --unix "$SOCK" --mount "$TMP/mnt" -- sh -c "[ -f $TMP/mnt/README ] && echo ok")" +expect_eq "--mount relative" "ok" "$(cd "$TMP" && timeout 60 "$NS" --unix "$SOCK" --mount rel -- sh -c "[ -f $TMP/rel/README ] && echo ok")" +expect_eq "--mount missing under /" "125" "$(timeout 60 "$NS" --unix "$SOCK" --mount /nonexistent-9ns-dir -- true 2>/dev/null; echo $?)" + +echo "# lifecycle" +START=$(date +%s) +expect_eq "background grandchild does not block exit" "3" "$(run_in 'sleep 30 >/dev/null 2>&1 & exit 3'; echo $?)" +expect_eq "exit was prompt" "yes" "$([ $(( $(date +%s) - START )) -lt 10 ] && echo yes)" +expect_eq "SIGTERM forwarded" "143" "$(timeout 60 "$NS" --unix "$SOCK" -- sh -c 'kill -TERM $PPID; sleep 5; echo alive' >/dev/null 2>&1; echo $?)" + +echo "# --spawn transport" +TRANSPORT=(--spawn "$PROC --stdio") +expect_eq "spawn: zig_version" "$(zig version)" "$(run_in "cat $M/build/zig_version")" +expect_eq "spawn: ctl" "7" "$(run_in "echo 'add 3 4' > $M/runtime/ctl && cat $M/runtime/ctl")" +expect_eq "spawn: stateful sequence in one session" 'hello world 12 moved 0' "$(run_in "cd $M/scratch && echo hello > a && echo world >> a && cat a && stat -c %s a && mkdir d && echo moved > d/f && mv d/f d/g && cat d/g && rm d/g && rmdir d && rm a && ls | wc -l" | tr '\n' ' ' | sed 's/ $//')" + +echo "# --tcp transport" +PORT=$(( 20000 + RANDOM % 20000 )) +"$PROC" --tcp "127.0.0.1:$PORT" & +PIDS+=($!) +sleep 0.3 +TRANSPORT=(--tcp "127.0.0.1:$PORT") +expect_eq "tcp: zig_version" "$(zig version)" "$(run_in "cat $M/build/zig_version")" +expect_eq "tcp: scratch" "tcp" "$(run_in "echo tcp > $M/scratch/t && cat $M/scratch/t && rm $M/scratch/t")" + +echo "# server death" +"$PROC" --unix "$TMP/dying.sock" & +DYING=$! +wait_socket "$TMP/dying.sock" +TRANSPORT=(--unix "$TMP/dying.sock") +OUT=$(run_in "cat $M/build/optimize >/dev/null && kill $DYING && sleep 0.3; cat $M/build/optimize 2>&1 >/dev/null | sed 's/.*: //'; echo status=\$?") +expect_contains "server death yields an error, not a hang" "status=0" "$OUT" +expect_eq "server death errno text" "yes" "$(case "$OUT" in *"Input/output error"*|*"Transport endpoint is not connected"*) echo yes;; *) echo "no: $OUT";; esac)" + +echo "# plan9port ramfs (independent 9P2000 implementation)" +if [ -x /usr/lib/plan9/bin/ramfs ]; then + mkdir -p "$TMP/p9ns" + NAMESPACE=$TMP/p9ns /usr/lib/plan9/bin/ramfs -s ramfs + wait_socket "$TMP/p9ns/ramfs" || echo "ramfs socket missing" + PIDS+=($(pgrep -f "9pserve -u unix!$TMP/p9ns/ramfs")) + TRANSPORT=(--unix "$TMP/p9ns/ramfs") + expect_eq "ramfs: mkdir scratch" "ok" "$(run_in "mkdir $M/scratch && echo ok")" + scratch_battery "ramfs" scratch +else + echo "skip - plan9port ramfs not installed" +fi + +echo +echo "passed=$PASSED failed=$FAILED" +[ "$FAILED" -eq 0 ] diff --git a/9player/README.md b/9player/README.md deleted file mode 100644 index 77081d6..0000000 --- a/9player/README.md +++ /dev/null @@ -1,156 +0,0 @@ -# 9player - -Mount a 9P2000 file tree into a fresh mount namespace and run a program in it, -as a plain user, without touching the host's mount table. - -```sh -9player --unix /run/user/1000/acme -- fish # a shell that sees the tree at /mnt/9p -9player --tcp 127.0.0.1:564 -- claude # an agent that sees it too -9player --spawn 'introspect --stdio' -- bash # start the server yourself, talk over a socketpair -``` - -Inside, the tree is ordinary files: `ls`, `cat`, `echo x > ctl`, editors, -`find`, `rsync`, whatever. `$NINEPLAYER_MOUNT` tells programs where it is -(default `/mnt/9p`). When the program exits, 9player exits with its status -and the namespace, mount and connection disappear. - -## How it works - -The kernel's own `9p` filesystem cannot be mounted inside an unprivileged user -namespace, so 9player is a small FUSE server that speaks 9P2000 to the real -server. There is no libfuse and no libc: `src/fuse.zig` implements the subset -of the kernel FUSE protocol needed, straight from `linux/fuse.h`. - -``` - program (fish/bash/claude) 9player (parent) 9P server - in a new user+mount namespace │ - /mnt/9p ── FUSE ──▶ kernel ────▶│ fuse.zig ─▶ bridge.zig ─▶ nine.zig ──▶ unix / tcp / socketpair - │ (framing) (translation) (cloud9 Client) -``` - -1. The parent connects to the 9P server (version + attach) so failures are - reported before anything is forked. -2. The child does `unshare(CLONE_NEWUSER|CLONE_NEWNS)`, maps its own uid/gid, - makes every mount private, opens `/dev/fuse` (it must be opened inside the - new user namespace), mounts it on the mountpoint and passes the descriptor - back to the parent over `SCM_RIGHTS`, then execs the program. -3. The parent serves FUSE requests by translating them into 9P transactions - (`Twalk`, `Topen`, `Tread`, `Twrite`, `Tcreate`, `Tremove`, `Tstat`, - `Twstat`) until the child exits or the namespace disappears. - -Files are opened with `FOPEN_DIRECT_IO`, so synthetic files that report length -0 (the 9P convention for control files) still read correctly, and `O_TRUNC` -travels inside the 9P open mode (`OTRUNC`) rather than as a separate -truncate. Repeated lookups of the same qid map to the same inode. - -If `/mnt/9p` does not exist and cannot be created (the normal case), 9player -mounts a tmpfs over `/mnt` *inside the namespace only* and bind-mounts every -existing entry of `/mnt` back into it, so nothing is hidden. Pass `--mount DIR` -to use any other directory. - -## Building and testing - -9player lives in the [cloud9](../) repository as `cloud9/9player/`, beside -the 9P2000 protocol library it is built on, and is wired into cloud9's -`build.zig` through the fragment `9player/build.zig`. Everything is run from -the cloud9 root with Zig 0.16: - -```sh -zig build # zig-out/bin/{9player,introspect,cloud9-http,cloud9-probe} -zig build 9player # build and install only zig-out/bin/9player -zig build 9player-test # unit tests (protocol structs, session, bridge, namespace helpers) -zig build 9player-itest # integration tests: real namespaces, real FUSE, - # introspect over unix/tcp/socketpair, and plan9port's - # ramfs when /usr/lib/plan9/bin/ramfs is installed -zig build 9player-adv # adversarial suites: hostile 9P servers, FUSE semantics, - # namespace/signal edge cases, stress (several minutes) -zig build -Doptimize=ReleaseSafe -zig build -D9player=false # leave 9player out (the default on non-Linux targets) -``` - -The integration suites mount the `introspect` demo server (`../introspect`), -so they need `-Dintrospect=true` (the default on Linux), unprivileged user -namespaces (`kernel.unprivileged_userns_clone=1` on distributions that have -the knob), `/dev/fuse` and Python 3; they skip themselves otherwise. -`zig build programs-test` and `programs-itest` run the unit and integration -steps of every program in the repository. - -## Usage - -``` -9player [options] -- PROGRAM [ARGS...] - -Transport (exactly one): - --unix PATH Unix stream socket - --tcp IP:PORT TCP (IPv4/IPv6 literal) - --fd N an already-connected inherited descriptor - --spawn CMD run CMD via /bin/sh -c with a socketpair on its stdin/stdout - -Options: - --mount PATH mountpoint inside the new namespace (default /mnt/9p) - --uname NAME 9P user name (default $USER) - --aname NAME 9P tree to attach (default "") - --msize BYTES maximum 9P message size to request (default 131072, max 16 MiB) - --cache SECONDS attr/entry cache validity, fractional allowed (default 1) - --no-direct-io let the kernel cache file pages (trusts stat length) - --debug trace FUSE and 9P operations on stderr - --help, --version -``` - -PROGRAM defaults to `$SHELL`. Exit status is the program's (`128+signal` if it -was killed); 125 means 9player itself failed (usage, connect, namespace, -mount); 126/127 are exec failures as usual. - -## introspect: a demo 9P server - -`introspect` is a single-binary 9P2000 server whose file tree is the binary -itself: build-time facts, `comptime` reflection and live runtime state. It is -the demo of the [introspect library](../introspect) (`../introspect/demo/main.zig`), -built and installed by `zig build introspect`. - -``` -/README -/build/{zig_version,target,optimize,time,change} captured by build.zig (jj change id, UTC time) -/comptime/types//{name,size,align,fields} @sizeOf/@alignOf/@typeInfo, generated at comptime -/comptime/decls pub declarations of the server module -/runtime/{pid,ppid,uptime,argv,cwd,env,clients} -/runtime/fn/ reading calls a Zig function (hostname, now, random, uname, fib30); - the directory is generated from @typeInfo of the Fns struct -/runtime/ctl write "fib N" | "add A B" | "echo TEXT" | "sleep-ms N", read the result -/scratch/ in-memory read/write tree -``` - -`/runtime/env` exposes the server's whole environment, so serve introspect on -a Unix socket or loopback only. - -```sh -zig-out/bin/introspect --unix /tmp/intro.sock & -zig-out/bin/9player --unix /tmp/intro.sock -- sh -c ' - cat $NINEPLAYER_MOUNT/build/zig_version; echo - cat $NINEPLAYER_MOUNT/comptime/types/Qid/fields - echo "fib 20" > $NINEPLAYER_MOUNT/runtime/ctl; cat $NINEPLAYER_MOUNT/runtime/ctl' -``` - -The same command with `claude -p "explore /mnt/9p ..."` as the program gives an -agent a live, file-shaped view into a running process; that is the intended -use. - -## Limitations - -* One 9P request is in flight at a time; a server read that blocks (event - files) stalls the mount until it returns. -* Base 9P2000 only: no symlinks, ownership, xattrs or locks. Every file is - reported as owned by the invoking user. Cross-directory rename is `EXDEV`. -* No PID namespace and no `/proc` remount. `--tcp` takes IP literals only - (no libc, no resolver). -* Linux only. - -## Relation to cloud9 - -9player consumes cloud9 as the module `cloud9` and keeps all mounting, -namespace and process policy on its side, which is what cloud9's design asks -of applications. It ships from the cloud9 repository as the sibling directory -`9player/` (sources in `src/`, suites in `test/`, this README and -`docs/DESIGN.md`) with a build fragment that the root `build.zig` enables with -`-D9player` on Linux targets; nothing in the code depends on that layout. -`docs/DESIGN.md` has the full module contracts. diff --git a/9player/build.zig b/9player/build.zig deleted file mode 100644 index f3d5c1b..0000000 --- a/9player/build.zig +++ /dev/null @@ -1,71 +0,0 @@ -//! Build fragment for 9player: mount a 9P2000 tree into a fresh mount -//! namespace via FUSE and run a program in it (Linux only, no libc). It is -//! `@import`ed by the root build.zig and called with the root builder, so -//! every `b.path(...)` here is relative to the cloud9 root (hence the -//! `9player/` prefix), every option is defined by the root (no -//! `standardTargetOptions` here) and every step it registers lands in the -//! root's step list under the `9player` prefix. -//! -//! Steps: 9player, 9player-test, 9player-itest, 9player-adv. -const std = @import("std"); - -/// What the root passes in. The root owns target/optimize resolution and the -/// cloud9 module; the integration suites also need a 9P server to mount, -/// which is the introspect demo built by the sibling fragment. -pub const Context = struct { - target: std.Build.ResolvedTarget, - optimize: std.builtin.OptimizeMode, - cloud9: *std.Build.Module, - /// The `introspect` demo server; null when introspect is disabled, in - /// which case the end-to-end steps exist but fail with a notice. - introspect_demo: ?*std.Build.Step.Compile, -}; - -pub const Artifacts = struct { - /// The 9player executable (also installed by the plain `zig build`). - exe: *std.Build.Step.Compile, - /// `9player-test`, `9player-itest`, `9player-adv`. - test_step: *std.Build.Step, - itest_step: *std.Build.Step, - adv_step: *std.Build.Step, -}; - -pub fn add(b: *std.Build, ctx: Context) Artifacts { - const target = ctx.target; - const optimize = ctx.optimize; - std.debug.assert(target.result.os.tag == .linux); // the root only enables 9player on Linux - - const player_mod = b.createModule(.{ - .root_source_file = b.path("9player/src/main.zig"), - .target = target, - .optimize = optimize, - .imports = &.{.{ .name = "cloud9", .module = ctx.cloud9 }}, - }); - const player = b.addExecutable(.{ .name = "9player", .root_module = player_mod }); - const install = b.addInstallArtifact(player, .{}); - b.getInstallStep().dependOn(&install.step); - b.step("9player", "Build and install only the 9player binary").dependOn(&install.step); - - // Unit tests: protocol structs, session, bridge, namespace helpers (main.zig - // reaches every module). - const test_step = b.step("9player-test", "Run 9player's unit tests"); - test_step.dependOn(&b.addRunArtifact(b.addTest(.{ .root_module = player_mod })).step); - - // End-to-end suites: real namespaces, real FUSE, a real 9P server. - const itest = b.step("9player-itest", "Run 9player/test/integration.sh (needs unprivileged user namespaces and /dev/fuse)"); - const adv = b.step("9player-adv", "Run 9player/test/adversarial.sh (hostile servers, namespaces, stress; several minutes)"); - const demo = ctx.introspect_demo orelse { - const fail = b.addFail("9player-itest and 9player-adv need the introspect demo server (build with -Dintrospect=true)"); - itest.dependOn(&fail.step); - adv.dependOn(&fail.step); - return .{ .exe = player, .test_step = test_step, .itest_step = itest, .adv_step = adv }; - }; - inline for (.{ .{ itest, "integration" }, .{ adv, "adversarial" } }) |pair| { - const run = b.addSystemCommand(&.{"bash"}); - run.addFileArg(b.path("9player/test/" ++ pair[1] ++ ".sh")); - run.addArtifactArg(player); - run.addArtifactArg(demo); - pair[0].dependOn(&run.step); - } - return .{ .exe = player, .test_step = test_step, .itest_step = itest, .adv_step = adv }; -} diff --git a/9player/docs/DESIGN.md b/9player/docs/DESIGN.md deleted file mode 100644 index b9b2fbe..0000000 --- a/9player/docs/DESIGN.md +++ /dev/null @@ -1,405 +0,0 @@ -# 9player design - -`9player` mounts a 9P2000 file tree served over a Unix or TCP stream socket -into a **fresh mount namespace** and runs a program inside it. The program -(fish, bash, `claude`, anything) sees the 9P tree as ordinary files, without -root and without touching the host's mount table. - -## Why FUSE - -The kernel's own `9p` filesystem is not mountable inside an unprivileged user -namespace (it lacks `FS_USERNS_MOUNT`) and loading it needs root. FUSE has been -user-namespace mountable since Linux 4.18, and `/dev/fuse` is world read/write. -So 9player is a tiny FUSE server that speaks 9P2000 to the real server: - -``` - program (fish/bash/claude) 9player (parent) 9P server - in new user+mount namespace │ (introspect, - /mnt/9p ─── FUSE ───▶ kernel ──▶│ fuse.zig ──▶ bridge.zig ──▶ nine.zig ──▶ ramfs, ...) - │ (framing) (translation) (cloud9 Client) -``` - -No libfuse: `src/fuse.zig` implements the small subset of the kernel FUSE -protocol we need directly against `/usr/include/linux/fuse.h`. - -## Toolchain facts (Zig 0.16) - -* Zig 0.16.0 at `/usr/bin/zig`, std at `/usr/lib/zig/std`. **Grep the std - tree before assuming an API exists**; 0.16 moved a lot of process/fs code - behind `std.Io`. Raw Linux syscalls in `std.os.linux` (`fork`, `execve`, - `mount`, `unshare`, `waitpid`, `pipe2`, `socketpair`, `poll`, `read`, `write`, - `open`, `openat`, `getdents64`, `sigaction`, `kill`, `readlinkat`, `mkdirat`, - `symlinkat`) are the intended low-level path. They return `usize`; decode - with `std.os.linux.errno(rc)` (an `E` enum, `.SUCCESS` when ok). -* `std.posix.poll`, `std.posix.sigaction`, `std.posix.read`, `std.posix.kill` - exist. `std.posix.fork/execve/waitpid/pipe2/socketpair` do **not**. -* `pub fn main() !void` and `pub fn main(init: std.process.Init) !void` are - both supported. Prefer `main(init: std.process.Init)`; `init.gpa` is a - general purpose allocator, `init.arena` an arena, `init.minimal.args` the - argv (`toSlice(allocator)`), `init.minimal.environ.block` the envp block. -* No libc is linked. Do not use `std.c.*`. Hostname lookups are therefore out - of scope: `--tcp` takes IP literals only. -* 9player lives in the cloud9 repository as `cloud9/9player/` and is built by - the root `build.zig` through the fragment `9player/build.zig` (steps - `9player`, `9player-test`, `9player-itest`, `9player-adv`; toggle - `-D9player`). cloud9 itself is imported as module `cloud9` - (`@import("cloud9")`). Read `../src/client.zig`, `Server.zig`, `wire.zig` - and `../docs/design.md`. Its core is allocation-free and caller-driven: you - push bytes in, take results out. The demo 9P server the tests mount is the - sibling program `../introspect` (`zig build introspect`). -* Standalone module tests while other files are missing (from the cloud9 - root): - `zig test --dep cloud9 -Mroot=9player/src/.zig -Mcloud9=src/root.zig`. -* Format everything with `zig fmt`. - -## Process model - -``` -9player [options] -- PROGRAM [ARGS...] -``` - -1. Parent parses args, probes that `/dev/fuse` exists, connects to the 9P - server, negotiates `version` and `attach`es (fid 0 = root). Connection - failures are reported before anything is forked. -2. Parent forks with a `socketpair` status channel. **Child**: - 1. `unshare(CLONE_NEWUSER | CLONE_NEWNS)`. - 2. Writes `/proc/self/setgroups` = `deny`, `/proc/self/uid_map` = - `" 1"`, `/proc/self/gid_map` = `" 1"` (same ids - inside as outside; the child creating the namespace holds full - capabilities in it until exec). - 3. `mount(NULL, "/", NULL, MS_REC|MS_PRIVATE, NULL)` so nothing propagates. - 4. Ensures the mountpoint exists (see below). - 5. Opens `/dev/fuse` (`O_RDWR|O_CLOEXEC`). The kernel refuses to mount a - fuse descriptor opened from a different user namespace than the mount - ("wrong user namespace for fuse device"), so this must happen here, not - in the parent. - 6. `mount("9player", mountpoint, "fuse", MS_NOSUID|MS_NODEV, - "fd=,rootmode=40000,user_id=,group_id=,max_read=")`. - 7. Sends the fuse fd to the parent over the status socket (`SCM_RIGHTS`). - 8. `statx` of the mountpoint: this forces one GETATTR, which the parent - serves. Without it the kernel keeps the root inode's initial uid 0 - (unmapped in the namespace) and every create in the root gets `EACCES`. - 9. Sets `NINEPLAYER_MOUNT=` in the environment. - 10. `execve` of PROGRAM with PATH search (implemented by hand; no libc). - Exec failures are reported through the `CLOEXEC` status socket - (errno + message); the parent prints them after the serve loop ends. -3. **Parent** receives the fuse fd, then runs the FUSE loop (`bridge.serve`) - until either the child exits (SIGCHLD via self-pipe) or the FUSE fd reports - `ENODEV` (last process in the namespace gone, mount destroyed). It then - closes the fuse fd and exits with the child's status (`128+sig` if - signalled). The self-pipe is also watched by the 9P session while a reply - is outstanding (`Session.stop_fd` → `error.Stopped`), so a server that - never answers cannot keep 9player alive after the child is gone; a 3 s - watchdog armed from the SIGCHLD handler is the last resort. -4. Signals in the parent: `SIGINT`/`SIGQUIT` ignored (the child owns the tty - and gets them itself); `SIGTERM`/`SIGHUP` forwarded to the child; - `SIGPIPE` ignored; `SIGCHLD` → self-pipe. - -The FUSE fd is shared with the child only until exec (CLOEXEC); the parent's -copy keeps the connection alive. - -### Mountpoint policy - -Default mountpoint: `/mnt/9p`. A relative `--mount` is resolved against cwd. - -* If the path is a directory: use it. -* Else try `mkdir`. If that fails with `EACCES`/`EPERM`/`EROFS` (the normal - case for `/mnt/9p` as a plain user), **shadow the parent directory**: - open an fd to the parent, mount a `tmpfs` over it, then recreate every - existing entry inside the tmpfs: directories → `mkdir` + bind mount from - `/proc/self/fd//`; symlinks → `readlinkat` + `symlink`; anything - else → empty regular file + bind mount. Then `mkdir` the target inside. - Refuse (with a clear message) if the parent has more than 4096 entries or - is `/`. This only affects the new namespace. -* Else fail with the errno and a hint to pass `--mount` an existing dir. - -## Module contracts - -### `src/fuse.zig` — kernel FUSE protocol (no policy) - -Extern structs mirroring `linux/fuse.h`, with `comptime` size asserts: -`InHeader` (40), `OutHeader` (16), `Attr` (88), `EntryOut` (128), -`AttrOut` (104), `GetattrIn` (16), `SetattrIn` (88), `OpenIn` (8), -`OpenOut` (16), `ReleaseIn` (24), `FlushIn` (24), `ReadIn` (40), -`WriteIn` (40), `WriteOut` (8), `CreateIn` (16), `MkdirIn` (8), -`RenameIn` (8), `Rename2In` (16), `ForgetIn` (8), `BatchForgetIn` (8), -`ForgetOne` (16), `FsyncIn` (16), `AccessIn` (8), `InterruptIn` (8), -`Kstatfs` (80), `StatfsOut` (80), `InitIn` (64), `InitOut` (64), -`Dirent` (24 header, name padded to 8), `LseekIn` (24). - -`pub const Opcode = enum(u32) { lookup = 1, forget = 2, getattr = 3, setattr = 4, -readlink = 5, symlink = 6, mknod = 8, mkdir = 9, unlink = 10, rmdir = 11, -rename = 12, link = 13, open = 14, read = 15, write = 16, statfs = 17, -release = 18, fsync = 20, setxattr = 21, getxattr = 22, listxattr = 23, -removexattr = 24, flush = 25, init = 26, opendir = 27, readdir = 28, -releasedir = 29, fsyncdir = 30, getlk = 31, setlk = 32, setlkw = 33, -access = 34, create = 35, interrupt = 36, bmap = 37, destroy = 38, -ioctl = 39, poll = 40, notify_reply = 41, batch_forget = 42, fallocate = 43, -readdirplus = 44, rename2 = 45, lseek = 46, copy_file_range = 47, -setupmapping = 48, removemapping = 49, syncfs = 50, tmpfile = 51, statx = 52, _ }` - -Constants: `kernel_version = 7`, `kernel_minor = 31` (what we answer; the -kernel adapts to the lower minor), `FOPEN_DIRECT_IO = 1`, `FOPEN_KEEP_CACHE = 2`, -`FOPEN_NONSEEKABLE = 4`, `FUSE_ASYNC_READ = 1`, `FUSE_MAX_PAGES = 1<<22`, -`FATTR_MODE=1, FATTR_UID=2, FATTR_GID=4, FATTR_SIZE=8, FATTR_ATIME=16, -FATTR_MTIME=32, FATTR_FH=64, FATTR_ATIME_NOW=128, FATTR_MTIME_NOW=256, -FATTR_LOCKOWNER=512, FATTR_CTIME=1024`. `root_id = 1`. - -I/O helpers (blocking fd, no allocation beyond the caller's buffer): - -```zig -pub const Request = struct { header: InHeader, body: []const u8 }; -/// One kernel request. Returns null on ENODEV (unmounted). Retries EINTR/EAGAIN/ENOENT. -pub fn readRequest(fd: i32, buf: []u8) !?Request; -/// Success reply: header + concatenated payload slices, single writev. -pub fn reply(fd: i32, unique: u64, payloads: []const []const u8) !void; -/// Error reply: negative errno. -pub fn replyError(fd: i32, unique: u64, err: std.os.linux.E) !void; -/// Append a fuse_dirent (8-byte padded) to `buf`; returns false if it doesn't fit. -pub fn addDirent(buf: []u8, used: *usize, ino: u64, off: u64, dtype: u32, name: []const u8) bool; -pub fn body(comptime T: type, req: Request) !*const T; // aligned copy-free view, checks size -pub fn nameAfter(comptime T: type, req: Request) ![]const u8; // NUL-terminated name after a struct -``` - -The request buffer must be ≥ `max_write + 4096`; 9player uses 1 MiB + 4 KiB. -Requests with an unknown/unsupported opcode get `ENOSYS`. - -### `src/nine.zig` — synchronous 9P2000 session on a blocking fd - -Thin, synchronous RPC layer over `cloud9.Client` (which is push/take, -non-blocking-agnostic). One outstanding request at a time (the FUSE loop is -single-threaded). Fids are allocated from a free list. - -```zig -pub const Address = union(enum) { unix: []const u8, tcp: struct { host: []const u8, port: u16 }, fd: i32 }; -pub const Session = struct { - pub const Error = error{ Nine, Protocol, Io, Closed, TooLarge, OutOfMemory }; - /// After error.Nine, `ename` holds the server's Rerror text (copied, bounded). - ename: [256]u8, ename_len: usize, - msize: u32, - - pub fn connect(gpa: std.mem.Allocator, address: Address, msize: u32) !Session; // socket+connect, version - pub fn deinit(s: *Session) void; - pub fn attach(s: *Session, fid: u32, uname: []const u8, aname: []const u8) Error!cloud9.Qid; - pub fn allocFid(s: *Session) u32; - pub fn freeFid(s: *Session, fid: u32) void; - /// Generic RPC. Result slices borrow the input buffer until the next call. - pub fn rpc(s: *Session, req: cloud9.Client.Request) Error!cloud9.Client.Result; - // Conveniences (all built on rpc): - pub fn walk(s, fid: u32, newfid: u32, names: []const []const u8) Error!Walk; // Walk = { nwqid, wqid[16] }; partial walk → error.Nine with ename "not found"-ish - pub fn clone(s, fid: u32) Error!u32; // allocFid + walk with no names - pub fn open(s, fid: u32, mode: u8) Error!Open; // Open = { qid, iounit } - pub fn create(s, fid: u32, name: []const u8, perm: u32, mode: u8) Error!Open; - pub fn read(s, fid: u32, offset: u64, buf: []u8) Error!usize; // chunks by maxRead/iounit; stops at short read - pub fn write(s, fid: u32, offset: u64, data: []const u8) Error!usize; // chunks; stops at short write - pub fn stat(s, fid: u32) Error!cloud9.Stat; // strings borrow the input buffer - pub fn wstat(s, fid: u32, st: cloud9.Stat) Error!void; - pub fn clunk(s, fid: u32) Error!void; // frees the fid even on error - pub fn remove(s, fid: u32) Error!void; // frees the fid even on error - pub fn errno(s: *const Session) std.os.linux.E; // map ename → errno (see below) -}; -pub const dontcare = cloud9.Stat{ .type = 0xFFFF, .dev = 0xFFFF_FFFF, .qid = .{ .type = 0xFF, .version = 0xFFFF_FFFF, .path = 0xFFFF_FFFF_FFFF_FFFF }, .mode = 0xFFFF_FFFF, .atime = 0xFFFF_FFFF, .mtime = 0xFFFF_FFFF, .length = 0xFFFF_FFFF_FFFF_FFFF, .name = "", .uid = "", .gid = "", .muid = "" }; -``` - -`connect`: for `.unix` and `.tcp` create a blocking `SOCK_STREAM|SOCK_CLOEXEC` -socket and connect (`TCP_NODELAY` on TCP); for `.fd` adopt it. Buffers of -`msize` bytes for in/out are heap allocated. Then submit `.version`, drain -output to the socket, read until `take()` yields the version result. The -negotiated msize is `result.version.msize`; if the server answered -`"unknown"`, fail with `error.Protocol`. - -`rpc`: submit, write all of `client.output()` (calling `wrote`), then loop: -`take()`; if null, `read` from the fd into a temp buffer and `push` (push -returns how much fit; the frame is at most msize so it always fits after a -`take`). If the fd returns 0 → `error.Closed`. If the client dies → -`error.Protocol`. A `.fail` result copies the ename and returns `error.Nine`. - -Rerror text → errno mapping (case-insensitive substring, in this order): -`"not exist"`, `"not found"`, `"no such"` → `ENOENT`; `"exists"` → `EEXIST`; -`"not empty"` → `ENOTEMPTY`; `"not a dir"` → `ENOTDIR`; -`"is a dir"` → `EISDIR`; `"permission"`, `"denied"` → `EACCES`; -`"read-only"`, `"read only"`, `"readonly"` → `EROFS`; `"no space"` → `ENOSPC`; -`"not allowed"`, `"not permitted"`, `"cannot"` → `EPERM`; -`"fid"` → `EBADF`; `"bad offset"`, `"invalid"`, `"bad "` → `EINVAL`; -`"busy"`, `"in use"` → `EBUSY`; `"too long"` → `ENAMETOOLONG`; -`"not supported"`, `"unsupported"` → `ENOTSUP`; otherwise `EIO`. - -### `src/bridge.zig` — FUSE ↔ 9P translation - -```zig -pub const Options = struct { - uid: u32, gid: u32, // reported owner of every file - attr_timeout_ns: u64 = 1e9, // attr/entry cache validity (0 = none) - direct_io: bool = true, // FOPEN_DIRECT_IO on every regular file - debug: bool = false, // trace to stderr -}; -/// Runs until the FUSE fd reports ENODEV or `stop_fd` becomes readable. -pub fn serve(gpa: std.mem.Allocator, fuse_fd: i32, nine: *nine.Session, root_fid: u32, stop_fd: i32, opts: Options) !void; -``` - -State: - -* `inodes: AutoHashMap(u64 /*nodeid*/, Inode{ fid: u32, qid: Qid, nlookup: u64 })`. - Node 1 is the root (`root_fid`, never forgotten). -* `by_qid: AutoHashMap(u64 /*qid.path*/, u64 /*nodeid*/)` so that repeated - lookups of the same file map to the same inode (the old fid is clunked and - the fresh one kept). Dedupe only merges when the qid type (dir bit) also - matches, so a server reusing a path across a file and a directory cannot - poison an inode. `ino` in attrs is `qid.path` (root, or anything carrying - the root's path: 1). -* `handles: AutoHashMap(u64 /*fh*/, Handle{ fid: u32, dir: ?DirList })`. - `DirList` is the entire directory read at first `READDIR` offset 0: - `[]Entry{ name: []u8, ino: u64, dtype: u32 }` with synthetic `.` and `..` - first. `READDIR` offsets are indices into that list; a `READDIR` at offset 0 - re-reads the directory (rewinddir). - -Op mapping (9P2000 has no symlinks, links, xattrs, locks, mknod): - -| FUSE | 9P | -|---|---| -| INIT | reply `InitOut{ major=7, minor=31, max_readahead=in.max_readahead, flags = FUSE_ASYNC_READ \| FUSE_ATOMIC_O_TRUNC \| FUSE_AUTO_INVAL_DATA \| FUSE_BIG_WRITES (plus FUSE_MAX_PAGES with max_pages=256 if offered), max_background=16, congestion_threshold=12, max_write=1 MiB, time_gran=1 }`. Atomic O_TRUNC matters: without it the kernel truncates via a separate SETATTR(size=0) that synthetic control files reject; with it `O_TRUNC` becomes 9P `OTRUNC` inside the open | -| LOOKUP(parent,name) | `walk(parent.fid → newfid, [name])`; `stat(newfid)`; dedupe by qid; `EntryOut` | -| FORGET / BATCH_FORGET | `nlookup -= n`; at 0 `clunk` and drop (no reply) | -| GETATTR | `stat(inode.fid)` → `AttrOut` | -| SETATTR | `stat` then `wstat` with a *dontcare* Stat: `FATTR_SIZE`→length; `FATTR_MODE`→`(old.mode & ~0o777) \| (mode & 0o777)`; `FATTR_MTIME`→mtime (`FATTR_MTIME_NOW` → now); `FATTR_ATIME` ignored; `FATTR_UID/GID` → `EPERM` unless unchanged; then `stat` again for the reply | -| OPEN | `clone(inode.fid)` then `open(newfid, mode)`; mode from `O_ACCMODE` (`oread/owrite/ordwr`), `O_TRUNC` → `otrunc`; reply `OpenOut{ fh, open_flags = FOPEN_DIRECT_IO }`; on failure clunk | -| OPENDIR | same with `oread`; `fh` with `dir = null` | -| READ | `read(fh.fid, offset, buf[0..min(size, 1 MiB)])`; reply data | -| WRITE | `write(fh.fid, offset, data)`; `WriteOut{ size = n }` | -| READDIR | fill `Dirent`s from the `DirList` starting at `offset`, up to `size` bytes | -| RELEASE / RELEASEDIR | `clunk(fh.fid)`; free DirList | -| FLUSH / FSYNC / FSYNCDIR | ok (no-op) | -| CREATE(parent,name,flags,mode) | `clone(parent)`; `create(fid, name, mode & 0o777, openmode)` → this fid is the **open** file; then `walk(parent → fid2, [name])` + `stat(fid2)` for the inode; reply `EntryOut ++ OpenOut` | -| MKDIR | `clone(parent)`; `create(fid, name, DMDIR \| (mode & 0o777), oread)`; `clunk`; then lookup as above | -| UNLINK / RMDIR | `walk(parent → tmp, [name])`; `remove(tmp)` | -| RENAME / RENAME2 | if `newdir != parent` → `EXDEV`; else `walk(parent → tmp, [oldname])`, `wstat(tmp, dontcare with .name = newname)`, `clunk`. 9P rename never replaces, POSIX does: when the target exists (and `RENAME_NOREPLACE` is not set) a directory target is removed first; a file target is parked under a temporary name, the rename retried, and the parked file removed only after success (restored on failure) | -| STATFS | constant `Kstatfs{ bsize = 4096, namelen = 255, frsize = 4096 }` | -| ACCESS | `ENOSYS` (kernel stops asking; the server enforces permissions on open) | -| READLINK, SYMLINK, LINK, MKNOD, *XATTR, *LK, IOCTL, POLL, BMAP, FALLOCATE, LSEEK, COPY_FILE_RANGE, TMPFILE, STATX | `ENOSYS` | -| INTERRUPT | ignored (reply nothing) | -| DESTROY | return from `serve` | - -Attr mapping from `cloud9.Stat`: `mode = (S_IFDIR if DMDIR else S_IFREG) | -(st.mode & 0o777)`; `nlink = 1`; `size = length`; `blocks = (length+511)/512`; -`blksize = 4096`; `atime/mtime/ctime = st.atime/st.mtime/st.mtime`; -`uid/gid = opts.uid/gid`. `Dirent.type` = `DT_DIR` (4) / `DT_REG` (8). - -Errors: `nine.Session.Error.Nine` → `nine.errno()`; `Closed`/`Protocol`/`Io` -→ `EIO` and, since the session is dead, `serve` returns `error.Closed` after -replying so 9player can report "9P server went away". - -With `direct_io` the kernel never trusts `length` for reads: synthetic files -that report length 0 (very common in 9P) still `cat` correctly, and reads run -until the server returns a short read. With `--no-direct-io` the bridge forces -`attr_timeout_ns = 0`, because a cached stale size truncates reads (observed -data loss on a 4 MiB copy otherwise). - -Hostile-server rules: directory listings are capped at 64 MiB (a server that -ignores read offsets otherwise loops forever); directory records with names -containing `/`, NUL, empty, `.`/`..` or longer than `FUSE_NAME_MAX` are dropped -rather than poisoning the whole READDIR reply; `length` near 2^64 is clamped -to `i64` max; the errno of a failing 9P call is latched before any cleanup -clunk overwrites the session's ename. - -### `src/ns.zig` — namespace and process plumbing - -```zig -pub const Spawn = struct { - argv: []const []const u8, // argv[0] is PATH-searched unless it contains '/' - envp: [*:null]const ?[*:0]const u8, // inherited environment - mountpoint: []const u8, // absolute - fuse_fd: i32, - uid: u32, gid: u32, - max_read: u32, -}; -pub const Child = struct { pid: i32 }; -/// fork; the child sets up the namespace, mounts, and execs. Returns once exec succeeded -/// (status pipe closed) or fails with the child's error (message on stderr). -pub fn spawn(gpa: std.mem.Allocator, s: Spawn) !Child; -pub fn ensureMountpoint(path: [:0]const u8) !void; // the shadowing logic, testable alone -pub fn resolveMountpoint(gpa, path: []const u8) ![:0]u8; // absolute, no trailing slash -pub fn findInPath(gpa, envp, name) ![:0]u8; -``` - -Also exports the signal plumbing used by `main.zig`: -`installSignals(child_pid_ptr: *i32) !i32` returning the SIGCHLD self-pipe -read end (used as `stop_fd` for `bridge.serve`), and -`waitChild(pid) !u8` → exit status (`128+sig` on signal death). - -### `src/main.zig` — CLI - -``` -Usage: 9player [options] -- PROGRAM [ARGS...] -Transport (exactly one): - --unix PATH Unix stream socket - --tcp IP:PORT TCP (IPv4/IPv6 literal) - --fd N already-connected inherited descriptor - --spawn CMD run CMD (via /bin/sh -c) with a socketpair on its stdin/stdout -Options: - --mount PATH mountpoint inside the new namespace (default /mnt/9p) - --uname NAME 9P user name (default $USER, else "none") - --aname NAME 9P tree to attach (default "") - --msize BYTES maximum 9P message size to request (default 131072, max 16 MiB) - --cache SECONDS attr/entry cache validity, may be fractional (default 1) - --no-direct-io let the kernel cache file pages (trusts stat length) - --debug trace FUSE and 9P operations on stderr - --help, --version -PROGRAM defaults to $SHELL (else /bin/sh). The mountpoint is exported as $NINEPLAYER_MOUNT. -``` - -Exit codes: child's status; 125 for 9player's own failures (usage, connect, -mount); 126/127 as usual for exec failures. - -### `../introspect/demo/main.zig` — demo 9P2000 server (binary `introspect`) - -The demo server is a separate program in this repository, built on the -introspect library; see `../introspect/docs/LIBRARY.md` for the library -contract (freestanding core, value renderers, Linux debug probe). The tree it serves keeps the paths the integration tests read -(`/build/*`, `/comptime/types//*`, `/comptime/decls`, `/runtime/fn/*`, -`/runtime/ctl`, `/runtime/{pid,ppid,uptime,argv,cwd,env,clients}`, -`/scratch/`) and adds `/vars`, `/threads`, `/addr`, `/mem`, `/hex`, -`/breakpoints` and `/panic`. - -## Integration test plan (`test/integration.sh`) - -Run by `zig build 9player-itest`; args: path to `9player`, path to `introspect`. -Everything under a temp dir. Skips (exit 0 with a notice) when -`unshare -Urm true` fails or `/dev/fuse` is missing. - -1. introspect on a Unix socket; `9player --unix … -- sh -c` scripts: - `cat /mnt/9p/build/zig_version` == `zig version`; `ls` listings; `stat` - sizes; `/runtime/fn/now` is numeric; `ctl` round trip; `/scratch`: create, - append (`>>`), overwrite, truncate, `mkdir -p a/b/c`, rename within dir, - `mv` across dirs fails with `EXDEV`-ish message, `rm`, `rmdir`, 1 MiB - random file round trip compared with `sha256sum`, `dd` with odd block - sizes, many small files, `find`, exit-status propagation (`exit 7` → 7), - `$NINEPLAYER_MOUNT` set, nested `9player` inside `9player`. -2. `--spawn " --stdio"` variant. -3. `--tcp 127.0.0.1:` variant. -4. If `/usr/lib/plan9/bin/ramfs` exists: `NAMESPACE=$tmp ramfs -s ramfs` - creates `$tmp/ramfs`; run the scratch battery against it. -5. `--mount` with an existing dir, with a relative path, and the default - `/mnt/9p` (exercises parent shadowing; verify `/mnt`'s other entries are - still visible inside). -6. Kill tests: 9player exits when the child exits; server death during use - yields `EIO`, not a hang. - -## Verification - -`zig build 9player-test` (unit), `zig build 9player-itest` (74 end-to-end -checks against introspect over unix/tcp/socketpair and against plan9port's -`ramfs`) and `zig build 9player-adv` (adversarial suites: a scriptable -hostile 9P server with ~30 misbehaviour modes, FUSE semantics through the -bridge, process/namespace/signal edge cases with 51 checks, and stress). The -suites that attack the introspect server itself (a hostile raw-9P client with -181 checks, the core, the Linux layer) moved with it to -`../introspect/test` (`zig build introspect-adv`). All pass in Debug and -ReleaseSafe. - -## Out of scope for v1 (documented, not hidden) - -* One 9P request in flight at a time: a 9P read that blocks (event files) - stalls the whole mount until it returns (but not past the child's exit). -* No 9P2000.u/.L: no symlinks, ownership, or extended attributes. -* No PID namespace, no `/proc` remount. `--tcp` needs an IP literal. -* Cross-directory rename returns `EXDEV` (9P2000 cannot move files). diff --git a/9player/src/bridge.zig b/9player/src/bridge.zig deleted file mode 100644 index 387e184..0000000 --- a/9player/src/bridge.zig +++ /dev/null @@ -1,971 +0,0 @@ -//! FUSE ↔ 9P2000 translation: the request loop that turns kernel FUSE requests -//! into synchronous 9P calls on a `nine.Session` and sends the replies back. -//! -//! Everything here is single-threaded and one request at a time. State is three -//! tables: inodes (nodeid → fid/qid, deduplicated by qid.path), open handles -//! (fh → fid plus a cached directory listing), and the reverse qid map. -const std = @import("std"); -const cloud9 = @import("cloud9"); -const fuse = @import("fuse.zig"); -const nine = @import("nine.zig"); -const linux = std.os.linux; - -pub const Options = struct { - /// Reported owner of every file. - uid: u32, - gid: u32, - /// attr/entry cache validity (0 = none). - attr_timeout_ns: u64 = 1_000_000_000, - /// FOPEN_DIRECT_IO on every regular file. - direct_io: bool = true, - /// Trace every request, reply and 9P call to stderr. - debug: bool = false, -}; - -/// Largest single READ/WRITE payload we accept from the kernel. -pub const max_write: u32 = 1 << 20; -/// Upper bound on the raw bytes of one directory listing (about a million entries); -/// past it the listing fails with EIO instead of eating memory. -pub const max_dir_bytes: u64 = 64 << 20; -/// The kernel refuses dirents longer than this (FUSE_NAME_MAX) with EIO. -pub const max_name_len: usize = 1024; -/// Request buffer: `max_write` plus room for the header and the largest in-struct. -pub const request_buf_len: usize = max_write + 4096; - -const Inode = struct { - fid: u32, - qid: cloud9.Qid, - nlookup: u64, - /// nodeid of the directory this inode was looked up in (root: itself). Used for "..". - parent: u64, -}; - -pub const Entry = struct { name: []u8, ino: u64, dtype: u32 }; - -pub const DirList = struct { - entries: std.ArrayList(Entry) = .empty, - - pub fn deinit(d: *DirList, gpa: std.mem.Allocator) void { - for (d.entries.items) |e| gpa.free(e.name); - d.entries.deinit(gpa); - } -}; - -const Handle = struct { - fid: u32, - nodeid: u64, - dir: ?DirList, -}; - -/// Errors a request handler may surface. Policy failures are ordinary errors -/// that the dispatcher maps to an errno; `FuseIo` means the kernel side is broken. -const HandlerError = nine.Session.Error || error{ - BadRequest, - NoEntry, - BadHandle, - Exdev, - Perm, - NotSup, - /// A directory listing the server sent could not be parsed (EIO, not fatal). - BadDir, - FuseIo, -}; - -/// Runs until the FUSE fd reports ENODEV, a DESTROY arrives, or `stop_fd` -/// becomes readable (also while a 9P reply is outstanding). Returns -/// `error.Closed` if the 9P server went away. -pub fn serve(gpa: std.mem.Allocator, fuse_fd: i32, session: *nine.Session, root_fid: u32, stop_fd: i32, opts: Options) !void { - var effective = opts; - // With page caching on, a nonzero attr cache lets the kernel trust a stale - // (often zero) size and truncate reads: 9P sizes are authoritative and change - // under us. direct_io ignores the cached size, so the cache is safe only there. - if (!effective.direct_io) effective.attr_timeout_ns = 0; - var b: Bridge = .{ - .gpa = gpa, - .fuse_fd = fuse_fd, - .nine = session, - .opts = effective, - }; - defer b.deinit(); - - b.req_buf = try gpa.alignedAlloc(u8, .@"8", request_buf_len); - b.data_buf = try gpa.alloc(u8, max_write); - - // Abandon any pending 9P reply once the child is gone (stop_fd readable), - // including the initial root stat below: a silent server must not pin us. - session.stop_fd = stop_fd; - defer session.stop_fd = -1; - - // Node 1 is the root; its qid comes from a stat so lookups resolving back to - // it (e.g. via a walk) dedupe onto node 1. - var root_qid: cloud9.Qid = .{ .type = cloud9.qtdir, .version = 0, .path = 0 }; - if (b.stat(root_fid)) |st| { - root_qid = st.qid; - b.root_path = st.qid.path; - try b.by_qid.put(gpa, st.qid.path, fuse.root_id); - } else |e| switch (e) { - error.Nine => {}, - error.Stopped => return, - else => return error.Closed, - } - try b.inodes.put(gpa, fuse.root_id, .{ .fid = root_fid, .qid = root_qid, .nlookup = 1, .parent = fuse.root_id }); - - var pfds = [_]linux.pollfd{ - .{ .fd = fuse_fd, .events = linux.POLL.IN, .revents = 0 }, - .{ .fd = stop_fd, .events = linux.POLL.IN, .revents = 0 }, - }; - while (true) { - pfds[0].revents = 0; - pfds[1].revents = 0; - const rc = linux.poll(&pfds, pfds.len, -1); - switch (linux.errno(rc)) { - .SUCCESS => {}, - .INTR, .AGAIN => continue, - else => return error.Io, - } - if (pfds[1].revents != 0) { - b.trace("stop_fd readable; leaving serve loop", .{}); - return; - } - if (pfds[0].revents == 0) continue; - const req = (fuse.readRequest(fuse_fd, b.req_buf) catch |e| switch (e) { - error.Protocol => return error.FuseProtocol, - else => return error.FuseIo, - }) orelse { - b.trace("fuse fd reports ENODEV; unmounted", .{}); - return; - }; - if (!try b.dispatch(req)) return; - } -} - -const Bridge = struct { - gpa: std.mem.Allocator, - fuse_fd: i32, - nine: *nine.Session, - opts: Options, - req_buf: []align(8) u8 = &.{}, - data_buf: []u8 = &.{}, - inodes: std.AutoHashMapUnmanaged(u64, Inode) = .empty, - by_qid: std.AutoHashMapUnmanaged(u64, u64) = .empty, - handles: std.AutoHashMapUnmanaged(u64, Handle) = .empty, - next_node: u64 = 2, - next_fh: u64 = 1, - /// qid.path of the root, reported as ino 1 wherever it shows up. - root_path: u64 = 0, - /// errno of the most recent Rerror that a handler did not swallow. Kept here - /// because `Session.rpc` clears its ename on every call, and error paths - /// clunk (an rpc) before the dispatcher maps the failure to an errno. - last_err: linux.E = .IO, - - fn deinit(b: *Bridge) void { - var it = b.handles.valueIterator(); - while (it.next()) |h| if (h.dir) |*d| d.deinit(b.gpa); - b.handles.deinit(b.gpa); - b.inodes.deinit(b.gpa); - b.by_qid.deinit(b.gpa); - if (b.req_buf.len != 0) b.gpa.free(b.req_buf); - if (b.data_buf.len != 0) b.gpa.free(b.data_buf); - } - - fn trace(b: *const Bridge, comptime fmt: []const u8, args: anytype) void { - if (b.opts.debug) std.debug.print("9player: " ++ fmt ++ "\n", args); - } - - // -- dispatch -------------------------------------------------------------------- - - /// Handles one request. Returns false when the loop should stop (DESTROY). - /// Fatal errors (dead 9P session, broken FUSE fd) propagate. - fn dispatch(b: *Bridge, req: fuse.Request) !bool { - const h = req.header; - const op = h.op(); - b.trace("<- {s} unique={d} nodeid={d} len={d} (fids={d} inodes={d} handles={d})", .{ opName(op), h.unique, h.nodeid, h.len, b.nine.fidsInUse(), b.inodes.count(), b.handles.count() }); - const wants_reply = switch (op) { - .forget, .batch_forget, .interrupt => false, - else => true, - }; - if (op == .destroy) { - b.reply(h.unique, &.{}) catch {}; - return false; - } - b.handle(req) catch |e| { - const code: linux.E = switch (e) { - error.Nine => b.last_err, - error.BadRequest => .INVAL, - error.NoEntry => .NOENT, - error.BadHandle => .BADF, - error.Exdev => .XDEV, - error.Perm => .PERM, - error.NotSup => .NOSYS, - error.OutOfMemory => .NOMEM, - error.TooLarge => .NAMETOOLONG, - error.BadDir => .IO, - error.Closed, error.Protocol, error.Io, error.Stopped => .IO, - error.FuseIo => return error.FuseIo, - }; - if (wants_reply) try b.replyError(h.unique, code); - switch (e) { - error.Closed, error.Protocol, error.Io => return error.Closed, - error.Stopped => return false, // the child is gone; the mount is being torn down - else => {}, - } - }; - return true; - } - - fn handle(b: *Bridge, req: fuse.Request) HandlerError!void { - const u = req.header.unique; - switch (req.header.op()) { - .init => { - const in = try body(fuse.InitIn, req); - const out = fuse.initReply(in, max_write); - try b.reply(u, &.{std.mem.asBytes(&out)}); - }, - .lookup => { - const name = try nameAfter(void, req); - const entry = try b.lookupEntry(req.header.nodeid, name); - try b.reply(u, &.{std.mem.asBytes(&entry)}); - }, - .forget => { - const in = try body(fuse.ForgetIn, req); - try b.forget(req.header.nodeid, in.nlookup); - }, - .batch_forget => { - const in = try body(fuse.BatchForgetIn, req); - const rest = req.body[@sizeOf(fuse.BatchForgetIn)..]; - const count: usize = in.count; - if (rest.len < count * @sizeOf(fuse.ForgetOne)) return error.BadRequest; - for (0..count) |i| { - const one = std.mem.bytesToValue(fuse.ForgetOne, rest[i * @sizeOf(fuse.ForgetOne) ..][0..@sizeOf(fuse.ForgetOne)]); - try b.forget(one.nodeid, one.nlookup); - } - }, - .getattr => { - const ino = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; - const st = try b.stat(ino.fid); - const out = b.attrOut(st, b.inoOf(req.header.nodeid, ino.qid)); - try b.reply(u, &.{std.mem.asBytes(&out)}); - }, - .setattr => try b.setattr(req), - .open => try b.openFile(req, false), - .opendir => try b.openFile(req, true), - .read => { - const in = try body(fuse.ReadIn, req); - const h = b.handles.get(in.fh) orelse return error.BadHandle; - const want: usize = @min(in.size, max_write); - const n = try b.read(h.fid, in.offset, b.data_buf[0..want]); - try b.reply(u, &.{b.data_buf[0..n]}); - }, - .write => { - const in = try body(fuse.WriteIn, req); - const h = b.handles.get(in.fh) orelse return error.BadHandle; - const rest = req.body[@sizeOf(fuse.WriteIn)..]; - if (rest.len < in.size) return error.BadRequest; - const n = try b.write(h.fid, in.offset, rest[0..in.size]); - const out = fuse.WriteOut{ .size = @intCast(n) }; - try b.reply(u, &.{std.mem.asBytes(&out)}); - }, - .readdir => try b.readdir(req), - .release, .releasedir => { - const in = try body(fuse.ReleaseIn, req); - const kv = b.handles.fetchRemove(in.fh) orelse return error.BadHandle; - var h = kv.value; - if (h.dir) |*d| d.deinit(b.gpa); - try b.clunk(h.fid); - try b.reply(u, &.{}); - }, - .flush, .fsync, .fsyncdir => try b.reply(u, &.{}), - .create => try b.create(req), - .mkdir => { - const in = try body(fuse.MkdirIn, req); - const name = try nameAfter(fuse.MkdirIn, req); - const parent = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; - const fid = try b.clone(parent.fid); - _ = b.create9(fid, name, cloud9.dmdir | (in.mode & 0o777), cloud9.oread) catch |e| { - b.clunkQuiet(fid); - return e; - }; - try b.clunk(fid); - const entry = try b.lookupEntry(req.header.nodeid, name); - try b.reply(u, &.{std.mem.asBytes(&entry)}); - }, - .unlink, .rmdir => { - const name = try nameAfter(void, req); - const parent = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; - const tmp = try b.walkName(parent.fid, name); - try b.remove(tmp); - try b.reply(u, &.{}); - }, - .rename => { - const in = try body(fuse.RenameIn, req); - const old = try nameAfter(fuse.RenameIn, req); - const new = try secondName(req, old, @sizeOf(fuse.RenameIn)); - try b.rename(req.header.nodeid, in.newdir, old, new, 0); - try b.reply(u, &.{}); - }, - .rename2 => { - const in = try body(fuse.Rename2In, req); - const old = try nameAfter(fuse.Rename2In, req); - const new = try secondName(req, old, @sizeOf(fuse.Rename2In)); - try b.rename(req.header.nodeid, in.newdir, old, new, in.flags); - try b.reply(u, &.{}); - }, - .statfs => { - const out = fuse.StatfsOut{ .st = .{ .bsize = 4096, .namelen = 255, .frsize = 4096 } }; - try b.reply(u, &.{std.mem.asBytes(&out)}); - }, - .interrupt => {}, - .destroy => unreachable, // handled in dispatch - .access => return error.NotSup, - else => return error.NotSup, - } - } - - // -- handlers ---------------------------------------------------------------------- - - /// walk(parent → new fid, [name]) + stat, deduplicated by qid.path. Bumps nlookup. - fn lookupEntry(b: *Bridge, parent_id: u64, name: []const u8) HandlerError!fuse.EntryOut { - const parent = b.inodes.get(parent_id) orelse return error.NoEntry; - const newfid = try b.walkName(parent.fid, name); - const st = b.stat(newfid) catch |e| { - b.clunkQuiet(newfid); - return e; - }; - const qid = st.qid; - var nodeid: u64 = undefined; - if (b.by_qid.get(qid.path)) |existing| { - // A directory and a file sharing a qid.path (a server bug) must not - // share a node: the kernel would mark the inode bad, and for the - // root that is fatal for the whole mount. - const merge = if (b.inodes.getPtr(existing)) |ino| (ino.qid.type & cloud9.qtdir) == (qid.type & cloud9.qtdir) else false; - if (merge) { - const ino = b.inodes.getPtr(existing).?; - ino.nlookup += 1; - ino.qid = qid; - nodeid = existing; - if (existing == fuse.root_id) { - b.clunkQuiet(newfid); - } else { - // Keep the fresh fid (it is bound to the current file at this - // name) and retire the older one. - const stale = ino.fid; - ino.fid = newfid; - b.clunkQuiet(stale); - } - } else { - // Stale reverse entry, or a type clash: bind a fresh node to it. - nodeid = try b.newInode(newfid, qid, parent_id); - } - } else { - nodeid = try b.newInode(newfid, qid, parent_id); - } - var out = fuse.EntryOut{ - .nodeid = nodeid, - .generation = 0, - .attr = b.attrFrom(st, b.inoOf(nodeid, qid)), - }; - out.entry_valid = b.opts.attr_timeout_ns / 1_000_000_000; - out.entry_valid_nsec = @intCast(b.opts.attr_timeout_ns % 1_000_000_000); - out.attr_valid = out.entry_valid; - out.attr_valid_nsec = out.entry_valid_nsec; - return out; - } - - fn newInode(b: *Bridge, fid: u32, qid: cloud9.Qid, parent: u64) HandlerError!u64 { - const nodeid = b.next_node; - b.inodes.put(b.gpa, nodeid, .{ .fid = fid, .qid = qid, .nlookup = 1, .parent = parent }) catch |e| { - b.clunkQuiet(fid); - return e; - }; - b.by_qid.put(b.gpa, qid.path, nodeid) catch |e| { - _ = b.inodes.remove(nodeid); - b.clunkQuiet(fid); - return e; - }; - b.next_node += 1; - return nodeid; - } - - fn forget(b: *Bridge, nodeid: u64, n: u64) HandlerError!void { - if (nodeid == fuse.root_id) return; - const ino = b.inodes.getPtr(nodeid) orelse return; - if (ino.nlookup > n) { - ino.nlookup -= n; - return; - } - const fid = ino.fid; - const path = ino.qid.path; - _ = b.inodes.remove(nodeid); - if (b.by_qid.get(path)) |mapped| { - if (mapped == nodeid) _ = b.by_qid.remove(path); - } - b.clunk(fid) catch |e| switch (e) { - error.Nine => {}, - else => return e, - }; - } - - fn setattr(b: *Bridge, req: fuse.Request) HandlerError!void { - const in = try body(fuse.SetattrIn, req); - const ino = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; - const old = try b.stat(ino.fid); - const old_mode = old.mode; - - var st = nine.dontcare; - var changed = false; - if (in.valid & fuse.FATTR_UID != 0 and in.uid != b.opts.uid) return error.Perm; - if (in.valid & fuse.FATTR_GID != 0 and in.gid != b.opts.gid) return error.Perm; - if (in.valid & fuse.FATTR_SIZE != 0) { - st.length = in.size; - changed = true; - } - if (in.valid & fuse.FATTR_MODE != 0) { - st.mode = (old_mode & ~@as(u32, 0o777)) | (in.mode & 0o777); - changed = true; - } - if (in.valid & fuse.FATTR_MTIME_NOW != 0) { - st.mtime = nowSeconds(); - changed = true; - } else if (in.valid & fuse.FATTR_MTIME != 0) { - st.mtime = @truncate(in.mtime); - changed = true; - } - if (changed) try b.wstat(ino.fid, st); - const fresh = try b.stat(ino.fid); - const out = b.attrOut(fresh, b.inoOf(req.header.nodeid, ino.qid)); - try b.reply(req.header.unique, &.{std.mem.asBytes(&out)}); - } - - fn openFile(b: *Bridge, req: fuse.Request, is_dir: bool) HandlerError!void { - const in = try body(fuse.OpenIn, req); - const ino = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; - const mode: u8 = if (is_dir) cloud9.oread else openMode(in.flags); - const fid = try b.clone(ino.fid); - _ = b.open9(fid, mode) catch |e| { - b.clunkQuiet(fid); - return e; - }; - const fh = try b.newHandle(fid, req.header.nodeid); - const out = fuse.OpenOut{ - .fh = fh, - .open_flags = if (!is_dir and b.opts.direct_io) fuse.FOPEN_DIRECT_IO else 0, - }; - try b.reply(req.header.unique, &.{std.mem.asBytes(&out)}); - } - - fn newHandle(b: *Bridge, fid: u32, nodeid: u64) HandlerError!u64 { - const fh = b.next_fh; - b.handles.put(b.gpa, fh, .{ .fid = fid, .nodeid = nodeid, .dir = null }) catch |e| { - b.clunkQuiet(fid); - return e; - }; - b.next_fh += 1; - return fh; - } - - fn create(b: *Bridge, req: fuse.Request) HandlerError!void { - const in = try body(fuse.CreateIn, req); - const name = try nameAfter(fuse.CreateIn, req); - const parent = b.inodes.get(req.header.nodeid) orelse return error.NoEntry; - // The created fid becomes the open file. - const fid = try b.clone(parent.fid); - _ = b.create9(fid, name, in.mode & 0o777, openMode(in.flags)) catch |e| { - b.clunkQuiet(fid); - return e; - }; - const entry = b.lookupEntry(req.header.nodeid, name) catch |e| { - b.clunkQuiet(fid); - return e; - }; - const fh = try b.newHandle(fid, entry.nodeid); - const oo = fuse.OpenOut{ - .fh = fh, - .open_flags = if (b.opts.direct_io) fuse.FOPEN_DIRECT_IO else 0, - }; - try b.reply(req.header.unique, &.{ std.mem.asBytes(&entry), std.mem.asBytes(&oo) }); - } - - fn rename(b: *Bridge, parent_id: u64, newdir: u64, old: []const u8, new: []const u8, flags: u32) HandlerError!void { - if (newdir != parent_id) return error.Exdev; - const rf: linux.RENAME = @bitCast(flags); - if (rf.EXCHANGE or rf.WHITEOUT) return error.BadRequest; - const parent = b.inodes.get(parent_id) orelse return error.NoEntry; - const tmp = try b.walkName(parent.fid, old); - defer b.clunkQuiet(tmp); - var st = nine.dontcare; - st.name = new; - b.wstat(tmp, st) catch |e| { - // 9P2000 rename never replaces an existing name; POSIX rename does. - if (e != error.Nine or rf.NOREPLACE or b.nine.errno() != .EXIST) return e; - try b.renameOver(parent.fid, tmp, new); - }; - } - - /// Replace `new` with the file behind `src`. An (empty) directory target is - /// removed first: it holds no data and the VFS already ruled out mismatched - /// types. A file target is parked under a temporary name so that a failing - /// second rename can put it back instead of having destroyed it. - fn renameOver(b: *Bridge, parent_fid: u32, src: u32, new: []const u8) HandlerError!void { - const victim = try b.walkName(parent_fid, new); - const vst = b.stat(victim) catch |e| { - b.clunkQuiet(victim); - return e; - }; - var st = nine.dontcare; - st.name = new; - if (vst.mode & cloud9.dmdir != 0) { - b.trace(" rename target is a directory; removing it and retrying", .{}); - try b.remove(victim); - return b.wstat(src, st); - } - var park_buf: [48]u8 = undefined; - const park = std.fmt.bufPrint(&park_buf, ".9player-rename-{x}", .{randomU64()}) catch unreachable; - b.trace(" rename target exists; parking it as {s} and retrying", .{park}); - var pst = nine.dontcare; - pst.name = park; - b.wstat(victim, pst) catch |e| { - b.clunkQuiet(victim); - return e; - }; - b.wstat(src, st) catch |e| { - b.trace(" rename still failed; restoring the target", .{}); - const saved = b.last_err; - b.wstat(victim, st) catch {}; - b.last_err = saved; - b.clunkQuiet(victim); - return e; - }; - b.remove(victim) catch b.trace(" could not remove the parked target {s}", .{park}); - } - - fn readdir(b: *Bridge, req: fuse.Request) HandlerError!void { - const in = try body(fuse.ReadIn, req); - const h = b.handles.getPtr(in.fh) orelse return error.BadHandle; - if (in.offset == 0 or h.dir == null) { - if (h.dir) |*d| d.deinit(b.gpa); - h.dir = null; - h.dir = try b.loadDir(h.fid, h.nodeid); - } - const dir = &h.dir.?; - const size: usize = @min(in.size, max_write); - const used = packDirents(dir.entries.items, in.offset, b.data_buf[0..size]); - try b.reply(req.header.unique, &.{b.data_buf[0..used]}); - } - - /// Reads the whole directory and builds its listing, "." and ".." first. - fn loadDir(b: *Bridge, fid: u32, nodeid: u64) HandlerError!DirList { - var list: DirList = .{}; - errdefer list.deinit(b.gpa); - const self_ino = b.inoOfNode(nodeid); - const parent_ino = if (b.inodes.get(nodeid)) |ino| b.inoOfNode(ino.parent) else self_ino; - try list.entries.append(b.gpa, .{ .name = try b.gpa.dupe(u8, "."), .ino = self_ino, .dtype = fuse.DT_DIR }); - try list.entries.append(b.gpa, .{ .name = try b.gpa.dupe(u8, ".."), .ino = parent_ino, .dtype = fuse.DT_DIR }); - - var offset: u64 = 0; - while (true) { - // A server that ignores the offset would otherwise feed us forever. - if (offset >= max_dir_bytes) return error.BadDir; - const n = try b.read(fid, offset, b.data_buf); - if (n == 0) break; - try parseDirRecords(b.gpa, b.data_buf[0..n], &list); - offset += n; - } - // Entries carrying the root's own qid.path get the root's ino (1), as GETATTR would report it. - for (list.entries.items[2..]) |*e| if (e.ino == b.root_path) { - e.ino = fuse.root_id; - }; - return list; - } - - // -- 9P wrappers (tracing) ----------------------------------------------------------- - - fn stat(b: *Bridge, fid: u32) nine.Session.Error!cloud9.Stat { - const st = b.nine.stat(fid) catch |e| return b.nineErr("stat", fid, e); - b.trace(" 9p stat fid={d} -> name={s} mode={o} len={d} qid={x}", .{ fid, st.name, st.mode, st.length, st.qid.path }); - return st; - } - - fn walkName(b: *Bridge, fid: u32, name: []const u8) nine.Session.Error!u32 { - const newfid = b.nine.allocFid(); - _ = b.nine.walk(fid, newfid, &.{name}) catch |e| { - b.nine.freeFid(newfid); - return b.nineErr("walk", fid, e); - }; - b.trace(" 9p walk fid={d} newfid={d} name={s} -> ok", .{ fid, newfid, name }); - return newfid; - } - - fn clone(b: *Bridge, fid: u32) nine.Session.Error!u32 { - const newfid = b.nine.clone(fid) catch |e| return b.nineErr("clone", fid, e); - b.trace(" 9p walk fid={d} newfid={d} (clone) -> ok", .{ fid, newfid }); - return newfid; - } - - fn open9(b: *Bridge, fid: u32, mode: u8) nine.Session.Error!nine.Session.Open { - const o = b.nine.open(fid, mode) catch |e| return b.nineErr("open", fid, e); - b.trace(" 9p open fid={d} mode={d} -> iounit={d}", .{ fid, mode, o.iounit }); - return o; - } - - fn create9(b: *Bridge, fid: u32, name: []const u8, perm: u32, mode: u8) nine.Session.Error!nine.Session.Open { - const o = b.nine.create(fid, name, perm, mode) catch |e| return b.nineErr("create", fid, e); - b.trace(" 9p create fid={d} name={s} perm={o} mode={d} -> iounit={d}", .{ fid, name, perm, mode, o.iounit }); - return o; - } - - fn read(b: *Bridge, fid: u32, offset: u64, buf: []u8) nine.Session.Error!usize { - const n = b.nine.read(fid, offset, buf) catch |e| return b.nineErr("read", fid, e); - b.trace(" 9p read fid={d} offset={d} count={d} -> {d}", .{ fid, offset, buf.len, n }); - return n; - } - - fn write(b: *Bridge, fid: u32, offset: u64, data: []const u8) nine.Session.Error!usize { - const n = b.nine.write(fid, offset, data) catch |e| return b.nineErr("write", fid, e); - b.trace(" 9p write fid={d} offset={d} count={d} -> {d}", .{ fid, offset, data.len, n }); - return n; - } - - fn wstat(b: *Bridge, fid: u32, st: cloud9.Stat) nine.Session.Error!void { - b.nine.wstat(fid, st) catch |e| return b.nineErr("wstat", fid, e); - b.trace(" 9p wstat fid={d} name={s} mode={x} len={x} mtime={x} -> ok", .{ fid, st.name, st.mode, st.length, st.mtime }); - } - - fn clunk(b: *Bridge, fid: u32) nine.Session.Error!void { - b.nine.clunk(fid) catch |e| return b.nineErr("clunk", fid, e); - b.trace(" 9p clunk fid={d} -> ok", .{fid}); - } - - /// Best-effort clunk during error unwinding; a dead session surfaces on the - /// next call. Does not disturb the errno of the failure being unwound. - fn clunkQuiet(b: *Bridge, fid: u32) void { - const saved = b.last_err; - defer b.last_err = saved; - b.clunk(fid) catch {}; - } - - fn remove(b: *Bridge, fid: u32) nine.Session.Error!void { - b.nine.remove(fid) catch |e| return b.nineErr("remove", fid, e); - b.trace(" 9p remove fid={d} -> ok", .{fid}); - } - - fn nineErr(b: *Bridge, what: []const u8, fid: u32, e: nine.Session.Error) nine.Session.Error { - if (e == error.Nine) { - b.last_err = b.nine.errno(); - b.trace(" 9p {s} fid={d} -> Rerror \"{s}\" ({s})", .{ what, fid, b.nine.ename[0..b.nine.ename_len], @tagName(b.nine.errno()) }); - } else { - b.trace(" 9p {s} fid={d} -> {s}", .{ what, fid, @errorName(e) }); - } - return e; - } - - // -- FUSE wrappers (tracing) -------------------------------------------------------- - - fn reply(b: *Bridge, unique: u64, payloads: []const []const u8) error{FuseIo}!void { - var total: usize = 0; - for (payloads) |p| total += p.len; - b.trace("-> unique={d} ok ({d} bytes)", .{ unique, total }); - fuse.reply(b.fuse_fd, unique, payloads) catch return error.FuseIo; - } - - fn replyError(b: *Bridge, unique: u64, code: linux.E) error{FuseIo}!void { - b.trace("-> unique={d} error E{s}", .{ unique, @tagName(code) }); - fuse.replyError(b.fuse_fd, unique, code) catch return error.FuseIo; - } - - // -- attrs --------------------------------------------------------------------------- - - fn inoOf(b: *const Bridge, nodeid: u64, qid: cloud9.Qid) u64 { - return if (nodeid == fuse.root_id or qid.path == b.root_path) fuse.root_id else qid.path; - } - - fn inoOfNode(b: *const Bridge, nodeid: u64) u64 { - if (nodeid == fuse.root_id) return fuse.root_id; - const ino = b.inodes.get(nodeid) orelse return nodeid; - return b.inoOf(nodeid, ino.qid); - } - - fn attrFrom(b: *const Bridge, st: cloud9.Stat, ino: u64) fuse.Attr { - return attrFromStat(st, ino, b.opts.uid, b.opts.gid); - } - - fn attrOut(b: *const Bridge, st: cloud9.Stat, ino: u64) fuse.AttrOut { - return .{ - .attr_valid = b.opts.attr_timeout_ns / 1_000_000_000, - .attr_valid_nsec = @intCast(b.opts.attr_timeout_ns % 1_000_000_000), - .attr = b.attrFrom(st, ino), - }; - } -}; - -// -- pure helpers (unit-tested) ------------------------------------------------------------ - -/// Attr from a 9P Stat: DMDIR → S_IFDIR else S_IFREG, low 9 permission bits kept. -pub fn attrFromStat(st: cloud9.Stat, ino: u64, uid: u32, gid: u32) fuse.Attr { - const ftype: u32 = if (st.mode & cloud9.dmdir != 0) fuse.S_IFDIR else fuse.S_IFREG; - return .{ - .ino = ino, - // The kernel marks an inode bad when size > LLONG_MAX; clamp hostile lengths. - .size = @min(st.length, std.math.maxInt(i64)), - // Saturating: a hostile length of 2^64-1 must not overflow. - .blocks = st.length / 512 + @intFromBool(st.length % 512 != 0), - .atime = st.atime, - .mtime = st.mtime, - .ctime = st.mtime, - .mode = ftype | (st.mode & 0o777), - .nlink = 1, - .uid = uid, - .gid = gid, - .blksize = 4096, - }; -} - -/// Kernel open(2) flags → 9P open mode. O_APPEND has no 9P equivalent and is ignored. -pub fn openMode(flags: u32) u8 { - const o: linux.O = @bitCast(flags); - var mode: u8 = switch (o.ACCMODE) { - .RDONLY => cloud9.oread, - .WRONLY => cloud9.owrite, - .RDWR => cloud9.ordwr, - }; - if (o.TRUNC) mode |= cloud9.otrunc; - return mode; -} - -/// Parses consecutive 9P directory records (2-byte size + Stat) and appends entries. -pub fn parseDirRecords(gpa: std.mem.Allocator, bytes: []const u8, list: *DirList) error{ OutOfMemory, BadDir }!void { - var pos: usize = 0; - while (pos < bytes.len) { - if (bytes.len - pos < 2) return error.BadDir; - const size: usize = std.mem.readInt(u16, bytes[pos..][0..2], .little); - if (bytes.len - pos < 2 + size) return error.BadDir; - const st = cloud9.Stat.decode(bytes[pos..][0 .. 2 + size]) catch return error.BadDir; - pos += 2 + size; - // The kernel rejects a whole READDIR reply (EIO) over one bad name, and - // "." and ".." are synthesised by loadDir: drop such records instead. - if (!validDirentName(st.name)) continue; - const name = try gpa.dupe(u8, st.name); - errdefer gpa.free(name); - try list.entries.append(gpa, .{ - .name = name, - .ino = st.qid.path, - .dtype = if (st.mode & cloud9.dmdir != 0) fuse.DT_DIR else fuse.DT_REG, - }); - } -} - -/// A name the kernel will accept in a dirent and that does not duplicate the synthetic "." / "..". -pub fn validDirentName(name: []const u8) bool { - if (name.len == 0 or name.len > max_name_len) return false; - if (std.mem.indexOfAny(u8, name, "/\x00") != null) return false; - if (std.mem.eql(u8, name, ".") or std.mem.eql(u8, name, "..")) return false; - return true; -} - -/// Packs dirents from `entries[offset..]` into `buf`; each record's `off` is its index + 1. -/// Returns the number of bytes used. -pub fn packDirents(entries: []const Entry, offset: u64, buf: []u8) usize { - var used: usize = 0; - var i: usize = @intCast(@min(offset, entries.len)); - while (i < entries.len) : (i += 1) { - const e = entries[i]; - if (!fuse.addDirent(buf, &used, e.ino, @as(u64, i) + 1, e.dtype, e.name)) break; - } - return used; -} - -fn randomU64() u64 { - var bytes: [8]u8 = undefined; - if (linux.errno(linux.getrandom(&bytes, bytes.len, 0)) == .SUCCESS) return std.mem.readInt(u64, &bytes, .little); - var ts: linux.timespec = undefined; - _ = linux.clock_gettime(.MONOTONIC, &ts); - return @as(u64, @bitCast(ts.nsec)) ^ (@as(u64, @bitCast(ts.sec)) << 32); -} - -fn nowSeconds() u32 { - var ts: linux.timespec = undefined; - if (linux.errno(linux.clock_gettime(.REALTIME, &ts)) != .SUCCESS) return 0; - return @intCast(@as(u64, @intCast(ts.sec)) & 0xFFFF_FFFF); -} - -fn opName(op: fuse.Opcode) []const u8 { - return switch (op) { - _ => "unknown", - else => @tagName(op), - }; -} - -// Thin adapters so fuse.zig's parse errors become HandlerError.BadRequest. -fn body(comptime T: type, req: fuse.Request) error{BadRequest}!*const T { - return fuse.body(T, req) catch error.BadRequest; -} - -fn nameAfter(comptime T: type, req: fuse.Request) error{BadRequest}![]const u8 { - return fuse.nameAfter(T, req) catch error.BadRequest; -} - -fn secondName(req: fuse.Request, first: []const u8, offset: usize) error{BadRequest}![]const u8 { - return fuse.secondName(req, first, offset) catch error.BadRequest; -} - -// -- tests ------------------------------------------------------------------------------ - -const testing = std.testing; - -test { - // Force semantic analysis of `serve` and the whole dispatch path, which no - // unit test can exercise without a FUSE mount. - testing.refAllDecls(@This()); -} - -fn testStat(name: []const u8, mode: u32, length: u64, path: u64) cloud9.Stat { - return .{ - .type = 0, - .dev = 0, - .qid = .{ .type = if (mode & cloud9.dmdir != 0) cloud9.qtdir else 0, .version = 0, .path = path }, - .mode = mode, - .atime = 100, - .mtime = 200, - .length = length, - .name = name, - .uid = "u", - .gid = "g", - .muid = "u", - }; -} - -test "attr mapping: DMDIR → S_IFDIR|perm, length → size/blocks" { - const d = attrFromStat(testStat("d", cloud9.dmdir | 0o755, 0, 9), 9, 1000, 1001); - try testing.expectEqual(fuse.S_IFDIR | 0o755, d.mode); - try testing.expectEqual(@as(u64, 9), d.ino); - try testing.expectEqual(@as(u64, 0), d.size); - try testing.expectEqual(@as(u64, 0), d.blocks); - try testing.expectEqual(@as(u32, 1000), d.uid); - try testing.expectEqual(@as(u32, 1001), d.gid); - try testing.expectEqual(@as(u32, 1), d.nlink); - - const f = attrFromStat(testStat("f", 0o640 | cloud9.dmappend, 1025, 4), 4, 0, 0); - try testing.expectEqual(fuse.S_IFREG | 0o640, f.mode); // dmappend bit not leaked - try testing.expectEqual(@as(u64, 1025), f.size); - try testing.expectEqual(@as(u64, 3), f.blocks); - try testing.expectEqual(@as(u32, 4096), f.blksize); - try testing.expectEqual(@as(u64, 100), f.atime); - try testing.expectEqual(@as(u64, 200), f.mtime); - try testing.expectEqual(@as(u64, 200), f.ctime); - - try testing.expectEqual(@as(u64, 1), attrFromStat(testStat("f", 0o600, 512, 4), 4, 0, 0).blocks); - try testing.expectEqual(@as(u64, 2), attrFromStat(testStat("f", 0o600, 513, 4), 4, 0, 0).blocks); -} - -test "open flag → 9P mode mapping" { - const rdonly: u32 = @bitCast(linux.O{ .ACCMODE = .RDONLY }); - const wronly: u32 = @bitCast(linux.O{ .ACCMODE = .WRONLY }); - const rdwr: u32 = @bitCast(linux.O{ .ACCMODE = .RDWR }); - const trunc: u32 = @bitCast(linux.O{ .TRUNC = true }); - const append: u32 = @bitCast(linux.O{ .APPEND = true }); - const creat: u32 = @bitCast(linux.O{ .CREAT = true }); - try testing.expectEqual(cloud9.oread, openMode(rdonly)); - try testing.expectEqual(cloud9.owrite, openMode(wronly)); - try testing.expectEqual(cloud9.ordwr, openMode(rdwr)); - try testing.expectEqual(cloud9.owrite | cloud9.otrunc, openMode(wronly | trunc)); - try testing.expectEqual(cloud9.ordwr | cloud9.otrunc, openMode(rdwr | trunc | creat)); - try testing.expectEqual(cloud9.owrite, openMode(wronly | append)); // O_APPEND ignored -} - -test "dirlist parsing from two hand-encoded Stat records" { - var buf: [512]u8 = undefined; - const a = try cloud9.Stat.encode(testStat("alpha", 0o644, 10, 0x11), &buf); - const bb = try cloud9.Stat.encode(testStat("beta", cloud9.dmdir | 0o755, 0, 0x22), buf[a.len..]); - const bytes = buf[0 .. a.len + bb.len]; - // Sanity: the record is prefixed by its own 2-byte size. - try testing.expectEqual(a.len - 2, std.mem.readInt(u16, bytes[0..2], .little)); - - var list: DirList = .{}; - defer list.deinit(testing.allocator); - try parseDirRecords(testing.allocator, bytes, &list); - try testing.expectEqual(@as(usize, 2), list.entries.items.len); - try testing.expectEqualStrings("alpha", list.entries.items[0].name); - try testing.expectEqual(@as(u64, 0x11), list.entries.items[0].ino); - try testing.expectEqual(fuse.DT_REG, list.entries.items[0].dtype); - try testing.expectEqualStrings("beta", list.entries.items[1].name); - try testing.expectEqual(@as(u64, 0x22), list.entries.items[1].ino); - try testing.expectEqual(fuse.DT_DIR, list.entries.items[1].dtype); - - // Truncated input is a protocol error and leaves earlier entries intact. - try testing.expectError(error.BadDir, parseDirRecords(testing.allocator, bytes[0 .. bytes.len - 1], &list)); - try testing.expectEqual(@as(usize, 3), list.entries.items.len); -} - -test "readdir packing and offset resumption" { - const names = [_][]const u8{ ".", "..", "one", "two", "three" }; - var entries: [names.len]Entry = undefined; - for (&entries, names, 0..) |*e, n, i| e.* = .{ .name = @constCast(n), .ino = 100 + i, .dtype = if (i < 2) fuse.DT_DIR else fuse.DT_REG }; - - // Everything fits: five records, off = index + 1. - var big: [1024]u8 = undefined; - const used = packDirents(&entries, 0, &big); - var pos: usize = 0; - var idx: usize = 0; - while (pos < used) : (idx += 1) { - const d = std.mem.bytesToValue(fuse.Dirent, big[pos..][0..@sizeOf(fuse.Dirent)]); - try testing.expectEqual(@as(u64, 100 + idx), d.ino); - try testing.expectEqual(@as(u64, idx + 1), d.off); - try testing.expectEqualStrings(names[idx], big[pos + @sizeOf(fuse.Dirent) ..][0..d.namelen]); - pos += (@sizeOf(fuse.Dirent) + d.namelen + 7) & ~@as(usize, 7); - } - try testing.expectEqual(names.len, idx); - - // A buffer that fits exactly two records ("." = 32, ".." = 32) stops there… - var small: [64]u8 = undefined; - const first_used = packDirents(&entries, 0, &small); - try testing.expectEqual(@as(usize, 64), first_used); - const last = std.mem.bytesToValue(fuse.Dirent, small[32..][0..@sizeOf(fuse.Dirent)]); - try testing.expectEqual(@as(u64, 2), last.off); - // …and resuming at the last `off` yields "one" next. - const second_used = packDirents(&entries, last.off, &small); - const next = std.mem.bytesToValue(fuse.Dirent, small[0..@sizeOf(fuse.Dirent)]); - try testing.expectEqualStrings("one", small[@sizeOf(fuse.Dirent)..][0..next.namelen]); - try testing.expectEqual(@as(u64, 3), next.off); - try testing.expect(second_used > 0); - - // Past the end: nothing (EOF for the kernel). - try testing.expectEqual(@as(usize, 0), packDirents(&entries, names.len, &big)); - try testing.expectEqual(@as(usize, 0), packDirents(&entries, 1000, &big)); -} - -test "attr mapping saturates hostile lengths instead of overflowing" { - const a = attrFromStat(testStat("f", 0o600, std.math.maxInt(u64), 4), 4, 0, 0); - try testing.expectEqual(@as(u64, std.math.maxInt(i64)), a.size); - try testing.expectEqual(@as(u64, std.math.maxInt(u64) / 512 + 1), a.blocks); - const b = attrFromStat(testStat("f", 0o600, 1024, 4), 4, 0, 0); - try testing.expectEqual(@as(u64, 2), b.blocks); - try testing.expectEqual(@as(u64, 1024), b.size); -} - -test "dirent names the kernel would reject are dropped from listings" { - try testing.expect(validDirentName("a")); - try testing.expect(validDirentName("x" ** 1024)); - try testing.expect(!validDirentName("")); - try testing.expect(!validDirentName("a/b")); - try testing.expect(!validDirentName("a\x00b")); - try testing.expect(!validDirentName(".")); - try testing.expect(!validDirentName("..")); - try testing.expect(!validDirentName("x" ** 1025)); - - var buf: [4096]u8 = undefined; - var n: usize = 0; - for ([_][]const u8{ ".", "..", "", "a/b", "keep", "x" ** 1025, "also" }) |name| { - n += (try cloud9.Stat.encode(testStat(name, 0o644, 1, 0x30), buf[n..])).len; - } - var list: DirList = .{}; - defer list.deinit(testing.allocator); - try parseDirRecords(testing.allocator, buf[0..n], &list); - try testing.expectEqual(@as(usize, 2), list.entries.items.len); - try testing.expectEqualStrings("keep", list.entries.items[0].name); - try testing.expectEqualStrings("also", list.entries.items[1].name); -} - -test "DirList frees its names" { - var list: DirList = .{}; - try list.entries.append(testing.allocator, .{ .name = try testing.allocator.dupe(u8, "x"), .ino = 1, .dtype = fuse.DT_REG }); - list.deinit(testing.allocator); -} diff --git a/9player/src/fuse.zig b/9player/src/fuse.zig deleted file mode 100644 index ef11873..0000000 --- a/9player/src/fuse.zig +++ /dev/null @@ -1,653 +0,0 @@ -//! Kernel FUSE protocol subset (no libfuse, no libc, no policy). -//! -//! Extern structs mirror `/usr/include/linux/fuse.h` (kernel header 7.45); -//! every layout is checked against the header's size at comptime. Only the -//! opcodes and structs 9player needs are here. The I/O helpers are blocking -//! and allocation-free: the caller owns a single request buffer. -//! -//! Wire rules worth remembering: -//! * The kernel delivers exactly one request per `read(2)` on `/dev/fuse`, -//! and a reply must be exactly one `write(2)`/`writev(2)`. -//! * Request bodies start right after the 40-byte `InHeader`; since every -//! in-struct is 8-byte aligned in the header, `body()` requires the caller's -//! buffer to be 8-byte aligned (`std.heap` page allocations and -//! `align(8)` arrays both qualify). -//! * A write that fails with `ENOENT` means the request was interrupted and -//! the kernel already forgot it: the reply is silently dropped. -//! * `ENODEV` on read means the filesystem was unmounted. - -const std = @import("std"); -const linux = std.os.linux; - -pub const kernel_version: u32 = 7; -/// The minor we answer; the kernel adapts to the lower of the two. -pub const kernel_minor: u32 = 31; -pub const root_id: u64 = 1; - -pub const FOPEN_DIRECT_IO: u32 = 1 << 0; -pub const FOPEN_KEEP_CACHE: u32 = 1 << 1; -pub const FOPEN_NONSEEKABLE: u32 = 1 << 2; - -pub const FUSE_ASYNC_READ: u32 = 1 << 0; -/// The kernel passes O_TRUNC in OPEN instead of a separate SETATTR(size=0); 9P has OTRUNC for exactly this. -pub const FUSE_ATOMIC_O_TRUNC: u32 = 1 << 3; -/// Without this the kernel's cached-write path (`--no-direct-io`) sends one 4 KiB WRITE per page. -pub const FUSE_BIG_WRITES: u32 = 1 << 5; -/// Re-fetch a cached inode's size/mtime and drop stale pages when they change. -/// Required for `--no-direct-io` correctness: 9P sizes change under us, and -/// without this the kernel trusts a stale cached size and truncates reads. -pub const FUSE_AUTO_INVAL_DATA: u32 = 1 << 12; -pub const FUSE_MAX_PAGES: u32 = 1 << 22; - -pub const FATTR_MODE: u32 = 1 << 0; -pub const FATTR_UID: u32 = 1 << 1; -pub const FATTR_GID: u32 = 1 << 2; -pub const FATTR_SIZE: u32 = 1 << 3; -pub const FATTR_ATIME: u32 = 1 << 4; -pub const FATTR_MTIME: u32 = 1 << 5; -pub const FATTR_FH: u32 = 1 << 6; -pub const FATTR_ATIME_NOW: u32 = 1 << 7; -pub const FATTR_MTIME_NOW: u32 = 1 << 8; -pub const FATTR_LOCKOWNER: u32 = 1 << 9; -pub const FATTR_CTIME: u32 = 1 << 10; - -/// File type bits for `Attr.mode` and `Dirent.type` (used by the bridge). -pub const S_IFDIR: u32 = linux.S.IFDIR; -pub const S_IFREG: u32 = linux.S.IFREG; -pub const DT_DIR: u32 = linux.DT.DIR; -pub const DT_REG: u32 = linux.DT.REG; - -pub const Opcode = enum(u32) { - lookup = 1, - forget = 2, - getattr = 3, - setattr = 4, - readlink = 5, - symlink = 6, - mknod = 8, - mkdir = 9, - unlink = 10, - rmdir = 11, - rename = 12, - link = 13, - open = 14, - read = 15, - write = 16, - statfs = 17, - release = 18, - fsync = 20, - setxattr = 21, - getxattr = 22, - listxattr = 23, - removexattr = 24, - flush = 25, - init = 26, - opendir = 27, - readdir = 28, - releasedir = 29, - fsyncdir = 30, - getlk = 31, - setlk = 32, - setlkw = 33, - access = 34, - create = 35, - interrupt = 36, - bmap = 37, - destroy = 38, - ioctl = 39, - poll = 40, - notify_reply = 41, - batch_forget = 42, - fallocate = 43, - readdirplus = 44, - rename2 = 45, - lseek = 46, - copy_file_range = 47, - setupmapping = 48, - removemapping = 49, - syncfs = 50, - tmpfile = 51, - statx = 52, - _, -}; - -// --------------------------------------------------------------------------- -// Structs (field order and widths follow linux/fuse.h exactly) -// --------------------------------------------------------------------------- - -pub const InHeader = extern struct { - len: u32, - opcode: u32, - unique: u64, - nodeid: u64, - uid: u32, - gid: u32, - pid: u32, - total_extlen: u16, - padding: u16, - - pub fn op(h: InHeader) Opcode { - return @enumFromInt(h.opcode); - } -}; - -pub const OutHeader = extern struct { - len: u32, - @"error": i32, - unique: u64, -}; - -pub const Attr = extern struct { - ino: u64 = 0, - size: u64 = 0, - blocks: u64 = 0, - atime: u64 = 0, - mtime: u64 = 0, - ctime: u64 = 0, - atimensec: u32 = 0, - mtimensec: u32 = 0, - ctimensec: u32 = 0, - mode: u32 = 0, - nlink: u32 = 0, - uid: u32 = 0, - gid: u32 = 0, - rdev: u32 = 0, - blksize: u32 = 0, - flags: u32 = 0, -}; - -pub const EntryOut = extern struct { - nodeid: u64 = 0, - generation: u64 = 0, - entry_valid: u64 = 0, - attr_valid: u64 = 0, - entry_valid_nsec: u32 = 0, - attr_valid_nsec: u32 = 0, - attr: Attr = .{}, -}; - -pub const AttrOut = extern struct { - attr_valid: u64 = 0, - attr_valid_nsec: u32 = 0, - dummy: u32 = 0, - attr: Attr = .{}, -}; - -pub const GetattrIn = extern struct { getattr_flags: u32, dummy: u32, fh: u64 }; - -pub const SetattrIn = extern struct { - valid: u32, - padding: u32, - fh: u64, - size: u64, - lock_owner: u64, - atime: u64, - mtime: u64, - ctime: u64, - atimensec: u32, - mtimensec: u32, - ctimensec: u32, - mode: u32, - unused4: u32, - uid: u32, - gid: u32, - unused5: u32, -}; - -pub const OpenIn = extern struct { flags: u32, open_flags: u32 }; -pub const OpenOut = extern struct { fh: u64 = 0, open_flags: u32 = 0, backing_id: i32 = 0 }; -pub const ReleaseIn = extern struct { fh: u64, flags: u32, release_flags: u32, lock_owner: u64 }; -pub const FlushIn = extern struct { fh: u64, unused: u32, padding: u32, lock_owner: u64 }; - -pub const ReadIn = extern struct { - fh: u64, - offset: u64, - size: u32, - read_flags: u32, - lock_owner: u64, - flags: u32, - padding: u32, -}; - -pub const WriteIn = extern struct { - fh: u64, - offset: u64, - size: u32, - write_flags: u32, - lock_owner: u64, - flags: u32, - padding: u32, -}; - -pub const WriteOut = extern struct { size: u32, padding: u32 = 0 }; -pub const CreateIn = extern struct { flags: u32, mode: u32, umask: u32, open_flags: u32 }; -pub const MkdirIn = extern struct { mode: u32, umask: u32 }; -pub const RenameIn = extern struct { newdir: u64 }; -pub const Rename2In = extern struct { newdir: u64, flags: u32, padding: u32 }; -pub const ForgetIn = extern struct { nlookup: u64 }; -pub const BatchForgetIn = extern struct { count: u32, dummy: u32 }; -pub const ForgetOne = extern struct { nodeid: u64, nlookup: u64 }; -pub const FsyncIn = extern struct { fh: u64, fsync_flags: u32, padding: u32 }; -pub const AccessIn = extern struct { mask: u32, padding: u32 }; -pub const InterruptIn = extern struct { unique: u64 }; -pub const LseekIn = extern struct { fh: u64, offset: u64, whence: u32, padding: u32 }; - -pub const Kstatfs = extern struct { - blocks: u64 = 0, - bfree: u64 = 0, - bavail: u64 = 0, - files: u64 = 0, - ffree: u64 = 0, - bsize: u32 = 0, - namelen: u32 = 0, - frsize: u32 = 0, - padding: u32 = 0, - spare: [6]u32 = [_]u32{0} ** 6, -}; - -pub const StatfsOut = extern struct { st: Kstatfs = .{} }; - -pub const InitIn = extern struct { - major: u32, - minor: u32, - max_readahead: u32, - flags: u32, - flags2: u32, - unused: [11]u32, -}; - -/// 64 bytes; the kernel accepts this size whenever the answered minor >= 23. -pub const InitOut = extern struct { - major: u32 = kernel_version, - minor: u32 = kernel_minor, - max_readahead: u32 = 0, - flags: u32 = 0, - max_background: u16 = 0, - congestion_threshold: u16 = 0, - max_write: u32 = 0, - time_gran: u32 = 0, - max_pages: u16 = 0, - map_alignment: u16 = 0, - flags2: u32 = 0, - max_stack_depth: u32 = 0, - request_timeout: u16 = 0, - unused: [11]u16 = [_]u16{0} ** 11, -}; - -/// Fixed 24-byte head of `fuse_dirent`; the name follows, padded to 8 bytes. -pub const Dirent = extern struct { ino: u64, off: u64, namelen: u32, type: u32 }; - -comptime { - std.debug.assert(@sizeOf(InHeader) == 40); - std.debug.assert(@sizeOf(OutHeader) == 16); - std.debug.assert(@sizeOf(Attr) == 88); - std.debug.assert(@sizeOf(EntryOut) == 128); - std.debug.assert(@sizeOf(AttrOut) == 104); - std.debug.assert(@sizeOf(GetattrIn) == 16); - std.debug.assert(@sizeOf(SetattrIn) == 88); - std.debug.assert(@sizeOf(OpenIn) == 8); - std.debug.assert(@sizeOf(OpenOut) == 16); - std.debug.assert(@sizeOf(ReleaseIn) == 24); - std.debug.assert(@sizeOf(FlushIn) == 24); - std.debug.assert(@sizeOf(ReadIn) == 40); - std.debug.assert(@sizeOf(WriteIn) == 40); - std.debug.assert(@sizeOf(WriteOut) == 8); - std.debug.assert(@sizeOf(CreateIn) == 16); - std.debug.assert(@sizeOf(MkdirIn) == 8); - std.debug.assert(@sizeOf(RenameIn) == 8); - std.debug.assert(@sizeOf(Rename2In) == 16); - std.debug.assert(@sizeOf(ForgetIn) == 8); - std.debug.assert(@sizeOf(BatchForgetIn) == 8); - std.debug.assert(@sizeOf(ForgetOne) == 16); - std.debug.assert(@sizeOf(FsyncIn) == 16); - std.debug.assert(@sizeOf(AccessIn) == 8); - std.debug.assert(@sizeOf(InterruptIn) == 8); - std.debug.assert(@sizeOf(Kstatfs) == 80); - std.debug.assert(@sizeOf(StatfsOut) == 80); - std.debug.assert(@sizeOf(InitIn) == 64); - std.debug.assert(@sizeOf(InitOut) == 64); - std.debug.assert(@sizeOf(Dirent) == 24); - std.debug.assert(@sizeOf(LseekIn) == 24); -} - -// --------------------------------------------------------------------------- -// Request / reply helpers -// --------------------------------------------------------------------------- - -pub const Error = error{ Protocol, Io, TooManyPayloads }; - -pub const Request = struct { - header: InHeader, - /// Bytes after the header; a slice into the caller's buffer. - body: []const u8, -}; - -/// Reads one kernel request with a single `read(2)`. Returns null on ENODEV -/// (unmounted). Retries EINTR/EAGAIN/ENOENT. `buf` should be at least -/// `max_write + 4096` bytes and 8-byte aligned so `body()` can view it. -pub fn readRequest(fd: i32, buf: []u8) Error!?Request { - while (true) { - const rc = linux.read(fd, buf.ptr, buf.len); - switch (linux.errno(rc)) { - .SUCCESS => { - const n: usize = rc; - if (n < @sizeOf(InHeader)) return error.Protocol; - const header = std.mem.bytesToValue(InHeader, buf[0..@sizeOf(InHeader)]); - if (header.len != n) return error.Protocol; - return .{ .header = header, .body = buf[@sizeOf(InHeader)..n] }; - }, - .INTR, .AGAIN, .NOENT => continue, - .NODEV => return null, - else => return error.Io, - } - } -} - -/// Maximum number of payload slices a single `reply` can carry. -pub const max_payloads = 7; - -/// Success reply: `OutHeader` followed by the concatenated `payloads`, sent in -/// one `writev(2)`. An ENOENT from the kernel means the request was -/// interrupted; the reply is dropped and this returns normally. -pub fn reply(fd: i32, unique: u64, payloads: []const []const u8) Error!void { - if (payloads.len > max_payloads) return error.TooManyPayloads; - var total: usize = @sizeOf(OutHeader); - for (payloads) |p| total += p.len; - if (total > std.math.maxInt(u32)) return error.Protocol; - const header = OutHeader{ .len = @intCast(total), .@"error" = 0, .unique = unique }; - var iov: [max_payloads + 1]std.posix.iovec_const = undefined; - iov[0] = .{ .base = @ptrCast(&header), .len = @sizeOf(OutHeader) }; - for (payloads, 1..) |p, i| iov[i] = .{ .base = p.ptr, .len = p.len }; - return writeAll(fd, &iov, payloads.len + 1, total); -} - -/// Error reply: an `OutHeader` carrying `-errno` and no payload. -pub fn replyError(fd: i32, unique: u64, err: linux.E) Error!void { - const code: i32 = @intCast(@intFromEnum(err)); - const header = OutHeader{ .len = @sizeOf(OutHeader), .@"error" = -code, .unique = unique }; - var iov = [_]std.posix.iovec_const{.{ .base = @ptrCast(&header), .len = @sizeOf(OutHeader) }}; - return writeAll(fd, &iov, 1, @sizeOf(OutHeader)); -} - -fn writeAll(fd: i32, iov: [*]const std.posix.iovec_const, count: usize, total: usize) Error!void { - while (true) { - const rc = linux.writev(fd, iov, count); - switch (linux.errno(rc)) { - .SUCCESS => return if (rc == total) {} else error.Protocol, - .INTR => continue, - .NOENT => return, // request was interrupted; reply dropped - else => return error.Io, - } - } -} - -/// Appends a `fuse_dirent` (head + name, padded to a multiple of 8) at -/// `buf[used.*..]`. Returns false and leaves `buf`/`used` unchanged if the -/// record does not fit. -pub fn addDirent(buf: []u8, used: *usize, ino: u64, off: u64, dtype: u32, name: []const u8) bool { - const raw = @sizeOf(Dirent) + name.len; - const rec = (raw + 7) & ~@as(usize, 7); - if (used.* > buf.len or buf.len - used.* < rec) return false; - const dst = buf[used.*..][0..rec]; - const head = Dirent{ .ino = ino, .off = off, .namelen = @intCast(name.len), .type = dtype }; - @memcpy(dst[0..@sizeOf(Dirent)], std.mem.asBytes(&head)); - @memcpy(dst[@sizeOf(Dirent)..raw], name); - @memset(dst[raw..rec], 0); - used.* += rec; - return true; -} - -/// Views the first `@sizeOf(T)` bytes of `req.body` as `T` (copy-free). -/// Fails with `error.Protocol` if the body is too short or misaligned. -pub fn body(comptime T: type, req: Request) Error!*const T { - if (req.body.len < @sizeOf(T)) return error.Protocol; - if (@intFromPtr(req.body.ptr) % @alignOf(T) != 0) return error.Protocol; - return @ptrCast(@alignCast(req.body.ptr)); -} - -/// The NUL-terminated string at `req.body[offset..]`, without the NUL. -pub fn nameAt(req: Request, offset: usize) Error![]const u8 { - if (offset > req.body.len) return error.Protocol; - const rest = req.body[offset..]; - const end = std.mem.indexOfScalar(u8, rest, 0) orelse return error.Protocol; - return rest[0..end]; -} - -/// The NUL-terminated string following a `T` body (or at offset 0 when -/// `T == void`), e.g. LOOKUP's name (`void`) or MKDIR's name (`MkdirIn`). -pub fn nameAfter(comptime T: type, req: Request) Error![]const u8 { - const offset = if (T == void) 0 else @sizeOf(T); - return nameAt(req, offset); -} - -/// The string that follows `first` (obtained via `nameAt(req, offset)`), -/// for "old\0new\0" pairs such as RENAME's. -pub fn secondName(req: Request, first: []const u8, offset: usize) Error![]const u8 { - return nameAt(req, offset + first.len + 1); -} - -/// Builds the INIT reply per docs/DESIGN.md. -pub fn initReply(in: *const InitIn, max_write: u32) InitOut { - var out = InitOut{ - .major = kernel_version, - .minor = @min(kernel_minor, in.minor), - .max_readahead = in.max_readahead, - .flags = FUSE_ASYNC_READ | FUSE_ATOMIC_O_TRUNC | FUSE_BIG_WRITES | FUSE_AUTO_INVAL_DATA, - .max_background = 16, - .congestion_threshold = 12, - .max_write = max_write, - .time_gran = 1, - }; - if (in.flags & FUSE_MAX_PAGES != 0) { - out.flags |= FUSE_MAX_PAGES; - out.max_pages = 256; - } - return out; -} - -// --------------------------------------------------------------------------- -// Tests -// --------------------------------------------------------------------------- - -const testing = std.testing; - -test "struct sizes match linux/fuse.h" { - // The comptime block above is the real check; this makes it run under - // `zig test` even if the module is otherwise unreferenced. - try testing.expectEqual(@as(usize, 40), @sizeOf(InHeader)); - try testing.expectEqual(@as(usize, 64), @sizeOf(InitOut)); - try testing.expectEqual(@as(usize, 24), @sizeOf(Dirent)); - try testing.expectEqual(@as(u32, 26), @intFromEnum(Opcode.init)); - try testing.expectEqual(Opcode.statx, @as(Opcode, @enumFromInt(52))); -} - -test "addDirent pads records to 8 bytes and refuses when full" { - var buf: [1024]u8 = undefined; - var used: usize = 0; - const name = "abcdefghijklmnopq"; // 17 chars - var expect_total: usize = 0; - var n: usize = 1; - while (n <= 17) : (n += 1) { - const before = used; - try testing.expect(addDirent(&buf, &used, n, n, DT_REG, name[0..n])); - const rec = used - before; - try testing.expectEqual(@as(usize, 0), rec % 8); - try testing.expectEqual((24 + n + 7) & ~@as(usize, 7), rec); - // check head fields and NUL padding - const head = std.mem.bytesToValue(Dirent, buf[before..][0..24]); - try testing.expectEqual(n, head.ino); - try testing.expectEqual(@as(u32, @intCast(n)), head.namelen); - try testing.expectEqualStrings(name[0..n], buf[before + 24 ..][0..n]); - for (buf[before + 24 + n .. used]) |b| try testing.expectEqual(@as(u8, 0), b); - expect_total += rec; - } - try testing.expectEqual(expect_total, used); - - // A record that does not fit leaves everything untouched. - var small: [40]u8 = undefined; - var used2: usize = 0; - try testing.expect(addDirent(&small, &used2, 1, 1, DT_DIR, "0123456789abcdef")); // 24+16 = 40 - try testing.expectEqual(@as(usize, 40), used2); - try testing.expect(!addDirent(&small, &used2, 2, 2, DT_DIR, "x")); - try testing.expectEqual(@as(usize, 40), used2); - var tight: [31]u8 = undefined; - var used3: usize = 0; - try testing.expect(!addDirent(&tight, &used3, 1, 1, DT_REG, "a")); // needs 32 - try testing.expectEqual(@as(usize, 0), used3); -} - -test "body/nameAfter/secondName on hand-built requests" { - var buf: [128]u8 align(8) = undefined; - // LOOKUP(parent=1, "hello") - const name = "hello"; - const hdr = InHeader{ - .len = @intCast(@sizeOf(InHeader) + name.len + 1), - .opcode = @intFromEnum(Opcode.lookup), - .unique = 7, - .nodeid = root_id, - .uid = 1000, - .gid = 1000, - .pid = 42, - .total_extlen = 0, - .padding = 0, - }; - @memcpy(buf[0..40], std.mem.asBytes(&hdr)); - @memcpy(buf[40..45], name); - buf[45] = 0; - const req = Request{ .header = hdr, .body = buf[40..hdr.len] }; - try testing.expectEqual(Opcode.lookup, req.header.op()); - try testing.expectEqualStrings("hello", try nameAfter(void, req)); - try testing.expectError(error.Protocol, body(MkdirIn, Request{ .header = hdr, .body = buf[40..44] })); - - // MKDIR(mode=0o755) + "dir" - const mk = MkdirIn{ .mode = 0o755, .umask = 0o22 }; - @memcpy(buf[40..48], std.mem.asBytes(&mk)); - @memcpy(buf[48..51], "dir"); - buf[51] = 0; - const mreq = Request{ .header = hdr, .body = buf[40..52] }; - const got = try body(MkdirIn, mreq); - try testing.expectEqual(@as(u32, 0o755), got.mode); - try testing.expectEqualStrings("dir", try nameAfter(MkdirIn, mreq)); - - // RENAME(newdir) + "old\0new\0" - const rn = RenameIn{ .newdir = 9 }; - @memcpy(buf[40..48], std.mem.asBytes(&rn)); - @memcpy(buf[48..56], "old\x00new\x00"); - const rreq = Request{ .header = hdr, .body = buf[40..56] }; - try testing.expectEqual(@as(u64, 9), (try body(RenameIn, rreq)).newdir); - const old = try nameAfter(RenameIn, rreq); - try testing.expectEqualStrings("old", old); - try testing.expectEqualStrings("new", try secondName(rreq, old, @sizeOf(RenameIn))); - try testing.expectError(error.Protocol, secondName(rreq, "new", @sizeOf(RenameIn) + 4)); - - // Missing NUL and misalignment are protocol errors. - try testing.expectError(error.Protocol, nameAt(Request{ .header = hdr, .body = buf[48..51] }, 0)); - try testing.expectError(error.Protocol, body(MkdirIn, Request{ .header = hdr, .body = buf[41..57] })); -} - -test "initReply fields" { - var in = InitIn{ .major = 7, .minor = 45, .max_readahead = 131072, .flags = 0, .flags2 = 0, .unused = [_]u32{0} ** 11 }; - const a = initReply(&in, 1 << 20); - try testing.expectEqual(@as(u32, 7), a.major); - try testing.expectEqual(@as(u32, 31), a.minor); - try testing.expectEqual(@as(u32, 131072), a.max_readahead); - try testing.expectEqual(FUSE_ASYNC_READ | FUSE_ATOMIC_O_TRUNC | FUSE_BIG_WRITES | FUSE_AUTO_INVAL_DATA, a.flags); - try testing.expectEqual(@as(u16, 0), a.max_pages); - try testing.expectEqual(@as(u16, 16), a.max_background); - try testing.expectEqual(@as(u16, 12), a.congestion_threshold); - try testing.expectEqual(@as(u32, 1 << 20), a.max_write); - try testing.expectEqual(@as(u32, 1), a.time_gran); - - in.flags = FUSE_MAX_PAGES | FUSE_ASYNC_READ; - in.minor = 27; - const b = initReply(&in, 4096); - try testing.expectEqual(@as(u32, 27), b.minor); - try testing.expectEqual(FUSE_ASYNC_READ | FUSE_ATOMIC_O_TRUNC | FUSE_BIG_WRITES | FUSE_AUTO_INVAL_DATA | FUSE_MAX_PAGES, b.flags); - try testing.expectEqual(@as(u16, 256), b.max_pages); - try testing.expectEqual(@as(u32, 4096), b.max_write); -} - -fn makePipe() ![2]i32 { - var fds: [2]i32 = undefined; - if (linux.errno(linux.pipe2(&fds, .{ .CLOEXEC = true })) != .SUCCESS) return error.Io; - return fds; -} - -fn readExact(fd: i32, out: []u8) !void { - var got: usize = 0; - while (got < out.len) { - const rc = linux.read(fd, out[got..].ptr, out.len - got); - if (linux.errno(rc) != .SUCCESS or rc == 0) return error.Io; - got += rc; - } -} - -test "reply writes header + payloads through a pipe" { - const fds = try makePipe(); - defer _ = linux.close(fds[0]); - defer _ = linux.close(fds[1]); - - const oo = OpenOut{ .fh = 0x1234, .open_flags = FOPEN_DIRECT_IO }; - try reply(fds[1], 99, &.{ std.mem.asBytes(&oo), "tail" }); - - var out: [16 + 16 + 4]u8 = undefined; - try readExact(fds[0], &out); - const h = std.mem.bytesToValue(OutHeader, out[0..16]); - try testing.expectEqual(@as(u32, 36), h.len); - try testing.expectEqual(@as(i32, 0), h.@"error"); - try testing.expectEqual(@as(u64, 99), h.unique); - try testing.expectEqualSlices(u8, std.mem.asBytes(&oo), out[16..32]); - try testing.expectEqualStrings("tail", out[32..36]); - - // Empty payload list: header only. - try reply(fds[1], 5, &.{}); - var only: [16]u8 = undefined; - try readExact(fds[0], &only); - try testing.expectEqual(@as(u32, 16), std.mem.bytesToValue(OutHeader, &only).len); - - var too_many: [max_payloads + 1][]const u8 = undefined; - for (&too_many) |*p| p.* = "x"; - try testing.expectError(error.TooManyPayloads, reply(fds[1], 1, &too_many)); -} - -test "replyError writes a negative errno" { - const fds = try makePipe(); - defer _ = linux.close(fds[0]); - defer _ = linux.close(fds[1]); - - try replyError(fds[1], 0xdead_beef, .NOENT); - var out: [16]u8 = undefined; - try readExact(fds[0], &out); - const h = std.mem.bytesToValue(OutHeader, &out); - try testing.expectEqual(@as(u32, 16), h.len); - try testing.expectEqual(@as(i32, -2), h.@"error"); - try testing.expectEqual(@as(u64, 0xdead_beef), h.unique); - - try replyError(fds[1], 1, .NOSYS); - try readExact(fds[0], &out); - try testing.expectEqual(-@as(i32, @intCast(@intFromEnum(linux.E.NOSYS))), std.mem.bytesToValue(OutHeader, &out).@"error"); -} - -test "readRequest parses one request from a pipe and rejects bad lengths" { - const fds = try makePipe(); - defer _ = linux.close(fds[0]); - defer _ = linux.close(fds[1]); - - var wire: [48]u8 align(8) = undefined; - const hdr = InHeader{ .len = 48, .opcode = @intFromEnum(Opcode.forget), .unique = 3, .nodeid = 2, .uid = 0, .gid = 0, .pid = 0, .total_extlen = 0, .padding = 0 }; - @memcpy(wire[0..40], std.mem.asBytes(&hdr)); - @memcpy(wire[40..48], std.mem.asBytes(&ForgetIn{ .nlookup = 11 })); - try testing.expectEqual(@as(usize, 48), linux.write(fds[1], &wire, wire.len)); - - var buf: [4096]u8 align(8) = undefined; - const req = (try readRequest(fds[0], &buf)) orelse return error.Io; - try testing.expectEqual(Opcode.forget, req.header.op()); - try testing.expectEqual(@as(u64, 2), req.header.nodeid); - try testing.expectEqual(@as(u64, 11), (try body(ForgetIn, req)).nlookup); - - // Header length disagreeing with what was read is a protocol error. - var bad = wire; - std.mem.bytesAsValue(InHeader, bad[0..40]).len = 40; - try testing.expectEqual(@as(usize, 48), linux.write(fds[1], &bad, bad.len)); - try testing.expectError(error.Protocol, readRequest(fds[0], &buf)); -} diff --git a/9player/src/main.zig b/9player/src/main.zig deleted file mode 100644 index 24990b9..0000000 --- a/9player/src/main.zig +++ /dev/null @@ -1,444 +0,0 @@ -//! 9player: mount a 9P2000 tree into a fresh user+mount namespace via FUSE -//! and run a program inside it. -//! -//! Exit codes: the child's status (128+sig if signalled); 125 for 9player's -//! own failures (usage, connect, attach, namespace/mount); 126/127 for exec -//! failures. - -const std = @import("std"); -const linux = std.os.linux; -const ns = @import("ns.zig"); -const nine = @import("nine.zig"); -const bridge = @import("bridge.zig"); - -const version_string = "9player 0.1.0"; - -const usage_text = - \\Usage: 9player [options] -- PROGRAM [ARGS...] - \\Transport (exactly one): - \\ --unix PATH Unix stream socket - \\ --tcp IP:PORT TCP (IPv4/IPv6 literal) - \\ --fd N already-connected inherited descriptor - \\ --spawn CMD run CMD (via /bin/sh -c) with a socketpair on its stdin/stdout - \\Options: - \\ --mount PATH mountpoint inside the new namespace (default /mnt/9p) - \\ --uname NAME 9P user name (default $USER, else "none") - \\ --aname NAME 9P tree to attach (default "") - \\ --msize BYTES maximum 9P message size to request (default 131072) - \\ --cache SECONDS attr/entry cache validity, may be fractional (default 1) - \\ --no-direct-io let the kernel cache file pages (trusts stat length) - \\ --debug trace FUSE and 9P operations on stderr - \\ --help, --version - \\PROGRAM defaults to $SHELL (else /bin/sh). The mountpoint is exported as $NINEPLAYER_MOUNT. - \\ -; - -const own_failure: u8 = 125; -/// Largest 9P message size we agree to request: the session allocates two -/// buffers of this size up front, before the server negotiates it down. -const max_msize: u32 = 16 * 1024 * 1024; - -/// Write `text` to stdout (informational output such as --help); errors are -/// ignored, there is nowhere better to report them. -fn printStdout(text: []const u8) void { - var off: usize = 0; - while (off < text.len) { - const rc = linux.write(1, text[off..].ptr, text.len - off); - switch (linux.errno(rc)) { - .SUCCESS => off += rc, - .INTR => continue, - else => return, - } - } -} - -const Config = struct { - address: ?nine.Address = null, - spawn_cmd: ?[]const u8 = null, - mount: []const u8 = "/mnt/9p", - uname: ?[]const u8 = null, - aname: []const u8 = "", - msize: u32 = 131072, - cache_ns: u64 = 1_000_000_000, - direct_io: bool = true, - debug: bool = false, - /// Empty means "default program". - program: []const []const u8 = &.{}, -}; - -const ParseResult = union(enum) { - run: Config, - /// Usage error, already reported on stderr; exit with this status. - exit: u8, - /// --help/--version: text for stdout, then exit 0. Printing is left to - /// `main` so that no test path writes to fd 1 (under `zig build test` - /// that is the test runner's protocol pipe). - info: []const u8, -}; - -fn usageError(comptime fmt: []const u8, args: anytype) ParseResult { - std.debug.print("9player: " ++ fmt ++ "\n(try 9player --help)\n", args); - return .{ .exit = own_failure }; -} - -fn parseArgs(arena: std.mem.Allocator, args: []const [:0]const u8) !ParseResult { - var cfg = Config{}; - var transports: usize = 0; - var i: usize = 1; - var program_start: ?usize = null; - while (i < args.len) : (i += 1) { - const arg: []const u8 = args[i]; - if (std.mem.eql(u8, arg, "--")) { - program_start = i + 1; - break; - } - if (!std.mem.startsWith(u8, arg, "--")) { - // A single-dash word is a typo for an option, not a program. - if (arg.len > 1 and arg[0] == '-') return usageError("unknown option {s} (options start with --)", .{arg}); - // A bare word starts PROGRAM, as if "--" were given. - program_start = i; - break; - } - // Split "--opt=value". - var name = arg; - var inline_value: ?[]const u8 = null; - if (std.mem.indexOfScalar(u8, arg, '=')) |eq| { - name = arg[0..eq]; - inline_value = arg[eq + 1 ..]; - } - const Opt = enum { unix, tcp, fd, spawn, mount, uname, aname, msize, cache, @"no-direct-io", debug, help, version, unknown }; - const opt = std.meta.stringToEnum(Opt, name[2..]) orelse .unknown; - switch (opt) { - .@"no-direct-io", .debug, .help, .version => if (inline_value != null) return usageError("{s} takes no value", .{name}), - .unknown => return usageError("unknown option {s}", .{name}), - else => {}, - } - const value: []const u8 = switch (opt) { - .@"no-direct-io", .debug, .help, .version, .unknown => "", - else => inline_value orelse blk: { - i += 1; - if (i >= args.len) return usageError("{s} needs a value", .{name}); - break :blk args[i]; - }, - }; - switch (opt) { - .unix => { - if (value.len == 0) return usageError("--unix wants a socket path", .{}); - cfg.address = .{ .unix = value }; - transports += 1; - }, - .tcp => { - cfg.address = parseTcp(value) orelse return usageError("--tcp wants IP:PORT (IPv6 as [ADDR]:PORT), got '{s}'", .{value}); - transports += 1; - }, - .fd => { - const n = std.fmt.parseInt(i32, value, 10) catch return usageError("--fd wants a number, got '{s}'", .{value}); - if (n < 0) return usageError("--fd wants a non-negative number", .{}); - cfg.address = .{ .fd = n }; - transports += 1; - }, - .spawn => { - if (value.len == 0) return usageError("--spawn wants a command", .{}); - cfg.spawn_cmd = value; - transports += 1; - }, - .mount => { - if (value.len == 0) return usageError("--mount wants a path", .{}); - cfg.mount = value; - }, - .uname => cfg.uname = value, - .aname => cfg.aname = value, - .msize => { - cfg.msize = std.fmt.parseInt(u32, value, 10) catch return usageError("--msize wants a number, got '{s}'", .{value}); - if (cfg.msize < 4096 or cfg.msize > max_msize) return usageError("--msize must be between 4096 and {d}", .{max_msize}); - }, - .cache => { - const secs = std.fmt.parseFloat(f64, value) catch return usageError("--cache wants seconds, got '{s}'", .{value}); - if (!(secs >= 0) or secs > 1e9) return usageError("--cache out of range", .{}); - cfg.cache_ns = @intFromFloat(secs * 1e9); - }, - .@"no-direct-io" => cfg.direct_io = false, - .debug => cfg.debug = true, - .help => return .{ .info = usage_text }, - .version => return .{ .info = version_string ++ "\n" }, - .unknown => unreachable, - } - } - if (transports == 0) return usageError("one transport is required (--unix, --tcp, --fd or --spawn)", .{}); - if (transports > 1) return usageError("exactly one transport is allowed", .{}); - if (program_start) |start| { - const prog = try arena.alloc([]const u8, args.len - start); - for (args[start..], 0..) |a, j| prog[j] = a; - cfg.program = prog; - } - return .{ .run = cfg }; -} - -fn parseTcp(spec: []const u8) ?nine.Address { - const colon = std.mem.lastIndexOfScalar(u8, spec, ':') orelse return null; - var host = spec[0..colon]; - if (host.len >= 2 and host[0] == '[' and host[host.len - 1] == ']') host = host[1 .. host.len - 1]; - if (host.len == 0) return null; - const port = std.fmt.parseInt(u16, spec[colon + 1 ..], 10) catch return null; - return .{ .tcp = .{ .host = host, .port = port } }; -} - -/// `--spawn`: run CMD under /bin/sh with one end of a socketpair as its -/// stdin/stdout; the other end is the 9P transport. -const Server = struct { pid: i32, fd: i32 }; - -fn spawnServer(cmd: [:0]const u8, envp: [*:null]const ?[*:0]const u8) !Server { - var sv: [2]i32 = undefined; - switch (linux.errno(linux.socketpair(linux.AF.UNIX, linux.SOCK.STREAM | linux.SOCK.CLOEXEC, 0, &sv))) { - .SUCCESS => {}, - else => |e| { - std.debug.print("9player: socketpair: E{t}\n", .{e}); - return error.SystemResources; - }, - } - const rc = linux.fork(); - switch (linux.errno(rc)) { - .SUCCESS => {}, - else => |e| { - _ = linux.close(sv[0]); - _ = linux.close(sv[1]); - std.debug.print("9player: fork: E{t}\n", .{e}); - return error.SystemResources; - }, - } - if (rc == 0) { - // Child: dup2 clears CLOEXEC on 0 and 1; everything else is CLOEXEC. - if (linux.errno(linux.dup2(sv[1], 0)) != .SUCCESS or linux.errno(linux.dup2(sv[1], 1)) != .SUCCESS) linux.exit_group(125); - // The server shares our process group, so a Ctrl-C meant for the - // program would kill it and take the mount down with it: ignore the - // tty signals (inherited across exec). SIGPIPE goes back to its - // default, we only ignore it for ourselves. - ignoreSignal(.INT); - ignoreSignal(.QUIT); - defaultSignal(.PIPE); - const argv = [_:null]?[*:0]const u8{ "sh", "-c", cmd.ptr }; - const e = linux.errno(linux.execve("/bin/sh", &argv, envp)); - std.debug.print("9player: --spawn: exec /bin/sh: E{t}\n", .{e}); - linux.exit_group(127); - } - _ = linux.close(sv[1]); - return .{ .pid = @intCast(rc), .fd = sv[0] }; -} - -fn stopServer(server: ?Server) void { - const s = server orelse return; - _ = linux.kill(s.pid, .TERM); - ns.reapAny(s.pid); -} - -/// Fail early (before spawning servers or forking) if /dev/fuse is unusable. -fn probeFuseDevice() bool { - const rc = linux.open("/dev/fuse", .{ .ACCMODE = .RDWR, .CLOEXEC = true }, 0); - switch (linux.errno(rc)) { - .SUCCESS => { - _ = linux.close(@intCast(rc)); - return true; - }, - .NOENT => std.debug.print("9player: /dev/fuse: ENOENT (is the fuse module loaded? try: modprobe fuse)\n", .{}), - else => |e| std.debug.print("9player: open /dev/fuse: E{t}\n", .{e}), - } - return false; -} - -fn ignoreSignal(sig: linux.SIG) void { - const ign = linux.Sigaction{ .handler = .{ .handler = linux.SIG.IGN }, .mask = linux.sigemptyset(), .flags = 0 }; - std.posix.sigaction(sig, &ign, null); -} - -fn defaultSignal(sig: linux.SIG) void { - const dfl = linux.Sigaction{ .handler = .{ .handler = linux.SIG.DFL }, .mask = linux.sigemptyset(), .flags = 0 }; - std.posix.sigaction(sig, &dfl, null); -} - -/// `--fd N`: the descriptor is ours from now on; it must not leak into the -/// program (which could otherwise read 9P replies meant for us). Fails on a -/// bad descriptor, which is the earliest place to report it. -fn adoptFd(fd: i32) bool { - switch (linux.errno(linux.fcntl(fd, linux.F.SETFD, linux.FD_CLOEXEC))) { - .SUCCESS => return true, - else => |e| { - std.debug.print("9player: --fd {d}: E{t}\n", .{ fd, e }); - return false; - }, - } -} - -fn describeAddress(a: nine.Address, buf: []u8) []const u8 { - return switch (a) { - .unix => |p| std.fmt.bufPrint(buf, "unix socket {s}", .{p}) catch "unix socket", - .tcp => |t| std.fmt.bufPrint(buf, "tcp {s}:{d}", .{ t.host, t.port }) catch "tcp", - .fd => |fd| std.fmt.bufPrint(buf, "fd {d}", .{fd}) catch "fd", - }; -} - -pub fn main(init: std.process.Init) !u8 { - const gpa = init.gpa; - const arena = init.arena.allocator(); - const args = try init.minimal.args.toSlice(arena); - const envp: [*:null]const ?[*:0]const u8 = init.minimal.environ.block.slice.ptr; - - var cfg = switch (try parseArgs(arena, args)) { - .exit => |code| return code, - .info => |text| { - printStdout(text); - return 0; - }, - .run => |c| c, - }; - - // Defaults that come from the environment. - if (cfg.program.len == 0) { - const env_shell = ns.getenv(envp, "SHELL") orelse ""; - const shell = if (env_shell.len == 0) "/bin/sh" else env_shell; - cfg.program = try arena.dupe([]const u8, &.{shell}); - } - const uname = cfg.uname orelse ns.getenv(envp, "USER") orelse "none"; - const mountpoint = ns.resolveMountpoint(gpa, cfg.mount) catch |err| { - std.debug.print("9player: --mount {s}: {t}\n", .{ cfg.mount, err }); - return own_failure; - }; - defer gpa.free(mountpoint); - - if (!probeFuseDevice()) return own_failure; - - // Writes to a dead server socket must not kill us. - ignoreSignal(.PIPE); - - var server: ?Server = null; - var address: nine.Address = undefined; - if (cfg.spawn_cmd) |cmd| { - const cmd_z = try arena.dupeZ(u8, cmd); - server = spawnServer(cmd_z, envp) catch return own_failure; - ns.watchServer(server.?.pid); - address = .{ .fd = server.?.fd }; - } else { - address = cfg.address.?; - if (address == .fd and !adoptFd(address.fd)) return own_failure; - } - - var addr_buf: [256]u8 = undefined; - var session = nine.Session.connect(gpa, address, cfg.msize) catch |err| { - std.debug.print("9player: connect to {s}: {t}\n", .{ describeAddress(address, &addr_buf), err }); - stopServer(server); - return own_failure; - }; - defer session.deinit(); - defer stopServer(server); - - _ = session.attach(0, uname, cfg.aname) catch |err| { - switch (err) { - error.Nine => std.debug.print("9player: attach (uname={s}, aname='{s}'): {s}\n", .{ uname, cfg.aname, session.ename[0..session.ename_len] }), - else => std.debug.print("9player: attach: {t}\n", .{err}), - } - return own_failure; - }; - if (cfg.debug) std.debug.print("9player: attached to {s} (msize {d}), mounting on {s}\n", .{ describeAddress(address, &addr_buf), session.msize, mountpoint }); - - var child_pid: i32 = 0; - const stop_fd = ns.installSignals(&child_pid) catch return own_failure; - - const uid = linux.getuid(); - const gid = linux.getgid(); - const child = ns.spawn(gpa, .{ - .argv = cfg.program, - .envp = envp, - .mountpoint = mountpoint, - .uid = uid, - .gid = gid, - .max_read = bridge.max_write, - }) catch return own_failure; - - bridge.serve(gpa, child.fuse_fd, &session, 0, stop_fd, .{ - .uid = uid, - .gid = gid, - .attr_timeout_ns = cfg.cache_ns, - .direct_io = cfg.direct_io, - .debug = cfg.debug, - }) catch |err| switch (err) { - error.Closed => std.debug.print("9player: 9P server connection closed\n", .{}), - else => std.debug.print("9player: fuse: {t}\n", .{err}), - }; - - // Closing the device aborts the FUSE connection: anything still using - // the mount gets ENOTCONN instead of hanging on an unserved request. - _ = linux.close(child.fuse_fd); - - const status = ns.reapIfExited(child.pid) orelse ns.waitChild(child.pid) catch own_failure; - // An exec failure (126/127) is already in `status`; this prints its message. - _ = ns.reportExecFailure(child); - return status; -} - -test "parseTcp" { - const a = parseTcp("127.0.0.1:564").?; - try std.testing.expectEqualStrings("127.0.0.1", a.tcp.host); - try std.testing.expectEqual(@as(u16, 564), a.tcp.port); - const b = parseTcp("[::1]:9999").?; - try std.testing.expectEqualStrings("::1", b.tcp.host); - try std.testing.expectEqual(@as(u16, 9999), b.tcp.port); - try std.testing.expect(parseTcp("nohost") == null); - try std.testing.expect(parseTcp(":564") == null); - try std.testing.expect(parseTcp("1.2.3.4:") == null); - try std.testing.expect(parseTcp("1.2.3.4:70000") == null); -} - -test "parseArgs" { - const arena = std.testing.allocator; - { - const args = [_][:0]const u8{ "9player", "--unix", "/s", "--cache", "0.5", "--msize=8192", "--no-direct-io", "--", "sh", "-c", "x" }; - const r = try parseArgs(arena, &args); - defer arena.free(r.run.program); - try std.testing.expectEqualStrings("/s", r.run.address.?.unix); - try std.testing.expectEqual(@as(u64, 500_000_000), r.run.cache_ns); - try std.testing.expectEqual(@as(u32, 8192), r.run.msize); - try std.testing.expect(!r.run.direct_io); - try std.testing.expectEqual(@as(usize, 3), r.run.program.len); - try std.testing.expectEqualStrings("x", r.run.program[2]); - } - { - const args = [_][:0]const u8{ "9player", "--fd", "3" }; - const r = try parseArgs(arena, &args); - try std.testing.expectEqual(@as(i32, 3), r.run.address.?.fd); - try std.testing.expectEqual(@as(usize, 0), r.run.program.len); - try std.testing.expectEqualStrings("/mnt/9p", r.run.mount); - } - { - // Two transports, no transport, unknown option, missing value: all 125. - const two = [_][:0]const u8{ "9player", "--fd", "3", "--unix", "/s" }; - try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &two)).exit); - const none = [_][:0]const u8{ "9player", "--", "sh" }; - try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &none)).exit); - const unknown = [_][:0]const u8{ "9player", "--bogus" }; - try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &unknown)).exit); - const missing = [_][:0]const u8{ "9player", "--unix" }; - try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &missing)).exit); - const badcache = [_][:0]const u8{ "9player", "--fd", "3", "--cache", "abc" }; - try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &badcache)).exit); - // Empty values, a single-dash typo, and an msize that would allocate gigabytes. - const emptyunix = [_][:0]const u8{ "9player", "--unix=", "--", "sh" }; - try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &emptyunix)).exit); - const emptymount = [_][:0]const u8{ "9player", "--fd", "3", "--mount", "" }; - try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &emptymount)).exit); - const singledash = [_][:0]const u8{ "9player", "--fd", "3", "-mount", "/x" }; - try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &singledash)).exit); - const hugemsize = [_][:0]const u8{ "9player", "--fd", "3", "--msize", "4294967295" }; - try std.testing.expectEqual(@as(u8, 125), (try parseArgs(arena, &hugemsize)).exit); - const okmsize = [_][:0]const u8{ "9player", "--fd", "3", "--msize", "16777216" }; - try std.testing.expectEqual(@as(u32, 16777216), (try parseArgs(arena, &okmsize)).run.msize); - } - { - const ver = [_][:0]const u8{ "9player", "--version" }; - try std.testing.expectEqualStrings(version_string ++ "\n", (try parseArgs(arena, &ver)).info); - const help = [_][:0]const u8{ "9player", "--help" }; - try std.testing.expect(std.mem.startsWith(u8, (try parseArgs(arena, &help)).info, "Usage: 9player")); - } -} - -test { - _ = ns; -} diff --git a/9player/src/nine.zig b/9player/src/nine.zig deleted file mode 100644 index 70633e6..0000000 --- a/9player/src/nine.zig +++ /dev/null @@ -1,756 +0,0 @@ -//! Synchronous 9P2000 session over a blocking file descriptor. -//! -//! A thin RPC layer over `cloud9.Client` (push/take, allocation-free). One request -//! is outstanding at a time: the FUSE loop that drives this is single-threaded, so -//! every call here blocks until its reply (or the connection's death) arrives. -//! Fids are handed out from a free list; fid 0 is reserved for the root. -const std = @import("std"); -const cloud9 = @import("cloud9"); -const linux = std.os.linux; - -pub const Address = union(enum) { - unix: []const u8, - tcp: struct { host: []const u8, port: u16 }, - fd: i32, -}; - -/// A Stat whose every field means "leave unchanged" in a Twstat. -pub const dontcare = cloud9.Stat{ - .type = 0xFFFF, - .dev = 0xFFFF_FFFF, - .qid = .{ .type = 0xFF, .version = 0xFFFF_FFFF, .path = 0xFFFF_FFFF_FFFF_FFFF }, - .mode = 0xFFFF_FFFF, - .atime = 0xFFFF_FFFF, - .mtime = 0xFFFF_FFFF, - .length = 0xFFFF_FFFF_FFFF_FFFF, - .name = "", - .uid = "", - .gid = "", - .muid = "", -}; - -pub const Session = struct { - pub const Error = error{ Nine, Protocol, Io, Closed, Stopped, TooLarge, OutOfMemory }; - - pub const Walk = struct { nwqid: u16, wqid: [cloud9.max_welem]cloud9.Qid }; - pub const Open = struct { qid: cloud9.Qid, iounit: u32 }; - - gpa: std.mem.Allocator, - fd: i32, - client: cloud9.Client, - in_buf: []u8, - out_buf: []u8, - /// After `error.Nine`, the server's Rerror text (copied, bounded). - ename: [256]u8 = undefined, - ename_len: usize = 0, - /// Negotiated maximum message size. - msize: u32, - next_fid: u32 = 1, - free_fids: std.ArrayList(u32) = .empty, - /// Per-fid iounit learned from open/create (0 = none); used to chunk read/write. - iounits: std.AutoHashMapUnmanaged(u32, u32) = .empty, - /// Optional descriptor watched while waiting for a reply: when it becomes - /// readable (the bridge's "child exited" pipe) the pending rpc fails with - /// `error.Stopped` instead of blocking on a server that never answers. - stop_fd: i32 = -1, - - /// Connect to `address`, then negotiate the protocol version. - /// `msize` is the maximum message size to ask for (0 = the buffers' size). - pub fn connect(gpa: std.mem.Allocator, address: Address, msize: u32) !Session { - const want: u32 = if (msize == 0) 8192 else @max(msize, 24); - const fd = try openTransport(address); - errdefer if (address != .fd) { - _ = linux.close(fd); - }; - - const in_buf = try gpa.alloc(u8, want); - errdefer gpa.free(in_buf); - const out_buf = try gpa.alloc(u8, want); - errdefer gpa.free(out_buf); - - var s: Session = .{ - .gpa = gpa, - .fd = fd, - .client = .init(.{ .in = in_buf, .out = out_buf }), - .in_buf = in_buf, - .out_buf = out_buf, - .msize = want, - }; - const r = try s.rpc(.{ .version = .{ .msize = want } }); - if (!std.mem.eql(u8, r.version.version, "9P2000")) return error.Protocol; - s.msize = r.version.msize; - return s; - } - - /// Closes the descriptor and frees the buffers. Fids are not clunked. - pub fn deinit(s: *Session) void { - _ = linux.close(s.fd); - s.free_fids.deinit(s.gpa); - s.iounits.deinit(s.gpa); - s.gpa.free(s.in_buf); - s.gpa.free(s.out_buf); - s.* = undefined; - } - - pub fn attach(s: *Session, fid: u32, uname: []const u8, aname: []const u8) Error!cloud9.Qid { - const r = try s.rpc(.{ .attach = .{ .fid = fid, .uname = uname, .aname = aname } }); - return r.attach; - } - - /// Fid 0 is never handed out: it belongs to the root attach. - pub fn allocFid(s: *Session) u32 { - if (s.free_fids.pop()) |fid| return fid; - const fid = s.next_fid; - s.next_fid += 1; - return fid; - } - - /// Fids currently bound (excluding fid 0); a debugging aid for leak hunting. - pub fn fidsInUse(s: *const Session) usize { - return (s.next_fid - 1) - s.free_fids.items.len; - } - - pub fn freeFid(s: *Session, fid: u32) void { - _ = s.iounits.remove(fid); - // If the free list cannot grow the fid is simply leaked; the counter keeps going. - s.free_fids.append(s.gpa, fid) catch {}; - } - - /// Generic RPC. Result slices borrow the input buffer until the next call. - pub fn rpc(s: *Session, req: cloud9.Client.Request) Error!cloud9.Client.Result { - s.ename_len = 0; - _ = s.client.submit(req) catch |e| switch (e) { - error.NoTags, error.Handshake, error.Dead => return error.Protocol, - error.NoSpace, error.TooLarge => return error.TooLarge, - error.BadRequest => { - s.setEname("bad request"); - return error.Nine; - }, - }; - try s.flush(); - var tmp: [64 * 1024]u8 = undefined; - while (true) { - if (s.client.take()) |done| { - switch (done.result) { - .fail => |ename| { - s.setEname(ename); - return error.Nine; - }, - else => return done.result, - } - } - if (s.client.dead) return error.Protocol; - // After take() returned null the previous frame is gone, so the free - // space is at least what the pending frame still needs. - const room = s.client.in.len - s.client.in_len; - if (room == 0) return error.Protocol; - const n = try readSome(s.fd, s.stop_fd, tmp[0..@min(room, tmp.len)]); - if (n == 0) return error.Closed; - const pushed = s.client.push(tmp[0..n]); - if (pushed != n) return error.Protocol; - } - } - - /// Walk `names` from `fid` to `newfid`. A partial walk leaves `newfid` unbound - /// (9P semantics) and reports `error.Nine` with ename "file does not exist". - pub fn walk(s: *Session, fid: u32, newfid: u32, names: []const []const u8) Error!Walk { - const r = try s.rpc(.{ .walk = .{ .fid = fid, .newfid = newfid, .names = names } }); - if (r.walk.nwqid < names.len) { - s.setEname("file does not exist"); - return error.Nine; - } - return .{ .nwqid = r.walk.nwqid, .wqid = r.walk.wqid }; - } - - /// allocFid + zero-element walk. The fid is released again on failure. - pub fn clone(s: *Session, fid: u32) Error!u32 { - const newfid = s.allocFid(); - errdefer s.freeFid(newfid); - _ = try s.walk(fid, newfid, &.{}); - return newfid; - } - - pub fn open(s: *Session, fid: u32, mode: u8) Error!Open { - const r = try s.rpc(.{ .open = .{ .fid = fid, .mode = mode } }); - s.noteIounit(fid, r.open.iounit); - return .{ .qid = r.open.qid, .iounit = r.open.iounit }; - } - - pub fn create(s: *Session, fid: u32, name: []const u8, perm: u32, mode: u8) Error!Open { - const r = try s.rpc(.{ .create = .{ .fid = fid, .name = name, .perm = perm, .mode = mode } }); - s.noteIounit(fid, r.create.iounit); - return .{ .qid = r.create.qid, .iounit = r.create.iounit }; - } - - /// Reads into `buf`, chunking by min(maxRead, iounit) and stopping at the first - /// short read. Returns the number of bytes read (0 at end of file). - pub fn read(s: *Session, fid: u32, offset: u64, buf: []u8) Error!usize { - return readWith(s, rpc, fid, offset, buf, s.chunk(fid)); - } - - /// Writes `data`, chunking like `read` and stopping at the first short write. - pub fn write(s: *Session, fid: u32, offset: u64, data: []const u8) Error!usize { - return writeWith(s, rpc, fid, offset, data, s.chunkWrite(fid)); - } - - /// The returned Stat's strings (name/uid/gid/muid) borrow the session's input - /// buffer: they are valid only until the next rpc. Copy what must outlive it. - pub fn stat(s: *Session, fid: u32) Error!cloud9.Stat { - const r = try s.rpc(.{ .stat = .{ .fid = fid } }); - return r.stat; - } - - pub fn wstat(s: *Session, fid: u32, st: cloud9.Stat) Error!void { - _ = try s.rpc(.{ .wstat = .{ .fid = fid, .stat = st } }); - } - - /// Frees the fid locally even when the server reports an error. - pub fn clunk(s: *Session, fid: u32) Error!void { - defer s.freeFid(fid); - _ = try s.rpc(.{ .clunk = .{ .fid = fid } }); - } - - /// Frees the fid locally even when the server reports an error. - pub fn remove(s: *Session, fid: u32) Error!void { - defer s.freeFid(fid); - _ = try s.rpc(.{ .remove = .{ .fid = fid } }); - } - - /// Maps the last Rerror text to an errno (case-insensitive substring match). - pub fn errno(s: *const Session) linux.E { - return enameToErrno(s.ename[0..s.ename_len]); - } - - // -- internals -------------------------------------------------------------- - - fn setEname(s: *Session, text: []const u8) void { - const n = @min(text.len, 255); - @memcpy(s.ename[0..n], text[0..n]); - s.ename_len = n; - } - - fn noteIounit(s: *Session, fid: u32, iounit: u32) void { - if (iounit == 0) { - _ = s.iounits.remove(fid); - } else { - s.iounits.put(s.gpa, fid, iounit) catch {}; - } - } - - fn chunk(s: *Session, fid: u32) u32 { - return chunkSize(s.client.maxRead(), s.iounits.get(fid) orelse 0); - } - - fn chunkWrite(s: *Session, fid: u32) u32 { - return chunkSize(s.client.maxWrite(), s.iounits.get(fid) orelse 0); - } - - /// Writes everything in the client's output buffer to the socket. - fn flush(s: *Session) Error!void { - while (s.client.output().len != 0) { - const out = s.client.output(); - const rc = linux.write(s.fd, out.ptr, out.len); - switch (linux.errno(rc)) { - .SUCCESS => { - if (rc == 0) return error.Closed; - s.client.wrote(rc); - }, - .INTR, .AGAIN => continue, - .PIPE, .CONNRESET => return error.Closed, - else => return error.Io, - } - } - } -}; - -fn chunkSize(max: u32, iounit: u32) u32 { - if (iounit != 0 and iounit < max) return iounit; - return max; -} - -/// Chunked read over any rpc-shaped function (injected so the loop is testable). -fn readWith( - s: anytype, - comptime rpcFn: anytype, - fid: u32, - offset: u64, - buf: []u8, - max_chunk: u32, -) Session.Error!usize { - if (max_chunk == 0) return error.Protocol; - var done: usize = 0; - while (done < buf.len) { - const want: u32 = @intCast(@min(buf.len - done, max_chunk)); - const r = try rpcFn(s, .{ .read = .{ .fid = fid, .offset = offset + done, .count = want } }); - const data = r.read; - @memcpy(buf[done..][0..data.len], data); - done += data.len; - if (data.len < want) break; - } - return done; -} - -/// Chunked write over any rpc-shaped function. -fn writeWith( - s: anytype, - comptime rpcFn: anytype, - fid: u32, - offset: u64, - data: []const u8, - max_chunk: u32, -) Session.Error!usize { - if (max_chunk == 0) return error.Protocol; - var done: usize = 0; - while (done < data.len) { - const want: usize = @min(data.len - done, max_chunk); - const r = try rpcFn(s, .{ .write = .{ .fid = fid, .offset = offset + done, .data = data[done..][0..want] } }); - done += r.write; - if (r.write < want) break; - } - return done; -} - -fn readSome(fd: i32, stop_fd: i32, buf: []u8) Session.Error!usize { - while (true) { - if (stop_fd >= 0) { - var pfds = [_]linux.pollfd{ - .{ .fd = fd, .events = linux.POLL.IN, .revents = 0 }, - .{ .fd = stop_fd, .events = linux.POLL.IN, .revents = 0 }, - }; - const prc = linux.poll(&pfds, pfds.len, -1); - switch (linux.errno(prc)) { - .SUCCESS => {}, - .INTR, .AGAIN => continue, - else => return error.Io, - } - if (pfds[1].revents != 0 and pfds[0].revents == 0) return error.Stopped; - } - const rc = linux.read(fd, buf.ptr, buf.len); - switch (linux.errno(rc)) { - .SUCCESS => return rc, - .INTR, .AGAIN => continue, - .CONNRESET => return error.Closed, - else => return error.Io, - } - } -} - -/// Rerror text → errno, per docs/DESIGN.md (first match wins). -pub fn enameToErrno(ename: []const u8) linux.E { - const Rule = struct { needle: []const u8, err: linux.E }; - const rules = [_]Rule{ - .{ .needle = "not exist", .err = .NOENT }, - .{ .needle = "not found", .err = .NOENT }, - .{ .needle = "no such", .err = .NOENT }, - .{ .needle = "exists", .err = .EXIST }, - .{ .needle = "not empty", .err = .NOTEMPTY }, - .{ .needle = "not a dir", .err = .NOTDIR }, - .{ .needle = "is a dir", .err = .ISDIR }, - .{ .needle = "permission", .err = .ACCES }, - .{ .needle = "denied", .err = .ACCES }, - .{ .needle = "read-only", .err = .ROFS }, - .{ .needle = "read only", .err = .ROFS }, - .{ .needle = "readonly", .err = .ROFS }, - .{ .needle = "no space", .err = .NOSPC }, - .{ .needle = "not allowed", .err = .PERM }, - .{ .needle = "not permitted", .err = .PERM }, - .{ .needle = "cannot", .err = .PERM }, - .{ .needle = "fid", .err = .BADF }, - .{ .needle = "bad offset", .err = .INVAL }, - .{ .needle = "invalid", .err = .INVAL }, - .{ .needle = "bad ", .err = .INVAL }, - .{ .needle = "busy", .err = .BUSY }, - .{ .needle = "in use", .err = .BUSY }, - .{ .needle = "too long", .err = .NAMETOOLONG }, - .{ .needle = "not supported", .err = .OPNOTSUPP }, - .{ .needle = "unsupported", .err = .OPNOTSUPP }, - }; - for (rules) |rule| { - if (std.ascii.findIgnoreCase(ename, rule.needle) != null) return rule.err; - } - return .IO; -} - -// -- transport ------------------------------------------------------------------ - -fn openTransport(address: Address) !i32 { - switch (address) { - .fd => |fd| return fd, - .unix => |path| { - if (path.len == 0 or path.len >= 108) return error.NameTooLong; - var sa: linux.sockaddr.un = .{ .path = @splat(0) }; - @memcpy(sa.path[0..path.len], path); - const fd = try newSocket(linux.AF.UNIX, 0); - errdefer _ = linux.close(fd); - try doConnect(fd, @ptrCast(&sa), @sizeOf(linux.sockaddr.un)); - return fd; - }, - .tcp => |t| { - const ip = std.Io.net.IpAddress.parse(t.host, t.port) catch return error.InvalidAddress; - switch (ip) { - .ip4 => |a| { - const sa: linux.sockaddr.in = .{ - .port = std.mem.nativeToBig(u16, t.port), - .addr = @bitCast(a.bytes), - }; - const fd = try newSocket(linux.AF.INET, linux.IPPROTO.TCP); - errdefer _ = linux.close(fd); - setNodelay(fd); - try doConnect(fd, @ptrCast(&sa), @sizeOf(linux.sockaddr.in)); - return fd; - }, - .ip6 => |a| { - const sa: linux.sockaddr.in6 = .{ - .port = std.mem.nativeToBig(u16, t.port), - .flowinfo = 0, - .addr = a.bytes, - .scope_id = 0, - }; - const fd = try newSocket(linux.AF.INET6, linux.IPPROTO.TCP); - errdefer _ = linux.close(fd); - setNodelay(fd); - try doConnect(fd, @ptrCast(&sa), @sizeOf(linux.sockaddr.in6)); - return fd; - }, - } - }, - } -} - -fn newSocket(domain: u32, protocol: u32) !i32 { - const rc = linux.socket(domain, linux.SOCK.STREAM | linux.SOCK.CLOEXEC, protocol); - switch (linux.errno(rc)) { - .SUCCESS => return @intCast(rc), - .MFILE, .NFILE => return error.ProcessFdQuotaExceeded, - .AFNOSUPPORT, .PROTONOSUPPORT => return error.AddressFamilyNotSupported, - .ACCES => return error.AccessDenied, - .NOMEM, .NOBUFS => return error.SystemResources, - else => return error.Unexpected, - } -} - -fn setNodelay(fd: i32) void { - const one: u32 = 1; - _ = linux.setsockopt(fd, linux.IPPROTO.TCP, linux.TCP.NODELAY, @ptrCast(&one), @sizeOf(u32)); -} - -fn doConnect(fd: i32, addr: *const linux.sockaddr, len: linux.socklen_t) !void { - while (true) { - const rc = linux.connect(fd, addr, len); - switch (linux.errno(rc)) { - .SUCCESS => return, - .INTR => continue, - .CONNREFUSED => return error.ConnectionRefused, - .NOENT, .NOTDIR => return error.FileNotFound, - .ACCES, .PERM => return error.AccessDenied, - .TIMEDOUT => return error.ConnectionTimedOut, - .NETUNREACH, .HOSTUNREACH => return error.NetworkUnreachable, - .ADDRNOTAVAIL => return error.AddressNotAvailable, - .AGAIN, .INPROGRESS => return error.WouldBlock, - else => return error.Unexpected, - } - } -} - -// -- tests ---------------------------------------------------------------------- - -const testing = std.testing; - -test { - testing.refAllDecls(@This()); -} - -test "ename → errno mapping" { - try testing.expectEqual(linux.E.NOENT, enameToErrno("file does not exist")); - try testing.expectEqual(linux.E.NOENT, enameToErrno("No Such File")); - try testing.expectEqual(linux.E.NOENT, enameToErrno("directory entry not found")); - try testing.expectEqual(linux.E.EXIST, enameToErrno("file already exists")); - try testing.expectEqual(linux.E.NOTEMPTY, enameToErrno("directory not empty")); - try testing.expectEqual(linux.E.NOTDIR, enameToErrno("not a directory")); - try testing.expectEqual(linux.E.ISDIR, enameToErrno("is a directory")); - try testing.expectEqual(linux.E.ACCES, enameToErrno("permission denied")); - try testing.expectEqual(linux.E.ACCES, enameToErrno("access denied")); - try testing.expectEqual(linux.E.ROFS, enameToErrno("read-only file system")); - try testing.expectEqual(linux.E.NOSPC, enameToErrno("no space left")); - try testing.expectEqual(linux.E.PERM, enameToErrno("operation not permitted")); - try testing.expectEqual(linux.E.PERM, enameToErrno("cannot remove root")); - try testing.expectEqual(linux.E.BADF, enameToErrno("unknown fid")); - try testing.expectEqual(linux.E.BADF, enameToErrno("fid in use")); // "fid" precedes "in use" - try testing.expectEqual(linux.E.INVAL, enameToErrno("bad offset")); - try testing.expectEqual(linux.E.INVAL, enameToErrno("invalid argument")); - try testing.expectEqual(linux.E.INVAL, enameToErrno("bad request")); - try testing.expectEqual(linux.E.BUSY, enameToErrno("device busy")); - try testing.expectEqual(linux.E.NAMETOOLONG, enameToErrno("name too long")); - try testing.expectEqual(linux.E.OPNOTSUPP, enameToErrno("operation not supported")); - try testing.expectEqual(linux.E.IO, enameToErrno("something odd happened")); - try testing.expectEqual(linux.E.IO, enameToErrno("")); -} - -test "fid allocator recycles and never hands out 0" { - var s: Session = undefined; - s.gpa = testing.allocator; - s.next_fid = 1; - s.free_fids = .empty; - s.iounits = .empty; - defer s.free_fids.deinit(s.gpa); - defer s.iounits.deinit(s.gpa); - - const a = s.allocFid(); - const b = s.allocFid(); - const c = s.allocFid(); - try testing.expectEqual(@as(u32, 1), a); - try testing.expectEqual(@as(u32, 2), b); - try testing.expectEqual(@as(u32, 3), c); - s.freeFid(b); - try testing.expectEqual(b, s.allocFid()); - s.freeFid(a); - s.freeFid(c); - const x = s.allocFid(); - const y = s.allocFid(); - try testing.expect((x == a and y == c) or (x == c and y == a)); - try testing.expectEqual(@as(u32, 4), s.allocFid()); - try testing.expect(a != 0 and b != 0 and c != 0); -} - -test "chunkSize honours iounit only when smaller" { - try testing.expectEqual(@as(u32, 100), chunkSize(100, 0)); - try testing.expectEqual(@as(u32, 40), chunkSize(100, 40)); - try testing.expectEqual(@as(u32, 100), chunkSize(100, 400)); -} - -/// Fake rpc for the chunked read/write loops: a file of `len` bytes where byte i == i & 0xff. -const FakeFile = struct { - len: usize, - calls: usize = 0, - max_count: u32 = 0, - short_write_at: ?usize = null, - scratch: [4096]u8 = undefined, - - fn rpc(f: *FakeFile, req: cloud9.Client.Request) Session.Error!cloud9.Client.Result { - f.calls += 1; - switch (req) { - .read => |r| { - f.max_count = @max(f.max_count, r.count); - if (r.offset >= f.len) return .{ .read = "" }; - const n: usize = @min(@as(usize, r.count), f.len - @as(usize, @intCast(r.offset))); - for (f.scratch[0..n], 0..) |*b, i| b.* = @truncate(r.offset + i); - return .{ .read = f.scratch[0..n] }; - }, - .write => |w| { - f.max_count = @max(f.max_count, @as(u32, @intCast(w.data.len))); - if (f.short_write_at) |at| { - if (w.offset + w.data.len > at) { - const n: usize = if (w.offset >= at) 0 else @intCast(at - w.offset); - return .{ .write = @intCast(n) }; - } - } - return .{ .write = @intCast(w.data.len) }; - }, - else => unreachable, - } - } -}; - -test "read chunks by max_chunk and stops at a short read" { - var f: FakeFile = .{ .len = 2500 }; - var buf: [4000]u8 = undefined; - const n = try readWith(&f, FakeFile.rpc, 7, 0, &buf, 1000); - try testing.expectEqual(@as(usize, 2500), n); - try testing.expectEqual(@as(usize, 3), f.calls); // 1000, 1000, 500 (short → stop) - try testing.expectEqual(@as(u32, 1000), f.max_count); - for (buf[0..n], 0..) |b, i| try testing.expectEqual(@as(u8, @truncate(i)), b); - - // Reading exactly up to a chunk boundary uses one call per chunk and no more. - f = .{ .len = 2000 }; - try testing.expectEqual(@as(usize, 2000), try readWith(&f, FakeFile.rpc, 7, 0, buf[0..2000], 1000)); - try testing.expectEqual(@as(usize, 2), f.calls); - - // Offset past EOF → 0. - f = .{ .len = 10 }; - try testing.expectEqual(@as(usize, 0), try readWith(&f, FakeFile.rpc, 7, 50, &buf, 1000)); -} - -test "write chunks and stops at a short write" { - var f: FakeFile = .{ .len = 0 }; - var data: [2500]u8 = undefined; - for (&data, 0..) |*b, i| b.* = @truncate(i); - try testing.expectEqual(@as(usize, 2500), try writeWith(&f, FakeFile.rpc, 7, 0, &data, 1000)); - try testing.expectEqual(@as(usize, 3), f.calls); - try testing.expectEqual(@as(u32, 1000), f.max_count); - - f = .{ .len = 0, .short_write_at = 1500 }; - try testing.expectEqual(@as(usize, 1500), try writeWith(&f, FakeFile.rpc, 7, 0, &data, 1000)); - try testing.expectEqual(@as(usize, 2), f.calls); -} - -// -- in-process server test --------------------------------------------------------- - -/// A tiny 9P2000 backend on a cloud9.Server: answers version/attach/walk/stat/open/ -/// read/clunk/remove with canned data. Runs in its own thread over a socketpair. -const FakeServer = struct { - fd: i32, - msize: u32, - max_read_count: u32 = 0, - file_len: usize, - - const file_qid: cloud9.Qid = .{ .type = 0, .version = 3, .path = 0x1234 }; - const dir_qid: cloud9.Qid = .{ .type = cloud9.qtdir, .version = 1, .path = 0x1 }; - - fn run(fs: *FakeServer) void { - fs.loop() catch |e| std.debug.print("fake server: {s}\n", .{@errorName(e)}); - _ = linux.close(fs.fd); - } - - fn loop(fs: *FakeServer) !void { - const gpa = testing.allocator; - const in = try gpa.alloc(u8, fs.msize); - defer gpa.free(in); - const out = try gpa.alloc(u8, fs.msize * 2); - defer gpa.free(out); - var srv: cloud9.Server = .init(.{ .in = in, .out = out }); - var tmp: [4096]u8 = undefined; - var data: [8192]u8 = undefined; - while (true) { - while (try srv.receive()) |req| { - const tag = req.tag; - switch (req.msg) { - .tversion => |m| try srv.negotiate(m.msize, m.version), - .tattach => try srv.reply(tag, .{ .rattach = .{ .qid = dir_qid } }), - .twalk => |m| { - var wq: [cloud9.max_welem]cloud9.Qid = @splat(dir_qid); - var n: u16 = 0; - for (m.wname[0..m.nwname]) |name| { - if (std.mem.eql(u8, name, "file")) { - wq[n] = file_qid; - } else if (std.mem.eql(u8, name, "dir")) { - wq[n] = dir_qid; - } else break; - n += 1; - } - if (n == 0 and m.nwname != 0) { - try srv.reply(tag, .{ .rerror = .{ .ename = "file does not exist" } }); - } else { - try srv.reply(tag, .{ .rwalk = .{ .nwqid = n, .wqid = wq } }); - } - }, - .tstat => try srv.reply(tag, .{ .rstat = .{ .stat = .{ - .type = 0, - .dev = 0, - .qid = file_qid, - .mode = 0o644, - .atime = 1, - .mtime = 2, - .length = fs.file_len, - .name = "file", - .uid = "u", - .gid = "g", - .muid = "u", - } } }), - .topen => |m| try srv.reply(tag, .{ .ropen = .{ .qid = file_qid, .iounit = if (m.mode == cloud9.owrite) 700 else 0 } }), - .tread => |m| { - fs.max_read_count = @max(fs.max_read_count, m.count); - var n: usize = 0; - if (m.offset < fs.file_len) n = @min(@as(usize, m.count), fs.file_len - @as(usize, @intCast(m.offset))); - n = @min(n, data.len); - for (data[0..n], 0..) |*b, i| b.* = @truncate(m.offset + i); - try srv.reply(tag, .{ .rread = .{ .data = data[0..n] } }); - }, - .twrite => |m| try srv.reply(tag, .{ .rwrite = .{ .count = @intCast(m.data.len) } }), - .tclunk => try srv.reply(tag, .rclunk), - .tremove => try srv.reply(tag, .{ .rerror = .{ .ename = "permission denied" } }), - .twstat => try srv.reply(tag, .rwstat), - // A flush is the test's "hang up now" signal. - .tflush => return, - else => try srv.reply(tag, .{ .rerror = .{ .ename = "not supported" } }), - } - srv.release(); - } - while (srv.output().len != 0) { - const o = srv.output(); - const rc = linux.write(fs.fd, o.ptr, o.len); - if (linux.errno(rc) != .SUCCESS) return error.Write; - srv.wrote(rc); - } - const rc = linux.read(fs.fd, &tmp, tmp.len); - if (linux.errno(rc) != .SUCCESS) return error.Read; - if (rc == 0) return; - if (srv.push(tmp[0..rc]) != rc) return error.Overflow; - } - } -}; - -test "session against an in-process cloud9.Server" { - var fds: [2]i32 = undefined; - try testing.expectEqual(linux.E.SUCCESS, linux.errno(linux.socketpair(linux.AF.UNIX, linux.SOCK.STREAM | linux.SOCK.CLOEXEC, 0, &fds))); - - var fs: FakeServer = .{ .fd = fds[1], .msize = 8192, .file_len = 20_000 }; - const th = try std.Thread.spawn(.{}, FakeServer.run, .{&fs}); - - var s = try Session.connect(testing.allocator, .{ .fd = fds[0] }, 8192); - defer { - s.deinit(); - th.join(); - } - try testing.expectEqual(@as(u32, 8192), s.msize); - - const root = try s.attach(0, "me", ""); - try testing.expectEqual(FakeServer.dir_qid.path, root.path); - - // Plain rpc + stat borrowing the input buffer. - const fid = s.allocFid(); - const w = try s.walk(0, fid, &.{"file"}); - try testing.expectEqual(@as(u16, 1), w.nwqid); - try testing.expectEqual(FakeServer.file_qid.path, w.wqid[0].path); - const st = try s.stat(fid); - try testing.expectEqualStrings("file", st.name); - try testing.expectEqual(@as(u64, 20_000), st.length); - - // Chunked read: 20000 bytes at maxRead = msize - 11 = 8181 per chunk. - _ = try s.open(fid, cloud9.oread); - const buf = try testing.allocator.alloc(u8, 30_000); - defer testing.allocator.free(buf); - const n = try s.read(fid, 0, buf); - try testing.expectEqual(@as(usize, 20_000), n); - for (buf[0..n], 0..) |b, i| try testing.expectEqual(@as(u8, @truncate(i)), b); - try testing.expectEqual(@as(u32, 8181), fs.max_read_count); - try testing.expectEqual(@as(usize, 0), try s.read(fid, 20_000, buf)); - - // iounit from open bounds the chunk. - const wfid = try s.clone(fid); - _ = try s.open(wfid, cloud9.owrite); - fs.max_read_count = 0; - _ = try s.read(wfid, 0, buf[0..3000]); - try testing.expectEqual(@as(u32, 700), fs.max_read_count); - try testing.expectEqual(@as(usize, 3000), try s.write(wfid, 0, buf[0..3000])); - - // Partial walk → error.Nine with a "not exist" ename → ENOENT. - const pfid = s.allocFid(); - try testing.expectError(error.Nine, s.walk(0, pfid, &.{ "dir", "nope" })); - try testing.expectEqual(linux.E.NOENT, s.errno()); - try testing.expectEqualStrings("file does not exist", s.ename[0..s.ename_len]); - s.freeFid(pfid); - - // Server Rerror → error.Nine, ename copied, fid freed by remove even on error. - try testing.expectError(error.Nine, s.remove(wfid)); - try testing.expectEqual(linux.E.ACCES, s.errno()); - try testing.expectEqual(wfid, s.allocFid()); // recycled - s.freeFid(wfid); - - // Unsupported op → "not supported" → ENOTSUP; a plain wstat succeeds. - try testing.expectError(error.Nine, s.rpc(.{ .auth = .{ .afid = 5, .uname = "me" } })); - try testing.expectEqual(linux.E.OPNOTSUPP, s.errno()); - try s.wstat(fid, dontcare); - try s.clunk(fid); - try testing.expectEqual(fid, s.allocFid()); - s.freeFid(fid); - - // A clone bound to a fid that then fails to walk must release the fid. - const before = s.next_fid; - const cfid = s.allocFid(); - s.freeFid(cfid); - try testing.expectError(error.Nine, s.walk(0, cfid, &.{"nope"})); - try testing.expectEqual(before, s.next_fid); - - // The server hanging up makes the pending rpc fail with error.Closed. - try testing.expectError(error.Closed, s.rpc(.{ .flush = .{ .oldtag = 0 } })); -} diff --git a/9player/src/ns.zig b/9player/src/ns.zig deleted file mode 100644 index 2c6f202..0000000 --- a/9player/src/ns.zig +++ /dev/null @@ -1,1082 +0,0 @@ -//! Namespace and process plumbing for 9player. -//! -//! Everything here is raw `std.os.linux` syscalls (no libc). The child side -//! of `spawn` runs between `fork` and `execve`; it does not allocate except -//! inside `ensureMountpoint` (the process is single-threaded by then, so the -//! inherited allocator is safe to use). -//! -//! Exit codes produced by the child before exec: 125 for namespace/mount -//! setup failures, 126 when the program was found but is not executable, -//! 127 when it was not found. - -const std = @import("std"); -const builtin = @import("builtin"); -const linux = std.os.linux; -const Allocator = std.mem.Allocator; -const E = linux.E; - -pub const Spawn = struct { - /// argv[0] is PATH-searched unless it contains '/'. - argv: []const []const u8, - /// Inherited environment; `NINEPLAYER_MOUNT` is added or replaced. - envp: [*:null]const ?[*:0]const u8, - /// Absolute mountpoint (see `resolveMountpoint`). - mountpoint: []const u8, - uid: u32, - gid: u32, - max_read: u32, - /// When false the namespace is set up (including mountpoint shadowing) - /// but `/dev/fuse` is not opened and nothing is mounted; `Child.fuse_fd` - /// is then -1. Only for smoke tests. - mount_fuse: bool = true, -}; - -pub const Child = struct { - pid: i32, - /// The `/dev/fuse` connection backing the mount, opened by the child - /// inside its user namespace (the kernel refuses to mount a fuse fd that - /// was opened from another user namespace) and handed back over the - /// status socket with SCM_RIGHTS. Owned by the caller; CLOEXEC. - fuse_fd: i32, - /// Parent end of the status socket. The child reports an exec failure - /// on it (see `reportExecFailure`); it reads EOF once exec succeeded. - status_fd: i32, -}; - -/// Exit status used by the child for setup failures (matches 9player's own). -pub const setup_failure_status: u8 = 125; -/// Refuse to shadow a directory with more entries than this. -pub const max_shadow_entries: usize = 4096; - -const default_path = "/usr/local/bin:/bin:/usr/bin"; -const path_max = 4096; - -// --------------------------------------------------------------------------- -// Mountpoint resolution -// --------------------------------------------------------------------------- - -/// Absolute path (relative paths resolved against cwd), duplicate slashes -/// collapsed, `.` and `..` components resolved lexically, no trailing slash. -/// `/` itself is rejected. -pub fn resolveMountpoint(gpa: Allocator, path: []const u8) ![:0]u8 { - var cwd_buf: [path_max]u8 = undefined; - var cwd: []const u8 = "/"; - if (path.len == 0 or path[0] != '/') { - const rc = linux.getcwd(&cwd_buf, cwd_buf.len); - switch (linux.errno(rc)) { - .SUCCESS => {}, - else => |e| { - std.debug.print("9player: getcwd: E{t}\n", .{e}); - return error.Cwd; - }, - } - // rc counts the terminating NUL. - cwd = cwd_buf[0 .. rc - 1]; - } - return normalizePath(gpa, cwd, path); -} - -/// Pure part of `resolveMountpoint`: `cwd` is only used when `path` is relative. -fn normalizePath(gpa: Allocator, cwd: []const u8, path: []const u8) ![:0]u8 { - if (path.len == 0) return error.InvalidMountpoint; - var out: std.ArrayList(u8) = .empty; - defer out.deinit(gpa); - if (path[0] != '/') try appendComponents(gpa, &out, cwd); - try appendComponents(gpa, &out, path); - if (out.items.len == 0) return error.InvalidMountpoint; // "/" or equivalent - return out.toOwnedSliceSentinel(gpa, 0); -} - -fn appendComponents(gpa: Allocator, out: *std.ArrayList(u8), path: []const u8) !void { - var it = std.mem.tokenizeScalar(u8, path, '/'); - while (it.next()) |comp| { - if (std.mem.eql(u8, comp, ".")) continue; - if (std.mem.eql(u8, comp, "..")) { - // Pop the last component (lexically; "/.." stays "/"). - const idx = std.mem.lastIndexOfScalar(u8, out.items, '/') orelse 0; - out.shrinkRetainingCapacity(idx); - continue; - } - try out.append(gpa, '/'); - try out.appendSlice(gpa, comp); - } -} - -// --------------------------------------------------------------------------- -// Environment helpers -// --------------------------------------------------------------------------- - -/// Look a variable up in a raw envp block. -pub fn getenv(envp: [*:null]const ?[*:0]const u8, name: []const u8) ?[]const u8 { - var i: usize = 0; - while (envp[i]) |entry| : (i += 1) { - const kv = std.mem.span(entry); - if (kv.len > name.len and kv[name.len] == '=' and std.mem.eql(u8, kv[0..name.len], name)) { - return kv[name.len + 1 ..]; - } - } - return null; -} - -/// Every path `execve` should try for `name`, in order: just `name` if it -/// contains a '/', else `/` for each `$PATH` element (an empty -/// element means the current directory; `$PATH` unset falls back to -/// `/usr/local/bin:/bin:/usr/bin`). -pub fn pathCandidates(gpa: Allocator, envp: [*:null]const ?[*:0]const u8, name: []const u8) ![]const [:0]const u8 { - if (name.len == 0) return error.EmptyProgramName; - var list: std.ArrayList([:0]const u8) = .empty; - errdefer { - for (list.items) |c| gpa.free(c); - list.deinit(gpa); - } - if (std.mem.indexOfScalar(u8, name, '/') != null) { - try list.append(gpa, try gpa.dupeZ(u8, name)); - return list.toOwnedSlice(gpa); - } - const path = getenv(envp, "PATH") orelse default_path; - var it = std.mem.splitScalar(u8, path, ':'); - while (it.next()) |dir| { - const d = if (dir.len == 0) "." else dir; - try list.append(gpa, try std.fmt.allocPrintSentinel(gpa, "{s}/{s}", .{ d, name }, 0)); - } - return list.toOwnedSlice(gpa); -} - -/// First PATH candidate that is an executable regular file, or the name -/// itself when it contains a '/'. Provided for completeness; `spawn` simply -/// tries `execve` on every candidate instead. -pub fn findInPath(gpa: Allocator, envp: [*:null]const ?[*:0]const u8, name: []const u8) ![:0]u8 { - const cands = try pathCandidates(gpa, envp, name); - defer { - for (cands) |c| gpa.free(c); - gpa.free(cands); - } - for (cands) |c| { - var stx: linux.Statx = undefined; - const rc = linux.statx(linux.AT.FDCWD, c.ptr, 0, .{ .TYPE = true, .MODE = true }, &stx); - if (linux.errno(rc) != .SUCCESS) continue; - if (stx.mode & linux.S.IFMT != linux.S.IFREG) continue; - if (stx.mode & 0o111 == 0) continue; - return gpa.dupeZ(u8, c); - } - return error.FileNotFound; -} - -/// New envp block: every entry of `envp` except `NINEPLAYER_MOUNT=...`, -/// followed by `NINEPLAYER_MOUNT=`. -fn buildEnvp(gpa: Allocator, envp: [*:null]const ?[*:0]const u8, mountpoint: []const u8) ![:null]?[*:0]const u8 { - const key = "NINEPLAYER_MOUNT="; - var keep: usize = 0; - var i: usize = 0; - while (envp[i]) |entry| : (i += 1) { - if (!std.mem.startsWith(u8, std.mem.span(entry), key)) keep += 1; - } - const out = try gpa.allocSentinel(?[*:0]const u8, keep + 1, null); - errdefer gpa.free(out); - var j: usize = 0; - i = 0; - while (envp[i]) |entry| : (i += 1) { - if (std.mem.startsWith(u8, std.mem.span(entry), key)) continue; - out[j] = entry; - j += 1; - } - const mount_entry = try std.fmt.allocPrintSentinel(gpa, key ++ "{s}", .{mountpoint}, 0); - out[j] = mount_entry.ptr; - return out; -} - -fn buildArgv(gpa: Allocator, argv: []const []const u8) ![:null]?[*:0]const u8 { - const out = try gpa.allocSentinel(?[*:0]const u8, argv.len, null); - for (argv, 0..) |a, i| out[i] = (try gpa.dupeZ(u8, a)).ptr; - return out; -} - -// --------------------------------------------------------------------------- -// Mountpoint policy -// --------------------------------------------------------------------------- - -/// Make sure `path` is a directory, inside the *current* mount namespace: -/// -/// * already a directory → done; -/// * else `mkdir`; on `EACCES`/`EPERM`/`EROFS` shadow the parent directory -/// with a tmpfs that re-exposes every existing entry (bind mounts for -/// directories and files, recreated symlinks) and `mkdir` inside it; -/// * anything else fails with the errno and a hint. -/// -/// Every failure prints `9player: : E` to stderr before -/// returning. Meant to be called in the child of `spawn` (or from a -/// throwaway namespace: `unshare -Urm`). -pub fn ensureMountpoint(gpa: Allocator, path: [:0]const u8) !void { - if (fileType(linux.AT.FDCWD, path, false)) |ft| { - if (ft == .dir) return; - std.debug.print("9player: mountpoint {s}: exists but is not a directory\n", .{path}); - return error.Mountpoint; - } - if (fileType(linux.AT.FDCWD, path, true) == .symlink) { - std.debug.print("9player: mountpoint {s}: dangling symlink\n", .{path}); - return error.Mountpoint; - } - const mk = linux.errno(linux.mkdirat(linux.AT.FDCWD, path, 0o755)); - switch (mk) { - .SUCCESS => return, - .ACCES, .PERM, .ROFS => {}, - else => |e| { - std.debug.print("9player: mkdir {s}: E{t} (pass --mount an existing directory)\n", .{ path, e }); - return error.Mountpoint; - }, - } - const parent = std.fs.path.dirname(path) orelse "/"; - if (std.mem.eql(u8, parent, "/") or isSameDirectory(parent, "/")) { - std.debug.print("9player: mkdir {s}: E{t}; refusing to shadow / (pass --mount an existing directory)\n", .{ path, mk }); - return error.Mountpoint; - } - // The shadow rebuilds entries from /proc/self/fd//; a tmpfs - // over /proc (or a subtree of it) would take that away from itself. - if (std.mem.eql(u8, parent, "/proc") or std.mem.startsWith(u8, parent, "/proc/")) { - std.debug.print("9player: mkdir {s}: E{t}; refusing to shadow {s} (pass --mount an existing directory)\n", .{ path, mk, parent }); - return error.Mountpoint; - } - const parent_z = try gpa.dupeZ(u8, parent); - defer gpa.free(parent_z); - try shadowDirectory(gpa, parent_z); - switch (linux.errno(linux.mkdirat(linux.AT.FDCWD, path, 0o755))) { - .SUCCESS => {}, - else => |e| { - std.debug.print("9player: mkdir {s} (in shadow tmpfs): E{t}\n", .{ path, e }); - return error.Mountpoint; - }, - } -} - -const FileType = enum { dir, symlink, other }; - -/// True when both paths resolve (following symlinks, including magic ones -/// such as /proc/self/root) to the same inode. -fn isSameDirectory(a: []const u8, b: [*:0]const u8) bool { - var a_buf: [path_max]u8 = undefined; - const a_z = std.fmt.bufPrintZ(&a_buf, "{s}", .{a}) catch return false; - var sa: linux.Statx = undefined; - var sb: linux.Statx = undefined; - if (linux.errno(linux.statx(linux.AT.FDCWD, a_z, 0, .{ .INO = true }, &sa)) != .SUCCESS) return false; - if (linux.errno(linux.statx(linux.AT.FDCWD, b, 0, .{ .INO = true }, &sb)) != .SUCCESS) return false; - return sa.ino == sb.ino and sa.dev_major == sb.dev_major and sa.dev_minor == sb.dev_minor; -} - -fn fileType(dirfd: i32, name: [*:0]const u8, nofollow: bool) ?FileType { - var stx: linux.Statx = undefined; - const flags: u32 = if (nofollow) linux.AT.SYMLINK_NOFOLLOW else 0; - const rc = linux.statx(dirfd, name, flags, .{ .TYPE = true }, &stx); - if (linux.errno(rc) != .SUCCESS) return null; - return switch (stx.mode & linux.S.IFMT) { - linux.S.IFDIR => .dir, - linux.S.IFLNK => .symlink, - else => .other, - }; -} - -const Entry = struct { name: [:0]u8, kind: FileType }; - -/// Read every entry of the directory open at `fd` (excluding `.` and `..`). -fn listDir(gpa: Allocator, fd: i32, dirpath: []const u8) ![]Entry { - var list: std.ArrayList(Entry) = .empty; - errdefer { - for (list.items) |e| gpa.free(e.name); - list.deinit(gpa); - } - var buf: [32 * 1024]u8 align(@alignOf(linux.dirent64)) = undefined; - while (true) { - const rc = linux.getdents64(fd, &buf, buf.len); - switch (linux.errno(rc)) { - .SUCCESS => {}, - else => |e| { - std.debug.print("9player: getdents64 {s}: E{t}\n", .{ dirpath, e }); - return error.Mountpoint; - }, - } - if (rc == 0) break; - var off: usize = 0; - while (off < rc) { - const d: *align(1) const linux.dirent64 = @ptrCast(&buf[off]); - const name_ptr: [*:0]const u8 = @ptrCast(&buf[off + @offsetOf(linux.dirent64, "name")]); - const name = std.mem.span(name_ptr); - const dtype = d.type; - off += d.reclen; - if (std.mem.eql(u8, name, ".") or std.mem.eql(u8, name, "..")) continue; - if (list.items.len >= max_shadow_entries) { - std.debug.print("9player: refusing to shadow {s}: more than {d} entries\n", .{ dirpath, max_shadow_entries }); - return error.TooManyEntries; - } - const kind: FileType = switch (dtype) { - linux.DT.DIR => .dir, - linux.DT.LNK => .symlink, - linux.DT.UNKNOWN => fileType(fd, name_ptr, true) orelse .other, - else => .other, - }; - try list.append(gpa, .{ .name = try gpa.dupeZ(u8, name), .kind = kind }); - } - } - return list.toOwnedSlice(gpa); -} - -fn shadowDirectory(gpa: Allocator, parent: [:0]const u8) !void { - const open_rc = linux.open(parent, .{ .ACCMODE = .RDONLY, .DIRECTORY = true, .CLOEXEC = true }, 0); - switch (linux.errno(open_rc)) { - .SUCCESS => {}, - else => |e| { - std.debug.print("9player: open {s}: E{t}\n", .{ parent, e }); - return error.Mountpoint; - }, - } - const pfd: i32 = @intCast(open_rc); - defer _ = linux.close(pfd); - - const entries = try listDir(gpa, pfd, parent); - defer { - for (entries) |e| gpa.free(e.name); - gpa.free(entries); - } - - const tmpfs_opts: [*:0]const u8 = "mode=755"; - switch (linux.errno(linux.mount("tmpfs", parent, "tmpfs", linux.MS.NOSUID | linux.MS.NODEV, @intFromPtr(tmpfs_opts)))) { - .SUCCESS => {}, - else => |e| { - std.debug.print("9player: mount tmpfs on {s}: E{t}\n", .{ parent, e }); - return error.Mountpoint; - }, - } - - // `pfd` still refers to the original directory underneath the tmpfs, so - // `/proc/self/fd//` reaches the hidden entries. - var src_buf: [path_max]u8 = undefined; - var dst_buf: [path_max]u8 = undefined; - var link_buf: [path_max]u8 = undefined; - for (entries) |e| { - const src = std.fmt.bufPrintZ(&src_buf, "/proc/self/fd/{d}/{s}", .{ pfd, e.name }) catch { - std.debug.print("9player: shadow {s}/{s}: name too long (skipped)\n", .{ parent, e.name }); - continue; - }; - const dst = std.fmt.bufPrintZ(&dst_buf, "{s}/{s}", .{ parent, e.name }) catch { - std.debug.print("9player: shadow {s}/{s}: name too long (skipped)\n", .{ parent, e.name }); - continue; - }; - switch (e.kind) { - .dir => { - if (!check("mkdir", dst, linux.mkdirat(linux.AT.FDCWD, dst, 0o755))) continue; - _ = check("bind", dst, linux.mount(src, dst, null, linux.MS.BIND | linux.MS.REC, 0)); - }, - .symlink => { - const rc = linux.readlinkat(pfd, e.name, &link_buf, link_buf.len - 1); - if (!check("readlink", dst, rc)) continue; - link_buf[rc] = 0; - const target: [*:0]const u8 = @ptrCast(&link_buf); - _ = check("symlink", dst, linux.symlinkat(target, linux.AT.FDCWD, dst)); - }, - .other => { - const rc = linux.openat(linux.AT.FDCWD, dst, .{ .ACCMODE = .WRONLY, .CREAT = true, .CLOEXEC = true }, 0o644); - if (!check("create", dst, rc)) continue; - _ = linux.close(@intCast(rc)); - _ = check("bind", dst, linux.mount(src, dst, null, linux.MS.BIND | linux.MS.REC, 0)); - }, - } - } -} - -/// Report a failed per-entry step as a warning (the entry is skipped; the -/// rest of the shadow is still useful). Returns true on success. -fn check(step: []const u8, path: [*:0]const u8, rc: usize) bool { - switch (linux.errno(rc)) { - .SUCCESS => return true, - else => |e| { - std.debug.print("9player: shadow: {s} {s}: E{t} (skipped)\n", .{ step, std.mem.span(path), e }); - return false; - }, - } -} - -// --------------------------------------------------------------------------- -// spawn -// --------------------------------------------------------------------------- - -const ChildArgs = struct { - gpa: Allocator, - status_sock: i32, - mountpoint: [:0]const u8, - fuse_opts_prefix: [:0]const u8, // everything after "fd=," - mount_fuse: bool, - uid_map: []const u8, - gid_map: []const u8, - argv: [:null]?[*:0]const u8, - envp: [:null]?[*:0]const u8, - candidates: []const [:0]const u8, - name: []const u8, -}; - -/// Status channel protocol (child → parent, over a CLOEXEC socketpair): -/// a 0 byte means "namespace and mount are up" and carries the fuse fd as -/// SCM_RIGHTS; a non-zero byte is an exit status followed by a message. -/// EOF ends the conversation (exec succeeded, or the child died). -const ok_byte: u8 = 0; - -/// fork; the child unshares user+mount namespaces, maps its uid/gid, -/// makes `/` private, ensures the mountpoint, opens `/dev/fuse`, mounts it -/// on the mountpoint and sends the fd back. `spawn` returns at that point -/// (with `error.ChildFailed` and a message on stderr if any step failed). -/// The child then stats the mountpoint, which makes the kernel fetch the -/// root's attributes once the parent serves (the kernel seeds the fuse root -/// with uid 0, unmapped in the new user namespace, so nothing could be -/// created in the root until then), sets `NINEPLAYER_MOUNT` and execs -/// `argv`. An exec failure is reported on `Child.status_fd` and ends the -/// child with 126/127; collect it with `reportExecFailure` after -/// `bridge.serve` returns. -/// -/// If `installSignals` was called, the pid is stored into the registered -/// variable as soon as fork returns so no SIGCHLD can be missed. -pub fn spawn(gpa: Allocator, s: Spawn) !Child { - if (s.argv.len == 0 or s.argv[0].len == 0) { - std.debug.print("9player: empty program name\n", .{}); - return error.EmptyProgramName; - } - - const mountpoint = try gpa.dupeZ(u8, s.mountpoint); - defer gpa.free(mountpoint); - const fuse_opts_prefix = try std.fmt.allocPrintSentinel(gpa, "rootmode=40000,user_id={d},group_id={d},max_read={d}", .{ s.uid, s.gid, s.max_read }, 0); - defer gpa.free(fuse_opts_prefix); - var uid_buf: [64]u8 = undefined; - var gid_buf: [64]u8 = undefined; - const uid_map = try std.fmt.bufPrint(&uid_buf, "{d} {d} 1\n", .{ s.uid, s.uid }); - const gid_map = try std.fmt.bufPrint(&gid_buf, "{d} {d} 1\n", .{ s.gid, s.gid }); - const argv = try buildArgv(gpa, s.argv); - defer { - for (argv) |a| gpa.free(std.mem.span(a.?)); - gpa.free(argv); - } - const envp = try buildEnvp(gpa, s.envp, s.mountpoint); - defer { - gpa.free(std.mem.span(envp[envp.len - 1].?)); // the NINEPLAYER_MOUNT entry we created - gpa.free(envp); - } - const candidates = try pathCandidates(gpa, s.envp, s.argv[0]); - defer { - for (candidates) |c| gpa.free(c); - gpa.free(candidates); - } - - var sv: [2]i32 = undefined; - switch (linux.errno(linux.socketpair(linux.AF.UNIX, linux.SOCK.STREAM | linux.SOCK.CLOEXEC, 0, &sv))) { - .SUCCESS => {}, - else => |e| { - std.debug.print("9player: socketpair: E{t}\n", .{e}); - return error.SystemResources; - }, - } - - const child_args = ChildArgs{ - .gpa = gpa, - .status_sock = sv[1], - .mountpoint = mountpoint, - .fuse_opts_prefix = fuse_opts_prefix, - .mount_fuse = s.mount_fuse, - .uid_map = uid_map, - .gid_map = gid_map, - .argv = argv, - .envp = envp, - .candidates = candidates, - .name = s.argv[0], - }; - - const fork_rc = linux.fork(); - switch (linux.errno(fork_rc)) { - .SUCCESS => {}, - else => |e| { - _ = linux.close(sv[0]); - _ = linux.close(sv[1]); - std.debug.print("9player: fork: E{t}\n", .{e}); - return error.SystemResources; - }, - } - if (fork_rc == 0) childMain(&child_args); - - const pid: i32 = @intCast(fork_rc); - if (child_pid_ptr) |p| @atomicStore(i32, p, pid, .seq_cst); - _ = linux.close(sv[1]); - - // First byte: ok (with the fuse fd attached) or a failure status. - var first: [1]u8 = .{ok_byte}; // defined even if recvmsg stores nothing - var fuse_fd: i32 = -1; - var n: usize = 0; - while (true) { - const rc = recvWithFd(sv[0], &first, &fuse_fd, 0); - switch (linux.errno(rc)) { - .SUCCESS => {}, - .INTR => continue, - else => break, - } - n = rc; - break; - } - if (n == 1 and first[0] == ok_byte and (fuse_fd >= 0 or !s.mount_fuse)) { - return .{ .pid = pid, .fuse_fd = fuse_fd, .status_fd = sv[0] }; - } - - // Failure. A status byte means the child is exiting on its own and a - // message follows. Anything else (EOF: the child died before reporting; - // an ok byte without the fd: the SCM_RIGHTS transfer was truncated, e.g. - // EMFILE) is a protocol violation: the child may be about to exec with a - // dead mount, so kill it before waiting rather than reading the status - // socket until an exec'd program eventually exits. - const reported = n == 1 and first[0] != ok_byte; - if (!reported) _ = linux.kill(pid, .KILL); - if (fuse_fd >= 0) _ = linux.close(fuse_fd); - var msg: [512]u8 = undefined; - var len: usize = 0; - while (reported and len < msg.len) { - const rc = linux.read(sv[0], msg[len..].ptr, msg.len - len); - switch (linux.errno(rc)) { - .SUCCESS => {}, - .INTR => continue, - else => break, - } - if (rc == 0) break; - len += rc; - } - _ = linux.close(sv[0]); - if (reported) { - std.debug.print("9player: {s}\n", .{msg[0..len]}); - } else if (n == 1) { - std.debug.print("9player: child handshake failed: no fuse fd received (out of file descriptors?)\n", .{}); - } else { - std.debug.print("9player: child exited before reporting\n", .{}); - } - _ = waitChild(pid) catch {}; - if (child_pid_ptr) |p| @atomicStore(i32, p, 0, .seq_cst); - const status: u8 = if (reported) first[0] else setup_failure_status; - return switch (status) { - 126 => error.ExecPermission, - 127 => error.ExecNotFound, - else => error.ChildFailed, - }; -} - -/// After the child is gone (or the mount is dead): print the exec failure -/// the child reported on `status_fd`, if any, and close it. Returns the -/// status byte the child announced, or null when exec succeeded / nothing -/// was reported. Never blocks. -pub fn reportExecFailure(child: Child) ?u8 { - defer _ = linux.close(child.status_fd); - var msg: [512]u8 = undefined; - var len: usize = 0; - while (len < msg.len) { - var iov = [_]std.posix.iovec{.{ .base = msg[len..].ptr, .len = msg.len - len }}; - var hdr = linux.msghdr{ - .name = null, - .namelen = 0, - .iov = &iov, - .iovlen = 1, - .control = null, - .controllen = 0, - .flags = 0, - }; - const rc = linux.recvmsg(child.status_fd, &hdr, linux.MSG.DONTWAIT); - switch (linux.errno(rc)) { - .SUCCESS => {}, - .INTR => continue, - else => break, - } - if (rc == 0) break; - len += rc; - } - if (len == 0) return null; - std.debug.print("9player: {s}\n", .{msg[1..len]}); - return msg[0]; -} - -const cmsg_fd_len = @sizeOf(linux.cmsghdr) + @sizeOf(i32); -const cmsg_fd_space = std.mem.alignForward(usize, cmsg_fd_len, @sizeOf(usize)); - -/// sendmsg one data byte, optionally with `fd` attached as SCM_RIGHTS. -fn sendWithFd(sock: i32, byte: u8, fd: ?i32) usize { - const data = [_]u8{byte}; - const iov = [_]std.posix.iovec_const{.{ .base = &data, .len = 1 }}; - var cbuf: [cmsg_fd_space]u8 align(@alignOf(linux.cmsghdr)) = @splat(0); - var msg = linux.msghdr_const{ - .name = null, - .namelen = 0, - .iov = &iov, - .iovlen = 1, - .control = null, - .controllen = 0, - .flags = 0, - }; - if (fd) |f| { - const hdr: *linux.cmsghdr = @ptrCast(&cbuf); - hdr.* = .{ .len = cmsg_fd_len, .level = linux.SOL.SOCKET, .type = linux.SCM.RIGHTS }; - @memcpy(cbuf[@sizeOf(linux.cmsghdr)..][0..@sizeOf(i32)], std.mem.asBytes(&f)); - msg.control = &cbuf; - msg.controllen = cmsg_fd_space; - } - return linux.sendmsg(sock, &msg, linux.MSG.NOSIGNAL); -} - -/// recvmsg into `buf`; an SCM_RIGHTS fd, if any, is stored in `fd_out`. -fn recvWithFd(sock: i32, buf: []u8, fd_out: *i32, flags: u32) usize { - var iov = [_]std.posix.iovec{.{ .base = buf.ptr, .len = buf.len }}; - var cbuf: [cmsg_fd_space]u8 align(@alignOf(linux.cmsghdr)) = @splat(0); - var msg = linux.msghdr{ - .name = null, - .namelen = 0, - .iov = &iov, - .iovlen = 1, - .control = &cbuf, - .controllen = cbuf.len, - .flags = 0, - }; - const rc = linux.recvmsg(sock, &msg, linux.MSG.CMSG_CLOEXEC | flags); - if (linux.errno(rc) != .SUCCESS) return rc; - if (msg.controllen >= cmsg_fd_len) { - const hdr: *const linux.cmsghdr = @ptrCast(&cbuf); - if (hdr.level == linux.SOL.SOCKET and hdr.type == linux.SCM.RIGHTS and hdr.len >= cmsg_fd_len) { - var fd: i32 = undefined; - @memcpy(std.mem.asBytes(&fd), cbuf[@sizeOf(linux.cmsghdr)..][0..@sizeOf(i32)]); - fd_out.* = fd; - } - } - return rc; -} - -/// Child side of `spawn`. Never returns. -fn childMain(c: *const ChildArgs) noreturn { - resetSignals(); - - const rc_unshare = linux.errno(linux.unshare(linux.CLONE.NEWUSER | linux.CLONE.NEWNS)); - if (rc_unshare != .SUCCESS) childFail(c, setup_failure_status, "unshare(CLONE_NEWUSER|CLONE_NEWNS)", rc_unshare, true); - writeProcFile(c, "/proc/self/setgroups", "deny", true); - writeProcFile(c, "/proc/self/uid_map", c.uid_map, false); - writeProcFile(c, "/proc/self/gid_map", c.gid_map, false); - - const root: [*:0]const u8 = "/"; - const rc_priv = linux.mount(null, root, null, linux.MS.REC | linux.MS.PRIVATE, 0); - if (linux.errno(rc_priv) != .SUCCESS) childFail(c, setup_failure_status, "mount(/, MS_REC|MS_PRIVATE)", linux.errno(rc_priv), true); - - ensureMountpoint(c.gpa, c.mountpoint) catch { - childFail(c, setup_failure_status, "mountpoint setup failed (pass --mount an existing directory)", .SUCCESS, false); - }; - - var fuse_fd: ?i32 = null; - if (c.mount_fuse) { - // Must be opened here, after unshare: the kernel only mounts a fuse - // device opened from the mount's own user namespace. - const rc_open = linux.open("/dev/fuse", .{ .ACCMODE = .RDWR, .CLOEXEC = true }, 0); - switch (linux.errno(rc_open)) { - .SUCCESS => {}, - .NOENT => childFail(c, setup_failure_status, "open /dev/fuse: ENOENT (is the fuse module loaded? try: modprobe fuse)", .SUCCESS, false), - else => |e| childFail(c, setup_failure_status, "open /dev/fuse", e, true), - } - const fd: i32 = @intCast(rc_open); - var opts_buf: [256]u8 = undefined; - const opts = std.fmt.bufPrintZ(&opts_buf, "fd={d},{s}", .{ fd, c.fuse_opts_prefix }) catch unreachable; - const rc = linux.mount("9player", c.mountpoint, "fuse", linux.MS.NOSUID | linux.MS.NODEV, @intFromPtr(opts.ptr)); - if (linux.errno(rc) != .SUCCESS) childFail(c, setup_failure_status, "mount fuse", linux.errno(rc), true); - fuse_fd = fd; - } - const sent = sendWithFd(c.status_sock, ok_byte, fuse_fd); - if (linux.errno(sent) != .SUCCESS) linux.exit_group(setup_failure_status); - if (fuse_fd) |fd| { - _ = linux.close(fd); // the parent holds the connection now - // Force one GETATTR of the root (served by the parent, which is - // entering its serve loop now); see `spawn`. Errors don't matter. - var stx: linux.Statx = undefined; - _ = linux.statx(linux.AT.FDCWD, c.mountpoint, 0, .{ .TYPE = true }, &stx); - } - - var last: E = .NOENT; - var saw_acces = false; - for (c.candidates) |cand| { - const rc = linux.execve(cand.ptr, c.argv.ptr, c.envp.ptr); - last = linux.errno(rc); - switch (last) { - .NOENT, .NOTDIR, .LOOP, .NAMETOOLONG => continue, - .ACCES => { - saw_acces = true; - continue; - }, - else => break, - } - } - var buf: [512]u8 = undefined; - // "Not found" covers every candidate that could not even be resolved - // (a PATH element that is a file gives ENOTDIR, a symlink loop ELOOP); - // a candidate that existed but was not executable wins over those. - const not_found = switch (last) { - .NOENT, .NOTDIR, .LOOP, .NAMETOOLONG => true, - else => false, - }; - if (not_found and saw_acces) last = .ACCES; - const status: u8 = if (not_found and !saw_acces) 127 else 126; - const text = std.fmt.bufPrint(&buf, "exec {s}", .{c.name}) catch "exec"; - childFail(c, status, text, last, true); -} - -fn writeProcFile(c: *const ChildArgs, path: [*:0]const u8, data: []const u8, ignore_missing: bool) void { - const rc = linux.open(path, .{ .ACCMODE = .WRONLY, .CLOEXEC = true }, 0); - switch (linux.errno(rc)) { - .SUCCESS => {}, - .NOENT => if (ignore_missing) return else childFail(c, setup_failure_status, std.mem.span(path), .NOENT, true), - else => |e| childFail(c, setup_failure_status, std.mem.span(path), e, true), - } - const fd: i32 = @intCast(rc); - const w = linux.write(fd, data.ptr, data.len); - const we = linux.errno(w); - _ = linux.close(fd); - if (we != .SUCCESS) childFail(c, setup_failure_status, std.mem.span(path), we, true); - if (w != data.len) childFail(c, setup_failure_status, std.mem.span(path), .IO, true); -} - -/// Write `[: E]` to the status socket and exit. -fn childFail(c: *const ChildArgs, status: u8, step: []const u8, e: E, with_errno: bool) noreturn { - var buf: [600]u8 = undefined; - buf[0] = status; - const rest = if (with_errno) - std.fmt.bufPrint(buf[1..], "{s}: E{t}", .{ step, e }) catch buf[1..1] - else - std.fmt.bufPrint(buf[1..], "{s}", .{step}) catch buf[1..1]; - const msg = buf[0 .. 1 + rest.len]; - var off: usize = 0; - while (off < msg.len) { - const rc = linux.write(c.status_sock, msg[off..].ptr, msg.len - off); - if (linux.errno(rc) == .INTR) continue; - if (linux.errno(rc) != .SUCCESS) break; - off += rc; - } - linux.exit_group(status); -} - -// --------------------------------------------------------------------------- -// Signals -// --------------------------------------------------------------------------- - -var child_pid_ptr: ?*i32 = null; -var chld_pipe_w: i32 = -1; -var reaped = std.atomic.Value(bool).init(false); -var reaped_status = std.atomic.Value(u32).init(0); -/// A second child (the `--spawn` server) that the SIGCHLD handler reaps so -/// it does not linger as a zombie when it dies mid-session. Its exit does -/// not stop the serve loop. 0 = none. -var server_pid = std.atomic.Value(i32).init(0); - -/// Register the `--spawn` server for reaping by the SIGCHLD handler. -pub fn watchServer(pid: i32) void { - server_pid.store(pid, .seq_cst); -} - -/// Seconds the serve loop gets to come back after the child died before -/// the watchdog ends the process anyway. -pub const exit_grace_seconds: isize = 3; - -/// The watched child is already dead but the serve loop has not come back -/// (it is stuck in a 9P request the server never answers): a terminal -/// signal, or the watchdog armed by `onChld`, then ends 9player with the -/// child's status instead of hanging. Nothing is lost: the mount is torn -/// down when the process exits. -fn bailIfChildGone() void { - if (!reaped.load(.acquire)) return; - const srv = server_pid.load(.seq_cst); - if (srv > 0) _ = linux.kill(srv, .TERM); - linux.exit_group(decodeStatus(reaped_status.load(.acquire))); -} - -fn armWatchdog() void { - // setitimer takes an itimerval; std declares it with itimerspec, which - // has the same layout on 64-bit targets (the sub-second field is 0). - const t = linux.itimerspec{ - .it_interval = .{ .sec = 0, .nsec = 0 }, - .it_value = .{ .sec = exit_grace_seconds, .nsec = 0 }, - }; - _ = linux.setitimer(@intFromEnum(linux.ITIMER.REAL), &t, null); -} - -fn onAlarm(_: linux.SIG) callconv(.c) void { - bailIfChildGone(); -} - -fn onForward(sig: linux.SIG) callconv(.c) void { - const p = child_pid_ptr orelse return; - const pid = @atomicLoad(i32, p, .seq_cst); - if (pid > 0) _ = linux.kill(pid, sig); - bailIfChildGone(); -} - -/// SIGINT/SIGQUIT: the child owns the tty and gets them itself; we only -/// react when the child is already gone (see `bailIfChildGone`). -fn onTerminal(_: linux.SIG) callconv(.c) void { - bailIfChildGone(); -} - -/// Only the watched child counts: reap it here (WNOHANG), remember its -/// status, forget its pid (so a later SIGTERM cannot hit a recycled pid) -/// and poke the self-pipe. The `--spawn` server is reaped too but does not -/// interrupt `bridge.serve`; SIGCHLD from anything else is ignored. -fn onChld(_: linux.SIG) callconv(.c) void { - const srv = server_pid.load(.seq_cst); - if (srv > 0) { - var sst: u32 = 0; - const src = linux.waitpid(srv, &sst, linux.W.NOHANG); - if (linux.errno(src) == .SUCCESS and src != 0) server_pid.store(0, .seq_cst); - } - const p = child_pid_ptr orelse return; - const pid = @atomicLoad(i32, p, .seq_cst); - if (pid <= 0) return; - var st: u32 = 0; - const rc = linux.waitpid(pid, &st, linux.W.NOHANG); - if (linux.errno(rc) != .SUCCESS or rc == 0) return; - reaped_status.store(st, .release); - reaped.store(true, .release); - @atomicStore(i32, p, 0, .seq_cst); - const b = [_]u8{'c'}; - _ = linux.write(chld_pipe_w, &b, 1); - armWatchdog(); -} - -/// SIGPIPE ignored; SIGINT/SIGQUIT effectively ignored (the child owns the -/// tty) unless the child is already dead; SIGTERM/SIGHUP forwarded to -/// `*child_pid`; SIGCHLD for `*child_pid` reaps it, writes a byte to a -/// nonblocking self-pipe whose read end is returned (use it as `stop_fd`) -/// and arms a watchdog (`exit_grace_seconds`, SIGALRM) that ends the -/// process with the child's status should the serve loop stay blocked. -/// `*child_pid` is filled in by `spawn`. -pub fn installSignals(child_pid: *i32) !i32 { - child_pid_ptr = child_pid; - var fds: [2]i32 = undefined; - switch (linux.errno(linux.pipe2(&fds, .{ .CLOEXEC = true, .NONBLOCK = true }))) { - .SUCCESS => {}, - else => |e| { - std.debug.print("9player: pipe2: E{t}\n", .{e}); - return error.SystemResources; - }, - } - chld_pipe_w = fds[1]; - - const ign = linux.Sigaction{ .handler = .{ .handler = linux.SIG.IGN }, .mask = linux.sigemptyset(), .flags = 0 }; - const term = linux.Sigaction{ .handler = .{ .handler = &onTerminal }, .mask = linux.sigemptyset(), .flags = linux.SA.RESTART }; - const fwd = linux.Sigaction{ .handler = .{ .handler = &onForward }, .mask = linux.sigemptyset(), .flags = linux.SA.RESTART }; - const chld = linux.Sigaction{ .handler = .{ .handler = &onChld }, .mask = linux.sigemptyset(), .flags = linux.SA.RESTART | linux.SA.NOCLDSTOP }; - const alrm = linux.Sigaction{ .handler = .{ .handler = &onAlarm }, .mask = linux.sigemptyset(), .flags = linux.SA.RESTART }; - std.posix.sigaction(.INT, &term, null); - std.posix.sigaction(.QUIT, &term, null); - std.posix.sigaction(.ALRM, &alrm, null); - std.posix.sigaction(.PIPE, &ign, null); - std.posix.sigaction(.TERM, &fwd, null); - std.posix.sigaction(.HUP, &fwd, null); - std.posix.sigaction(.CHLD, &chld, null); - return fds[0]; -} - -/// Restore default dispositions in the child before exec (ignored signals -/// would otherwise survive execve). -fn resetSignals() void { - const dfl = linux.Sigaction{ .handler = .{ .handler = linux.SIG.DFL }, .mask = linux.sigemptyset(), .flags = 0 }; - inline for (.{ linux.SIG.INT, linux.SIG.QUIT, linux.SIG.PIPE, linux.SIG.TERM, linux.SIG.HUP, linux.SIG.CHLD, linux.SIG.ALRM }) |sig| { - _ = linux.sigaction(sig, &dfl, null); - } -} - -// --------------------------------------------------------------------------- -// Waiting -// --------------------------------------------------------------------------- - -fn takeReaped() ?u32 { - if (!reaped.load(.acquire)) return null; - return reaped_status.load(.acquire); -} - -/// waitpid status → exit code (`128+sig` when killed by a signal). -pub fn decodeStatus(st: u32) u8 { - if (linux.W.IFEXITED(st)) return linux.W.EXITSTATUS(st); - if (linux.W.IFSIGNALED(st)) return 128 +% @as(u8, @truncate(@intFromEnum(linux.W.TERMSIG(st)))); - return 1; -} - -/// Block until `pid` exits (the SIGCHLD handler may have reaped it already). -pub fn waitChild(pid: i32) !u8 { - while (true) { - if (takeReaped()) |st| return decodeStatus(st); - var st: u32 = 0; - const rc = linux.waitpid(pid, &st, 0); - switch (linux.errno(rc)) { - .SUCCESS => return decodeStatus(st), - .INTR => continue, - .CHILD => { - if (takeReaped()) |s| return decodeStatus(s); - return error.NoChild; - }, - else => |e| { - std.debug.print("9player: waitpid: E{t}\n", .{e}); - return error.Wait; - }, - } - } -} - -/// Non-blocking: the exit status of `pid` if it has exited, else null. -pub fn reapIfExited(pid: i32) ?u8 { - if (takeReaped()) |st| return decodeStatus(st); - var st: u32 = 0; - const rc = linux.waitpid(pid, &st, linux.W.NOHANG); - switch (linux.errno(rc)) { - .SUCCESS => return if (rc == 0) null else decodeStatus(st), - .CHILD => return if (takeReaped()) |s| decodeStatus(s) else null, - else => return null, - } -} - -/// Reap any child (used for the `--spawn` server at exit). Non-blocking. -pub fn reapAny(pid: i32) void { - var st: u32 = 0; - _ = linux.waitpid(pid, &st, linux.W.NOHANG); -} - -// --------------------------------------------------------------------------- -// Tests (no namespaces needed; `ensureMountpoint` is exercised by -// test/integration.sh through the 9player binary) -// --------------------------------------------------------------------------- - -const testing = std.testing; - -test "normalizePath: absolute paths" { - const gpa = testing.allocator; - const cases = [_]struct { in: []const u8, out: []const u8 }{ - .{ .in = "/mnt/9p", .out = "/mnt/9p" }, - .{ .in = "/mnt/9p/", .out = "/mnt/9p" }, - .{ .in = "//mnt///9p//", .out = "/mnt/9p" }, - .{ .in = "/mnt/./9p/.", .out = "/mnt/9p" }, - .{ .in = "/mnt/x/../9p", .out = "/mnt/9p" }, - .{ .in = "/../mnt/9p", .out = "/mnt/9p" }, - .{ .in = "/a/b/c/../..", .out = "/a" }, - }; - for (cases) |c| { - const got = try normalizePath(gpa, "/cwd", c.in); - defer gpa.free(got); - try testing.expectEqualStrings(c.out, got); - try testing.expectEqual(@as(u8, 0), got[got.len]); - } -} - -test "normalizePath: relative paths use cwd" { - const gpa = testing.allocator; - const cases = [_]struct { cwd: []const u8, in: []const u8, out: []const u8 }{ - .{ .cwd = "/home/me", .in = "mnt", .out = "/home/me/mnt" }, - .{ .cwd = "/home/me", .in = "./mnt/", .out = "/home/me/mnt" }, - .{ .cwd = "/home/me", .in = "../mnt", .out = "/home/mnt" }, - .{ .cwd = "/home/me/", .in = ".", .out = "/home/me" }, - .{ .cwd = "/", .in = "x", .out = "/x" }, - }; - for (cases) |c| { - const got = try normalizePath(gpa, c.cwd, c.in); - defer gpa.free(got); - try testing.expectEqualStrings(c.out, got); - } -} - -test "normalizePath: rejects root and empty" { - const gpa = testing.allocator; - try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/cwd", "/")); - try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/cwd", "///")); - try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/cwd", "/mnt/..")); - try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/cwd", "")); - try testing.expectError(error.InvalidMountpoint, normalizePath(gpa, "/", "..")); -} - -test "resolveMountpoint: relative resolves against the real cwd" { - const gpa = testing.allocator; - const got = try resolveMountpoint(gpa, "sub/dir"); - defer gpa.free(got); - try testing.expect(got[0] == '/'); - try testing.expect(std.mem.endsWith(u8, got, "/sub/dir")); -} - -test "getenv" { - const env = [_:null]?[*:0]const u8{ "PATH=/a:/b", "X=", "PATHX=no", "NINEPLAYER_MOUNT=/m" }; - const envp: [*:null]const ?[*:0]const u8 = &env; - try testing.expectEqualStrings("/a:/b", getenv(envp, "PATH").?); - try testing.expectEqualStrings("", getenv(envp, "X").?); - try testing.expectEqualStrings("/m", getenv(envp, "NINEPLAYER_MOUNT").?); - try testing.expect(getenv(envp, "NOPE") == null); - try testing.expect(getenv(envp, "PAT") == null); -} - -test "pathCandidates: PATH search" { - const gpa = testing.allocator; - const env = [_:null]?[*:0]const u8{ "PATH=/usr/local/bin::/usr/bin", "HOME=/h" }; - const cands = try pathCandidates(gpa, &env, "fish"); - defer { - for (cands) |c| gpa.free(c); - gpa.free(cands); - } - try testing.expectEqual(@as(usize, 3), cands.len); - try testing.expectEqualStrings("/usr/local/bin/fish", cands[0]); - try testing.expectEqualStrings("./fish", cands[1]); - try testing.expectEqualStrings("/usr/bin/fish", cands[2]); -} - -test "pathCandidates: slash means no search; default PATH" { - const gpa = testing.allocator; - const env = [_:null]?[*:0]const u8{"HOME=/h"}; - { - const cands = try pathCandidates(gpa, &env, "./bin/x"); - defer { - for (cands) |c| gpa.free(c); - gpa.free(cands); - } - try testing.expectEqual(@as(usize, 1), cands.len); - try testing.expectEqualStrings("./bin/x", cands[0]); - } - { - const cands = try pathCandidates(gpa, &env, "sh"); - defer { - for (cands) |c| gpa.free(c); - gpa.free(cands); - } - try testing.expectEqual(@as(usize, 3), cands.len); - try testing.expectEqualStrings("/usr/local/bin/sh", cands[0]); - try testing.expectEqualStrings("/bin/sh", cands[1]); - } - try testing.expectError(error.EmptyProgramName, pathCandidates(gpa, &env, "")); -} - -test "findInPath finds sh" { - const gpa = testing.allocator; - const env = [_:null]?[*:0]const u8{"PATH=/nonexistent:/bin:/usr/bin"}; - const p = try findInPath(gpa, &env, "sh"); - defer gpa.free(p); - try testing.expect(std.mem.endsWith(u8, p, "/sh")); - try testing.expectError(error.FileNotFound, findInPath(gpa, &env, "definitely-not-a-program-9player")); -} - -test "buildEnvp replaces NINEPLAYER_MOUNT" { - const gpa = testing.allocator; - const env = [_:null]?[*:0]const u8{ "A=1", "NINEPLAYER_MOUNT=/old", "B=2" }; - const out = try buildEnvp(gpa, &env, "/mnt/9p"); - defer { - gpa.free(std.mem.span(out[out.len - 1].?)); - gpa.free(out); - } - try testing.expectEqual(@as(usize, 3), out.len); - try testing.expectEqualStrings("A=1", std.mem.span(out[0].?)); - try testing.expectEqualStrings("B=2", std.mem.span(out[1].?)); - try testing.expectEqualStrings("NINEPLAYER_MOUNT=/mnt/9p", std.mem.span(out[2].?)); - try testing.expect(out[3] == null); - try testing.expectEqualStrings("/mnt/9p", getenv(out.ptr, "NINEPLAYER_MOUNT").?); -} - -test "decodeStatus" { - try testing.expectEqual(@as(u8, 0), decodeStatus(0)); - try testing.expectEqual(@as(u8, 7), decodeStatus(7 << 8)); - try testing.expectEqual(@as(u8, 255), decodeStatus(255 << 8)); - try testing.expectEqual(@as(u8, 128 + 9), decodeStatus(9)); // SIGKILL - try testing.expectEqual(@as(u8, 128 + 15), decodeStatus(15)); // SIGTERM -} - -test "ensureMountpoint: existing directory is accepted, plain file rejected" { - const gpa = testing.allocator; - try ensureMountpoint(gpa, "/tmp"); - try testing.expectError(error.Mountpoint, ensureMountpoint(gpa, "/proc/self/status")); -} diff --git a/9player/test/adv_bridge_hostile.py b/9player/test/adv_bridge_hostile.py deleted file mode 100755 index b842a1a..0000000 --- a/9player/test/adv_bridge_hostile.py +++ /dev/null @@ -1,487 +0,0 @@ -#!/usr/bin/env python3 -"""A scriptable, hostile 9P2000 server on a Unix socket (stdlib only). - -Usage: adv_bridge_hostile.py SOCKET MODE - -Serves a tiny in-memory tree: - /f "hello world\\n" - /d/g "in d\\n" - /fids reading it returns the number of fids currently bound - /big 1 MiB of pseudo-random bytes -plus create/write/remove/wstat so the scratch battery can run in `ok` mode. - -MODE selects one misbehaviour (see MODES below). Everything not covered by -the mode behaves normally, so 9player gets through version/attach/stat(root). -""" -import os -import random -import socket -import struct -import sys -import time - -NOTAG = 0xFFFF -NOFID = 0xFFFFFFFF -QTDIR = 0x80 -DMDIR = 0x80000000 - -Tversion, Rversion = 100, 101 -Tauth, Rauth = 102, 103 -Tattach, Rattach = 104, 105 -Rerror = 107 -Tflush, Rflush = 108, 109 -Twalk, Rwalk = 110, 111 -Topen, Ropen = 112, 113 -Tcreate, Rcreate = 114, 115 -Tread, Rread = 116, 117 -Twrite, Rwrite = 118, 119 -Tclunk, Rclunk = 120, 121 -Tremove, Rremove = 122, 123 -Tstat, Rstat = 124, 125 -Twstat, Rwstat = 126, 127 - -MODES = """ -ok behave (qid paths are recycled LIFO after remove, like many servers) -trunc Rread on /f: send half the frame, then close -short_frame Rread on /f: frame whose size field is 3 -huge_frame Rread on /f: frame whose size field is msize+1 -wrong_tag Rread on /f: reply carries tag+1 -wrong_type Tstat on /f: answer with an Rwalk -rread_big Rread on /f: count = requested+1 -rwalk_many Twalk to f: nwqid = nwname+1 -rwalk_zero Twalk to nope: Rwalk nwqid=0 instead of Rerror -rstat_garbage Tstat on /f: random bytes as the stat -rstat_overlong Tstat on /f: inner stat size disagrees with outer -dir_split Tread on /: a stat record split across two Rreads -dir_forever Tread on /: ignore offset, always return the same records -qid_collide every file and dir shares qid.path 7 (root keeps its own) -qid_zero every qid.path is 0, including the root -name_slash / has an entry "a/b" -name_empty / has an entry "" -name_huge / has an entry with a 60000-byte name -name_dots / lists "." and ".." too -rerror_big Twalk to nope: Rerror with 65535 bytes of text -extra_reply Rread on /f: an unsolicited Rclunk (tag 9) precedes the real reply -never Tread on /f: never reply (hang) -close_mid Tread on /f: close the socket without replying -renegotiate Tread on /f: an unsolicited Rversion precedes the real reply -length_max Tstat on /f: length = 2**64-1 -iounit_one Ropen: iounit = 1 -rwrite_big Rwrite: count = requested+1 -msize_tiny Rversion msize = 64 -version_unknown Rversion "unknown" -rename_fail Twstat with a new name always fails "file already exists" -slow every reply delayed 20 ms (for interrupt tests) -""" - - -def s8(x): return struct.pack('= 4: - n = struct.unpack('= n: - msg, buf = buf[:n], buf[n:] - out = self.handle(msg) - if out is None: - return # hang up / hang - if self.mode == 'slow': - time.sleep(0.02) - conn.sendall(out) - continue - data = conn.recv(65536) - if not data: - return - buf += data - - def handle(self, msg): - typ = msg[4] - tag = struct.unpack(' len(node.content): - node.content.extend(b'\0' * (off - len(node.content))) - node.content[off:off + len(data)] = data - node.mtime = int(time.time()) - n = len(data) + 1 if self.mode == 'rwrite_big' else len(data) - return self.frame(Rwrite, tag, s32(n)) - if typ == Tclunk: - fid = r.u32() - if fid not in self.fids: - return self.err(tag, 'unknown fid') - del self.fids[fid] - return self.frame(Rclunk, tag, b'') - if typ == Tremove: - fid = r.u32() - if fid not in self.fids: - return self.err(tag, 'unknown fid') - node = self.fids[fid][0] - del self.fids[fid] - if node is self.root: - return self.err(tag, 'cannot remove root') - if node.isdir and node.children: - return self.err(tag, 'directory not empty') - parent = self.find_parent(self.root, node) - if parent is not None: - del parent.children[node.name] - self.free_paths.append(node.path) - node.removed = True - return self.frame(Rremove, tag, b'') - if typ == Tstat: - fid = r.u32() - if fid not in self.fids: - return self.err(tag, 'unknown fid') - node = self.fids[fid][0] - if node is self.root.children.get('f'): - m = self.mode - if m == 'wrong_type': - return self.frame(Rwalk, tag, s16(0)) - if m == 'rstat_garbage': - junk = bytes([0xAB] * 60) - return self.frame(Rstat, tag, s16(len(junk)) + junk) - if m == 'rstat_overlong': - st = node.stat_bytes(self) - inner = st[2:] - return self.frame(Rstat, tag, s16(len(inner) + 5) + inner) - if m == 'length_max': - st = node.stat_bytes(self, length=2 ** 64 - 1) - return self.frame(Rstat, tag, s16(len(st)) + st) - st = node.stat_bytes(self) - return self.frame(Rstat, tag, s16(len(st)) + st) - if typ == Twstat: - fid = r.u32() - r.u16() - st = r.bytes(r.u16()) - if fid not in self.fids: - return self.err(tag, 'unknown fid') - node = self.fids[fid][0] - sr = Reader(st) - sr.u16(); sr.u32(); sr.bytes(13) - mode = sr.u32(); sr.u32(); mtime = sr.u32(); length = sr.u64() - name = sr.str() - if name and name != node.name: - if self.mode == 'rename_fail': - return self.err(tag, 'file already exists') - parent = self.find_parent(self.root, node) - if name in parent.children: - return self.err(tag, 'file already exists') - del parent.children[node.name] - node.name = name - parent.children[name] = node - if mode != 0xFFFFFFFF: - node.mode = mode & 0o777 - if mtime != 0xFFFFFFFF: - node.mtime = mtime - if length != 0xFFFFFFFFFFFFFFFF and not node.isdir: - if length < len(node.content): - del node.content[length:] - else: - node.content.extend(b'\0' * (length - len(node.content))) - return self.frame(Rwstat, tag, b'') - return self.err(tag, 'unsupported message') - - def find_parent(self, cur, node): - for c in cur.children.values(): - if c is node: - return cur - if c.isdir: - p = self.find_parent(c, node) - if p is not None: - return p - return None - - def readdir(self, tag, node, off, count): - recs = [] - if node is self.root: - m = self.mode - if m == 'name_slash': - recs.append(node.stat_bytes(self, name='a/b')) - if m == 'name_empty': - recs.append(node.stat_bytes(self, name='')) - if m == 'name_huge': - recs.append(node.stat_bytes(self, name='h' * 60000)) - if m == 'name_dots': - recs.append(node.stat_bytes(self, name='.')) - recs.append(node.stat_bytes(self, name='..')) - for c in node.children.values(): - recs.append(c.stat_bytes(self)) - blob = b''.join(recs) - if node is self.root and self.mode == 'dir_forever': - return self.frame(Rread, tag, s32(len(blob)) + blob) - if node is self.root and self.mode == 'dir_split': - # first read: up to the middle of the second record; second read: the rest - cut = len(recs[0]) + len(recs[1]) // 2 - if off == 0: - data = blob[:cut] - elif off == cut: - data = blob[cut:] - else: - data = b'' - return self.frame(Rread, tag, s32(len(data)) + data) - # 9P rule: offset 0 or previous offset+count; never split a record. - out = b'' - pos = 0 - for rec in recs: - if pos >= off and len(out) + len(rec) <= count: - out += rec - elif pos >= off: - break - pos += len(rec) - return self.frame(Rread, tag, s32(len(out)) + out) - - -class Reader: - def __init__(self, b): - self.b = b - self.i = 0 - - def bytes(self, n): - v = self.b[self.i:self.i + n] - self.i += n - return v - - def u8(self): return struct.unpack(' (introspect unused; part of zig build 9player-adv) -set -u -PLAYER=$(realpath "${1:?path to 9player}") -HERE=$(cd "$(dirname "$0")" && pwd) -SRV=$HERE/adv_bridge_hostile.py -TMP=$(mktemp -d "${TMPDIR:-/tmp}/9player-adv.XXXXXX") -M=/mnt/9p -FAILED=0 -PASSED=0 -SRVPID= - -cleanup() { [ -n "$SRVPID" ] && kill "$SRVPID" 2>/dev/null; pkill -f "adv_bridge_hostile.py $TMP" 2>/dev/null; rm -rf "$TMP"; } -trap cleanup EXIT - -if ! unshare -Urm true 2>/dev/null || [ ! -c /dev/fuse ]; then echo "SKIP: no user namespaces or /dev/fuse"; exit 0; fi - -pass() { PASSED=$((PASSED + 1)); echo "ok - $1"; } -fail() { FAILED=$((FAILED + 1)); echo "FAIL - $1"; shift; [ $# -gt 0 ] && printf ' %s\n' "$@"; } -expect_eq() { if [ "$2" = "$3" ]; then pass "$1"; else fail "$1" "expected: $(printf %q "$2")" "actual: $(printf %q "$3")"; fi; } -expect_contains() { case "$3" in *"$2"*) pass "$1" ;; *) fail "$1" "missing: $(printf %q "$2")" "in: $(printf %q "$3")" ;; esac; } - -start_server() { # mode - [ -n "$SRVPID" ] && { kill "$SRVPID" 2>/dev/null; wait "$SRVPID" 2>/dev/null; } - SOCK=$TMP/$1.sock - rm -f "$SOCK" - python3 "$SRV" "$SOCK" "$1" >"$TMP/$1.srv.out" 2>&1 "$TMP/stderr") - RC=$? - STDERR=$(cat "$TMP/stderr") -} - -# 9player must not die of a signal or panic. RC 124 = timeout(1) fired. -no_crash() { # name - if [ "$RC" -ge 128 ] || [ "$RC" -eq 124 ]; then fail "$1: 9player exit $RC" "$STDERR"; return; fi - case "$STDERR" in *panic*|*"Segmentation"*|*"integer overflow"*|*"reached unreachable"*|*"index out of bounds"*) fail "$1: crash text in stderr" "$STDERR";; *) pass "$1: no crash (exit $RC)";; esac -} - -echo "# sanity: the hostile server behaves in 'ok' mode" -run ok "cat $M/f"; expect_eq "ok: cat f" "hello world" "$OUT" -run ok "cat $M/d/g"; expect_eq "ok: nested" "in d" "$OUT" -run ok "cat $M/nope 2>&1 | sed 's/.*: //'"; expect_eq "ok: ENOENT" "No such file or directory" "$OUT" -run ok "head -c 1048576 $M/big | wc -c | grep -q 1048576 && echo yes"; expect_eq "ok: 1 MiB read matches" "yes" "$OUT" - -echo "# unlink + recreate with a recycled qid.path" -run ok "echo 1 > $M/a; rm $M/a; echo 2 > $M/a; cat $M/a; rm $M/a"; expect_eq "qid reuse: new content, not stale" "2" "$OUT" -run ok "echo 1 > $M/x; rm $M/x; mkdir $M/x; stat -c %F $M/x; rmdir $M/x"; expect_eq "qid reuse: file→dir on the same path" "directory" "$OUT" - -echo "# fids do not grow with the number of operations" -loop='i=0; while [ $i -lt N ]; do echo hi > M/t; cat M/t >/dev/null; mkdir M/dd; rmdir M/dd; rm M/t; i=$((i+1)); done; cat M/fids' -run ok "$(echo "$loop" | sed "s|N|20|; s|M/|$M/|g")"; a=$OUT -run ok "$(echo "$loop" | sed "s|N|200|; s|M/|$M/|g")"; b=$OUT -expect_eq "fids after 20 == after 200 iterations ($a)" "$a" "$b" -floop='i=0; while [ $i -lt N ]; do cat M/nope 2>/dev/null; echo x > M/fids 2>/dev/null; mkdir M/f 2>/dev/null; rm M/d 2>/dev/null; mv M/f M/d 2>/dev/null; i=$((i+1)); done; cat M/fids' -run ok "$(echo "$floop" | sed "s|N|20|; s|M/|$M/|g")"; a=$OUT -run ok "$(echo "$floop" | sed "s|N|200|; s|M/|$M/|g")"; b=$OUT -expect_eq "fids after 20 == after 200 failing iterations ($a)" "$a" "$b" - -echo "# rename over an existing file must not lose the target when the rename fails" -run rename_fail "echo A > $M/a; echo B > $M/b; mv $M/a $M/b 2>/dev/null; echo mv=\$?; cat $M/b; cat $M/a" -expect_contains "rename_fail: mv reports failure" "mv=1" "$OUT" -expect_contains "rename_fail: target b still has its content" "B" "$OUT" -expect_contains "rename_fail: source a still has its content" "A" "$OUT" - -echo "# protocol violations on a data read must yield an error, not a crash" -for mode in trunc short_frame huge_frame wrong_tag rread_big extra_reply close_mid renegotiate rwrite_big; do - if [ "$mode" = rwrite_big ]; then script="dd if=/dev/zero of=$M/f bs=10 count=1 2>&1; echo status=\$?"; else script="cat $M/f 2>&1; echo status=\$?"; fi - run "$mode" "$script" - no_crash "$mode" - expect_contains "$mode: child sees an error" "status=1" "$OUT" - case "$OUT" in *"Input/output error"*|*"not connected"*) pass "$mode: EIO/ENOTCONN";; *) fail "$mode: errno text" "$OUT";; esac -done - -echo "# protocol violations on lookup/stat" -for mode in wrong_type rwalk_many rstat_garbage rstat_overlong; do - run "$mode" "stat -c %s $M/f 2>&1; echo status=\$?" - no_crash "$mode" - expect_contains "$mode: child sees an error" "status=1" "$OUT" -done -run rwalk_zero "cat $M/nope 2>&1; echo status=\$?; cat $M/f 2>&1" -no_crash "rwalk_zero" -expect_contains "rwalk_zero: child sees an error" "status=1" "$OUT" -run rerror_big "cat $M/nope 2>&1; echo status=\$?; cat $M/f" -no_crash "rerror_big" -expect_contains "rerror_big: child sees an error" "status=1" "$OUT" -expect_contains "rerror_big: session survives a 64 KiB Rerror" "hello world" "$OUT" - -echo "# hostile stat contents" -run length_max "stat -c '%s %b' $M/f 2>&1; echo status=\$?" -no_crash "length_max" -expect_contains "length_max: stat succeeds with a saturated block count" "status=0" "$OUT" -run iounit_one "cat $M/f; head -c 3000 $M/big | wc -c" -no_crash "iounit_one" -expect_contains "iounit_one: read still complete" "hello world" "$OUT" -expect_contains "iounit_one: 3000 bytes" "3000" "$OUT" - -echo "# hostile directory listings" -run dir_split "ls $M 2>&1; echo status=\$?; cat $M/f" -no_crash "dir_split" -expect_contains "dir_split: readdir fails with EIO" "Input/output error" "$OUT" -expect_contains "dir_split: session survives" "hello world" "$OUT" -run dir_forever "ls $M 2>&1 | tail -c 200; echo status=\$?; cat $M/f; grep VmRSS /proc/\$PPID/status" -no_crash "dir_forever" -expect_contains "dir_forever: infinite directory is cut off with EIO" "Input/output error" "$OUT" -expect_contains "dir_forever: session survives" "hello world" "$OUT" -for mode in name_slash name_empty name_huge name_dots; do - run "$mode" "ls -a $M | tr '\n' ' '; echo; cat $M/f" - no_crash "$mode" - expect_contains "$mode: listing still works" "big d f fids" "$OUT" - expect_contains "$mode: file readable" "hello world" "$OUT" - case "$mode" in - name_slash) expect_eq "$mode: slash entry dropped" "" "$(printf '%s' "$OUT" | grep -o 'a/b')";; - name_dots) expect_eq "$mode: exactly one . and one .." "1 1" "$(printf '%s %s' "$(printf '%s\n' "$OUT" | head -1 | tr ' ' '\n' | grep -c '^\.$')" "$(printf '%s\n' "$OUT" | head -1 | tr ' ' '\n' | grep -c '^\.\.$')")";; - esac -done - -echo "# qid collisions" -run qid_collide "cat $M/f; cat $M/d/g; ls $M/d; stat -c %i $M/f $M/d 2>&1; echo status=\$?" -no_crash "qid_collide" -expect_contains "qid_collide: reads work" "hello world" "$OUT" -run qid_zero "cat $M/f; ls $M | tr '\n' ' '; echo; cat $M/d/g; echo status=\$?" -no_crash "qid_zero" -expect_contains "qid_zero: file with root's qid.path is still a readable file" "hello world" "$OUT" -expect_contains "qid_zero: root still lists" "big d f fids" "$OUT" -expect_contains "qid_zero: nested file readable" "in d" "$OUT" - -echo "# version negotiation" -run version_unknown "echo ran" -expect_eq "version_unknown: 9player refuses (125)" "125" "$RC" -run msize_tiny "cat $M/f 2>&1; echo status=\$?" -no_crash "msize_tiny" - -echo "# a server that never replies" -start_server never -# SIGTERM is forwarded to the child; once the child is gone 9player must leave the -# pending 9P reply behind and exit even though the server stays silent. -timeout -s TERM 3 "$PLAYER" --unix "$SOCK" -- sh -c "cat $M/f; echo unreachable" >"$TMP/never.out" 2>"$TMP/never.err" & -TPID=$! -sleep 4 -if kill -0 "$TPID" 2>/dev/null; then - fail "never: SIGTERM did not end 9player while a reply was outstanding"; kill -9 "$TPID" -else - pass "never: SIGTERM ends 9player even while the server is silent" -fi -wait "$TPID" 2>/dev/null -# Without a signal the mount hangs (documented v1 limitation) until the server dies. -timeout 30 "$PLAYER" --unix "$SOCK" -- sh -c "cat $M/f; echo unreachable" >"$TMP/never.out" 2>"$TMP/never.err" & -TPID=$! -sleep 1.5 -if kill -0 "$TPID" 2>/dev/null; then - pass "never: mount hangs while the server is silent (documented v1 limitation)" - kill "$SRVPID"; wait "$SRVPID" 2>/dev/null; SRVPID= - for _ in $(seq 1 50); do kill -0 "$TPID" 2>/dev/null || break; sleep 0.1; done - if kill -0 "$TPID" 2>/dev/null; then fail "never: 9player still alive after its server died"; kill -9 "$TPID"; else pass "never: killing the server unblocks 9player"; fi -else - wait "$TPID"; fail "never: 9player exited early ($?)" "$(cat "$TMP/never.err")" -fi - -echo "# interrupting a slow read (INTERRUPT must not confuse reply matching)" -run slow "cat $M/big > /dev/null; cat $M/f; cat $M/fids" --msize 8192 -base=$(printf '%s\n' "$OUT" | tail -1) -run slow "(cat $M/big > /dev/null & sleep 0.3; kill -INT \$!; wait \$!) 2>/dev/null; cat $M/f; cat $M/fids" --msize 8192 -no_crash "slow" -expect_contains "slow: read after interrupted read works" "hello world" "$OUT" -expect_eq "slow: fids after an interrupted read == after a complete one ($base)" "$base" "$(printf '%s\n' "$OUT" | tail -1)" - -echo -echo "passed=$PASSED failed=$FAILED" -[ "$FAILED" -eq 0 ] diff --git a/9player/test/adv_bridge_semantics.sh b/9player/test/adv_bridge_semantics.sh deleted file mode 100755 index f13fd57..0000000 --- a/9player/test/adv_bridge_semantics.sh +++ /dev/null @@ -1,205 +0,0 @@ -#!/usr/bin/env bash -# FUSE semantics through the bridge against introspect's /scratch tree. -# Usage: bash 9player/test/adv_bridge_semantics.sh <9player> (part of zig build 9player-adv) -set -u -PLAYER=$(realpath "${1:?path to 9player}") -INTROSPECT=$(realpath "${2:?path to introspect}") -TMP=$(mktemp -d "${TMPDIR:-/tmp}/9player-sem.XXXXXX") -M=/mnt/9p -S=$M/scratch -FAILED=0 -PASSED=0 -SRVPID= -cleanup() { [ -n "$SRVPID" ] && kill "$SRVPID" 2>/dev/null; rm -rf "$TMP"; } -trap cleanup EXIT -if ! unshare -Urm true 2>/dev/null || [ ! -c /dev/fuse ]; then echo "SKIP: no user namespaces or /dev/fuse"; exit 0; fi - -pass() { PASSED=$((PASSED + 1)); echo "ok - $1"; } -fail() { FAILED=$((FAILED + 1)); echo "FAIL - $1"; shift; [ $# -gt 0 ] && printf ' %s\n' "$@"; } -expect_eq() { if [ "$2" = "$3" ]; then pass "$1"; else fail "$1" "expected: $(printf %q "$2")" "actual: $(printf %q "$3")"; fi; } -expect_contains() { case "$3" in *"$2"*) pass "$1" ;; *) fail "$1" "missing: $(printf %q "$2")" "in: $(printf %q "$3")" ;; esac; } - -SOCK=$TMP/i.sock -"$INTROSPECT" --unix "$SOCK" >"$TMP/srv.out" 2>&1 & -SRVPID=$! -for _ in $(seq 1 100); do [ -S "$SOCK" ] && break; sleep 0.02; done -# Each run is a fresh session; state persists in the server, so tests clean up after themselves. -run() { OUT=$(timeout 120 "$PLAYER" --unix "$SOCK" "${EXTRA[@]}" -- sh -c "$1" 2>"$TMP/stderr"); RC=$?; STDERR=$(cat "$TMP/stderr"); } -EXTRA=() -py() { run "python3 - <<'PYEOF' -$1 -PYEOF"; } - -echo "# open/create flags" -py " -import os, errno -p='$S/excl' -fd=os.open(p, os.O_CREAT|os.O_WRONLY, 0o644); os.write(fd, b'x'); os.close(fd) -try: - os.open(p, os.O_CREAT|os.O_EXCL|os.O_WRONLY, 0o644); print('no error') -except OSError as e: print(errno.errorcode[e.errno]) -os.unlink(p) -fd=os.open('$S/ro', os.O_CREAT|os.O_RDONLY, 0o644); print(os.read(fd, 10)); os.close(fd) -print(os.path.exists('$S/ro')); os.unlink('$S/ro') -" -expect_eq "O_EXCL on an existing file is EEXIST" "EEXIST" "$(printf '%s\n' "$OUT" | sed -n 1p)" -expect_eq "create with O_RDONLY works and reads empty" $'b\'\'\nTrue' "$(printf '%s\n' "$OUT" | sed -n 2,3p)" - -run "mkdir -m 700 $S/m7 && stat -c %a $S/m7; chmod 755 $S/m7 && stat -c %a $S/m7; rmdir $S/m7" -expect_eq "mkdir -m 700 then chmod 755" $'700\n755' "$OUT" -run "echo x > $S/c && chmod 600 $S/c && stat -c %a $S/c; chmod 444 $S/c && stat -c %a $S/c; rm -f $S/c" -expect_eq "chmod on a file" $'600\n444' "$OUT" -run "echo x > $S/t && touch -d @1000000000 $S/t && stat -c %Y $S/t; touch $S/t && [ \$(stat -c %Y $S/t) -gt 1000000000 ] && echo now; rm $S/t" -expect_eq "utimes (explicit) and touch (now)" $'1000000000\nnow' "$OUT" -run "echo abc > $S/tr && truncate -s 10 $S/tr && stat -c %s $S/tr && od -An -c $S/tr | tr -s ' ' | tr -d '\n'; echo; rm $S/tr" -expect_eq "truncate to larger zero-fills" $'10\n a b c \\n \\0 \\0 \\0 \\0 \\0 \\0' "$OUT" -run "echo a > $S/ap && echo b >> $S/ap && echo c >> $S/ap && cat $S/ap | tr '\n' ' '; rm $S/ap" -expect_eq "shell append" "a b c " "$OUT" -py " -import os -p='$S/ap2' -f1=os.open(p, os.O_CREAT|os.O_WRONLY|os.O_APPEND, 0o644) -f2=os.open(p, os.O_WRONLY|os.O_APPEND) -os.write(f1, b'one '); os.write(f2, b'two '); os.write(f1, b'three') -os.close(f1); os.close(f2) -print(open(p).read()); os.unlink(p) -" -expect_eq "O_APPEND from two descriptors interleaves in order" "one two three" "$OUT" -run "echo 0123456789 > $S/tt && (echo X > $S/tt) && cat $S/tt && stat -c %s $S/tt; rm $S/tt" -expect_eq "O_TRUNC (atomic_o_trunc) truncates before write" $'X\n2' "$OUT" - -echo "# reads at the edges" -run "printf hello > $S/e; dd if=$S/e bs=1 skip=100 count=5 2>/dev/null | wc -c; head -c 0 $S/e | wc -c; dd if=/dev/null of=$S/e bs=1 count=0 conv=notrunc 2>/dev/null; cat $S/e; echo; rm $S/e" -expect_eq "read past EOF is 0 bytes; 0-byte read/write are no-ops" $'0\n0\nhello' "$OUT" -head -c 4194304 /dev/urandom >"$TMP/four" -SUM=$(sha256sum <"$TMP/four" | cut -d' ' -f1) -run "dd if=$TMP/four of=$S/four bs=4M status=none && dd if=$S/four bs=4M status=none | sha256sum | cut -d' ' -f1; stat -c %s $S/four; rm $S/four" -expect_eq "4 MiB single-request dd round trip" "$SUM"$'\n4194304' "$OUT" -py " -import os -p='$S/lseek' -open(p,'w').write('0123456789') -f=open(p,'rb'); f.seek(0, 2); print(f.tell()); f.seek(-3, 2); print(f.read()); f.close() -os.unlink(p) -" -expect_eq "lseek SEEK_END on a direct_io file" $'10\nb\'789\'' "$OUT" - -echo "# unlink of an open file" -py " -import os -p='$S/unl' -fd=os.open(p, os.O_CREAT|os.O_RDWR, 0o644) -os.write(fd, b'before') -os.unlink(p) -print(os.path.exists(p)) -os.lseek(fd, 0, 0); print(os.read(fd, 100)) -os.write(fd, b'-after'); os.lseek(fd, 0, 0); print(os.read(fd, 100)) -os.close(fd) -" -expect_eq "read/write through the fd after unlink" $'False\nb\'before\'\nb\'before-after\'' "$OUT" - -echo "# rename" -run "echo A > $S/ra; echo B > $S/rb; mv $S/ra $S/rb && cat $S/rb; ls $S | tr '\n' ' '; echo; rm $S/rb" -expect_eq "rename over an existing file replaces it, no leftovers" $'A\nrb ' "$OUT" -run "mkdir $S/rd1 && echo x > $S/rd1/f && mv $S/rd1 $S/rd2 && cat $S/rd2/f && ls $S/rd2; rm -r $S/rd2; ls $S | wc -l" -expect_eq "rename of a directory" $'x\nf\n0' "$OUT" -run "mkdir $S/e1 $S/e2 && mv -T $S/e1 $S/e2 && ls $S | tr '\n' ' '; rmdir $S/e2" -expect_eq "rename dir over an empty dir" "e2 " "$OUT" -run "mkdir $S/n1 $S/n2 && echo x > $S/n2/f && mv -T $S/n1 $S/n2 2>&1 | sed 's/.*: //'; rm -r $S/n1 $S/n2" -expect_eq "rename dir over a non-empty dir is ENOTEMPTY" "Directory not empty" "$OUT" -py " -import os -p='$S/same'; open(p,'w').write('x'); os.rename(p, p); print(open(p).read()); os.unlink(p) -" -expect_eq "rename onto itself is a no-op" "x" "$OUT" -py " -import os, ctypes, errno -libc = ctypes.CDLL(None, use_errno=True) -a, b = b'$S/nra', b'$S/nrb' -open(a,'w').write('A'); open(b,'w').write('B') -r = libc.renameat2(-100, a, -100, b, 1) # RENAME_NOREPLACE -print('rc', r, errno.errorcode.get(ctypes.get_errno())) -print(open(b).read()) -os.unlink(a); os.unlink(b) -" -expect_eq "RENAME_NOREPLACE keeps the target" $'rc -1 EEXIST\nB' "$OUT" - -echo "# directories" -run "mkdir $S/many && cd $S/many && i=0; while [ \$i -lt 5000 ]; do : > f\$i; i=\$((i+1)); done; ls | wc -l; ls -l | wc -l; grep VmRSS /proc/\$PPID/status | awk '{print \$2}' > $TMP/rss1; rm -f $S/many/*; rmdir $S/many; ls | wc -l; grep VmRSS /proc/\$PPID/status | awk '{print \$2}' > $TMP/rss2" -expect_eq "5000 entries: ls and ls -l" $'5000\n5001\n0' "$OUT" -r1=$(cat "$TMP/rss1"); r2=$(cat "$TMP/rss2") -if [ "$r2" -le $((r1 + 2048)) ]; then pass "RSS after cleanup ($r2 KiB) <= after listing ($r1 KiB)+2 MiB"; else fail "RSS grew after cleanup: $r1 -> $r2 KiB"; fi -py " -import os -d='$S/chg'; os.mkdir(d) -for i in range(50): open(f'{d}/a{i}','w').close() -it = os.scandir(d); first = next(it).name -for i in range(3000): open(f'{d}/b{i}','w').close() -rest = [e.name for e in it] -print(first[0], len(rest) >= 49, len(set(rest)) == len(rest)) -for n in os.listdir(d): os.unlink(f'{d}/{n}') -os.rmdir(d) -" -expect_eq "readdir of a directory that changes mid-iteration" "a True True" "$OUT" -run "ls $M/.. > /dev/null && echo ok; stat -c %i $M $M/. $M/scratch/..; cd $M/scratch && ls .. | grep -c scratch" -expect_eq ".. of the root and of a subdir" $'ok\n1\n1\n1\n1' "$OUT" -run "cd $M && find . -type d | wc -l && find . -type f | head -1 && find $S -type f | wc -l" -expect_contains "find -type works" "./README" "$OUT" -run "stat -f -c '%T %S %l' $M; df -P $M | tail -1 | awk '{print \$1}'; sync -f $M && echo synced; sync && echo synced2" -expect_eq "statfs, df, syncfs, sync" $'fuse 4096 255\n9player\nsynced\nsynced2' "$OUT" - -echo "# server refusals keep their errno through the error path" -run "echo x > $M/build/zig_version; a=\$?; mkdir $M/build/x 2>/dev/null; b=\$?; rm $M/README 2>/dev/null; c=\$?; rmdir $M/build 2>/dev/null; d=\$?; echo \$a\$b\$c\$d" 2>/dev/null -expect_eq "open-for-write / mkdir / rm / rmdir on read-only nodes all fail (errno preserved through error path)" "1111" "$(printf '%s\n' "$OUT" | tail -1)" - -echo "# unsupported operations fail cleanly" -run "echo x > $S/l1; ln $S/l1 $S/l2 2>&1 | sed 's/.*: //'; ln -s l1 $S/l3 2>&1 | sed 's/.*: //'; mkfifo $S/p 2>&1 | sed 's/.*: //'; ls $S | tr '\n' ' '; echo; rm $S/l1" -# The kernel turns ENOSYS from LINK into EPERM (fuse_link); symlink/mknod keep ENOSYS. -expect_eq "link/symlink/mknod fail cleanly" $'Operation not permitted\nFunction not implemented\nFunction not implemented\nl1 ' "$OUT" -run "echo x > $S/x1; setfattr -n user.a -v 1 $S/x1 2>&1 | sed 's/.*: //'; getfattr -n user.a $S/x1 2>&1 | sed 's/.*: //'; getfattr -d $S/x1 2>&1 | sed 's/.*: //'; rm $S/x1" -expect_eq "xattr ops are EOPNOTSUPP" $'Operation not supported\nOperation not supported\nOperation not supported' "$OUT" -py " -import os, fcntl, mmap, errno -p='$S/mm'; open(p,'w').write('mapme') -fd=os.open(p, os.O_RDWR) -fcntl.flock(fd, fcntl.LOCK_EX); fcntl.flock(fd, fcntl.LOCK_UN); fcntl.lockf(fd, fcntl.LOCK_EX); fcntl.lockf(fd, fcntl.LOCK_UN); print('locks ok') -try: - m = mmap.mmap(fd, 5); print('shared', bytes(m)); m.close() -except OSError as e: print('shared', errno.errorcode[e.errno]) -try: - m = mmap.mmap(fd, 5, flags=mmap.MAP_PRIVATE, prot=mmap.PROT_READ); print('private', bytes(m)); m.close() -except OSError as e: print('private', errno.errorcode[e.errno]) -os.close(fd); os.unlink(p) -" -expect_contains "flock/lockf work (local locks)" "locks ok" "$OUT" -case "$OUT" in *"shared ENODEV"*|*"shared b'mapme'"*) pass "shared mmap: clean result ($(printf '%s\n' "$OUT" | sed -n 2p))";; *) fail "shared mmap" "$OUT";; esac -expect_contains "private mmap reads the file" "private b'mapme'" "$OUT" - -echo "# tools" -mkdir -p "$TMP/tree/sub/deeper"; echo one > "$TMP/tree/a"; echo two > "$TMP/tree/sub/b"; head -c 70000 /dev/urandom > "$TMP/tree/sub/deeper/blob"; chmod 640 "$TMP/tree/a" -run "cp -a $TMP/tree $S/tree 2>&1; diff -r $TMP/tree $S/tree && echo same; stat -c %a $S/tree/a; cp -a $S/tree $TMP/back && diff -r $TMP/tree $TMP/back && echo back; rm -r $S/tree" -expect_eq "cp -a there and back" $'same\n640\nback' "$OUT" -run "cd $TMP && tar cf $S/t.tar tree && cd $S && mkdir tx && tar xf t.tar -C tx && diff -r $TMP/tree tx/tree && echo tar-ok; rm -r $S/tx $S/t.tar" -expect_eq "tar into and out of the mount" "tar-ok" "$OUT" -run "rsync -a $TMP/tree/ $S/rs/ && diff -r $TMP/tree $S/rs && echo rsync-ok; sleep 1.1; echo mod > $TMP/tree/a; rsync -a $TMP/tree/ $S/rs/ && cat $S/rs/a; rm -r $S/rs" -expect_eq "rsync -a twice" $'rsync-ok\nmod' "$OUT" -run "cd $S && mkdir repo && cd repo && git init -q . && git config user.email a@b && git config user.name n && echo hi > f && git add f && git commit -qm init && git log --oneline | wc -l && git status --porcelain | wc -l; cd $S && rm -rf repo; ls $S | wc -l" -expect_eq "git init/add/commit inside the mount" $'1\n0\n0' "$OUT" - -echo "# --no-direct-io" -EXTRA=(--no-direct-io) -run "cp $TMP/four $S/nd && cmp $TMP/four $S/nd && echo same; stat -c %s $S/nd; rm $S/nd" -expect_eq "no-direct-io: 4 MiB round trip" $'same\n4194304' "$OUT" -[ "$RC" -eq 0 ] || echo " stderr: $STDERR" -# Buffered writes are per-page without a writeback cache (kernel behaviour); the -# point here is only that a large buffered write is delivered intact. -EXTRA=(--no-direct-io) -head -c 262144 /dev/urandom > "$TMP/w" -WSUM=$(sha256sum <"$TMP/w" | cut -d' ' -f1) -run "cp $TMP/w $S/w && sha256sum < $S/w | cut -d' ' -f1; stat -c %s $S/w; rm $S/w" -expect_eq "no-direct-io: 256 KiB buffered write is intact" "$WSUM"$'\n262144' "$OUT" -EXTRA=() - -echo -echo "passed=$PASSED failed=$FAILED" -[ "$FAILED" -eq 0 ] diff --git a/9player/test/adv_bridge_stress.sh b/9player/test/adv_bridge_stress.sh deleted file mode 100755 index 3d1203a..0000000 --- a/9player/test/adv_bridge_stress.sh +++ /dev/null @@ -1,64 +0,0 @@ -#!/usr/bin/env bash -# Resource and concurrency stress through the bridge against introspect. -# Usage: bash 9player/test/adv_bridge_stress.sh <9player> (~1-2 min; part of zig build 9player-adv) -set -u -PLAYER=$(realpath "${1:?path to 9player}") -INTROSPECT=$(realpath "${2:?path to introspect}") -TMP=$(mktemp -d "${TMPDIR:-/tmp}/9player-stress.XXXXXX") -M=/mnt/9p -S=$M/scratch -FAILED=0 -PASSED=0 -SRVPID= -cleanup() { [ -n "$SRVPID" ] && kill "$SRVPID" 2>/dev/null; rm -rf "$TMP"; } -trap cleanup EXIT -if ! unshare -Urm true 2>/dev/null || [ ! -c /dev/fuse ]; then echo "SKIP: no user namespaces or /dev/fuse"; exit 0; fi -pass() { PASSED=$((PASSED + 1)); echo "ok - $1"; } -fail() { FAILED=$((FAILED + 1)); echo "FAIL - $1"; shift; [ $# -gt 0 ] && printf ' %s\n' "$@"; } -expect_eq() { if [ "$2" = "$3" ]; then pass "$1"; else fail "$1" "expected: $(printf %q "$2")" "actual: $(printf %q "$3")"; fi; } - -SOCK=$TMP/i.sock -"$INTROSPECT" --unix "$SOCK" >"$TMP/srv.out" 2>&1 & -SRVPID=$! -for _ in $(seq 1 100); do [ -S "$SOCK" ] && break; sleep 0.02; done -run() { OUT=$(timeout 600 "$PLAYER" --unix "$SOCK" -- sh -c "$1" 2>"$TMP/stderr"); RC=$?; STDERR=$(cat "$TMP/stderr"); } - -echo "# 100k+ 9P operations in one session; RSS must plateau" -# Each iteration: create+write+close, open+read+close, unlink, plus a failing lookup: ~15 RPCs. -run "rss() { grep VmRSS /proc/\$PPID/status | awk '{print \$2}'; } -i=0; while [ \$i -lt 8000 ]; do echo \$i > $S/s; cat $S/s > /dev/null; rm $S/s; cat $S/none 2>/dev/null; i=\$((i+1)); if [ \$i -eq 2000 ]; then rss; fi; done; rss; ls $S | wc -l" -r1=$(printf '%s\n' "$OUT" | sed -n 1p); r2=$(printf '%s\n' "$OUT" | sed -n 2p); left=$(printf '%s\n' "$OUT" | sed -n 3p) -expect_eq "scratch left clean" "0" "$left" -if [ -n "$r1" ] && [ -n "$r2" ] && [ "$r2" -le $((r1 + 4096)) ]; then pass "RSS at 2000 iterations = $r1 KiB, at 8000 = $r2 KiB"; else fail "RSS grows: $r1 -> $r2 KiB" "$STDERR"; fi -case "$STDERR" in *leak*) fail "allocator reported leaks" "$STDERR";; *) pass "no leak report from the debug allocator";; esac - -echo "# eight processes hammering the mount concurrently" -run "mkdir $S/par; for p in 1 2 3 4 5 6 7 8; do ( - d=$S/par/p\$p; mkdir \$d; i=0; bad=0 - while [ \$i -lt 300 ]; do - printf '%s-%s' \$p \$i > \$d/f\$((i % 7)); v=\$(cat \$d/f\$((i % 7))); [ \"\$v\" = \"\$p-\$i\" ] || bad=\$((bad+1)) - mkdir \$d/dd; rmdir \$d/dd; ls \$d > /dev/null; i=\$((i+1)) - done; rm -r \$d; echo \$p:\$bad ) & done; wait; ls $S/par | wc -l; rmdir $S/par" -expect_eq "all workers verified their own data" "1:0 2:0 3:0 4:0 5:0 6:0 7:0 8:0" "$(printf '%s\n' "$OUT" | grep ':' | sort | tr '\n' ' ' | sed 's/ $//')" -expect_eq "parallel tree fully removed" "0" "$(printf '%s\n' "$OUT" | grep -v ':')" - -echo "# a process killed mid-read and mid-write" -run "head -c 8000000 /dev/urandom > $S/kb; (cat $S/kb > /dev/null & sleep 0.05; kill -9 \$!; wait \$!) 2>/dev/null; (cat /dev/zero > $S/kw & sleep 0.05; kill -9 \$!; wait \$!) 2>/dev/null; sha256sum < $S/kb | cut -c1-8 > /dev/null && echo readable; [ -f $S/kw ] && echo written; rm $S/kb $S/kw; ls $S | wc -l" -expect_eq "survives SIGKILL mid-read/mid-write" $'readable\nwritten\n0' "$OUT" - -echo "# many open handles at once (fh counter, fid table)" -run "python3 - <<'EOF' -import os -d='$S/fh'; os.mkdir(d) -fds=[] -for i in range(1500): - fd=os.open(f'{d}/h{i%50}', os.O_CREAT|os.O_RDWR, 0o644); os.write(fd, b'z'); fds.append(fd) -for fd in fds: os.close(fd) -for i in range(50): os.unlink(f'{d}/h{i}') -os.rmdir(d); print('ok') -EOF" -expect_eq "1500 simultaneous handles" "ok" "$OUT" - -echo -echo "passed=$PASSED failed=$FAILED" -[ "$FAILED" -eq 0 ] diff --git a/9player/test/adv_ns_process.sh b/9player/test/adv_ns_process.sh deleted file mode 100755 index 4ce0ae8..0000000 --- a/9player/test/adv_ns_process.sh +++ /dev/null @@ -1,202 +0,0 @@ -#!/usr/bin/env bash -# Adversarial regression tests for 9player/src/ns.zig and 9player/src/main.zig: process, -# namespace, signal and CLI handling. Real namespaces, real FUSE. -# Usage: bash 9player/test/adv_ns_process.sh <9player> (part of zig build 9player-adv) -# Exit 0 on success (or when the machine cannot run the tests), 1 on failure. -set -u - -PLAYER=$(realpath "${1:?path to 9player}") -INTROSPECT=$(realpath "${2:?path to introspect}") -# Unix socket paths are limited to ~107 bytes; keep the temp dir short. -TMP=$(mktemp -d "${TMPDIR:-/tmp}/9padv.XXXXXX") -PIDS=() -FAILED=0 -PASSED=0 - -cleanup() { - for p in "${PIDS[@]:-}"; do [ -n "$p" ] && kill "$p" 2>/dev/null; done - rm -rf "$TMP" -} -trap cleanup EXIT - -if ! unshare -Urm true 2>/dev/null; then echo "SKIP: unprivileged user namespaces unavailable"; exit 0; fi -if [ ! -c /dev/fuse ]; then echo "SKIP: /dev/fuse missing"; exit 0; fi - -pass() { PASSED=$((PASSED + 1)); echo "ok - $1"; } -fail() { FAILED=$((FAILED + 1)); echo "FAIL - $1"; shift; [ $# -gt 0 ] && printf ' %s\n' "$@"; } -expect_eq() { if [ "$2" = "$3" ]; then pass "$1"; else fail "$1" "expected: $(printf %q "$2")" "actual: $(printf %q "$3")"; fi; } -expect_contains() { case "$3" in *"$2"*) pass "$1" ;; *) fail "$1" "missing: $(printf %q "$2")" "in: $(printf %q "$3")" ;; esac; } - -SOCK=$TMP/s -"$INTROSPECT" --unix "$SOCK" & -PIDS+=($!) -for _ in $(seq 1 100); do [ -S "$SOCK" ] && break; sleep 0.05; done -[ -S "$SOCK" ] || { echo "introspect did not create $SOCK"; exit 1; } -MI_BEFORE=$(grep -v " $TMP" /proc/self/mountinfo | sort) - -TIMEOUT=$(command -v timeout) -run() { "$TIMEOUT" 60 "$PLAYER" --unix "$SOCK" "$@"; } - -echo "# CLI" -expect_eq "--help goes to stdout, exit 0" "Usage: 9player" "$(run --help 2>/dev/null | head -1 | cut -c1-14; )" -expect_eq "--help exit code" "0" "$("$PLAYER" --help >/dev/null 2>&1; echo $?)" -expect_eq "--version on stdout" "9player" "$("$PLAYER" --version 2>/dev/null | cut -d' ' -f1)" -expect_eq "single-dash typo is a usage error, not a program" "125" "$(run -mount /x -- true 2>/dev/null; echo $?)" -expect_contains "single-dash typo message" "unknown option -mount" "$(run -mount /x -- true 2>&1)" -expect_eq "--unix= empty is a usage error" "125" "$("$PLAYER" --unix= -- true 2>/dev/null; echo $?)" -expect_contains "--unix= message" "socket path" "$("$PLAYER" --unix= -- true 2>&1)" -expect_eq "--mount '' is a usage error" "125" "$(run --mount '' -- true 2>/dev/null; echo $?)" -expect_eq "--msize huge rejected" "125" "$(run --msize 4294967295 -- true 2>/dev/null; echo $?)" -expect_eq "--msize 16 MiB accepted" "ok" "$(run --msize 16777216 -- sh -c 'echo ok')" -expect_contains "empty program name is reported" "empty program name" "$(run -- '' 2>&1)" -expect_eq "empty program name exit" "125" "$(run -- '' 2>/dev/null; echo $?)" -expect_eq "empty \$SHELL falls back to /bin/sh" "0" "$(SHELL= run -- /dev/null 2>&1; echo $?)" -expect_eq "--fd with a closed descriptor fails early" "125" "$("$PLAYER" --fd 987 -- true 2>/dev/null; echo $?)" -expect_contains "--fd bad descriptor message" "--fd 987: EBADF" "$("$PLAYER" --fd 987 -- true 2>&1)" - -echo "# exec failures" -expect_eq "not found is 127" "127" "$(run -- no-such-program-9player 2>/dev/null; echo $?)" -expect_eq "PATH element that is a file: still 127" "127" "$(PATH=/etc/passwd run -- true 2>/dev/null; echo $?)" -expect_contains "PATH element that is a file: message" "exec true: E" "$(PATH=/etc/passwd run -- true 2>&1)" -printf '#!/bin/sh\necho no\n' >"$TMP/nx"; chmod 644 "$TMP/nx" -expect_eq "non-executable is 126" "126" "$(run -- "$TMP/nx" 2>/dev/null; echo $?)" -mkdir -p "$TMP/p1" "$TMP/p2"; cp "$TMP/nx" "$TMP/p1/prog"; printf '#!/bin/sh\necho right\n' >"$TMP/p2/prog"; chmod 755 "$TMP/p2/prog" -expect_eq "non-executable first in PATH, executable later" "right" "$(PATH=$TMP/p1:$TMP/p2 run -- prog)" -expect_eq "non-executable only in PATH is 126" "126" "$(PATH=$TMP/p1 run -- prog 2>/dev/null; echo $?)" -expect_eq "argv[0] preserved" "sh" "$(run -- sh -c 'echo $0')" -expect_eq "PATH unset uses default" "ok" "$(env -u PATH "$PLAYER" --unix "$SOCK" -- sh -c 'echo ok')" - -echo "# fd hygiene" -# 9player passes inherited descriptors through untouched, so compare with what a -# plain child of this script sees (the runner may itself hold extra fds). -FD_LIST='ls /proc/self/fd | grep -v "^3$" | sort -n | tr "\n" " " | sed "s/ $//"' -FD_BASE=$(sh -c "$FD_LIST") -expect_eq "no extra fds in the program (unix)" "$FD_BASE" "$(run -- sh -c "$FD_LIST")" -expect_eq "no extra fds in the program (spawn)" "$FD_BASE" "$("$TIMEOUT" 60 "$PLAYER" --spawn "$INTROSPECT --stdio" -- sh -c "$FD_LIST")" -expect_eq "--fd transport does not leak into the program" "0 1 2" "$(python3 - "$PLAYER" "$SOCK" <<'EOF' -import socket, subprocess, sys, os -s = socket.socket(socket.AF_UNIX, socket.SOCK_STREAM); s.connect(sys.argv[2]) -r = subprocess.run([sys.argv[1], "--fd", str(s.fileno()), "--", "sh", "-c", - 'ls /proc/self/fd | grep -v "^3$" | sort -n | tr "\n" " " | sed "s/ $//"'], - pass_fds=[s.fileno()], capture_output=True, text=True) -print(r.stdout.strip()) -EOF -)" - -echo "# signals" -expect_eq "SIGTERM forwarded" "143" "$(run -- sh -c 'kill -TERM $PPID; sleep 5; echo alive' >/dev/null 2>&1; echo $?)" -expect_eq "SIGHUP forwarded" "129" "$(run -- sh -c 'kill -HUP $PPID; sleep 5; echo alive' >/dev/null 2>&1; echo $?)" -expect_eq "SIGINT to 9player is ignored while the child lives" "still-here" "$(run -- sh -c 'kill -INT $PPID; sleep 0.3; echo still-here')" -# Ctrl-C from the tty must not kill the --spawn server (same process group). -expect_eq "Ctrl-C on the tty leaves the --spawn server alive" "ok" "$(timeout 30 python3 - "$PLAYER" "$INTROSPECT" <<'EOF' -import os, pty, sys, time, select -P, I = sys.argv[1], sys.argv[2] -prog = ["python3", "-c", """ -import os, signal, sys, time -signal.signal(signal.SIGINT, lambda *a: None) -m = os.environ['NINEPLAYER_MOUNT'] -open(m + '/build/optimize').read() -sys.stdin.readline() -try: - open(m + '/build/optimize').read(); print('ok', flush=True) -except Exception as e: - print('mount dead:', e, flush=True) -"""] -pid, fd = pty.fork() -if pid == 0: - os.execv(P, [P, "--spawn", I + " --stdio", "--"] + prog) -out = b"" -def rd(t): - global out - end = time.time() + t - while time.time() < end: - r, _, _ = select.select([fd], [], [], 0.1) - if r: - try: d = os.read(fd, 4096) - except OSError: return - if not d: return - out += d -rd(1.5); os.write(fd, b"\x03"); rd(0.7); os.write(fd, b"\n"); rd(3) -os.waitpid(pid, 0) -print(out.decode(errors="replace").replace("^C", "").strip().splitlines()[-1] if out.strip() else "no output") -EOF -)" -# The --spawn server dying mid-session is reaped (no zombie) and does not end the session. -OUT=$("$TIMEOUT" 60 "$PLAYER" --spawn "$INTROSPECT --stdio" -- sh -c 'srv=$(cat $NINEPLAYER_MOUNT/runtime/pid); kill -TERM $srv; sleep 0.5; st=$(ps -o stat= -p $srv 2>/dev/null); echo "${st:-gone}"; exit 5' 2>/dev/null); RC=$? -expect_eq "server death mid-session: exit status still the child's, server reaped (no zombie)" "5 gone" "$RC $OUT" -# A server that never answers: once the child is dead, SIGTERM must end 9player. -cat >"$TMP/hang.py" <<'EOF' -import struct, os, sys, time -def rd(n): - b = b"" - while len(b) < n: - c = os.read(0, n - len(b)) - if not c: sys.exit(0) - b += c - return b -while True: - size, = struct.unpack("/dev/null & -HP=$! -sleep 1; kill -TERM $HP -START=$(date +%s) -for _ in $(seq 1 100); do kill -0 $HP 2>/dev/null || break; sleep 0.1; done -if kill -0 $HP 2>/dev/null; then kill -KILL $HP; RC=hung; else wait $HP; RC=$?; fi -expect_eq "hung server: one SIGTERM ends 9player once the child is dead (watchdog)" "143" "$RC" -expect_eq "hung server: exit was prompt" "yes" "$([ $(( $(date +%s) - START )) -lt 8 ] && echo yes)" -pkill -f "$TMP/hang.py" 2>/dev/null - -echo "# child/parent protocol" -if command -v strace >/dev/null 2>&1 && strace -qq -e trace=none true 2>/dev/null; then - expect_eq "child killed before handoff" "125" "$(strace -f -qq -e trace=unshare -e inject=unshare:signal=KILL -o /dev/null timeout 20 "$PLAYER" --unix "$SOCK" -- true 2>/dev/null; echo $?)" - expect_contains "child killed before handoff: message" "child exited before reporting" "$(strace -f -qq -e trace=unshare -e inject=unshare:signal=KILL -o /dev/null timeout 20 "$PLAYER" --unix "$SOCK" -- true 2>&1)" - expect_eq "status handoff fails" "125" "$(strace -f -qq -e trace=sendmsg -e inject=sendmsg:error=EPIPE -o /dev/null timeout 20 "$PLAYER" --unix "$SOCK" -- true 2>/dev/null; echo $?)" - # recvmsg skipped (returns 1 without the fd): the child must be killed, not exec'd onto a dead mount. - OUT=$(strace -f -qq -e trace=recvmsg -e inject=recvmsg:retval=1:when=1 -o /dev/null timeout 20 "$PLAYER" --unix "$SOCK" -- sh -c 'echo child-ran' 2>&1; echo "rc=$?") - expect_contains "truncated fd handoff: child not exec'd" "rc=125" "$OUT" - expect_eq "truncated fd handoff: program never ran" "no" "$(case "$OUT" in *child-ran*) echo yes;; *) echo no;; esac)" - expect_contains "fuse mount failure is reported" "mount fuse: EPERM" "$(strace -f -qq -e trace=mount -e inject=mount:error=EPERM:when=2 -o /dev/null timeout 20 "$PLAYER" --unix "$SOCK" --mount "$TMP/mp" -- true 2>&1)" -else - echo "skip - strace unavailable (child failure injection)" -fi -expect_contains "fork failure is reported" "fork: E" "$(python3 -c " -import resource, os -resource.setrlimit(resource.RLIMIT_NPROC, (1, 1)) -os.execv('$PLAYER', ['$PLAYER', '--unix', '$SOCK', '--', 'true'])" 2>&1)" - -echo "# mountpoint policy" -ln -s /nonexistent "$TMP/dangling" -expect_contains "dangling symlink mountpoint" "dangling symlink" "$(run --mount "$TMP/dangling" -- true 2>&1)" -expect_eq "refuse to shadow / via /proc/self/root" "125" "$(run --mount /proc/self/root/x9p -- true 2>/dev/null; echo $?)" -expect_contains "refuse to shadow / via /proc/self/root: message" "refusing to shadow /" "$(run --mount /proc/self/root/x9p -- true 2>&1)" -expect_eq "refuse to shadow under /proc" "125" "$(run --mount /proc/self/fd/x9p -- true 2>/dev/null; echo $?)" -if [ "$(ls -A /usr/lib | wc -l)" -gt 4096 ]; then - expect_contains "parent with >4096 entries refused" "more than 4096 entries" "$(run --mount /usr/lib/x9p -- true 2>&1)" -else - echo "skip - no root-owned directory with >4096 entries" -fi -expect_eq "shadowed /run keeps its entries" "$(ls -A /run | sort | tr '\n' ' ')" "$(run --mount /run/x9p -- sh -c 'ls -A /run | grep -v "^x9p$" | sort | tr "\n" " "')" -expect_eq "mountpoint with spaces" "ok" "$(mkdir -p "$TMP/with space" && run --mount "$TMP/with space" -- sh -c '[ -f "$NINEPLAYER_MOUNT/README" ] && echo ok')" -expect_eq "mountpoint is a file" "125" "$(run --mount "$TMP/nx" -- true 2>/dev/null; echo $?)" - -echo "# leaks" -for i in $(seq 1 30); do run -- sh -c 'cat $NINEPLAYER_MOUNT/build/optimize >/dev/null' 2>/dev/null; done -BG=(); for i in $(seq 1 10); do ( run -- sh -c 'cat $NINEPLAYER_MOUNT/build/optimize >/dev/null' 2>/dev/null ) & BG+=($!); done; wait "${BG[@]}" # not a bare wait: that would also wait for the server -sleep 0.3 -expect_eq "no stray 9player processes" "" "$(pgrep -f "^$PLAYER " | tr '\n' ' ')" -expect_eq "no stray --stdio servers" "" "$(pgrep -f "$INTROSPECT --stdio" | tr '\n' ' ')" -expect_eq "host mount table untouched" "same" "$([ "$MI_BEFORE" = "$(grep -v " $TMP" /proc/self/mountinfo | sort)" ] && echo same || echo changed)" - -echo -echo "passed=$PASSED failed=$FAILED" -[ "$FAILED" -eq 0 ] diff --git a/9player/test/adversarial.sh b/9player/test/adversarial.sh deleted file mode 100755 index 857af14..0000000 --- a/9player/test/adversarial.sh +++ /dev/null @@ -1,15 +0,0 @@ -#!/usr/bin/env bash -# Runs every 9player adversarial suite in sequence (hostile servers, FUSE -# semantics, process/namespace edge cases, stress). The suites aimed at the -# introspect server itself live in introspect/test (zig build introspect-adv). -# Usage: bash 9player/test/adversarial.sh <9player> (zig build 9player-adv) -set -u -PLAYER=${1:?path to 9player} -INTROSPECT=${2:?path to introspect} -HERE=$(cd "$(dirname "$0")" && pwd) -status=0 -for suite in adv_ns_process adv_bridge_hostile adv_bridge_semantics adv_bridge_stress; do - echo "### $suite" - if bash "$HERE/$suite.sh" "$PLAYER" "$INTROSPECT"; then echo "### $suite: ok"; else echo "### $suite: FAILED"; status=1; fi -done -exit $status diff --git a/9player/test/integration.sh b/9player/test/integration.sh deleted file mode 100755 index e12cb5f..0000000 --- a/9player/test/integration.sh +++ /dev/null @@ -1,160 +0,0 @@ -#!/usr/bin/env bash -# Integration tests for 9player: real user+mount namespaces, real FUSE, real 9P servers. -# Usage: bash 9player/test/integration.sh <9player> (zig build 9player-itest) -# Exit 0 on success (or when the machine cannot run the tests), 1 on failure. -set -u - -PLAYER=$(realpath "${1:?path to 9player}") -INTROSPECT=$(realpath "${2:?path to introspect}") -TMP=$(mktemp -d "${TMPDIR:-/tmp}/9player-itest.XXXXXX") -PIDS=() -FAILED=0 -PASSED=0 -M=/mnt/9p - -cleanup() { - for p in "${PIDS[@]:-}"; do [ -n "$p" ] && kill "$p" 2>/dev/null; done - rm -rf "$TMP" -} -trap cleanup EXIT - -if ! unshare -Urm true 2>/dev/null; then - echo "SKIP: unprivileged user namespaces unavailable"; exit 0 -fi -if [ ! -c /dev/fuse ]; then - echo "SKIP: /dev/fuse missing"; exit 0 -fi - -pass() { PASSED=$((PASSED + 1)); echo "ok - $1"; } -fail() { FAILED=$((FAILED + 1)); echo "FAIL - $1"; shift; [ $# -gt 0 ] && printf ' %s\n' "$@"; } -expect_eq() { # name expected actual - if [ "$2" = "$3" ]; then pass "$1"; else fail "$1" "expected: $(printf %q "$2")" "actual: $(printf %q "$3")"; fi -} -expect_contains() { # name needle haystack - case "$3" in *"$2"*) pass "$1" ;; *) fail "$1" "missing: $(printf %q "$2")" "in: $(printf %q "$3")" ;; esac -} - -wait_socket() { # path - for _ in $(seq 1 100); do [ -S "$1" ] && return 0; sleep 0.05; done - return 1 -} - -# run_in "" — run inside a namespace with the current transport ($TRANSPORT array). -run_in() { timeout 60 "$PLAYER" "${TRANSPORT[@]}" -- sh -c "$1" 2>"$TMP/stderr"; } - -# --- scratch battery: works against any writable 9P tree rooted at $1 (relative to mount) --- -scratch_battery() { # label scratchdir - local label=$1 S=$2 - - expect_eq "$label: create+append+read" $'hello\nworld' "$(run_in "echo hello > $M/$S/a && echo world >> $M/$S/a && cat $M/$S/a")" - expect_eq "$label: overwrite" "x" "$(run_in "echo x > $M/$S/a && cat $M/$S/a")" - expect_eq "$label: stat size after overwrite" "2" "$(run_in "stat -c %s $M/$S/a")" - expect_eq "$label: truncate" "0" "$(run_in "truncate -s 0 $M/$S/a && stat -c %s $M/$S/a")" - expect_eq "$label: truncate extend" "10" "$(run_in "truncate -s 10 $M/$S/a && stat -c %s $M/$S/a")" - expect_eq "$label: mkdir -p nested" "directory" "$(run_in "mkdir -p $M/$S/d1/d2/d3 && stat -c %F $M/$S/d1/d2/d3")" - expect_eq "$label: rename within dir" "moved" "$(run_in "echo moved > $M/$S/d1/d2/f && mv $M/$S/d1/d2/f $M/$S/d1/d2/g && cat $M/$S/d1/d2/g")" - # mv(1) silently falls back to copy+delete on EXDEV, so probe rename(2) directly. - expect_contains "$label: rename across dirs is EXDEV" "EXDEV" "$(run_in "python3 -c 'import os,errno -try: os.rename(\"$M/$S/d1/d2/g\", \"$M/$S/d1/g\") -except OSError as e: print(errno.errorcode[e.errno]) -'")" - expect_eq "$label: rm file" "gone" "$(run_in "rm $M/$S/d1/d2/g && [ ! -e $M/$S/d1/d2/g ] && echo gone")" - expect_eq "$label: rmdir non-empty fails" "1" "$(run_in "rmdir $M/$S/d1 2>/dev/null; echo \$?")" - expect_eq "$label: rmdir chain" "ok" "$(run_in "rmdir $M/$S/d1/d2/d3 $M/$S/d1/d2 $M/$S/d1 && echo ok")" - expect_eq "$label: ENOENT" "1" "$(run_in "cat $M/$S/nope 2>/dev/null; echo \$?")" - expect_eq "$label: ENOENT errno text" "No such file or directory" "$(run_in "cat $M/$S/nope 2>&1 | sed 's/.*: //'")" - - head -c 1048576 /dev/urandom >"$TMP/rand" - local sum; sum=$(sha256sum <"$TMP/rand" | cut -d' ' -f1) - expect_eq "$label: 1 MiB round trip (cp)" "$sum" "$(run_in "cp $TMP/rand $M/$S/big && sha256sum < $M/$S/big | cut -d' ' -f1")" - expect_eq "$label: 1 MiB size" "1048576" "$(run_in "stat -c %s $M/$S/big")" - expect_eq "$label: odd block sizes (dd bs=1000)" "$sum" "$(run_in "dd if=$M/$S/big of=$M/$S/big2 bs=1000 status=none && sha256sum < $M/$S/big2 | cut -d' ' -f1")" - expect_eq "$label: partial read at offset" "$(tail -c 12345 "$TMP/rand" | sha256sum | cut -d' ' -f1)" "$(run_in "tail -c 12345 $M/$S/big | sha256sum | cut -d' ' -f1")" - expect_eq "$label: many small files" "200" "$(run_in "mkdir $M/$S/many && for i in \$(seq 1 200); do echo \$i > $M/$S/many/f\$i; done; ls $M/$S/many | wc -l")" - expect_eq "$label: find count" "201" "$(run_in "find $M/$S/many | wc -l")" - expect_eq "$label: readdir contents" "f1 f100 f200" "$(run_in "cd $M/$S/many && ls f1 f100 f200 | tr '\n' ' ' | sed 's/ \$//'")" - expect_eq "$label: cleanup many" "0" "$(run_in "rm -r $M/$S/many $M/$S/big $M/$S/big2 $M/$S/a; ls $M/$S | wc -l")" -} - -# ============================================================================ -echo "# introspect over a Unix socket" -SOCK=$TMP/introspect.sock -"$INTROSPECT" --unix "$SOCK" & -PIDS+=($!) -wait_socket "$SOCK" || { echo "introspect did not create $SOCK"; exit 1; } -TRANSPORT=(--unix "$SOCK") - -expect_eq "zig_version" "$(zig version)" "$(run_in "cat $M/build/zig_version")" -expect_eq "mount exported" "$M" "$(run_in 'echo $NINEPLAYER_MOUNT')" -expect_contains "root listing" "build" "$(run_in "ls $M")" -expect_contains "root listing has scratch" "scratch" "$(run_in "ls $M")" -expect_eq "README size > 0" "yes" "$(run_in "[ \$(stat -c %s $M/README) -gt 0 ] && echo yes")" -expect_eq "README readable" "yes" "$(run_in "[ -s $M/README ] && head -c 1 $M/README >/dev/null && echo yes")" -expect_eq "fn/now numeric" "num" "$(run_in "cat $M/runtime/fn/now | grep -Eq '^[0-9]+\$' && echo num")" -expect_eq "fn listing from comptime" "yes" "$(run_in "ls $M/runtime/fn | grep -q hostname && echo yes")" -expect_eq "ctl round trip" "5" "$(run_in "echo 'add 2 3' > $M/runtime/ctl && cat $M/runtime/ctl")" -expect_eq "ctl echo" "hi there" "$(run_in "echo 'echo hi there' > $M/runtime/ctl && cat $M/runtime/ctl")" -expect_contains "comptime types" "Qid" "$(run_in "ls $M/comptime/types")" -expect_eq "comptime size of Qid" "16" "$(run_in "cat $M/comptime/types/Qid/size")" -expect_eq "runtime pid is server pid" "${PIDS[-1]}" "$(run_in "cat $M/runtime/pid")" -expect_eq "exit status propagates" "7" "$(run_in 'exit 7'; echo $?)" -expect_eq "mount is fuse" "yes" "$(run_in "grep -q \"^9player $M fuse\" /proc/mounts && echo yes")" -expect_eq "host /mnt entries still visible" "$(ls -A /mnt | sort | tr '\n' ' ')" "$(run_in "ls -A /mnt | grep -v '^9p\$' | sort | tr '\n' ' '")" -expect_eq "host mount table untouched" "no" "$(grep -q " $M " /proc/self/mountinfo && echo yes || echo no)" -scratch_battery "introspect" scratch - -echo "# nested 9player" -expect_eq "nested mount" "$(zig version)" "$(run_in "$PLAYER --unix $SOCK --mount $TMP/inner -- sh -c 'cat \$NINEPLAYER_MOUNT/build/zig_version'")" - -echo "# --mount variants" -mkdir -p "$TMP/mnt" -expect_eq "--mount existing dir" "ok" "$(timeout 60 "$PLAYER" --unix "$SOCK" --mount "$TMP/mnt" -- sh -c "[ -f $TMP/mnt/README ] && echo ok")" -expect_eq "--mount relative" "ok" "$(cd "$TMP" && timeout 60 "$PLAYER" --unix "$SOCK" --mount rel -- sh -c "[ -f $TMP/rel/README ] && echo ok")" -expect_eq "--mount missing under /" "125" "$(timeout 60 "$PLAYER" --unix "$SOCK" --mount /nonexistent-9player-dir -- true 2>/dev/null; echo $?)" - -echo "# lifecycle" -START=$(date +%s) -expect_eq "background grandchild does not block exit" "3" "$(run_in 'sleep 30 >/dev/null 2>&1 & exit 3'; echo $?)" -expect_eq "exit was prompt" "yes" "$([ $(( $(date +%s) - START )) -lt 10 ] && echo yes)" -expect_eq "SIGTERM forwarded" "143" "$(timeout 60 "$PLAYER" --unix "$SOCK" -- sh -c 'kill -TERM $PPID; sleep 5; echo alive' >/dev/null 2>&1; echo $?)" - -echo "# --spawn transport" -TRANSPORT=(--spawn "$INTROSPECT --stdio") -expect_eq "spawn: zig_version" "$(zig version)" "$(run_in "cat $M/build/zig_version")" -expect_eq "spawn: ctl" "7" "$(run_in "echo 'add 3 4' > $M/runtime/ctl && cat $M/runtime/ctl")" -expect_eq "spawn: stateful sequence in one session" 'hello world 12 moved 0' "$(run_in "cd $M/scratch && echo hello > a && echo world >> a && cat a && stat -c %s a && mkdir d && echo moved > d/f && mv d/f d/g && cat d/g && rm d/g && rmdir d && rm a && ls | wc -l" | tr '\n' ' ' | sed 's/ $//')" - -echo "# --tcp transport" -PORT=$(( 20000 + RANDOM % 20000 )) -"$INTROSPECT" --tcp "127.0.0.1:$PORT" & -PIDS+=($!) -sleep 0.3 -TRANSPORT=(--tcp "127.0.0.1:$PORT") -expect_eq "tcp: zig_version" "$(zig version)" "$(run_in "cat $M/build/zig_version")" -expect_eq "tcp: scratch" "tcp" "$(run_in "echo tcp > $M/scratch/t && cat $M/scratch/t && rm $M/scratch/t")" - -echo "# server death" -"$INTROSPECT" --unix "$TMP/dying.sock" & -DYING=$! -wait_socket "$TMP/dying.sock" -TRANSPORT=(--unix "$TMP/dying.sock") -OUT=$(run_in "cat $M/build/optimize >/dev/null && kill $DYING && sleep 0.3; cat $M/build/optimize 2>&1 >/dev/null | sed 's/.*: //'; echo status=\$?") -expect_contains "server death yields an error, not a hang" "status=0" "$OUT" -expect_eq "server death errno text" "yes" "$(case "$OUT" in *"Input/output error"*|*"Transport endpoint is not connected"*) echo yes;; *) echo "no: $OUT";; esac)" - -echo "# plan9port ramfs (independent 9P2000 implementation)" -if [ -x /usr/lib/plan9/bin/ramfs ]; then - mkdir -p "$TMP/p9ns" - NAMESPACE=$TMP/p9ns /usr/lib/plan9/bin/ramfs -s ramfs - wait_socket "$TMP/p9ns/ramfs" || echo "ramfs socket missing" - PIDS+=($(pgrep -f "9pserve -u unix!$TMP/p9ns/ramfs")) - TRANSPORT=(--unix "$TMP/p9ns/ramfs") - expect_eq "ramfs: mkdir scratch" "ok" "$(run_in "mkdir $M/scratch && echo ok")" - scratch_battery "ramfs" scratch -else - echo "skip - plan9port ramfs not installed" -fi - -echo -echo "passed=$PASSED failed=$FAILED" -[ "$FAILED" -eq 0 ] diff --git a/9proc/README.md b/9proc/README.md new file mode 100644 index 0000000..0fd3ddb --- /dev/null +++ b/9proc/README.md @@ -0,0 +1,130 @@ +# 9proc + +A 9P2000 debug/introspection server as a Zig 0.16 library, built on +[cloud9](../): a debugger-shaped interface where the protocol is just files. +Anything that can read a filesystem (a shell, an agent, an editor, `9p`, +[9ns](../9ns)) can inspect a running program: build facts, comptime +type layouts, live values, threads and their stacks, memory, breakpoints, +panics. + +The core (`core`, `vars`) is freestanding: no allocator, no OS, no threads, +caller-owned static `Storage`, fixed-capacity tables sized at comptime. It +compiles for `riscv32-freestanding-none`. `scratch` (an in-memory read/write +tree) takes an allocator; `linux` is the platform layer (listeners, a poll +loop on one background thread, threads/stacks/registers, memory, breakpoints +and panics via `std.debug`). [docs/LIBRARY.md](docs/LIBRARY.md) has the full +contract. + +``` +9proc/ + build.zig fragment imported by cloud9's root build.zig (steps below) + src/root.zig pub const core, vars, scratch, linux; Config, Server(cfg), Provider + src/core.zig Tree/Server engine on cloud9.Server: fids, walks, dir reads, providers + src/vars.zig comptime value renderers (@typeInfo) for /vars + src/scratch.zig in-memory read/write tree provider (takes an Allocator) + src/freestanding_check.zig root for the riscv32-freestanding-none compile check + src/linux/probe.zig background thread + poll loop + unix/tcp/fd listeners + src/linux/debug.zig threads, stacks, registers, addr→source, memory, breakpoints, panic + src/linux/provider.zig the debug provider (/threads, /addr, /mem, /hex, /breakpoints, /panic) + src/linux/runtime.zig /runtime generators (pid, uptime, argv, cwd, env, clients) + demo/main.zig the `9proc-demo` binary (below) + test/ debug.sh, adv_9proc_hostile, adv_core_hostile, adv_linux_probe, adversarial.sh + docs/LIBRARY.md design rules and the module contracts +``` + +## Using the library + +cloud9's `build.zig` exports two modules: `cloud9` (the protocol) and +`9proc` (this library, which imports `cloud9` itself). A package that +depends on cloud9 takes both from the one dependency: + +```zig +const cloud9_dep = b.dependency("cloud9", .{ .target = target, .optimize = optimize }); +exe.root_module.addImport("cloud9", cloud9_dep.module("cloud9")); +exe.root_module.addImport("9proc", cloud9_dep.module("9proc")); +``` + +Embedding the core is three static objects and a push/step/output loop, the +same shape as cloud9's `Server`: + +```zig +const proc9 = @import("9proc"); // the module is `9proc`; identifiers cannot start with a digit + +const State = struct { ticks: u32, phase: enum { idle, busy } }; +const cfg: proc9.Config = .{ .name = "fw", .types = &.{State}, .msize = 2048, .max_fids = 16 }; +const S = proc9.Server(cfg); + +var state: State = .{ .ticks = 0, .phase = .idle }; +var storage: S.Storage = undefined; // per connection: in/out frames + snapshot slots +var shared: S.Shared = undefined; // once: providers and exposed variables + +pub fn main() void { + shared = .init(&state); + shared.expose("state", &state) catch unreachable; // /vars/state/{value,type,size,addr,raw,f/...} + var conn: S.Conn = .init(&shared, &storage, cfg.msize); + // transport loop: conn.push(bytes) ... while (try conn.step()) {} ... send conn.output(), conn.wrote(n) +} +``` + +On Linux, `proc9.linux.Probe` runs that loop for you on one background +thread over a Unix, TCP or inherited listener, and adds the debug provider; +`pub const panic = std.debug.FullPanic(proc9.linux.debug.panicHook);` +in the root module publishes panics under `/panic`. `demo/main.zig` shows +every piece together. + +## The demo (`zig build 9proc`, binary `9proc-demo`) + +A single-binary 9P2000 server whose file tree is the binary itself: build-time +facts, `comptime` reflection, live runtime state, a worker thread whose state +is exposed under `/vars`, and the Linux debug layer. + +``` +/README +/build/{zig_version,target,optimize,time,change} captured by 9proc/build.zig (jj change id, UTC time) +/comptime/types//{name,size,align,fields} @sizeOf/@alignOf/@typeInfo, generated at comptime +/comptime/decls pub declarations of the server module +/runtime/{pid,ppid,uptime,argv,cwd,env,clients} +/runtime/fn/ reading calls a Zig function (hostname, now, random, uname, fib30) +/runtime/ctl write "fib N" | "add A B" | "echo TEXT" | "sleep-ms N" | "trap" | "panic" +/scratch/ in-memory read/write tree +/vars/state/... the worker's State (readable and writable leaves) +/threads//{name,stat,stack,regs} /addr/ /mem/{maps,} /hex/ +/breakpoints//{stack,regs,ctl} /panic/{message,stack,ctl} +``` + +`/runtime/env` exposes the server's whole environment, so serve it on a Unix +socket or loopback only. + +```sh +zig-out/bin/9proc-demo --unix /tmp/intro.sock & # or --tcp IP:PORT, --stdio, --no-hold +zig-out/bin/9ns --unix /tmp/intro.sock -- sh -c ' + cat $NINE_MOUNT/build/zig_version; echo + cat $NINE_MOUNT/comptime/types/Qid/fields + echo "fib 20" > $NINE_MOUNT/runtime/ctl; cat $NINE_MOUNT/runtime/ctl + cat $NINE_MOUNT/threads/*/stack' +``` + +## Building and testing + +9proc lives in the cloud9 repository as `cloud9/9proc/` and is +wired into cloud9's `build.zig` through the fragment `9proc/build.zig`. +Everything is run from the cloud9 root: + +```sh +zig build # installs zig-out/bin/9proc-demo with the other binaries +zig build 9proc # build and install only the demo +zig build 9proc-test # library unit tests (core, vars, scratch, linux) and the demo's +zig build 9proc-check-freestanding # compile the core for riscv32-freestanding-none +zig build 9proc-debug-test # src/linux/debug.zig unit tests +zig build 9proc-debug-itest # test/debug.sh: threads, stacks, breakpoints, panic through 9ns +zig build 9proc-adv # hostile raw-9P clients against the demo and the core, + # the Linux layer through a 9ns mount (several minutes) +zig build -D9proc=false # leave 9proc out +zig build 9proc-check-freestanding -Dtarget=riscv32-freestanding-none -D9proc=true +``` + +`-D9proc` (default: on for Linux targets) enables the module and the +freestanding check on any target; the demo and the Linux suites are added +only when the target OS is Linux. The end-to-end suites also need 9ns +(`-D9ns=true`, the Linux default), unprivileged user namespaces, +`/dev/fuse` and Python 3, and skip themselves otherwise. diff --git a/9proc/build.zig b/9proc/build.zig new file mode 100644 index 0000000..670ea27 --- /dev/null +++ b/9proc/build.zig @@ -0,0 +1,169 @@ +//! Build fragment for 9proc: the 9P debug/introspection library (module +//! `9proc`), its freestanding check, the `9proc-demo` server and the +//! test suites under 9proc/test. It is `@import`ed by the root build.zig +//! and called with the root builder, so every `b.path(...)` here is relative +//! to the cloud9 root (hence the `9proc/` prefix), every option is +//! defined by the root (no `standardTargetOptions` here) and every step it +//! registers lands in the root's step list under the `9proc` prefix. +//! +//! Steps: 9proc, 9proc-test, 9proc-check-freestanding, +//! 9proc-debug-test, 9proc-debug-itest, 9proc-adv. +const std = @import("std"); + +/// What the root passes in. The root owns target/optimize resolution and the +/// cloud9 module; this fragment derives everything else from them. +pub const Context = struct { + target: std.Build.ResolvedTarget, + optimize: std.builtin.OptimizeMode, + /// The cloud9 library module for `target`. Its `root_source_file` is also + /// used to instantiate cloud9 for the freestanding check target. + cloud9: *std.Build.Module, +}; + +pub const Artifacts = struct { + /// The `9proc` module, exported with `b.addModule` so dependents of + /// cloud9 can `.module("9proc")`. Built for any target; the Linux + /// layer is compiled in only when the target OS is Linux. + module: *std.Build.Module, + /// The demo 9P2000 server (binary `9proc-demo`); null when the target is + /// not Linux. 9ns's integration tests use it as their server. + demo: ?*std.Build.Step.Compile, + /// `9proc-test`: library (and demo) unit tests. + test_step: *std.Build.Step, + /// `9proc-check-freestanding`: the core compiled for riscv32-freestanding-none. + check_step: *std.Build.Step, + /// `9proc-debug-test`: linux/debug.zig unit tests; null when not Linux. + debug_test_step: ?*std.Build.Step, + /// `9proc-adv`: the hostile-client suites are attached by `add`, the + /// Linux-layer suite (which needs a 9ns mount) by `addNsTests`. + adv_step: *std.Build.Step, +}; + +pub fn add(b: *std.Build, ctx: Context) Artifacts { + const target = ctx.target; + const optimize = ctx.optimize; + const is_linux = target.result.os.tag == .linux; + + // Build-time facts embedded into the demo (/build/*): jj change id, UTC + // time, optimize mode, target triple. `jj` is pointed at the build root + // (the cloud9 checkout) explicitly, so the result does not depend on the + // directory `zig build` was invoked from. + const build_options = b.addOptions(); + var code: u8 = 0; + const repo = b.build_root.path orelse "."; + const change_id = b.runAllowFail(&.{ "jj", "-R", repo, "log", "--no-graph", "-r", "@", "-T", "change_id.short()", "--ignore-working-copy" }, &code, .ignore) catch "unknown"; + build_options.addOption([]const u8, "change_id", std.mem.trim(u8, change_id, " \n\r\t")); + const stamp = b.runAllowFail(&.{ "date", "-u", "+%Y-%m-%dT%H:%M:%SZ" }, &code, .ignore) catch "unknown"; + build_options.addOption([]const u8, "build_time", std.mem.trim(u8, stamp, " \n\r\t")); + build_options.addOption([]const u8, "optimize", @tagName(optimize)); + build_options.addOption([]const u8, "target", target.result.zigTriple(b.allocator) catch "unknown"); + + // The library: freestanding core + vars, scratch (allocator), Linux layer. + // Exported under the name `9proc` for packages that depend on cloud9. + const lib_mod = b.addModule("9proc", .{ + .root_source_file = b.path("9proc/src/root.zig"), + .target = target, + .optimize = optimize, + .imports = &.{.{ .name = "cloud9", .module = ctx.cloud9 }}, + }); + + const test_step = b.step("9proc-test", "Run the 9proc library's unit tests (and the demo's)"); + test_step.dependOn(&b.addRunArtifact(b.addTest(.{ .root_module = lib_mod })).step); + + // The core must compile without an OS (rule 1 of 9proc/docs/LIBRARY.md). + // cloud9 is re-instantiated for that target from the same root source. + const fs_target = b.resolveTargetQuery(.{ .cpu_arch = .riscv32, .os_tag = .freestanding, .abi = .none }); + const fs_cloud9 = b.createModule(.{ + .root_source_file = ctx.cloud9.root_source_file.?, + .target = fs_target, + .optimize = optimize, + }); + const fs_check = b.addObject(.{ + .name = "9proc-freestanding", + .root_module = b.createModule(.{ + .root_source_file = b.path("9proc/src/freestanding_check.zig"), + .target = fs_target, + .optimize = optimize, + .imports = &.{.{ .name = "cloud9", .module = fs_cloud9 }}, + }), + }); + const check_step = b.step("9proc-check-freestanding", "Compile the 9proc core for riscv32-freestanding-none"); + check_step.dependOn(&fs_check.step); + + // `9proc-adv` exists on every target so the step list is stable; its + // suites are attached below (Linux only) and by addNsTests. + const adv_step = b.step("9proc-adv", "Run 9proc's adversarial suites (hostile clients, Linux layer; several minutes)"); + + if (!is_linux) { + adv_step.dependOn(&b.addFail("9proc-adv needs a Linux target (the demo server is Linux-only)").step); + return .{ .module = lib_mod, .demo = null, .test_step = test_step, .check_step = check_step, .debug_test_step = null, .adv_step = adv_step }; + } + + // The demo: a 9P2000 server whose tree is the binary itself (build-time, + // comptime and runtime facts, a worker thread, the Linux debug layer). + const demo_mod = b.createModule(.{ + .root_source_file = b.path("9proc/demo/main.zig"), + .target = target, + .optimize = optimize, + .imports = &.{ + .{ .name = "cloud9", .module = ctx.cloud9 }, + .{ .name = "build_options", .module = build_options.createModule() }, + .{ .name = "9proc", .module = lib_mod }, + }, + }); + const demo = b.addExecutable(.{ .name = "9proc-demo", .root_module = demo_mod }); + const install_demo = b.addInstallArtifact(demo, .{}); + b.getInstallStep().dependOn(&install_demo.step); + b.step("9proc", "Build and install only the 9proc-demo server").dependOn(&install_demo.step); + test_step.dependOn(&b.addRunArtifact(b.addTest(.{ .root_module = demo_mod })).step); + + // Linux debug facilities (threads, stacks, breakpoints, panic): self-contained unit tests. + const debug_mod = b.createModule(.{ + .root_source_file = b.path("9proc/src/linux/debug.zig"), + .target = target, + .optimize = optimize, + }); + const debug_test_step = b.step("9proc-debug-test", "Run the 9proc/src/linux/debug.zig unit tests"); + debug_test_step.dependOn(&b.addRunArtifact(b.addTest(.{ .root_module = debug_mod })).step); + + // Hostile raw-9P clients against the demo (framing, tags, floods; the core's + // /vars tree, snapshots, fid table). `--fast` as in the umbrella script. + inline for (.{ "adv_9proc_hostile", "adv_core_hostile" }) |suite| { + const run = b.addSystemCommand(&.{"bash"}); + run.addFileArg(b.path("9proc/test/" ++ suite ++ ".sh")); + run.addArtifactArg(demo); + run.addArg("--fast"); + adv_step.dependOn(&run.step); + } + + return .{ .module = lib_mod, .demo = demo, .test_step = test_step, .check_step = check_step, .debug_test_step = debug_test_step, .adv_step = adv_step }; +} + +/// The suites that drive the demo through a 9ns mount: test/debug.sh +/// (threads, stacks, breakpoints, panic end to end) and +/// test/adv_linux_probe.sh (memory endpoints, signal machinery, poll loop). +/// Called by the root after the 9ns fragment; `ns` is null when +/// 9ns is disabled, in which case the steps exist but fail with a notice. +/// Returns the `9proc-debug-itest` step (null when the target is not Linux). +pub fn addNsTests(b: *std.Build, arts: Artifacts, ns: ?*std.Build.Step.Compile) ?*std.Build.Step { + const demo = arts.demo orelse return null; // not Linux: nothing to drive + const debug_itest = b.step("9proc-debug-itest", "Run 9proc/test/debug.sh (threads, stacks, breakpoints, panic through 9ns)"); + const exe = ns orelse { + const fail = b.addFail("9proc-debug-itest and the Linux-layer adversarial suite need 9ns (build with -D9ns=true)"); + debug_itest.dependOn(&fail.step); + arts.adv_step.dependOn(&fail.step); + return debug_itest; + }; + const dbg = b.addSystemCommand(&.{"bash"}); + dbg.addFileArg(b.path("9proc/test/debug.sh")); + dbg.addArtifactArg(exe); + dbg.addArtifactArg(demo); + debug_itest.dependOn(&dbg.step); + + const adv_linux = b.addSystemCommand(&.{"bash"}); + adv_linux.addFileArg(b.path("9proc/test/adv_linux_probe.sh")); + adv_linux.addArtifactArg(exe); + adv_linux.addArtifactArg(demo); + arts.adv_step.dependOn(&adv_linux.step); + return debug_itest; +} diff --git a/9proc/demo/main.zig b/9proc/demo/main.zig new file mode 100644 index 0000000..df06369 --- /dev/null +++ b/9proc/demo/main.zig @@ -0,0 +1,423 @@ +//! 9proc: the demo 9P2000 server, built on the 9proc library. +//! +//! /README, /build/*, /comptime/{types,decls}, /runtime/{pid,ppid,uptime,argv,cwd,env,clients}, +//! /runtime/fn/{fib30,hostname,now,random,uname}, /runtime/ctl (echo|fib|sleep-ms|add|trap|panic), +//! /scratch (in-memory tree), /vars/state (the worker's exposed state), +//! /threads, /addr, /mem, /hex, /breakpoints, /panic (the Linux debug layer). +//! +//! A worker thread ("worker") runs `workerLoop`, incrementing `state.ticks` +//! every ~10 ms; `trap` makes it execute `@breakpoint()` on its next tick and +//! `panic` makes it panic from inside `workerLoop`. Panics go through the +//! library's hook, so the message and stack are published under /panic and +//! the process is held there until /panic/ctl says "continue" (`--no-hold` +//! disables the hold). +//! +//! Static memory: every buffer is a global; the only heap user is /scratch +//! (init.gpa, 512 MiB budget). With `max_clients` = 16 and msize = 1 MiB the +//! per-client core Storage is 3 MiB + 8 x 64 KiB snapshots and the Conn's fid +//! table ~1.88 MiB (32768 fids plus their hash index, needed for the 20000-fid +//! adversarial test), so `probe_storage` is ~86 MiB of BSS; untouched pages +//! cost nothing (an idle server has an RSS of ~7 MiB). +const std = @import("std"); +const builtin = @import("builtin"); +const cloud9 = @import("cloud9"); +const build_options = @import("build_options"); +const proc9 = @import("9proc"); +const linux = std.os.linux; +const Writer = std.Io.Writer; +const plinux = proc9.linux; +const runtime = plinux.runtime; + +pub const panic = std.debug.FullPanic(plinux.debug.panicHook); + +/// Simultaneous 9P clients; further connections are closed (see probe.zig for +/// the idle-eviction rule). +pub const max_clients = 16; +pub const max_msize: u32 = 1 << 20; + +pub const State = struct { + ticks: u64, + phase: enum { idle, working, trapped }, + last_job: struct { id: u32, cost: f32 }, +}; + +const Build = struct { + pub const zig_version: []const u8 = builtin.zig_version_string; + pub const target: []const u8 = build_options.target; + pub const optimize: []const u8 = build_options.optimize; + pub const time: []const u8 = build_options.build_time; + pub const change: []const u8 = build_options.change_id; +}; + +/// What every generator and the ctl handler see (`Shared.ctx`). +const App = struct { + info: runtime.Info, + probe: *ProbeT, + shared: *S.Shared, +}; + +/// /runtime/fn/: reading the file calls the function. +pub const Fns = struct { + pub fn hostname(_: *anyopaque, w: *Writer) anyerror!void { + var u: linux.utsname = undefined; + if (linux.errno(linux.uname(&u)) != .SUCCESS) return error.Uname; + try w.writeAll(std.mem.sliceTo(&u.nodename, 0)); + } + + pub fn now(_: *anyopaque, w: *Writer) anyerror!void { + try w.print("{d}", .{runtime.realtimeSecs()}); + } + + pub fn random(_: *anyopaque, w: *Writer) anyerror!void { + var b: [8]u8 = undefined; + var got: usize = 0; + while (got < b.len) { + const rc = linux.getrandom(b[got..].ptr, b.len - got, 0); + switch (linux.errno(rc)) { + .SUCCESS => got += rc, + .INTR => continue, + else => return error.Random, + } + } + try w.print("{x:0>16}", .{std.mem.readInt(u64, &b, .little)}); + } + + pub fn uname(_: *anyopaque, w: *Writer) anyerror!void { + var u: linux.utsname = undefined; + if (linux.errno(linux.uname(&u)) != .SUCCESS) return error.Uname; + try w.writeAll(std.mem.sliceTo(&u.release, 0)); + } + + pub fn fib30(_: *anyopaque, w: *Writer) anyerror!void { + try w.print("{d}", .{fib(30)}); + } +}; + +const cfg: proc9.Config = .{ + .name = "9proc-demo", + .build = Build, + .types = &.{ cloud9.Qid, cloud9.Stat, cloud9.Msg, proc9.core.Node, linux.Statx }, + .decls_of = @This(), + .fns = Fns, + .runtime = runtime.Fns(App, "info"), + .ctl = &ctl, + .ctl_dir = .runtime, + .ctl_bytes = 64 * 1024, + .msize = max_msize, + .max_fids = 32768, + .max_providers = 8, + .snapshot_slots = 8, + .snapshot_bytes = 64 * 1024, +}; +const S = proc9.Server(cfg); +const ProbeT = plinux.Probe(S); +const ProbeStorage = ProbeT.Storage(max_clients); + +// -- static state ------------------------------------------------------------- + +pub var state: State = .{ .ticks = 0, .phase = .idle, .last_job = .{ .id = 0, .cost = 0 } }; +var trap_requested: std.atomic.Value(bool) = .init(false); +var panic_requested: std.atomic.Value(bool) = .init(false); + +/// Zero-filled static memory for `T`. (An `= undefined` global is emitted as +/// 0xAA-filled .data in Debug builds, which would make the binary 120 MiB; +/// zeros go to .bss and cost nothing until touched.) +fn Bss(comptime T: type) type { + return struct { + bytes: [@sizeOf(T)]u8 align(@alignOf(T)) = @splat(0), + fn get(b: *@This()) *T { + return @ptrCast(&b.bytes); + } + }; +} +var app_mem: Bss(App) = .{}; +var shared_mem: Bss(S.Shared) = .{}; +var probe_storage_mem: Bss(ProbeStorage) = .{}; +var probe_mem: Bss(ProbeT) = .{}; +var scratch_mem: Bss(proc9.Scratch) = .{}; + +/// Sizes of the static pieces, for the report and `--help`. +pub const static_bytes = @sizeOf(ProbeStorage) + @sizeOf(S.Shared) + @sizeOf(ProbeT); + +// -- the worker --------------------------------------------------------------- + +/// Ticks every ~10 ms; honours `trap` and `panic` requests from /runtime/ctl. +pub noinline fn workerLoop() void { + plinux.setThreadName("worker"); + var job: u32 = 0; + while (true) { + napMs(10); + state.ticks +%= 1; + if (panic_requested.swap(false, .acq_rel)) { + state.phase = .working; + @panic("demo panic requested over 9p"); + } + if (trap_requested.swap(false, .acq_rel)) { + state.phase = .trapped; + @breakpoint(); + state.phase = .idle; + } + if (state.ticks % 100 == 0) { + job +%= 1; + state.phase = .working; + state.last_job = .{ .id = job, .cost = @as(f32, @floatFromInt(job % 7)) * 0.5 }; + state.phase = .idle; + } + } +} + +/// The worker's sleep, issued as a raw syscall from this file so that the +/// thread's innermost frame (the first line of /threads//stack, which +/// 9proc/test/debug.sh resolves through /addr) is in demo/main.zig rather than in std. +/// EINTR (a capture signal) just ends the nap early. +inline fn napMs(ms: u64) void { + var req: linux.timespec = .{ .sec = @intCast(ms / 1000), .nsec = @intCast((ms % 1000) * 1_000_000) }; + switch (builtin.cpu.arch) { + .x86_64 => _ = asm volatile ("syscall" + : [ret] "={rax}" (-> usize), + : [number] "{rax}" (@intFromEnum(linux.SYS.nanosleep)), + [arg1] "{rdi}" (@intFromPtr(&req)), + [arg2] "{rsi}" (@as(usize, 0)), + : .{ .rcx = true, .r11 = true, .memory = true }), + .aarch64 => _ = asm volatile ("svc #0" + : [ret] "={x0}" (-> usize), + : [number] "{x8}" (@intFromEnum(linux.SYS.nanosleep)), + [arg1] "{x0}" (@intFromPtr(&req)), + [arg2] "{x1}" (@as(usize, 0)), + : .{ .memory = true }), + else => plinux.sleepMs(ms), + } +} + +// -- /runtime/ctl ------------------------------------------------------------- + +/// The /runtime/ctl handler. The core stages the output and commits it only +/// on success, so a failed command leaves the previous result in place +/// (test/adv_9proc_hostile.py checks that). +fn ctl(ctx: *anyopaque, cmd: []const u8, out: *Writer) anyerror!void { + const a: *App = @ptrCast(@alignCast(ctx)); + const line = std.mem.trim(u8, cmd, " \t\r\n\x00"); + var it = std.mem.tokenizeScalar(u8, line, ' '); + const verb = it.next() orelse return error.BadCommand; + if (std.mem.eql(u8, verb, "echo")) { + try out.writeAll(std.mem.trimStart(u8, line[verb.len..], " \t")); + } else if (std.mem.eql(u8, verb, "fib")) { + const n = std.fmt.parseInt(u32, it.next() orelse return error.BadCommand, 10) catch return error.BadCommand; + if (n > 93) return error.BadCommand; // fib(94) overflows u64 + try out.print("{d}", .{fib(n)}); + } else if (std.mem.eql(u8, verb, "sleep-ms")) { + const n = std.fmt.parseInt(u64, it.next() orelse return error.BadCommand, 10) catch return error.BadCommand; + const ms = @min(n, 10_000); + a.probe.sleepServing(ms); + try out.print("slept {d} ms", .{ms}); + } else if (std.mem.eql(u8, verb, "add")) { + const x = std.fmt.parseInt(i64, it.next() orelse return error.BadCommand, 10) catch return error.BadCommand; + const y = std.fmt.parseInt(i64, it.next() orelse return error.BadCommand, 10) catch return error.BadCommand; + try out.print("{d}", .{x +% y}); + } else if (std.mem.eql(u8, verb, "trap")) { + trap_requested.store(true, .release); + try out.writeAll("trap armed: the worker stops in @breakpoint() on its next tick"); + } else if (std.mem.eql(u8, verb, "panic")) { + panic_requested.store(true, .release); + try out.writeAll("panic armed: the worker panics on its next tick"); + } else return error.BadCommand; +} + +/// fib(n) for n <= 93 (fib(93) is the largest that fits u64). +fn fib(n: u32) u64 { + std.debug.assert(n <= 93); + if (n == 0) return 0; + var a: u64 = 0; + var b: u64 = 1; + for (1..n) |_| { + const c = a + b; + a = b; + b = c; + } + return b; +} + +// -- main --------------------------------------------------------------------- + +const usage_text = + \\usage: 9proc-demo [--unix PATH | --tcp IP:PORT | --stdio] [--no-hold] + \\ + \\A demo 9P2000 file server exposing this binary's build-time, comptime and + \\runtime facts, plus a debugger-shaped view of the process (threads, stacks, + \\memory, breakpoints, panics). Default is --stdio (9P on fd 0/1). + \\--no-hold lets a panic abort at once instead of waiting for /panic/ctl. + \\ +; + +pub fn main(init: std.process.Init) !void { + run(init) catch |e| switch (e) { + // Already reported on stderr; no stack trace wanted. + error.Usage, error.Syscall => std.process.exit(1), + else => return e, + }; +} + +fn run(init: std.process.Init) !void { + // Transparent huge pages would back the first touched page of every + // client buffer with 2 MiB. Best effort: ignore failure. + _ = linux.prctl(@intFromEnum(linux.PR.SET_THP_DISABLE), 1, 0, 0, 0); + const arena = init.arena.allocator(); + const args = try init.minimal.args.toSlice(arena); + + const Mode = enum { stdio, unix, tcp }; + var mode: Mode = .stdio; + var address: []const u8 = ""; + var hold = true; + var i: usize = 1; + while (i < args.len) : (i += 1) { + const a = args[i]; + if (std.mem.eql(u8, a, "--stdio")) { + mode = .stdio; + } else if (std.mem.eql(u8, a, "--unix") or std.mem.eql(u8, a, "--tcp")) { + i += 1; + if (i >= args.len) { + std.debug.print("9proc-demo: {s} needs an argument\n{s}", .{ a, usage_text }); + return error.Usage; + } + mode = if (a[2] == 'u') .unix else .tcp; + address = args[i]; + } else if (std.mem.eql(u8, a, "--no-hold")) { + hold = false; + } else if (std.mem.eql(u8, a, "--help") or std.mem.eql(u8, a, "-h")) { + std.debug.print("{s}\nstatic memory: {d} bytes ({d} clients, msize {d})\n", .{ usage_text, static_bytes, max_clients, max_msize }); + return; + } else { + std.debug.print("9proc-demo: unknown argument {s}\n{s}", .{ a, usage_text }); + return error.Usage; + } + } + + // argv, env and cwd are gathered once, into the arena. + var argv_text: std.ArrayList(u8) = .empty; + for (args) |a| { + try argv_text.appendSlice(arena, a); + try argv_text.append(arena, '\n'); + } + var env_text: std.ArrayList(u8) = .empty; + for (init.minimal.environ.block.view().slice) |entry| { + try env_text.appendSlice(arena, std.mem.span(entry)); + try env_text.append(arena, '\n'); + } + var cwd_buf: [4096]u8 = undefined; + const cwd_rc = linux.getcwd(&cwd_buf, cwd_buf.len); + const cwd_text: []const u8 = if (linux.errno(cwd_rc) == .SUCCESS) + try arena.dupe(u8, std.mem.sliceTo(cwd_buf[0..cwd_rc], 0)) + else + ""; + + const app = app_mem.get(); + const shared = shared_mem.get(); + const probe = probe_mem.get(); + const scratch = scratch_mem.get(); + app.* = .{ + .info = .{ .argv = argv_text.items, .env = env_text.items, .cwd = cwd_text, .start_mono = runtime.monotonicSecs() }, + .probe = probe, + .shared = shared, + }; + shared.* = .init(app); + try shared.expose("state", &state); + scratch.* = try proc9.Scratch.init(init.gpa, 512 << 20); + // Freed on the way out so a Debug build's allocator does not report the + // tree as leaked (with a stack trace on stderr) after a clean --stdio EOF. + defer scratch.deinit(); + scratch.max_file = 64 << 20; + scratch.now = &runtime.realtimeSecs; + try shared.addProvider(scratch.provider("scratch")); + + const listen: plinux.Listen = switch (mode) { + .stdio => .{ .client = .{ .in = 0, .out = 1 } }, + .unix => .{ .unix = address }, + .tcp => .{ .tcp = address }, + }; + probe.init(shared, probe_storage_mem.get(), .{ .io = init.io, .listen = listen, .msize = max_msize, .hold_on_panic = hold }) catch |e| { + switch (e) { + error.PathTooLong => std.debug.print("9proc-demo: unix socket path too long\n", .{}), + error.BadAddress => std.debug.print("9proc-demo: --tcp wants an IPv4 literal a.b.c.d:port\n", .{}), + error.Syscall => std.debug.print("9proc-demo: {t} ({t})\n", .{ e, probe.last_errno }), + else => std.debug.print("9proc-demo: {t}\n", .{e}), + } + return if (e == error.PathTooLong or e == error.BadAddress) error.Usage else error.Syscall; + }; + // Failure past this point (thread spawn) still closes the listener and + // restores the signal dispositions. + errdefer probe.stop(); + app.info.clients = probe.clientCounter(); + + const worker = std.Thread.spawn(.{}, workerLoop, .{}) catch |e| { + std.debug.print("9proc-demo: worker thread: {t}\n", .{e}); + return error.Syscall; + }; + worker.detach(); + + switch (mode) { + .unix => std.debug.print("9proc-demo: listening on unix!{s}\n", .{address}), + .tcp => std.debug.print("9proc-demo: listening on tcp!{s}\n", .{address}), + .stdio => {}, + } + probe.start() catch |e| { + std.debug.print("9proc-demo: thread spawn failed: {t}\n", .{e}); + return error.Syscall; + }; + // SIGTERM/SIGINT end the loop cleanly: the socket file is unlinked, the + // signal dispositions restored and the scratch tree freed. + // A disposition of SIG_IGN inherited from the parent is left alone (Unix + // convention): 9ns runs a --spawn server with SIGINT ignored so that + // Ctrl-C on the terminal reaches only the program, not its file server. + const term: linux.Sigaction = .{ .handler = .{ .handler = onTerm }, .mask = linux.sigemptyset(), .flags = 0 }; + for ([_]linux.SIG{ .TERM, .INT }) |sig| { + var old: linux.Sigaction = undefined; + _ = linux.sigaction(sig, null, &old); + if (old.handler.handler != linux.SIG.IGN) _ = linux.sigaction(sig, &term, null); + } + probe.wait(); + probe.stop(); +} + +/// Async-signal-safe: an atomic store and one eventfd write. +fn onTerm(_: linux.SIG) callconv(.c) void { + probe_mem.get().requestStop(); +} + +// -- tests -------------------------------------------------------------------- + +test "ctl commands" { + const probe = probe_mem.get(); + const shared = shared_mem.get(); + var a: App = .{ .info = .{}, .probe = probe, .shared = shared }; + probe.thread_tid = .init(0); + probe.serving = .initEmpty(); + probe.nested = 0; + var buf: [128]u8 = undefined; + var w: Writer = .fixed(&buf); + try ctl(&a, "add 2 3\n", &w); + try std.testing.expectEqualStrings("5", w.buffered()); + try std.testing.expectError(error.BadCommand, ctl(&a, "nope", &w)); + w = .fixed(&buf); + try ctl(&a, "fib 93", &w); + try std.testing.expectEqualStrings("12200160415121876738", w.buffered()); + w = .fixed(&buf); + try std.testing.expectError(error.BadCommand, ctl(&a, "fib 94", &w)); + try std.testing.expectError(error.BadCommand, ctl(&a, "frobnicate", &w)); + w = .fixed(&buf); + try ctl(&a, "echo hi there ", &w); + try std.testing.expectEqualStrings("hi there", w.buffered()); + w = .fixed(&buf); + try ctl(&a, "sleep-ms 1", &w); + try std.testing.expectEqualStrings("slept 1 ms", w.buffered()); + w = .fixed(&buf); + try ctl(&a, "trap", &w); + try std.testing.expect(trap_requested.swap(false, .acq_rel)); + w = .fixed(&buf); + try ctl(&a, "panic", &w); + try std.testing.expect(panic_requested.swap(false, .acq_rel)); +} + +test "static footprint is what the file comment says" { + try std.testing.expect(@sizeOf(ProbeStorage) > 16 * (3 << 20)); + try std.testing.expect(@sizeOf(ProbeStorage) < 100 << 20); +} diff --git a/9proc/docs/LIBRARY.md b/9proc/docs/LIBRARY.md new file mode 100644 index 0000000..eef7dad --- /dev/null +++ b/9proc/docs/LIBRARY.md @@ -0,0 +1,287 @@ +# 9proc: a 9P debug/introspection server as a library + +The demo server that 9ns's tests use grows into a library any Zig program +can embed: a debugger-shaped interface where the protocol is just files. +Anything that can read a filesystem (a shell, an agent, an editor, `9p`, +9ns) can inspect a running process: build facts, comptime type layouts, +live values, threads and their stacks, memory, breakpoints, panics. + +Design rules (non-negotiable, they mirror cloud9): + +1. **The core is freestanding.** No allocator, no OS, no threads, no `std.Io`. + Caller-owned buffers, fixed-capacity tables sized at comptime. It must + compile for `riscv32-freestanding-none` (the ESP32-P4 firmware target, + `../05-zig-p4`), and `zig build 9proc-check-freestanding` proves it. +2. **Every dependency on a runtime is an explicit argument.** Features that + truly need an `Allocator` or an `std.Io` take them in their `init`; nothing + reaches for `std.heap.page_allocator` or a global `Io`. Where memory is + needed it is preferably a caller-provided `[]u8` or a comptime-sized + `Storage` struct the caller places in static memory. +3. **All allocation happens up front**, at init, from what the caller passed. + Steady-state operation does not allocate. +4. **Platform layers are separate modules** (`9proc.linux`) and are the + only places that touch sockets, threads, signals, `/proc` or `std.debug`. + +``` + 9proc/src/root.zig pub const core, vars, scratch, linux (linux only), Server(cfg) + 9proc/src/core.zig Tree/Server engine on cloud9.Server: fids, walks, dir reads, providers + 9proc/src/vars.zig comptime value renderers (@typeInfo) for /vars + 9proc/src/scratch.zig in-memory read/write tree provider (takes an Allocator) + 9proc/src/linux/probe.zig background thread + poll loop + unix/tcp/fd listeners + 9proc/src/linux/debug.zig threads, stacks, registers, addr→source, memory, breakpoints, panic + 9proc/demo/main.zig the `9proc-demo` binary: embeds everything, worker thread, exposed vars + +(paths from the cloud9 root; the library is the module `9proc` that +cloud9's `build.zig` exports next to `cloud9`, wired by `9proc/build.zig`) +``` + +## Core (`core.zig`) + +```zig +pub const Config = struct { + name: []const u8 = "9proc", // /README and Stat uid/gid + types: []const type = &.{}, // /comptime/types//... + decls_of: ?type = null, // /comptime/decls lists this type's pub decls + fns: type = struct {}, // /runtime/fn/: pub fn (ctx: *anyopaque, w: *std.Io.Writer) anyerror!void + ctl: ?*const fn (ctx: *anyopaque, cmd: []const u8, out: *std.Io.Writer) anyerror!void = null, // /ctl + max_fids: u16 = 64, + max_providers: u8 = 8, + max_vars: u8 = 32, + /// Dynamic file contents are generated at open time into per-fid snapshot + /// slots so that reads at arbitrary offsets are consistent. + snapshot_slots: u8 = 8, + snapshot_bytes: u32 = 16 * 1024, +}; + +pub fn Server(comptime cfg: Config) type { + return struct { + pub const Storage = struct { // caller places this in static memory + in: [msize]u8, out: [msize]u8, snapshots: [cfg.snapshot_slots][cfg.snapshot_bytes]u8, + }; + pub const Shared = struct { // state common to all connections (providers, vars) + pub fn init(name_ctx: *anyopaque) Shared; + pub fn addProvider(s: *Shared, p: Provider) error{Full}!void; + pub fn expose(s: *Shared, name: []const u8, ptr: anytype) error{Full}!void; // typed value → /vars/ + }; + pub const Conn = struct { // one 9P connection, push/step/output like cloud9 + pub fn init(shared: *Shared, storage: *Storage, msize: u32) Conn; + pub fn push(c: *Conn, bytes: []const u8) usize; // feed transport bytes + pub fn step(c: *Conn) error{Protocol}!bool; // handle ≤ 1 request; false = nothing to do + pub fn output(c: *const Conn) []const u8; // bytes to send + pub fn wrote(c: *Conn, n: usize) void; + pub fn hangup(c: *Conn) void; // drop fids, tell providers + }; + }; +} +``` + +`step` drives `cloud9.Server.receive/reply/negotiate` and the backend: the +static tree (comptime-generated from `cfg`: `/README`, `/build/*` via a +`build_options`-like struct passed in `cfg.build`, `/comptime/types/*`, +`/comptime/decls`, `/runtime/fn/*`, `/ctl`, `/vars/*`) plus **providers**. + +A provider is a runtime vtable mounted at a top-level name. It owns a subtree +with its own naming (dynamic directories such as `/threads/` or +`/addr/` cannot be enumerated at comptime): + +```zig +pub const Provider = struct { + name: []const u8, + ctx: *anyopaque, + vtable: *const VTable, + pub const Handle = u64; // provider-defined node id; 0 = provider root + pub const VTable = struct { + walk: *const fn (ctx, parent: Handle, name: []const u8) Error!Handle, + stat: *const fn (ctx, h: Handle, out: *NodeStat) Error!void, // kind (dir/file), mode, length, mtime + list: *const fn (ctx, dir: Handle, index: usize, out: *NodeStat) Error!bool, // nth entry; false when done + open: *const fn (ctx, h: Handle, mode: u8) Error!void, + read: *const fn (ctx, h: Handle, offset: u64, buf: []u8) Error!usize, + write: *const fn (ctx, h: Handle, offset: u64, data: []const u8) Error!usize, + create: ?*const fn (ctx, dir: Handle, name: []const u8, perm: u32, mode: u8) Error!Handle, + remove: ?*const fn (ctx, h: Handle) Error!void, + wstat: ?*const fn (ctx, h: Handle, st: *const cloud9.Stat) Error!void, + clunk: *const fn (ctx, h: Handle) void, // fid released (also on hangup) + }; + pub const Error = error{ NotFound, Exists, Perm, NotDir, IsDir, NotEmpty, BadOffset, NoSpace, Io, Unsupported }; +}; +``` + +Error → Rerror text mapping lives in one place in the core, using the Plan 9 +strings 9ns's bridge already understands (`file does not exist`, +`permission denied`, `file already exists`, `directory not empty`, +`not a directory`, `is a directory`, `bad offset`, `no space`, `i/o error`, +`not supported`). + +Directory reads follow the 9P rule (offset 0 or previous offset+count, never +split a record). Dynamic file reads: on open the content is generated once +into a snapshot slot (`open` runs the generator; `read` serves the slot; a +read at offset 0 regenerates); no free slot → Rerror `too many open dynamic +files`. Stats of dynamic files report length 0. + +Qids: static nodes get comptime paths; provider nodes get +`(provider index << 56) | handle`. + +Static memory: `Server(cfg).Storage` per connection, `Shared` once. No heap. +The core has unit tests driven through `cloud9.Client` in memory (like today). + +## Value renderers (`vars.zig`) + +`expose(name, ptr: anytype)` builds at comptime a `VTable` for +`@TypeOf(ptr.*)`: + +``` +/vars//value rendered text (structs: "field: value" lines, nested indented; unions: tag + payload; + optionals: "null" or the value; enums: tag; ints/floats/bools; []const u8 and [*:0]const u8 + as quoted strings (≤ 256 bytes); other pointers as 0x… never followed; arrays/slices ≤ 64 elements) +/vars//type @typeName +/vars//size @sizeOf +/vars//addr 0x… +/vars//raw the bytes (length = @sizeOf) +/vars//f//... same layout recursively for struct fields (depth ≤ 4), leaves writable: + writing text to a scalar's `value` parses and stores it (ints: decimal/0x, bools, floats, enums by tag) +``` + +Rendering is by a comptime-generated function table; no allocation. +Writes to scalars are plain stores (not atomic; documented). + +## Scratch provider (`scratch.zig`) + +The in-memory read/write tree from the current server, as a provider, with +`init(allocator, budget_bytes)`; the only core-level component that takes an +allocator, and it is optional. + +## Linux layer (`linux/probe.zig`) + +```zig +pub const Probe = struct { + pub const Options = struct { + io: std.Io, // for std.debug symbolization + listen: union(enum) { unix: []const u8, tcp: []const u8, fd: i32 }, + max_clients: u8 = 8, + msize: u32 = 64 * 1024, + hold_on_panic: bool = true, + capture_signal: u8 = SIGRTMIN + 3, // used to snapshot other threads + breakpoints: bool = true, // install the SIGTRAP handler + }; + pub fn Storage(comptime max_clients: u8) type; // static: per-client Server.Storage + poll table + pub fn init(p: *Probe, shared: *Server.Shared, storage: *Storage, opts: Options) !void; // listens, registers the debug provider + pub fn start(p: *Probe) !void; // spawns ONE background thread running a poll loop over listener + clients + pub fn stop(p: *Probe) void; // closes, joins +}; +``` + +One thread, `poll()` over the listener and every connection; each connection +is a core `Conn` fed with `push`/`step`/`output`. No per-connection threads. +Symbolization uses `std.debug.getSelfDebugInfo()` with the `io` passed in and +a caller-provided fixed buffer as the text arena. + +## Debug provider (`linux/debug.zig`) + +Mounted as `/threads`, `/addr`, `/mem`, `/hex`, `/breakpoints`, `/panic`. + +``` +/threads/ one directory per tid, enumerated from /proc/self/task at list time +/threads//name comm +/threads//stat state letter + a few fields from /proc/self/task//stat +/threads//stack "#n 0x in (::)" per frame +/threads//regs " 0x" per general register, from the captured cpu context +/addr/ dynamic dir: walk of any hex address yields a file "fn\nfile:line:col\nmodule\n" +/mem/maps /proc/self/maps served by pread at the requested offset (any size) +/mem/ raw bytes at address+offset via process_vm_readv/writev (never faults); writable +/hex/ hexdump text of 256 bytes at address (+offset), like std.debug.dumpHex +/breakpoints/ directory of tids currently stopped in @breakpoint() +/breakpoints//stack, regs as above +/breakpoints//ctl write "continue" (or "step"? no: continue only) to resume +/panic/message the panic message, empty before any panic +/panic/stack frames of the panicking thread +/panic/ctl write "continue" to let the default panic handler run (abort) +``` + +**Capturing another thread** (`stack`, `regs`): the server thread `tgkill`s +the target with `capture_signal`. The handler (SA_SIGINFO, async-signal-safe: +no allocation, no locks) copies the `cpu_context.Native` obtained through +`std.debug.cpu_context.fromPosixSignalContext` into a slot and futex-waits. +The server thread unwinds with `std.debug.StackIterator.init(&ctx)` while the +target is parked, symbolizes, then releases the slot; the target resumes. The +server's own thread unwinds itself directly. Timeout 250 ms → Rerror +`thread did not respond`. Threads blocked in uninterruptible syscalls simply +time out. A target parked while holding std.debug's `SelfInfo` lock (it was +printing a stack trace itself) cannot be unwound without deadlocking; the +probe detects that with `tryLock`, releases the target and answers +`i/o error` (registers still work). + +**Breakpoints**: `@breakpoint()` raises SIGTRAP on the executing thread only. +The installed handler stores the context in a slot, marks the thread paused, +and futex-waits until `/breakpoints//ctl` receives `continue`. On x86_64 +the saved PC already points past `int3`; on aarch64 the handler advances PC by +4 (`brk`) before returning, but only for kernel-generated traps +(`si_code > 0`); a user-sent `SIGTRAP` (`kill -TRAP`, `tgkill`) parks the +thread exactly where it was, which makes it a usable "pause this thread" +request. Other threads keep running; a slot table (`max_paused`, default 16) +bounds simultaneous pauses and, when it is full, the trapping thread simply +steps over the breakpoint (`traps_skipped` counts these). The probe's own +serving thread is never parked or held: a trap or panic on it goes straight +to the default behaviour, since nobody could write its `ctl` files. + +**Panics**: `pub const panic = proc9.linux.panic;` in the root module +(built with `std.debug.FullPanic`). The first panic records message and a +stack capture (`captureCurrentStackTrace` with `first_address`), publishes +them, and, if `hold_on_panic` and the probe is running, futex-waits until +`/panic/ctl` says `continue`; then `std.debug.defaultPanic` runs (prints the +trace and aborts). A nested or second panic goes straight to the default. + +Signal handlers are installed by `Probe.init` (breakpoints optional) and +restored by `stop`. + +## Demo (`demo/main.zig`, binary `9proc-demo`) + +Keeps every path the existing tests read (`/build/*`, `/comptime/types/Qid/*`, +`/comptime/decls`, `/runtime/fn/now|hostname|…`, `/runtime/ctl` with +`add|echo|fib|sleep-ms`, `/runtime/pid|ppid|uptime|argv|cwd|env|clients`, +`/scratch`), served by the library. Adds: + +* a worker thread running `workerLoop` that increments an exposed + `State { ticks: u64, phase: enum, last_job: Job }` (`/vars/state/...`); +* `/runtime/ctl` commands `trap` (the worker executes `@breakpoint()` on its + next tick) and `panic` (the worker panics with a message); +* `--stdio | --unix PATH | --tcp IP:PORT` as today, `--no-hold` to disable + panic holding. + +`main` passes `init.io` and an explicit allocator to the pieces that need one; +the demo's Storage is a global. + +## Verification + +* Unit tests: core (in-memory client drives every op incl. providers and + snapshots), vars (render/set for every category), scratch, debug (capture + own thread and a helper thread; breakpoint pause/continue on a helper + thread; panic record path without holding). +* `zig build 9proc-check-freestanding`: compiles `core.zig` + `vars.zig` + for `riscv32-freestanding-none` with a tiny freestanding root that + instantiates `Server(cfg)` with static Storage. +* `zig build 9proc-test` (library and demo unit tests) and + `zig build 9proc-debug-test` (linux/debug.zig). +* 9ns's `test/integration.sh` unchanged and passing; `test/debug.sh` + (`zig build 9proc-debug-itest`) through 9ns: read the worker's + stack (contains `workerLoop` and `demo/main.zig:`), + resolve a frame through `/addr`, dump `/hex` of the exposed state, read and + write `/vars/state/f/ticks/value`, trap → `/breakpoints` lists the worker, + its stack shows `workerLoop`, `continue` resumes (ticks keep increasing), + panic → `/panic/message`, `/panic/stack`, `continue` → server exits + non-zero. +* Adversarial pass afterwards (`zig build 9proc-adv`: hostile client + against the server and the core, signal races, memory reads of unmapped + addresses, panic while a capture is in flight; `test/adversarial.sh` runs + the same suites by hand). + +## Known upstream issue (Zig 0.16 std.debug) + +`std.debug.SelfInfo` for ELF (`std/debug/SelfInfo/Elf.zig`, `findModule`) +rebuilds its module list whenever it is asked about an address outside every +known module. That frees each module's `Dwarf.Unwind` and CIE list but leaves +`unwind_cache` entries pointing into the freed memory, so later unwinds read +freed data: empty traces, "unwind info invalid", or segfaults once the arena +reuses the block. `linux/debug.zig` records the executable's `PT_LOAD` ranges +at init and refuses to hand std an address outside them (`knownCode`), which +is why `/addr/` of a bogus address renders `?` instead of poisoning the +process. Worth reporting upstream; the guard can go once std clears the cache. diff --git a/9proc/src/core.zig b/9proc/src/core.zig new file mode 100644 index 0000000..cead8f3 --- /dev/null +++ b/9proc/src/core.zig @@ -0,0 +1,2656 @@ +//! The freestanding 9P2000 introspection engine: a static tree generated at +//! comptime from a `Config` (README, /build, /comptime, /runtime/fn, /ctl, +//! /vars) plus runtime `Provider`s mounted at the top level, served over a +//! `cloud9.Server` connection. No allocator, no OS, no threads: every buffer is +//! caller-owned (`Storage`, `Shared`, `Conn`), every table is sized at comptime. +//! See docs/LIBRARY.md. +const std = @import("std"); +const builtin = @import("builtin"); +const cloud9 = @import("cloud9"); +const vars = @import("vars.zig"); +const Writer = std.Io.Writer; + +/// Longest file name accepted in a create or rename. +pub const max_name: usize = 255; + +/// A dynamic-file generator (`Config.fns`, `Config.runtime`): writes the file's +/// content into `w` at open time (and again at each read from offset 0). +pub const Gen = *const fn (ctx: *anyopaque, w: *Writer) anyerror!void; +/// The /ctl command handler: `cmd` is the written text, `out` receives the +/// result that later reads of /ctl return. +pub const Ctl = *const fn (ctx: *anyopaque, cmd: []const u8, out: *Writer) anyerror!void; + +pub const Config = struct { + /// Appears in /README and as uid/gid/muid of every Stat. + name: []const u8 = "9proc", + /// A type whose pub decls `zig_version`, `target`, `optimize`, `time` + /// and `change` (all `[]const u8`) become the files of /build. `null` + /// omits /build. + build: ?type = null, + /// /comptime/types//{name,size,align,fields}. + types: []const type = &.{}, + /// /comptime/decls lists this type's pub decls (empty when null). + decls_of: ?type = null, + /// /runtime/fn/: every pub decl is a `fn (ctx: *anyopaque, w: *std.Io.Writer) anyerror!void`. + fns: type = struct {}, + /// /runtime/: same signature as `fns`, one level up (pid, uptime, ...). + runtime: type = struct {}, + /// The /ctl handler; `null` omits /ctl. + ctl: ?Ctl = null, + /// Where /ctl lives: the root or /runtime/ctl. + ctl_dir: enum { root, runtime } = .root, + /// Capacity of the ctl result (in `Shared`). + ctl_bytes: u32 = 4096, + /// Largest negotiable msize; sizes `Storage.in/out/data`. + msize: u32 = 8192, + max_fids: u16 = 64, + max_providers: u8 = 8, + max_vars: u8 = 32, + /// Dynamic file contents are generated at open time into per-fid snapshot + /// slots so that reads at arbitrary offsets are consistent. + snapshot_slots: u8 = 8, + snapshot_bytes: u32 = 16 * 1024, +}; + +/// Attributes of a provider node, filled by `VTable.stat` and `VTable.list`. +pub const NodeStat = struct { + /// Permission bits plus `cloud9.dmdir`/`dmappend`/`dmexcl`. + mode: u32, + length: u64 = 0, + atime: u32 = 0, + mtime: u32 = 0, + /// Becomes the qid version. + version: u32 = 0, + /// The entry name (`list`) or the node's own name (`stat`; ignored for the + /// provider root, whose name is the mount name). Must stay valid until the + /// provider's next call. + name: []const u8 = "", + /// Filled by `list`: the entry's handle. Not retained by the core. + handle: Provider.Handle = 0, + /// A stable identity for the qid path (low 56 bits), for providers whose + /// handles are not stable across the node's life (e.g. memory addresses + /// that an allocator may reuse). 0 means "the handle is the path". + path: u64 = 0, + + pub fn isDir(s: NodeStat) bool { + return s.mode & cloud9.dmdir != 0; + } +}; + +/// A runtime subtree mounted at a top-level name. +/// +/// Handle lifetime: every handle returned by `walk` or `create` is released by +/// the core with exactly one `clunk` (after `close` if the fid was open). The +/// root handle 0 is never obtained through `walk`, so providers must treat +/// `clunk(0)` as a no-op. `walk` must accept "." on any node, file or directory +/// (a fresh reference to the same node; the core clones fids with it), and ".." +/// on directories (except at the root, which the core resolves itself). +pub const Provider = struct { + name: []const u8, + ctx: *anyopaque, + vtable: *const VTable, + + /// Provider-defined node id; 0 = provider root. + pub const Handle = u64; + pub const root: Handle = 0; + + pub const Error = error{ NotFound, Exists, Perm, NotDir, IsDir, NotEmpty, BadOffset, NoSpace, Io, Unsupported, Excl }; + + pub const VTable = struct { + walk: *const fn (ctx: *anyopaque, parent: Handle, name: []const u8) Error!Handle, + stat: *const fn (ctx: *anyopaque, h: Handle, out: *NodeStat) Error!void, + /// The `index`-th entry of `dir`; false when done. + list: *const fn (ctx: *anyopaque, dir: Handle, index: usize, out: *NodeStat) Error!bool, + open: *const fn (ctx: *anyopaque, h: Handle, mode: u8) Error!void, + read: *const fn (ctx: *anyopaque, h: Handle, offset: u64, buf: []u8) Error!usize, + write: *const fn (ctx: *anyopaque, h: Handle, offset: u64, data: []const u8) Error!usize, + /// Returns the new node, already open with `mode`. + create: ?*const fn (ctx: *anyopaque, dir: Handle, name: []const u8, perm: u32, mode: u8) Error!Handle = null, + remove: ?*const fn (ctx: *anyopaque, h: Handle) Error!void = null, + /// Only name, length, mode and mtime can differ from the current stat + /// (the core has already checked the immutable fields and the name). + wstat: ?*const fn (ctx: *anyopaque, h: Handle, st: *const cloud9.Stat) Error!void = null, + /// An open fid on `h` was released (before `clunk`). + close: ?*const fn (ctx: *anyopaque, h: Handle) void = null, + /// A fid holding `h` was released (also on hangup and Tversion). + clunk: *const fn (ctx: *anyopaque, h: Handle) void, + }; +}; + +/// The Plan 9 error string for any error the engine or a provider can raise. +pub fn ename(err: anyerror) []const u8 { + return switch (err) { + error.NotFound, error.NoFile => "file does not exist", + error.Perm => "permission denied", + error.Exists => "file already exists", + error.NotEmpty => "directory not empty", + error.NotDir => "not a directory", + error.IsDir => "is a directory", + error.BadOffset => "bad offset", + error.NoSpace => "no space left on device", + error.Io => "i/o error", + error.Unsupported => "not supported", + error.Excl => "exclusive use file already open", + error.FidInUse => "fid in use", + error.UnknownFid => "unknown fid", + error.NotOpen => "file not open", + error.AlreadyOpen => "file already open", + error.AuthNotRequired => "authentication not required", + error.BadCommand => "bad command", + error.BadName => "bad file name", + error.Invalid, error.BadValue => "bad value", + error.TooManyFids => "too many fids", + error.NoSnapshot => "too many open dynamic files", + error.WriteFailed => "no space in buffer", + error.ReplyTooLarge => "reply too large for msize", + error.OutOfMemory => "out of memory", + else => "i/o error", + }; +} + +/// A "don't care" Twstat: every field left as it is. +pub const stat_dontcare: cloud9.Stat = .{ + .type = 0xFFFF, + .dev = 0xFFFF_FFFF, + .qid = .{ .type = 0xFF, .version = 0xFFFF_FFFF, .path = 0xFFFF_FFFF_FFFF_FFFF }, + .mode = 0xFFFF_FFFF, + .atime = 0xFFFF_FFFF, + .mtime = 0xFFFF_FFFF, + .length = 0xFFFF_FFFF_FFFF_FFFF, + .name = "", + .uid = "", + .gid = "", + .muid = "", +}; + +pub fn validName(name: []const u8) error{BadName}!void { + if (name.len == 0 or name.len > max_name) return error.BadName; + if (std.mem.eql(u8, name, ".") or std.mem.eql(u8, name, "..")) return error.BadName; + if (std.mem.indexOfAny(u8, name, "/\x00") != null) return error.BadName; +} + +/// "YYYY-MM-DDTHH:MM:SSZ" as unix seconds, or null. +pub fn parseIso8601(s: []const u8) ?u32 { + if (s.len != 20 or s[4] != '-' or s[7] != '-' or s[10] != 'T' or s[13] != ':' or s[16] != ':' or s[19] != 'Z') return null; + const y = std.fmt.parseInt(i64, s[0..4], 10) catch return null; + const mo = std.fmt.parseInt(i64, s[5..7], 10) catch return null; + const d = std.fmt.parseInt(i64, s[8..10], 10) catch return null; + const h = std.fmt.parseInt(i64, s[11..13], 10) catch return null; + const mi = std.fmt.parseInt(i64, s[14..16], 10) catch return null; + const sec = std.fmt.parseInt(i64, s[17..19], 10) catch return null; + if (mo < 1 or mo > 12 or d < 1 or d > 31 or h > 23 or mi > 59 or sec > 60) return null; + // Howard Hinnant's days_from_civil. + const yy = if (mo <= 2) y - 1 else y; + const era = @divFloor(yy, 400); + const yoe = yy - era * 400; + const mp = if (mo > 2) mo - 3 else mo + 9; + const doy = @divFloor(153 * mp + 2, 5) + d - 1; + const doe = yoe * 365 + @divFloor(yoe, 4) - @divFloor(yoe, 100) + doy; + const days = era * 146097 + doe - 719468; + const total = days * 86400 + h * 3600 + mi * 60 + sec; + if (total < 0 or total > std.math.maxInt(u32)) return null; + return @intCast(total); +} + +/// The last component of @typeName(T): "wire.Qid" -> "Qid". +pub fn shortTypeName(comptime T: type) []const u8 { + const full = @typeName(T); + const dot = std.mem.lastIndexOfScalar(u8, full, '.') orelse return full; + return full[dot + 1 ..]; +} + +/// The /comptime/types//fields text: "name: type @offset" per line. +pub fn fieldsText(comptime T: type) []const u8 { + comptime { + @setEvalBranchQuota(200_000); + var s: []const u8 = ""; + switch (@typeInfo(T)) { + .@"struct" => |info| for (info.fields) |f| { + if (info.layout == .@"packed") { + s = s ++ std.fmt.comptimePrint("{s}: {s} @{d}b\n", .{ f.name, @typeName(f.type), @bitOffsetOf(T, f.name) }); + } else if (f.is_comptime) { + s = s ++ std.fmt.comptimePrint("{s}: {s} (comptime)\n", .{ f.name, @typeName(f.type) }); + } else { + s = s ++ std.fmt.comptimePrint("{s}: {s} @{d}\n", .{ f.name, @typeName(f.type), @offsetOf(T, f.name) }); + } + }, + .@"union" => |info| for (info.fields) |f| { + s = s ++ f.name ++ ": " ++ @typeName(f.type) ++ "\n"; + }, + .@"enum" => |info| for (info.fields) |f| { + s = s ++ std.fmt.comptimePrint("{s} = {d}\n", .{ f.name, f.value }); + }, + else => s = @typeName(T) ++ "\n", + } + return s; + } +} + +fn declsText(comptime T: type) []const u8 { + comptime { + @setEvalBranchQuota(20_000); + const decls = switch (@typeInfo(T)) { + inline .@"struct", .@"union", .@"enum", .@"opaque" => |info| info.decls, + else => &[_]std.builtin.Type.Declaration{}, + }; + var s: []const u8 = ""; + for (decls) |d| s = s ++ d.name ++ "\n"; + return s; + } +} + +/// Wraps `Fns.` in a function of exactly the `Gen` signature. +fn genFor(comptime Fns: type, comptime name: []const u8) Gen { + return &struct { + fn g(ctx: *anyopaque, w: *Writer) anyerror!void { + return @field(Fns, name)(ctx, w); + } + }.g; +} + +/// A node of the static tree, described at comptime. +pub const Node = struct { + name: []const u8, + kind: Kind, + children: []const Node = &.{}, + content: []const u8 = "", + gen: ?Gen = null, + + pub const Kind = enum(u8) { dir, static, dynamic, ctl, vars }; + + fn isDir(n: Node) bool { + return n.kind == .dir or n.kind == .vars; + } +}; + +fn genNodes(comptime Fns: type) [@typeInfo(Fns).@"struct".decls.len]Node { + const decls = @typeInfo(Fns).@"struct".decls; + var arr: [decls.len]Node = undefined; + for (decls, 0..) |d, i| arr[i] = .{ .name = d.name, .kind = .dynamic, .gen = genFor(Fns, d.name) }; + return arr; +} + +pub fn Server(comptime cfg: Config) type { + return struct { + const Self = @This(); + + // -- the static tree ------------------------------------------------ + + pub const readme_text = std.fmt.comptimePrint( + \\{s}: a 9P2000 introspection server (built on cloud9). + \\ + \\/build facts baked in at build time (zig version, target, optimize, time, change id) + \\/comptime facts computed by the Zig compiler: type layouts under types//, pub decls + \\/runtime live facts; fn/ calls a Zig function on every read + \\/ctl write a command, read the result + \\/vars exposed variables: /{{value,type,size,addr,raw,f//...}} + \\ + \\Other top-level directories are providers mounted at runtime. + \\ + , .{cfg.name}); + + fn typeDir(comptime T: type) Node { + return .{ .name = shortTypeName(T), .kind = .dir, .children = &.{ + .{ .name = "name", .kind = .static, .content = @typeName(T) }, + .{ .name = "size", .kind = .static, .content = std.fmt.comptimePrint("{d}", .{@sizeOf(T)}) }, + .{ .name = "align", .kind = .static, .content = std.fmt.comptimePrint("{d}", .{@alignOf(T)}) }, + .{ .name = "fields", .kind = .static, .content = fieldsText(T) }, + } }; + } + + const type_dirs: [cfg.types.len]Node = blk: { + @setEvalBranchQuota(200_000); + var arr: [cfg.types.len]Node = undefined; + for (cfg.types, 0..) |T, i| arr[i] = typeDir(T); + for (arr, 0..) |a, i| for (arr[i + 1 ..]) |b| { + if (std.mem.eql(u8, a.name, b.name)) @compileError("duplicate short type name " ++ a.name); + }; + break :blk arr; + }; + + const decls_text: []const u8 = if (cfg.decls_of) |T| declsText(T) else ""; + const fn_nodes = genNodes(cfg.fns); + const runtime_nodes = genNodes(cfg.runtime); + const ctl_node: Node = .{ .name = "ctl", .kind = .ctl }; + + const build_nodes: []const Node = if (cfg.build) |B| &[_]Node{ + .{ .name = "zig_version", .kind = .static, .content = B.zig_version }, + .{ .name = "target", .kind = .static, .content = B.target }, + .{ .name = "optimize", .kind = .static, .content = B.optimize }, + .{ .name = "time", .kind = .static, .content = B.time }, + .{ .name = "change", .kind = .static, .content = B.change }, + } else &.{}; + + /// Build time as unix seconds (for static atime/mtime), or 0. + pub const build_secs: u32 = if (cfg.build) |B| (parseIso8601(B.time) orelse 0) else 0; + + const runtime_children: []const Node = blk: { + var list: []const Node = &runtime_nodes; + list = list ++ &[_]Node{.{ .name = "fn", .kind = .dir, .children = &fn_nodes }}; + if (cfg.ctl != null and cfg.ctl_dir == .runtime) list = list ++ &[_]Node{ctl_node}; + break :blk list; + }; + + const root_children: []const Node = blk: { + var list: []const Node = &[_]Node{.{ .name = "README", .kind = .static, .content = readme_text }}; + if (cfg.build != null) list = list ++ &[_]Node{.{ .name = "build", .kind = .dir, .children = build_nodes }}; + list = list ++ &[_]Node{ + .{ .name = "comptime", .kind = .dir, .children = &.{ + .{ .name = "types", .kind = .dir, .children = &type_dirs }, + .{ .name = "decls", .kind = .static, .content = decls_text }, + } }, + .{ .name = "runtime", .kind = .dir, .children = runtime_children }, + }; + if (cfg.ctl != null and cfg.ctl_dir == .root) list = list ++ &[_]Node{ctl_node}; + list = list ++ &[_]Node{.{ .name = "vars", .kind = .vars }}; + break :blk list; + }; + + pub const root_node: Node = .{ .name = "/", .kind = .dir, .children = root_children }; + + /// The static tree flattened so nodes can be referenced by index; the + /// children of a node occupy consecutive slots `first..first+count`. + const Flat = struct { node: Node, parent: u32, first: u32, count: u32 }; + + fn countNodes(n: Node) usize { + var c: usize = 1; + for (n.children) |ch| c += countNodes(ch); + return c; + } + + fn fillFlat(arr: []Flat, next: *usize, idx: usize, n: Node, parent: u32) void { + const first = next.*; + next.* += n.children.len; + arr[idx] = .{ .node = n, .parent = parent, .first = @intCast(first), .count = @intCast(n.children.len) }; + for (n.children, 0..) |ch, i| fillFlat(arr, next, first + i, ch, @intCast(idx)); + } + + pub const flat_len = countNodes(root_node); + pub const flat: [flat_len]Flat = blk: { + @setEvalBranchQuota(100_000); + var arr: [flat_len]Flat = undefined; + var next: usize = 1; + fillFlat(&arr, &next, 0, root_node, 0); + break :blk arr; + }; + const vars_idx: u32 = blk: { + for (flat, 0..) |f, i| if (f.node.kind == .vars) break :blk @intCast(i); + @compileError("no vars node"); + }; + + comptime { + for (flat) |f| if (f.node.kind == .dynamic and f.node.gen == null) @compileError("dynamic node without generator"); + std.debug.assert(cfg.msize >= cloud9.Server.msize_min); + std.debug.assert(cfg.snapshot_slots > 0 and cfg.max_fids > 0); + // Provider index 0xFE/0xFF would collide with the var/static qid tags. + std.debug.assert(cfg.max_providers < 0xFE); + } + + // -- qid paths ------------------------------------------------------ + + const static_tag: u64 = 0xFF << 56; + const var_tag: u64 = 0xFE << 56; + const handle_mask: u64 = (1 << 56) - 1; + + // -- storage -------------------------------------------------------- + + /// Per-connection buffers; the caller places one in static memory. + pub const Storage = struct { + in: [cfg.msize]u8, + out: [cfg.msize]u8, + /// Staging area for read replies (directory records, provider and raw reads). + data: [cfg.msize]u8, + snapshots: [cfg.snapshot_slots][cfg.snapshot_bytes]u8, + }; + + const Var = struct { + name: []const u8, + ptr: *anyopaque, + vt: *const vars.VTable, + }; + + /// State common to all connections: providers, exposed variables, the + /// ctl result. Not internally synchronized: one thread serves all + /// connections (or the caller serializes). + pub const Shared = struct { + ctx: *anyopaque, + providers: [cfg.max_providers]Provider = undefined, + nprov: u8 = 0, + vars: [cfg.max_vars]Var = undefined, + nvars: u8 = 0, + /// The ctl result is double-buffered: a command writes into the + /// buffer that is not current and commits it only on success, so + /// a failed command leaves the previous result intact. + ctl_bufs: [2][cfg.ctl_bytes]u8 = undefined, + ctl_cur: u1 = 0, + /// Length of the current ctl result (in `ctl_bufs[ctl_cur]`). + ctl_len: u32 = 0, + ctl_version: u32 = 0, + /// Entropy for the per-connection fid hash. The core mixes in a + /// connection counter and buffer addresses; a platform layer with a + /// random source may set this once after `init` to make the seed + /// unpredictable even where addresses are static. + hash_seed: u32 = 0, + conn_seq: u32 = 0, + + /// `ctx` is passed to every `fns`/`runtime` generator and to `ctl`. + pub fn init(ctx: *anyopaque) Shared { + return .{ .ctx = ctx }; + } + + /// Mounts `p` at `/`. The name must not collide with a + /// static entry or another provider. + pub fn addProvider(s: *Shared, p: Provider) error{Full}!void { + if (s.nprov == cfg.max_providers) return error.Full; + std.debug.assert(validName(p.name) != error.BadName); + std.debug.assert(s.findProvider(p.name) == null); + std.debug.assert(staticChild(0, p.name) == null); + s.providers[s.nprov] = p; + s.nprov += 1; + } + + /// Publishes `ptr.*` as /vars/. `name` and the pointee must + /// outlive the server. + pub fn expose(s: *Shared, name: []const u8, ptr: anytype) error{Full}!void { + const P = @TypeOf(ptr); + const info = @typeInfo(P); + if (info != .pointer or info.pointer.size != .one or info.pointer.is_const) @compileError("expose wants a *T, got " ++ @typeName(P)); + if (s.nvars == cfg.max_vars) return error.Full; + std.debug.assert(validName(name) != error.BadName); + std.debug.assert(s.findVar(name) == null); + s.vars[s.nvars] = .{ .name = name, .ptr = @ptrCast(ptr), .vt = vars.vtableFor(info.pointer.child) }; + s.nvars += 1; + } + + /// The result of the last successful ctl command. + pub fn ctlResult(s: *const Shared) []const u8 { + return s.ctl_bufs[s.ctl_cur][0..s.ctl_len]; + } + + fn findProvider(s: *const Shared, name: []const u8) ?u8 { + for (s.providers[0..s.nprov], 0..) |p, i| if (std.mem.eql(u8, p.name, name)) return @intCast(i); + return null; + } + + fn findVar(s: *const Shared, name: []const u8) ?u8 { + for (s.vars[0..s.nvars], 0..) |v, i| if (std.mem.eql(u8, v.name, name)) return @intCast(i); + return null; + } + }; + + fn staticChild(idx: u32, name: []const u8) ?u32 { + const f = flat[idx]; + for (f.first..f.first + f.count) |ci| { + if (std.mem.eql(u8, flat[ci].node.name, name)) return @intCast(ci); + } + return null; + } + + // -- connection ----------------------------------------------------- + + const NodeRef = union(enum) { + static: u32, + prov: struct { idx: u8, h: Provider.Handle }, + @"var": struct { idx: u8, node: u32 }, + }; + + const Fid = struct { + id: u32 = 0, + used: bool = false, + node: NodeRef = .{ .static = 0 }, + is_dir: bool = true, + open: bool = false, + mode: u8 = 0, + rclose: bool = false, + dir_offset: u64 = 0, + dir_index: usize = 0, + /// Snapshot slot of an open dynamic file. + snap: ?u8 = null, + /// Free-list link, meaningful while `!used`. + next_free: u16 = no_slot, + }; + + const no_slot: u16 = std.math.maxInt(u16); + /// The fid index is an open-addressing (linear probing) table from fid + /// number to a slot of `Conn.fids`, sized to stay at most half full so + /// that lookups are O(1) with any number of fids. + const index_len: usize = std.math.ceilPowerOfTwoAssert(usize, @as(usize, cfg.max_fids) * 2); + const index_mask: usize = index_len - 1; + const index_shift: u5 = @intCast(32 - @as(usize, std.math.log2_int(usize, index_len))); + + /// MurmurHash3's 32-bit finalizer: every input bit affects every output bit. + fn fmix32(x: u32) u32 { + var h = x; + h ^= h >> 16; + h *%= 0x85EB_CA6B; + h ^= h >> 13; + h *%= 0xC2B2_AE35; + h ^= h >> 16; + return h; + } + + /// Everything the engine needs to know about a node for qid/stat. + const Info = struct { + is_dir: bool, + mode: u32, + length: u64, + atime: u32, + mtime: u32, + version: u32, + path: u64, + name: []const u8, + + fn qid(i: Info) cloud9.Qid { + var t: u8 = if (i.is_dir) cloud9.qtdir else cloud9.qtfile; + if (i.mode & cloud9.dmappend != 0) t |= cloud9.qtappend; + if (i.mode & cloud9.dmexcl != 0) t |= cloud9.qtexcl; + return .{ .type = t, .version = i.version, .path = i.path }; + } + + fn stat(i: Info) cloud9.Stat { + return .{ + .type = 0, + .dev = 0, + .qid = i.qid(), + .mode = i.mode, + .atime = i.atime, + .mtime = i.mtime, + .length = i.length, + .name = i.name, + .uid = cfg.name, + .gid = cfg.name, + .muid = cfg.name, + }; + } + }; + + /// One 9P connection: a cloud9.Server plus a fid table and snapshot slots. + pub const Conn = struct { + shared: *Shared, + storage: *Storage, + server: cloud9.Server, + /// Largest msize this connection negotiates. + msize_cap: u32, + fids: [cfg.max_fids]Fid = @splat(.{}), + /// XORed into every fid number before hashing so that a client + /// cannot precompute fid numbers that collide (which would turn the + /// index back into a linear scan). + hash_seed: u32, + /// fid number -> slot of `fids` (`no_slot` = empty bucket). + index: [index_len]u16 = @splat(no_slot), + /// Head of the free list threaded through `Fid.next_free`. + free_head: u16 = no_slot, + /// Slots `high_water..` have never been used (bump allocation). + high_water: u16 = 0, + nfids: u16 = 0, + slot_used: [cfg.snapshot_slots]bool = @splat(false), + slot_len: [cfg.snapshot_slots]u32 = @splat(0), + name_buf: [max_name]u8 = undefined, + + pub fn init(shared: *Shared, storage: *Storage, msize: u32) Conn { + shared.conn_seq +%= 1; + const addr = @intFromPtr(storage) ^ (@intFromPtr(shared) << 7); + const seed = fmix32(shared.hash_seed ^ (shared.conn_seq *% 0x9E37_79B1) ^ @as(u32, @truncate(addr)) ^ @as(u32, @truncate(addr >> 16))); + return .{ + .shared = shared, + .storage = storage, + .server = .init(.{ .in = &storage.in, .out = &storage.out }), + .msize_cap = @max(@min(msize, cfg.msize), cloud9.Server.msize_min), + .hash_seed = seed, + }; + } + + /// Fibonacci hashing of the (seeded) fid number into `index_len` buckets. + fn fidHome(c: *const Conn, id: u32) usize { + return @intCast(((id ^ c.hash_seed) *% 0x9E37_79B1) >> index_shift); + } + + /// Feeds transport bytes; returns how many were taken. + pub fn push(c: *Conn, bytes: []const u8) usize { + return c.server.push(bytes); + } + + /// Bytes to send to the client. + pub fn output(c: *const Conn) []const u8 { + return c.server.output(); + } + + pub fn wrote(c: *Conn, n: usize) void { + c.server.wrote(n); + } + + /// Drops every fid (telling providers) and kills the session. + pub fn hangup(c: *Conn) void { + c.resetFids(); + c.server.hangup(); + } + + /// Handles at most one request. Returns false when more input (or + /// output drainage) is needed. `error.Protocol` is terminal. + pub fn step(c: *Conn) error{Protocol}!bool { + const req = (c.server.receive() catch return error.Protocol) orelse return false; + defer c.server.release(); + switch (req.msg) { + .tversion => |m| { + c.resetFids(); + c.server.negotiate(@min(m.msize, c.msize_cap), m.version) catch return error.Protocol; + }, + else => { + const reply = c.dispatch(req.msg) catch |e| cloud9.Msg{ .rerror = .{ .ename = ename(e) } }; + c.server.reply(req.tag, reply) catch |e| switch (e) { + // The reply does not fit the negotiated msize (Rstat or a long + // Rwalk at a tiny msize). receive() guarantees room for one + // msize-sized message, so this is never backpressure: answer with + // an Rerror (truncated to fit by cloud9). Rwalk, the only + // variable-size reply that follows a state change, is size-checked + // in walk() before anything is mutated. + error.TooLarge => c.server.reply(req.tag, .{ .rerror = .{ .ename = ename(error.ReplyTooLarge) } }) catch return error.Protocol, + else => return error.Protocol, + }; + }, + } + return true; + } + + /// Number of fids currently held. + pub fn fidCount(c: *const Conn) usize { + return c.nfids; + } + + fn dispatch(c: *Conn, msg: cloud9.Msg) anyerror!cloud9.Msg { + return switch (msg) { + .tauth => error.AuthNotRequired, + .tattach => |m| c.attach(m), + .tflush => .rflush, + .twalk => |m| c.walk(m), + .topen => |m| c.open(m), + .tcreate => |m| c.create(m), + .tread => |m| c.read(m), + .twrite => |m| c.write(m), + .tclunk => |m| c.clunk(m), + .tremove => |m| c.remove(m), + .tstat => |m| c.stat(m), + .twstat => |m| c.wstat(m), + else => error.Protocol, + }; + } + + // -- fid table -- + + /// The index bucket holding `id`, if any. + fn findBucket(c: *const Conn, id: u32) ?usize { + var pos = c.fidHome(id); + while (true) : (pos = (pos + 1) & index_mask) { + const slot = c.index[pos]; + if (slot == no_slot) return null; + if (c.fids[slot].id == id) return pos; + } + } + + fn findFid(c: *Conn, id: u32) ?*Fid { + const pos = c.findBucket(id) orelse return null; + return &c.fids[c.index[pos]]; + } + + fn allocFid(c: *Conn, id: u32) !*Fid { + if (c.findBucket(id) != null) return error.FidInUse; + if (c.nfids >= cfg.max_fids) return error.TooManyFids; + const slot: u16 = if (c.free_head != no_slot) blk: { + const slot = c.free_head; + c.free_head = c.fids[slot].next_free; + break :blk slot; + } else blk: { + const slot = c.high_water; + c.high_water += 1; + break :blk slot; + }; + c.fids[slot] = .{ .id = id, .used = true }; + var pos = c.fidHome(id); + while (c.index[pos] != no_slot) pos = (pos + 1) & index_mask; + c.index[pos] = slot; + c.nfids += 1; + return &c.fids[slot]; + } + + /// Removes `id` from the index (backward-shift deletion: no tombstones). + fn unlinkFid(c: *Conn, id: u32) void { + var i = c.findBucket(id).?; + var j = i; + while (true) { + j = (j + 1) & index_mask; + const slot = c.index[j]; + if (slot == no_slot) break; + const k = c.fidHome(c.fids[slot].id); + // The entry at j may move into the hole at i unless its home + // lies in the cyclic interval (i, j]. + const stays = if (i <= j) (k > i and k <= j) else (k > i or k <= j); + if (!stays) { + c.index[i] = slot; + i = j; + } + } + c.index[i] = no_slot; + } + + /// Releases everything a fid holds; the slot stays allocated. + fn dropContents(c: *Conn, f: *Fid) void { + if (f.snap) |s| c.slot_used[s] = false; + f.snap = null; + if (f.node == .prov) { + const p = c.shared.providers[f.node.prov.idx]; + if (f.open) if (p.vtable.close) |close| close(p.ctx, f.node.prov.h); + if (f.rclose and f.open) if (p.vtable.remove) |rm| rm(p.ctx, f.node.prov.h) catch {}; + p.vtable.clunk(p.ctx, f.node.prov.h); + } + f.open = false; + f.rclose = false; + } + + fn freeFid(c: *Conn, f: *Fid) void { + c.dropContents(f); + c.unlinkFid(f.id); + const slot: u16 = @intCast((@intFromPtr(f) - @intFromPtr(&c.fids)) / @sizeOf(Fid)); + f.* = .{ .next_free = c.free_head }; + c.free_head = slot; + c.nfids -= 1; + } + + fn resetFids(c: *Conn) void { + for (c.fids[0..c.high_water]) |*f| { + if (f.used) c.dropContents(f); + f.* = .{}; + } + @memset(&c.index, no_slot); + c.free_head = no_slot; + c.high_water = 0; + c.nfids = 0; + } + + /// Releases a provider handle that is not held by any fid. + fn releaseRef(c: *Conn, ref: NodeRef) void { + if (ref == .prov) { + const p = c.shared.providers[ref.prov.idx]; + p.vtable.clunk(p.ctx, ref.prov.h); + } + } + + // -- node helpers -- + + fn provider(c: *Conn, idx: u8) Provider { + return c.shared.providers[idx]; + } + + fn varBase(c: *Conn, idx: u8, node: u32) [*]u8 { + const v = c.shared.vars[idx]; + return @as([*]u8, @ptrCast(v.ptr)) + v.vt.nodes[node].offset; + } + + fn info(c: *Conn, ref: NodeRef) !Info { + switch (ref) { + .static => |idx| { + const n = flat[idx].node; + return .{ + .is_dir = n.isDir(), + .mode = switch (n.kind) { + .dir, .vars => cloud9.dmdir | 0o555, + .ctl => 0o666, + else => 0o444, + }, + .length = switch (n.kind) { + .static => n.content.len, + .ctl => c.shared.ctl_len, + else => 0, + }, + .atime = build_secs, + .mtime = build_secs, + .version = if (n.kind == .ctl) c.shared.ctl_version else 0, + .path = static_tag | idx, + .name = n.name, + }; + }, + .@"var" => |v| { + const sv = c.shared.vars[v.idx]; + const n = sv.vt.nodes[v.node]; + return .{ + .is_dir = n.isDir(), + .mode = if (n.isDir()) cloud9.dmdir | 0o555 else if (n.writable()) 0o644 else 0o444, + .length = switch (n.kind) { + .type_name, .size => n.content.len, + .raw => n.size, + else => 0, + }, + .atime = 0, + .mtime = 0, + .version = 0, + .path = var_tag | (@as(u64, v.idx) << 32) | v.node, + .name = if (v.node == 0) sv.name else n.name, + }; + }, + .prov => |p| { + const pr = c.provider(p.idx); + var st: NodeStat = .{ .mode = 0 }; + try pr.vtable.stat(pr.ctx, p.h, &st); + return provInfo(pr, p.idx, p.h, st); + }, + } + } + + fn provInfo(pr: Provider, idx: u8, h: Provider.Handle, st: NodeStat) Info { + return .{ + .is_dir = st.isDir(), + .mode = st.mode, + .length = if (st.isDir()) 0 else st.length, + .atime = st.atime, + .mtime = st.mtime, + .version = st.version, + .path = (@as(u64, idx) << 56) | ((if (st.path != 0) st.path else h) & handle_mask), + .name = if (h == Provider.root) pr.name else st.name, + }; + } + + /// Copies `st.name` into the connection so the reply cannot dangle. + fn pinName(c: *Conn, st: cloud9.Stat) cloud9.Stat { + var out = st; + const n = @min(st.name.len, c.name_buf.len); + @memcpy(c.name_buf[0..n], st.name[0..n]); + out.name = c.name_buf[0..n]; + return out; + } + + const Looked = struct { ref: NodeRef, info: Info }; + + /// Resolves `name` in the directory `ref`. A returned provider ref + /// is a fresh handle the caller must release or retain. + fn lookup(c: *Conn, ref: NodeRef, name: []const u8) !Looked { + switch (ref) { + .static => |idx| { + const dot = std.mem.eql(u8, name, "."); + const dotdot = std.mem.eql(u8, name, ".."); + var next: NodeRef = undefined; + if (dot) { + next = ref; + } else if (dotdot) { + next = .{ .static = flat[idx].parent }; + } else if (flat[idx].node.kind == .vars) { + const vi = c.shared.findVar(name) orelse return error.NotFound; + next = .{ .@"var" = .{ .idx = vi, .node = 0 } }; + } else if (staticChild(idx, name)) |ci| { + next = .{ .static = ci }; + } else if (idx == 0) { + const pi = c.shared.findProvider(name) orelse return error.NotFound; + next = .{ .prov = .{ .idx = pi, .h = Provider.root } }; + } else return error.NotFound; + return .{ .ref = next, .info = try c.info(next) }; + }, + .@"var" => |v| { + const vt = c.shared.vars[v.idx].vt; + var next = ref; + if (std.mem.eql(u8, name, ".")) { + // unchanged + } else if (std.mem.eql(u8, name, "..")) { + next = if (v.node == 0) .{ .static = vars_idx } else .{ .@"var" = .{ .idx = v.idx, .node = vt.nodes[v.node].parent } }; + } else { + const ci = vt.child(v.node, name) orelse return error.NotFound; + next = .{ .@"var" = .{ .idx = v.idx, .node = ci } }; + } + return .{ .ref = next, .info = try c.info(next) }; + }, + .prov => |p| { + if (p.h == Provider.root and std.mem.eql(u8, name, "..")) { + const next: NodeRef = .{ .static = 0 }; + return .{ .ref = next, .info = try c.info(next) }; + } + const pr = c.provider(p.idx); + const h = try pr.vtable.walk(pr.ctx, p.h, name); + const next: NodeRef = .{ .prov = .{ .idx = p.idx, .h = h } }; + errdefer c.releaseRef(next); + return .{ .ref = next, .info = try c.info(next) }; + }, + } + } + + /// The i-th entry of directory `ref` as a Stat, or null past the end. + /// The name borrows either static memory or the provider's NodeStat. + fn entryStat(c: *Conn, ref: NodeRef, i: usize) !?cloud9.Stat { + switch (ref) { + .static => |idx| { + const f = flat[idx]; + if (f.node.kind == .vars) { + if (i >= c.shared.nvars) return null; + return (try c.info(.{ .@"var" = .{ .idx = @intCast(i), .node = 0 } })).stat(); + } + if (i < f.count) return (try c.info(.{ .static = f.first + @as(u32, @intCast(i)) })).stat(); + if (idx == 0) { + const pi = i - f.count; + if (pi >= c.shared.nprov) return null; + return (try c.info(.{ .prov = .{ .idx = @intCast(pi), .h = Provider.root } })).stat(); + } + return null; + }, + .@"var" => |v| { + const n = c.shared.vars[v.idx].vt.nodes[v.node]; + if (i >= n.count) return null; + return (try c.info(.{ .@"var" = .{ .idx = v.idx, .node = n.first + @as(u32, @intCast(i)) } })).stat(); + }, + .prov => |p| { + const pr = c.provider(p.idx); + var st: NodeStat = .{ .mode = 0 }; + if (!try pr.vtable.list(pr.ctx, p.h, i, &st)) return null; + return provInfo(pr, p.idx, st.handle, st).stat(); + }, + } + } + + // -- snapshots -- + + fn takeSlot(c: *Conn) !u8 { + for (&c.slot_used, 0..) |*u, i| if (!u.*) { + u.* = true; + return @intCast(i); + }; + return error.NoSnapshot; + } + + /// (Re)generates the content of a dynamic file into its slot. + fn generate(c: *Conn, f: *Fid) !void { + const s = f.snap.?; + var w: Writer = .fixed(&c.storage.snapshots[s]); + c.slot_len[s] = 0; + switch (f.node) { + .static => |idx| try flat[idx].node.gen.?(c.shared.ctx, &w), + .@"var" => |v| { + const n = c.shared.vars[v.idx].vt.nodes[v.node]; + const base = c.varBase(v.idx, v.node); + switch (n.kind) { + .value => try n.render.?(base, &w), + .addr => try w.print("0x{x}", .{@intFromPtr(base)}), + else => unreachable, + } + }, + .prov => unreachable, + } + c.slot_len[s] = @intCast(w.buffered().len); + } + + fn isDynamic(c: *Conn, ref: NodeRef) bool { + return switch (ref) { + .static => |idx| flat[idx].node.kind == .dynamic, + .@"var" => |v| switch (c.shared.vars[v.idx].vt.nodes[v.node].kind) { + .value, .addr => true, + else => false, + }, + .prov => false, + }; + } + + // -- request handlers -- + + fn attach(c: *Conn, m: anytype) !cloud9.Msg { + const f = try c.allocFid(m.fid); + f.node = .{ .static = 0 }; + f.is_dir = true; + return .{ .rattach = .{ .qid = (try c.info(f.node)).qid() } }; + } + + fn walk(c: *Conn, m: anytype) !cloud9.Msg { + const f = c.findFid(m.fid) orelse return error.UnknownFid; + if (m.newfid != m.fid and c.findFid(m.newfid) != null) return error.FidInUse; + if (m.newfid != m.fid and c.nfids >= cfg.max_fids) return error.TooManyFids; + if (m.nwname > 0 and f.open) return error.AlreadyOpen; + // Cloning a fid onto itself changes nothing; in particular it must not + // close an open fid or discard generated content. + if (m.nwname == 0 and m.newfid == m.fid) return .{ .rwalk = .{ .nwqid = 0 } }; + // A full Rwalk must fit the negotiated msize; check before binding anything. + if (cloud9.header_len + 2 + cloud9.qid_len * @as(usize, m.nwname) > c.server.msize) return error.ReplyTooLarge; + var cur = f.node; + var cur_is_dir = f.is_dir; + var held = false; // cur is a provider handle obtained here, not the fid's + var reply: cloud9.Msg = .{ .rwalk = .{ .nwqid = 0 } }; + const names = m.wname[0..m.nwname]; + for (names, 0..) |name, i| { + if (!cur_is_dir) { + if (i == 0) return error.NotDir; + break; + } + const next = c.lookup(cur, name) catch |e| { + if (i == 0) return e; + break; + }; + if (held) c.releaseRef(cur); + cur = next.ref; + cur_is_dir = next.info.is_dir; + held = cur == .prov; + reply.rwalk.wqid[i] = next.info.qid(); + reply.rwalk.nwqid += 1; + } + if (reply.rwalk.nwqid != names.len) { + if (held) c.releaseRef(cur); + return reply; + } + if (names.len == 0 and cur == .prov) { + // A clone of a provider handle needs its own reference. + const dup = try c.lookup(cur, "."); + cur = dup.ref; + cur_is_dir = dup.info.is_dir; + held = true; + } + const target = if (m.newfid == m.fid) f else c.allocFid(m.newfid) catch |e| { + if (held) c.releaseRef(cur); + return e; + }; + if (target == f) c.dropContents(f); + target.node = cur; + target.is_dir = cur_is_dir; + return reply; + } + + fn open(c: *Conn, m: anytype) !cloud9.Msg { + const f = c.findFid(m.fid) orelse return error.UnknownFid; + if (f.open) return error.AlreadyOpen; + const acc = m.mode & 3; + const want_write = acc == cloud9.owrite or acc == cloud9.ordwr; + const trunc = m.mode & cloud9.otrunc != 0; + if (f.is_dir and (want_write or trunc)) return error.IsDir; + switch (f.node) { + .static => |idx| switch (flat[idx].node.kind) { + .dir, .vars, .ctl => {}, + .static, .dynamic => if (want_write or trunc) return error.Perm, + }, + .@"var" => |v| { + const n = c.shared.vars[v.idx].vt.nodes[v.node]; + if ((want_write or trunc) and !n.writable()) return error.Perm; + }, + .prov => |p| { + const pr = c.provider(p.idx); + try pr.vtable.open(pr.ctx, p.h, m.mode); + }, + } + const qid = (c.info(f.node) catch |e| { + // The provider's open succeeded but its stat did not: undo the open. + if (f.node == .prov) { + const pr = c.provider(f.node.prov.idx); + if (pr.vtable.close) |close| close(pr.ctx, f.node.prov.h); + } + return e; + }).qid(); + if (c.isDynamic(f.node)) { + f.snap = try c.takeSlot(); + c.generate(f) catch |e| { + c.slot_used[f.snap.?] = false; + f.snap = null; + if (f.node == .prov) unreachable; + return e; + }; + } + f.open = true; + f.mode = m.mode; + f.rclose = m.mode & cloud9.orclose != 0; + f.dir_offset = 0; + f.dir_index = 0; + return .{ .ropen = .{ .qid = qid, .iounit = 0 } }; + } + + fn create(c: *Conn, m: anytype) !cloud9.Msg { + const f = c.findFid(m.fid) orelse return error.UnknownFid; + if (f.open) return error.AlreadyOpen; + const p = switch (f.node) { + .prov => |p| p, + else => return error.Perm, + }; + if (!f.is_dir) return error.NotDir; + const pr = c.provider(p.idx); + const create_fn = pr.vtable.create orelse return error.Perm; + try validName(m.name); + const is_dir = m.perm & cloud9.dmdir != 0; + const acc = m.mode & 3; + if (is_dir and (acc != cloud9.oread or m.mode & cloud9.otrunc != 0)) return error.IsDir; + const h = try create_fn(pr.ctx, p.h, m.name, m.perm, m.mode); + const node: NodeRef = .{ .prov = .{ .idx = p.idx, .h = h } }; + const qid = (c.info(node) catch |e| { + if (pr.vtable.close) |close| close(pr.ctx, h); + pr.vtable.clunk(pr.ctx, h); + return e; + }).qid(); + c.dropContents(f); + f.node = node; + f.is_dir = is_dir; + f.open = true; + f.mode = m.mode; + f.rclose = m.mode & cloud9.orclose != 0; + f.dir_offset = 0; + f.dir_index = 0; + return .{ .rcreate = .{ .qid = qid, .iounit = 0 } }; + } + + fn read(c: *Conn, m: anytype) !cloud9.Msg { + const f = c.findFid(m.fid) orelse return error.UnknownFid; + if (!f.open or (f.mode & 3) == cloud9.owrite) return error.NotOpen; + const count: usize = @min(m.count, c.server.msize -| cloud9.iohdrsz, c.storage.data.len); + if (f.is_dir) return c.readDir(f, m.offset, count); + const data = &c.storage.data; + const src: []const u8 = switch (f.node) { + .static => |idx| blk: { + const n = flat[idx].node; + switch (n.kind) { + .static => break :blk n.content, + .ctl => break :blk c.shared.ctlResult(), + .dynamic => { + if (m.offset == 0) try c.generate(f); + break :blk c.storage.snapshots[f.snap.?][0..c.slot_len[f.snap.?]]; + }, + .dir, .vars => unreachable, + } + }, + .@"var" => |v| blk: { + const n = c.shared.vars[v.idx].vt.nodes[v.node]; + switch (n.kind) { + .type_name, .size => break :blk n.content, + .value, .addr => { + if (m.offset == 0) try c.generate(f); + break :blk c.storage.snapshots[f.snap.?][0..c.slot_len[f.snap.?]]; + }, + .raw => break :blk c.varBase(v.idx, v.node)[0..n.size], + .dir, .fields => unreachable, + } + }, + .prov => |p| { + const pr = c.provider(p.idx); + const n = try pr.vtable.read(pr.ctx, p.h, m.offset, data[0..count]); + return .{ .rread = .{ .data = data[0..@min(n, count)] } }; + }, + }; + if (m.offset >= src.len) return .{ .rread = .{ .data = "" } }; + const off: usize = @intCast(m.offset); + const n = @min(count, src.len - off); + if (f.node == .@"var" and c.shared.vars[f.node.@"var".idx].vt.nodes[f.node.@"var".node].kind == .raw) { + // Copy out of the variable so the reply does not read live memory twice. + @memcpy(data[0..n], src[off..][0..n]); + return .{ .rread = .{ .data = data[0..n] } }; + } + return .{ .rread = .{ .data = src[off..][0..n] } }; + } + + fn readDir(c: *Conn, f: *Fid, offset: u64, count: usize) !cloud9.Msg { + if (offset == 0) { + f.dir_offset = 0; + f.dir_index = 0; + } else if (offset != f.dir_offset) return error.BadOffset; + const data = &c.storage.data; + var used: usize = 0; + var i = f.dir_index; + while (try c.entryStat(f.node, i)) |st| : (i += 1) { + const rec = st.encode(data[used..count]) catch |e| switch (e) { + error.NoSpace => break, + else => return error.Io, + }; + used += rec.len; + } + f.dir_offset += used; + f.dir_index = i; + return .{ .rread = .{ .data = data[0..used] } }; + } + + fn write(c: *Conn, m: anytype) !cloud9.Msg { + const f = c.findFid(m.fid) orelse return error.UnknownFid; + const acc = f.mode & 3; + if (!f.open or (acc != cloud9.owrite and acc != cloud9.ordwr)) return error.NotOpen; + if (f.is_dir) return error.IsDir; + switch (f.node) { + .static => |idx| switch (flat[idx].node.kind) { + .ctl => try c.ctlCommand(m.data), + else => return error.Perm, + }, + .@"var" => |v| { + const n = c.shared.vars[v.idx].vt.nodes[v.node]; + const set = n.set orelse return error.Perm; + try set(c.varBase(v.idx, v.node), m.data); + }, + .prov => |p| { + const pr = c.provider(p.idx); + const n = try pr.vtable.write(pr.ctx, p.h, m.offset, m.data); + return .{ .rwrite = .{ .count = @intCast(@min(n, m.data.len)) } }; + }, + } + return .{ .rwrite = .{ .count = @intCast(m.data.len) } }; + } + + /// Runs `cfg.ctl`; on success its output becomes the ctl result. + /// On failure the previous result (and its qid version) survive: + /// the handler writes into the staging half of `ctl_bufs`. + fn ctlCommand(c: *Conn, line: []const u8) !void { + const s = c.shared; + const next = s.ctl_cur ^ 1; + var w: Writer = .fixed(&s.ctl_bufs[next]); + try cfg.ctl.?(s.ctx, line, &w); + s.ctl_cur = next; + s.ctl_len = @intCast(w.buffered().len); + s.ctl_version +%= 1; + } + + fn clunk(c: *Conn, m: anytype) !cloud9.Msg { + const f = c.findFid(m.fid) orelse return error.UnknownFid; + c.freeFid(f); + return .rclunk; + } + + fn remove(c: *Conn, m: anytype) !cloud9.Msg { + const f = c.findFid(m.fid) orelse return error.UnknownFid; + defer c.freeFid(f); // Tremove always clunks + f.rclose = false; + switch (f.node) { + .prov => |p| { + const pr = c.provider(p.idx); + const rm = pr.vtable.remove orelse return error.Perm; + try rm(pr.ctx, p.h); + }, + else => return error.Perm, + } + return .rremove; + } + + fn stat(c: *Conn, m: anytype) !cloud9.Msg { + const f = c.findFid(m.fid) orelse return error.UnknownFid; + return .{ .rstat = .{ .stat = c.pinName((try c.info(f.node)).stat()) } }; + } + + fn wstat(c: *Conn, m: anytype) !cloud9.Msg { + const f = c.findFid(m.fid) orelse return error.UnknownFid; + const p = switch (f.node) { + .prov => |p| p, + else => return error.Perm, + }; + const pr = c.provider(p.idx); + const ws = pr.vtable.wstat orelse return error.Perm; + const cur = try c.info(f.node); + const st = m.stat; + const q = cur.qid(); + // Fields we cannot change must be "don't care" or unchanged. + if (st.type != 0xFFFF and st.type != 0) return error.Perm; + if (st.dev != 0xFFFF_FFFF and st.dev != 0) return error.Perm; + if (st.qid.type != 0xFF and st.qid.type != q.type) return error.Perm; + if (st.qid.version != 0xFFFF_FFFF and st.qid.version != q.version) return error.Perm; + if (st.qid.path != 0xFFFF_FFFF_FFFF_FFFF and st.qid.path != q.path) return error.Perm; + if (st.uid.len != 0 and !std.mem.eql(u8, st.uid, cfg.name)) return error.Perm; + if (st.gid.len != 0 and !std.mem.eql(u8, st.gid, cfg.name)) return error.Perm; + if (st.muid.len != 0 and !std.mem.eql(u8, st.muid, cfg.name)) return error.Perm; + if (st.name.len != 0 and !std.mem.eql(u8, st.name, cur.name)) { + if (p.h == Provider.root) return error.Perm; + try validName(st.name); + } + if (st.length != 0xFFFF_FFFF_FFFF_FFFF and st.length != cur.length and cur.is_dir) return error.IsDir; + if (st.mode != 0xFFFF_FFFF and (st.mode & cloud9.dmdir) != (cur.mode & cloud9.dmdir)) return error.Perm; + try ws(pr.ctx, p.h, &st); + return .rwstat; + } + }; + + // -- in-memory test harness ------------------------------------------- + + /// Drives a `Conn` with a `cloud9.Client` in memory. Test-only (uses + /// std.testing.allocator); never referenced by non-test code. + pub const Harness = struct { + shared: *Shared, + storage: *Storage, + conn: Conn, + client: cloud9.Client, + cin: []u8, + cout: []u8, + + pub fn init(h: *Harness, shared: *Shared, storage: *Storage) !void { + h.shared = shared; + h.storage = storage; + h.conn = .init(shared, storage, cfg.msize); + h.cin = try testing.allocator.alloc(u8, cfg.msize); + errdefer testing.allocator.free(h.cin); + h.cout = try testing.allocator.alloc(u8, cfg.msize); + errdefer testing.allocator.free(h.cout); + h.client = .init(.{ .in = h.cin, .out = h.cout }); + try h.version(cfg.msize); + _ = try h.ok(.{ .attach = .{ .fid = 0, .uname = "tester" } }); + } + + pub fn deinit(h: *Harness) void { + h.conn.hangup(); + testing.allocator.free(h.cin); + testing.allocator.free(h.cout); + } + + pub fn version(h: *Harness, msize: u32) !void { + const v = try h.rpc(.{ .version = .{ .msize = msize } }); + try testing.expectEqual(msize, v.version.msize); + try testing.expectEqualStrings("9P2000", v.version.version); + } + + /// One round trip; the result borrows the client input buffer until the next call. + pub fn rpc(h: *Harness, req: cloud9.Client.Request) !cloud9.Client.Result { + _ = try h.client.submit(req); + while (true) { + var moved = false; + while (h.client.output().len > 0) { + const k = h.conn.push(h.client.output()); + h.client.wrote(k); + moved = moved or k > 0; + while (try h.conn.step()) {} + while (h.conn.output().len > 0) { + const n = h.client.push(h.conn.output()); + h.conn.wrote(n); + moved = moved or n > 0; + } + if (k == 0) break; + } + while (try h.conn.step()) {} + while (h.conn.output().len > 0) { + const n = h.client.push(h.conn.output()); + h.conn.wrote(n); + moved = moved or n > 0; + } + if (h.client.take()) |done| return done.result; + if (!moved) return error.Stuck; + } + } + + pub fn ok(h: *Harness, req: cloud9.Client.Request) !cloud9.Client.Result { + const r = try h.rpc(req); + if (r == .fail) { + std.debug.print("unexpected Rerror: {s}\n", .{r.fail}); + return error.Rerror; + } + return r; + } + + pub fn expectFail(h: *Harness, req: cloud9.Client.Request, msg: []const u8) !void { + const r = try h.rpc(req); + if (r != .fail) return error.ExpectedRerror; + try testing.expectEqualStrings(msg, r.fail); + } + + pub fn walkTo(h: *Harness, newfid: u32, names: []const []const u8) !void { + const r = try h.ok(.{ .walk = .{ .fid = 0, .newfid = newfid, .names = names } }); + try testing.expectEqual(@as(u16, @intCast(names.len)), r.walk.nwqid); + } + + /// Opens `fid` for reading and reads it whole (across consecutive offsets); caller frees. + pub fn readAll(h: *Harness, fid: u32) ![]u8 { + _ = try h.ok(.{ .open = .{ .fid = fid, .mode = cloud9.oread } }); + return h.readOpen(fid); + } + + pub fn readOpen(h: *Harness, fid: u32) ![]u8 { + var acc: std.ArrayList(u8) = .empty; + errdefer acc.deinit(testing.allocator); + while (true) { + const r = try h.ok(.{ .read = .{ .fid = fid, .offset = acc.items.len, .count = 1024 } }); + if (r.read.len == 0) break; + try acc.appendSlice(testing.allocator, r.read); + } + return acc.toOwnedSlice(testing.allocator); + } + + pub fn readPath(h: *Harness, names: []const []const u8) ![]u8 { + try h.walkTo(99, names); + defer _ = h.rpc(.{ .clunk = .{ .fid = 99 } }) catch {}; + return h.readAll(99); + } + + pub fn writePath(h: *Harness, names: []const []const u8, data: []const u8) !void { + try h.walkTo(98, names); + defer _ = h.rpc(.{ .clunk = .{ .fid = 98 } }) catch {}; + _ = try h.ok(.{ .open = .{ .fid = 98, .mode = cloud9.owrite } }); + const w = try h.ok(.{ .write = .{ .fid = 98, .offset = 0, .data = data } }); + try testing.expectEqual(@as(u32, @intCast(data.len)), w.write); + } + + /// Reads a whole directory in `count`-byte reads at consecutive offsets; returns owned names. + pub fn listDir(h: *Harness, fid: u32, count: u32) ![][]u8 { + var names: std.ArrayList([]u8) = .empty; + errdefer { + for (names.items) |n| testing.allocator.free(n); + names.deinit(testing.allocator); + } + var offset: u64 = 0; + while (true) { + const r = try h.ok(.{ .read = .{ .fid = fid, .offset = offset, .count = count } }); + if (r.read.len == 0) break; + offset += r.read.len; + var rest = r.read; + while (rest.len > 0) { + const n = std.mem.readInt(u16, rest[0..2], .little) + 2; + const st = try cloud9.Stat.decode(rest[0..n]); + try names.append(testing.allocator, try testing.allocator.dupe(u8, st.name)); + rest = rest[n..]; + } + } + return names.toOwnedSlice(testing.allocator); + } + + pub fn listPath(h: *Harness, names: []const []const u8) ![][]u8 { + try h.walkTo(97, names); + defer _ = h.rpc(.{ .clunk = .{ .fid = 97 } }) catch {}; + _ = try h.ok(.{ .open = .{ .fid = 97, .mode = cloud9.oread } }); + return h.listDir(97, 1024); + } + + pub fn freeNames(names: [][]u8) void { + for (names) |n| testing.allocator.free(n); + testing.allocator.free(names); + } + + pub fn hasName(names: []const []const u8, want: []const u8) bool { + for (names) |n| if (std.mem.eql(u8, n, want)) return true; + return false; + } + }; + }; +} + +// --------------------------------------------------------------------------- +// Tests +// --------------------------------------------------------------------------- + +const testing = std.testing; + +const TestBuild = struct { + pub const zig_version: []const u8 = builtin.zig_version_string; + pub const target: []const u8 = "test-target"; + pub const optimize: []const u8 = "Debug"; + pub const time: []const u8 = "2023-11-14T22:13:20Z"; + pub const change: []const u8 = "abc123"; +}; + +const Layout = struct { a: u8, b: u32, c: u64 }; +const Decls = struct { + pub const one = 1; + pub const two = 2; + pub fn three() void {} +}; + +/// The context every generator and the ctl handler receive in tests. +const TestCtx = struct { + calls: u32 = 0, + ctl_state: i64 = 0, +}; + +const TestFns = struct { + pub fn counter(ctx: *anyopaque, w: *Writer) anyerror!void { + const t: *TestCtx = @ptrCast(@alignCast(ctx)); + t.calls += 1; + try w.print("{d}", .{t.calls}); + } + pub fn fib30(_: *anyopaque, w: *Writer) anyerror!void { + try w.print("{d}", .{fib(30)}); + } + pub fn failing(_: *anyopaque, _: *Writer) anyerror!void { + return error.BadCommand; + } + pub fn huge(_: *anyopaque, w: *Writer) anyerror!void { + try w.splatByteAll('x', 1 << 20); + } +}; + +const TestRuntime = struct { + pub fn pid(_: *anyopaque, w: *Writer) anyerror!void { + try w.writeAll("4242"); + } +}; + +fn fib(n: u32) u64 { + if (n == 0) return 0; + var a: u64 = 0; + var b: u64 = 1; + for (1..n) |_| { + const c = a + b; + a = b; + b = c; + } + return b; +} + +fn testCtl(ctx: *anyopaque, cmd: []const u8, out: *Writer) anyerror!void { + const t: *TestCtx = @ptrCast(@alignCast(ctx)); + const line = std.mem.trim(u8, cmd, " \t\r\n\x00"); + var it = std.mem.tokenizeScalar(u8, line, ' '); + const verb = it.next() orelse return error.BadCommand; + if (std.mem.eql(u8, verb, "echo")) { + try out.writeAll(std.mem.trimStart(u8, line[verb.len..], " \t")); + } else if (std.mem.eql(u8, verb, "add")) { + const a = std.fmt.parseInt(i64, it.next() orelse return error.BadCommand, 10) catch return error.BadCommand; + const b = std.fmt.parseInt(i64, it.next() orelse return error.BadCommand, 10) catch return error.BadCommand; + t.ctl_state = a +% b; + try out.print("{d}", .{t.ctl_state}); + } else if (std.mem.eql(u8, verb, "partial")) { + try out.writeAll("half-written"); + return error.BadCommand; + } else return error.BadCommand; +} + +const test_cfg: Config = .{ + .name = "tester", + .build = TestBuild, + .types = &.{ Layout, cloud9.Qid }, + .decls_of = Decls, + .fns = TestFns, + .runtime = TestRuntime, + .ctl = &testCtl, + .msize = 8192, + .max_fids = 8, + .max_providers = 2, + .max_vars = 4, + .snapshot_slots = 2, + .snapshot_bytes = 512, +}; + +const TS = Server(test_cfg); + +/// A small in-memory provider: /prov/{hello,dir/{inner}} with create/remove/wstat, +/// counting every handle reference so tests can check clunk discipline. +const TestProv = struct { + const max_nodes = 16; + const Entry = struct { + used: bool = false, + name: [max_name]u8 = undefined, + name_len: u8 = 0, + parent: u32 = 0, + is_dir: bool = false, + mode: u32 = 0o644, + data: [64]u8 = undefined, + len: usize = 0, + refs: u32 = 0, + opens: u32 = 0, + mtime: u32 = 0, + + fn nameSlice(e: *const Entry) []const u8 { + return e.name[0..e.name_len]; + } + }; + nodes: [max_nodes]Entry = @splat(.{}), + total_refs: u32 = 0, + clunks: u32 = 0, + fail_io: bool = false, + fail_stat: bool = false, + + fn init() TestProv { + var p: TestProv = .{}; + p.nodes[0] = .{ .used = true, .is_dir = true, .mode = cloud9.dmdir | 0o755 }; + _ = p.add(0, "hello", false, 0o644); + p.nodes[1].len = 5; + @memcpy(p.nodes[1].data[0..5], "hello"); + const d = p.add(0, "dir", true, cloud9.dmdir | 0o755); + _ = p.add(d, "inner", false, 0o600); + _ = p.add(0, "locked", false, 0o000); + return p; + } + + fn add(p: *TestProv, parent: u32, name: []const u8, is_dir: bool, mode: u32) u32 { + for (&p.nodes, 0..) |*e, i| if (!e.used) { + e.* = .{ .used = true, .parent = parent, .is_dir = is_dir, .mode = mode }; + @memcpy(e.name[0..name.len], name); + e.name_len = @intCast(name.len); + return @intCast(i); + }; + unreachable; + } + + fn self(ctx: *anyopaque) *TestProv { + return @ptrCast(@alignCast(ctx)); + } + + fn node(p: *TestProv, h: Provider.Handle) Provider.Error!*Entry { + if (h >= max_nodes or !p.nodes[h].used) return error.NotFound; + return &p.nodes[h]; + } + + fn retain(p: *TestProv, h: Provider.Handle) Provider.Handle { + if (h != 0) { + p.nodes[h].refs += 1; + p.total_refs += 1; + } + return h; + } + + fn walk(ctx: *anyopaque, parent: Provider.Handle, name: []const u8) Provider.Error!Provider.Handle { + const p = self(ctx); + if (p.fail_io) return error.Io; + const d = try p.node(parent); + if (std.mem.eql(u8, name, ".")) return p.retain(parent); + if (!d.is_dir) return error.NotDir; + if (std.mem.eql(u8, name, "..")) return p.retain(d.parent); + for (p.nodes[0..], 0..) |*e, i| { + if (e.used and e.parent == parent and i != 0 and std.mem.eql(u8, e.nameSlice(), name)) return p.retain(@intCast(i)); + } + return error.NotFound; + } + + fn fillStat(e: *const Entry, h: Provider.Handle, out: *NodeStat) void { + out.* = .{ .mode = e.mode, .length = e.len, .mtime = e.mtime, .name = e.nameSlice(), .handle = h }; + } + + fn stat(ctx: *anyopaque, h: Provider.Handle, out: *NodeStat) Provider.Error!void { + const p = self(ctx); + if (p.fail_stat) return error.Io; + fillStat(try p.node(h), h, out); + } + + fn list(ctx: *anyopaque, dir: Provider.Handle, index: usize, out: *NodeStat) Provider.Error!bool { + const p = self(ctx); + const d = try p.node(dir); + if (!d.is_dir) return error.NotDir; + var k: usize = 0; + for (p.nodes[0..], 0..) |*e, i| { + if (!e.used or e.parent != dir or i == 0) continue; + if (k == index) { + fillStat(e, @intCast(i), out); + return true; + } + k += 1; + } + return false; + } + + fn open(ctx: *anyopaque, h: Provider.Handle, mode: u8) Provider.Error!void { + const p = self(ctx); + const e = try p.node(h); + const acc = mode & 3; + if (acc != cloud9.owrite and e.mode & 0o400 == 0) return error.Perm; + if (acc != cloud9.oread and e.mode & 0o200 == 0) return error.Perm; + if (mode & cloud9.otrunc != 0) e.len = 0; + e.opens += 1; + } + + fn close(ctx: *anyopaque, h: Provider.Handle) void { + const p = self(ctx); + p.nodes[h].opens -= 1; + } + + fn read(ctx: *anyopaque, h: Provider.Handle, offset: u64, buf: []u8) Provider.Error!usize { + const p = self(ctx); + const e = try p.node(h); + if (offset >= e.len) return 0; + const n = @min(buf.len, e.len - @as(usize, @intCast(offset))); + @memcpy(buf[0..n], e.data[@intCast(offset)..][0..n]); + return n; + } + + fn write(ctx: *anyopaque, h: Provider.Handle, offset: u64, data: []const u8) Provider.Error!usize { + const p = self(ctx); + const e = try p.node(h); + if (offset + data.len > e.data.len) return error.NoSpace; + const off: usize = @intCast(offset); + @memcpy(e.data[off..][0..data.len], data); + e.len = @max(e.len, off + data.len); + e.mtime += 1; + return data.len; + } + + fn create(ctx: *anyopaque, dir: Provider.Handle, name: []const u8, perm: u32, mode: u8) Provider.Error!Provider.Handle { + const p = self(ctx); + const d = try p.node(dir); + if (!d.is_dir) return error.NotDir; + for (p.nodes[0..]) |*e| if (e.used and e.parent == dir and std.mem.eql(u8, e.nameSlice(), name)) return error.Exists; + var free: ?u32 = null; + for (p.nodes[0..], 0..) |*e, i| if (!e.used) { + free = @intCast(i); + break; + }; + const idx = free orelse return error.NoSpace; + const h = p.add(@intCast(dir), name, perm & cloud9.dmdir != 0, perm); + std.debug.assert(h == idx); + p.nodes[h].opens = 1; + _ = mode; + return p.retain(h); + } + + fn remove(ctx: *anyopaque, h: Provider.Handle) Provider.Error!void { + const p = self(ctx); + const e = try p.node(h); + if (h == 0) return error.Perm; + for (p.nodes[0..]) |*c| if (c.used and c.parent == h) return error.NotEmpty; + e.used = false; // refs still keep the slot "alive" for clunk accounting + e.used = true; + e.parent = std.math.maxInt(u32); // unlinked + } + + fn wstat(ctx: *anyopaque, h: Provider.Handle, st: *const cloud9.Stat) Provider.Error!void { + const p = self(ctx); + const e = try p.node(h); + if (st.name.len != 0) { + @memcpy(e.name[0..st.name.len], st.name); + e.name_len = @intCast(st.name.len); + } + if (st.length != 0xFFFF_FFFF_FFFF_FFFF) { + if (st.length > e.data.len) return error.NoSpace; + e.len = @intCast(st.length); + } + if (st.mode != 0xFFFF_FFFF) e.mode = st.mode; + if (st.mtime != 0xFFFF_FFFF) e.mtime = st.mtime; + } + + fn clunk(ctx: *anyopaque, h: Provider.Handle) void { + const p = self(ctx); + p.clunks += 1; + if (h != 0) { + p.nodes[h].refs -= 1; + p.total_refs -= 1; + } + } + + const vtable: Provider.VTable = .{ + .walk = &walk, + .stat = &stat, + .list = &list, + .open = &open, + .read = &read, + .write = &write, + .create = &create, + .remove = &remove, + .wstat = &wstat, + .close = &close, + .clunk = &clunk, + }; + + fn provider(p: *TestProv) Provider { + return .{ .name = "prov", .ctx = p, .vtable = &vtable }; + } +}; + +const Inner = struct { x: f32 }; +const Exposed = struct { a: u32, b: bool, name: []const u8, inner: Inner }; + +/// Everything a core test needs, in one place; `harness.init` runs version+attach. +const Fixture = struct { + ctx: TestCtx = .{}, + shared: TS.Shared = undefined, + storage: TS.Storage = undefined, + prov: TestProv = undefined, + exposed: Exposed = .{ .a = 1, .b = true, .name = "hello", .inner = .{ .x = 0.5 } }, + counter: u64 = 7, + h: TS.Harness = undefined, + + fn init(x: *Fixture) !void { + x.shared = .init(&x.ctx); + x.prov = TestProv.init(); + try x.shared.addProvider(x.prov.provider()); + try x.shared.expose("state", &x.exposed); + try x.shared.expose("counter", &x.counter); + try x.h.init(&x.shared, &x.storage); + } + + fn deinit(x: *Fixture) void { + x.h.deinit(); + } +}; + +test "README, /build and the static tree read as expected" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + const readme = try x.h.readPath(&.{"README"}); + defer testing.allocator.free(readme); + try testing.expect(std.mem.startsWith(u8, readme, "tester: a 9P2000 introspection server")); + const zv = try x.h.readPath(&.{ "build", "zig_version" }); + defer testing.allocator.free(zv); + try testing.expectEqualStrings(builtin.zig_version_string, zv); + const ch = try x.h.readPath(&.{ "build", "change" }); + defer testing.allocator.free(ch); + try testing.expectEqualStrings("abc123", ch); + try testing.expectEqual(@as(u32, 1_700_000_000), TS.build_secs); + const names = try x.h.listPath(&.{}); + defer TS.Harness.freeNames(names); + for ([_][]const u8{ "README", "build", "comptime", "runtime", "ctl", "vars", "prov" }) |n| try testing.expect(TS.Harness.hasName(names, n)); + try testing.expectEqual(@as(usize, 7), names.len); + // static files are read-only; the static tree admits no creates or removes + try x.h.walkTo(1, &.{ "build", "target" }); + try x.h.expectFail(.{ .open = .{ .fid = 1, .mode = cloud9.owrite } }, "permission denied"); + try x.h.expectFail(.{ .remove = .{ .fid = 1 } }, "permission denied"); + try x.h.walkTo(2, &.{"build"}); + try x.h.expectFail(.{ .create = .{ .fid = 2, .name = "nope", .perm = 0o644, .mode = cloud9.owrite } }, "permission denied"); + try x.h.expectFail(.{ .wstat = .{ .fid = 2, .stat = stat_dontcare } }, "permission denied"); + const st = try x.h.ok(.{ .stat = .{ .fid = 2 } }); + try testing.expectEqualStrings("build", st.stat.name); + try testing.expectEqualStrings("tester", st.stat.uid); + try testing.expect(st.stat.qid.type & cloud9.qtdir != 0); + try testing.expectEqual(TS.build_secs, st.stat.mtime); + try testing.expectEqual(@as(u32, 0), parseIso8601("1970-01-01T00:00:00Z").?); + try testing.expectEqual(@as(?u32, null), parseIso8601("unknown")); +} + +test "comptime/types fields carry @offsetOf and comptime/decls lists pub decls" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + const names = try x.h.listPath(&.{ "comptime", "types" }); + defer TS.Harness.freeNames(names); + try testing.expectEqual(@as(usize, 2), names.len); + try testing.expect(TS.Harness.hasName(names, "Layout")); + try testing.expect(TS.Harness.hasName(names, "Qid")); + const fields = try x.h.readPath(&.{ "comptime", "types", "Layout", "fields" }); + defer testing.allocator.free(fields); + var expect_buf: [128]u8 = undefined; + const expect = try std.fmt.bufPrint(&expect_buf, "a: u8 @{d}\nb: u32 @{d}\nc: u64 @{d}\n", .{ @offsetOf(Layout, "a"), @offsetOf(Layout, "b"), @offsetOf(Layout, "c") }); + try testing.expectEqualStrings(expect, fields); + const size = try x.h.readPath(&.{ "comptime", "types", "Layout", "size" }); + defer testing.allocator.free(size); + try testing.expectEqualStrings(std.fmt.comptimePrint("{d}", .{@sizeOf(Layout)}), size); + const name = try x.h.readPath(&.{ "comptime", "types", "Qid", "name" }); + defer testing.allocator.free(name); + try testing.expectEqualStrings(@typeName(cloud9.Qid), name); + const decls = try x.h.readPath(&.{ "comptime", "decls" }); + defer testing.allocator.free(decls); + try testing.expectEqualStrings("one\ntwo\nthree\n", decls); +} + +test "runtime/fn calls the function at open and at each read from offset 0" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + const names = try x.h.listPath(&.{ "runtime", "fn" }); + defer TS.Harness.freeNames(names); + try testing.expectEqual(@typeInfo(TestFns).@"struct".decls.len, names.len); + const fib_text = try x.h.readPath(&.{ "runtime", "fn", "fib30" }); + defer testing.allocator.free(fib_text); + try testing.expectEqualStrings("832040", fib_text); + const pid = try x.h.readPath(&.{ "runtime", "pid" }); + defer testing.allocator.free(pid); + try testing.expectEqualStrings("4242", pid); + // the generator runs at open, then again at each read from offset 0, not at offset > 0 + try x.h.walkTo(1, &.{ "runtime", "fn", "counter" }); + _ = try x.h.ok(.{ .open = .{ .fid = 1, .mode = cloud9.oread } }); + try testing.expectEqual(@as(u32, 1), x.ctx.calls); + const r1 = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = 100 } }); + try testing.expectEqualStrings("2", r1.read); + const r2 = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 1, .count = 100 } }); + try testing.expectEqualStrings("", r2.read); + try testing.expectEqual(@as(u32, 2), x.ctx.calls); + // stat of a dynamic file reports length 0 + const st = try x.h.ok(.{ .stat = .{ .fid = 1 } }); + try testing.expectEqual(@as(u64, 0), st.stat.length); + try testing.expectEqual(@as(u32, 0o444), st.stat.mode); + // a generator error is the file's Rerror; a generator that overflows the slot too + try x.h.walkTo(2, &.{ "runtime", "fn", "failing" }); + try x.h.expectFail(.{ .open = .{ .fid = 2, .mode = cloud9.oread } }, "bad command"); + try x.h.walkTo(3, &.{ "runtime", "fn", "huge" }); + try x.h.expectFail(.{ .open = .{ .fid = 3, .mode = cloud9.oread } }, "no space in buffer"); + // a failed open frees its slot: two more dynamic opens still succeed + try x.h.walkTo(4, &.{ "runtime", "fn", "fib30" }); + _ = try x.h.ok(.{ .open = .{ .fid = 4, .mode = cloud9.oread } }); + _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); + _ = try x.h.ok(.{ .walk = .{ .fid = 4, .newfid = 5, .names = &.{} } }); + _ = try x.h.ok(.{ .open = .{ .fid = 5, .mode = cloud9.oread } }); +} + +test "snapshot slot exhaustion is an Rerror and clunk frees the slot" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + try x.h.walkTo(1, &.{ "runtime", "fn", "fib30" }); + try x.h.walkTo(2, &.{ "runtime", "fn", "fib30" }); + try x.h.walkTo(3, &.{ "vars", "counter", "value" }); + _ = try x.h.ok(.{ .open = .{ .fid = 1, .mode = cloud9.oread } }); + _ = try x.h.ok(.{ .open = .{ .fid = 2, .mode = cloud9.oread } }); + try x.h.expectFail(.{ .open = .{ .fid = 3, .mode = cloud9.oread } }, "too many open dynamic files"); + // static and provider files need no slot + const t = try x.h.readPath(&.{ "vars", "counter", "type" }); + defer testing.allocator.free(t); + try testing.expectEqualStrings("u64", t); + _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); + _ = try x.h.ok(.{ .open = .{ .fid = 3, .mode = cloud9.oread } }); + const r = try x.h.ok(.{ .read = .{ .fid = 3, .offset = 0, .count = 100 } }); + try testing.expectEqualStrings("7", r.read); + // cloning an open fid onto itself keeps it open and its content + const w = try x.h.ok(.{ .walk = .{ .fid = 3, .newfid = 3, .names = &.{} } }); + try testing.expectEqual(@as(u16, 0), w.walk.nwqid); + const r2 = try x.h.ok(.{ .read = .{ .fid = 3, .offset = 0, .count = 100 } }); + try testing.expectEqualStrings("7", r2.read); + try x.h.expectFail(.{ .walk = .{ .fid = 3, .newfid = 4, .names = &.{".."} } }, "file already open"); + _ = try x.h.ok(.{ .walk = .{ .fid = 3, .newfid = 4, .names = &.{} } }); + try x.h.expectFail(.{ .read = .{ .fid = 4, .offset = 0, .count = 100 } }, "file not open"); +} + +test "ctl round trip" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + try x.h.walkTo(1, &.{"ctl"}); + _ = try x.h.ok(.{ .open = .{ .fid = 1, .mode = cloud9.ordwr } }); + const w = try x.h.ok(.{ .write = .{ .fid = 1, .offset = 0, .data = "add 2 3\n" } }); + try testing.expectEqual(@as(u32, 8), w.write); + const r = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = 100 } }); + try testing.expectEqualStrings("5", r.read); + try testing.expectEqual(@as(i64, 5), x.ctx.ctl_state); + const st = try x.h.ok(.{ .stat = .{ .fid = 1 } }); + try testing.expectEqual(@as(u64, 1), st.stat.length); + try testing.expectEqualStrings("ctl", st.stat.name); + try testing.expectEqual(@as(u32, 0o666), st.stat.mode); + const v1 = st.stat.qid.version; + _ = try x.h.ok(.{ .write = .{ .fid = 1, .offset = 0, .data = "echo hello world" } }); + const r2 = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = 100 } }); + try testing.expectEqualStrings("hello world", r2.read); + const r3 = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 6, .count = 100 } }); + try testing.expectEqualStrings("world", r3.read); + try testing.expect((try x.h.ok(.{ .stat = .{ .fid = 1 } })).stat.qid.version != v1); + const v2 = (try x.h.ok(.{ .stat = .{ .fid = 1 } })).stat.qid.version; + try x.h.expectFail(.{ .write = .{ .fid = 1, .offset = 0, .data = "frobnicate" } }, "bad command"); + // a failed command leaves the previous result, length and version in place + const r4 = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = 100 } }); + try testing.expectEqualStrings("hello world", r4.read); + const st4 = try x.h.ok(.{ .stat = .{ .fid = 1 } }); + try testing.expectEqual(@as(u64, 11), st4.stat.length); + try testing.expectEqual(v2, st4.stat.qid.version); + // even when the handler wrote part of a result before failing + try x.h.expectFail(.{ .write = .{ .fid = 1, .offset = 0, .data = "partial" } }, "bad command"); + const r5 = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = 100 } }); + try testing.expectEqualStrings("hello world", r5.read); + try testing.expectEqualStrings("hello world", x.shared.ctlResult()); + // the empty result is a legitimate result too + _ = try x.h.ok(.{ .write = .{ .fid = 1, .offset = 0, .data = "echo" } }); + try testing.expectEqualStrings("", (try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = 100 } })).read); + try testing.expectEqual(@as(u64, 0), (try x.h.ok(.{ .stat = .{ .fid = 1 } })).stat.length); + // Tversion resets the ctl fid like any other + try x.h.version(4096); + try x.h.expectFail(.{ .clunk = .{ .fid = 1 } }, "unknown fid"); +} + +test "auth is not required and flush is answered" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + try x.h.expectFail(.{ .auth = .{ .afid = 5, .uname = "tester" } }, "authentication not required"); + const f = try x.h.ok(.{ .flush = .{ .oldtag = 1 } }); + try testing.expect(f == .flush); + try x.h.expectFail(.{ .attach = .{ .fid = 0, .uname = "tester" } }, "fid in use"); +} + +test "vars: value/type/size/addr/raw, fields and writes" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + const names = try x.h.listPath(&.{"vars"}); + defer TS.Harness.freeNames(names); + try testing.expectEqual(@as(usize, 2), names.len); + try testing.expectEqualStrings("state", names[0]); + const entries = try x.h.listPath(&.{ "vars", "state" }); + defer TS.Harness.freeNames(entries); + for ([_][]const u8{ "value", "type", "size", "addr", "raw", "f" }) |n| try testing.expect(TS.Harness.hasName(entries, n)); + try testing.expectEqual(@as(usize, 6), entries.len); + const value = try x.h.readPath(&.{ "vars", "state", "value" }); + defer testing.allocator.free(value); + try testing.expectEqualStrings("a: 1\nb: true\nname: \"hello\"\ninner:\n x: 0.5\n", value); + const tn = try x.h.readPath(&.{ "vars", "state", "type" }); + defer testing.allocator.free(tn); + try testing.expectEqualStrings(@typeName(Exposed), tn); + const size = try x.h.readPath(&.{ "vars", "state", "size" }); + defer testing.allocator.free(size); + try testing.expectEqualStrings(std.fmt.comptimePrint("{d}", .{@sizeOf(Exposed)}), size); + const addr = try x.h.readPath(&.{ "vars", "state", "addr" }); + defer testing.allocator.free(addr); + var addr_buf: [32]u8 = undefined; + try testing.expectEqualStrings(try std.fmt.bufPrint(&addr_buf, "0x{x}", .{@intFromPtr(&x.exposed)}), addr); + const raw = try x.h.readPath(&.{ "vars", "state", "raw" }); + defer testing.allocator.free(raw); + try testing.expectEqualSlices(u8, std.mem.asBytes(&x.exposed), raw); + try x.h.walkTo(1, &.{ "vars", "state", "raw" }); + const raw_st = try x.h.ok(.{ .stat = .{ .fid = 1 } }); + try testing.expectEqual(@as(u64, @sizeOf(Exposed)), raw_st.stat.length); + try testing.expectEqual(@as(u32, 0o444), raw_st.stat.mode); + _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); + // fields + const fnames = try x.h.listPath(&.{ "vars", "state", "f" }); + defer TS.Harness.freeNames(fnames); + try testing.expectEqual(@as(usize, 4), fnames.len); + const a_value = try x.h.readPath(&.{ "vars", "state", "f", "a", "value" }); + defer testing.allocator.free(a_value); + try testing.expectEqualStrings("1", a_value); + const a_type = try x.h.readPath(&.{ "vars", "state", "f", "a", "type" }); + defer testing.allocator.free(a_type); + try testing.expectEqualStrings("u32", a_type); + const xv = try x.h.readPath(&.{ "vars", "state", "f", "inner", "f", "x", "value" }); + defer testing.allocator.free(xv); + try testing.expectEqualStrings("0.5", xv); + const b_raw = try x.h.readPath(&.{ "vars", "state", "f", "b", "raw" }); + defer testing.allocator.free(b_raw); + try testing.expectEqualSlices(u8, &.{1}, b_raw); + // writes + try x.h.writePath(&.{ "vars", "state", "f", "a", "value" }, "42"); + try testing.expectEqual(@as(u32, 42), x.exposed.a); + try x.h.writePath(&.{ "vars", "state", "f", "b", "value" }, "false\n"); + try testing.expect(!x.exposed.b); + try x.h.writePath(&.{ "vars", "state", "f", "inner", "f", "x", "value" }, "2.25"); + try testing.expectEqual(@as(f32, 2.25), x.exposed.inner.x); + try x.h.writePath(&.{ "vars", "counter", "value" }, "0x10"); + try testing.expectEqual(@as(u64, 16), x.counter); + try x.h.walkTo(2, &.{ "vars", "state", "f", "a", "value" }); + _ = try x.h.ok(.{ .open = .{ .fid = 2, .mode = cloud9.ordwr } }); + try x.h.expectFail(.{ .write = .{ .fid = 2, .offset = 0, .data = "abc" } }, "bad value"); + const rd = try x.h.ok(.{ .read = .{ .fid = 2, .offset = 0, .count = 100 } }); + try testing.expectEqualStrings("42", rd.read); + const a_st = try x.h.ok(.{ .stat = .{ .fid = 2 } }); + try testing.expectEqual(@as(u32, 0o644), a_st.stat.mode); + try testing.expectEqualStrings("value", a_st.stat.name); + // non-scalar values, type/size/addr/raw and directories are read-only + try x.h.walkTo(3, &.{ "vars", "state", "value" }); + try x.h.expectFail(.{ .open = .{ .fid = 3, .mode = cloud9.owrite } }, "permission denied"); + _ = try x.h.ok(.{ .clunk = .{ .fid = 3 } }); + try x.h.walkTo(4, &.{ "vars", "state", "f", "name", "value" }); + try x.h.expectFail(.{ .open = .{ .fid = 4, .mode = cloud9.owrite } }, "permission denied"); + _ = try x.h.ok(.{ .clunk = .{ .fid = 4 } }); + try x.h.walkTo(5, &.{ "vars", "state" }); + try x.h.expectFail(.{ .open = .{ .fid = 5, .mode = cloud9.owrite } }, "is a directory"); + try x.h.expectFail(.{ .create = .{ .fid = 5, .name = "z", .perm = 0o644, .mode = cloud9.owrite } }, "permission denied"); + // .. climbs back out of the var tree; unknown names fail + const up = try x.h.ok(.{ .walk = .{ .fid = 5, .newfid = 6, .names = &.{ "f", "inner", "..", "..", "..", "..", "README" } } }); + try testing.expectEqual(@as(u16, 7), up.walk.nwqid); + _ = try x.h.ok(.{ .clunk = .{ .fid = 6 } }); + try x.h.walkTo(7, &.{"vars"}); + try x.h.expectFail(.{ .walk = .{ .fid = 7, .newfid = 8, .names = &.{"nope"} } }, "file does not exist"); + try x.h.expectFail(.{ .walk = .{ .fid = 2, .newfid = 8, .names = &.{"x"} } }, "file already open"); + _ = try x.h.ok(.{ .clunk = .{ .fid = 2 } }); + try x.h.walkTo(2, &.{ "vars", "state", "f", "a", "value" }); + try x.h.expectFail(.{ .walk = .{ .fid = 2, .newfid = 8, .names = &.{"x"} } }, "not a directory"); +} + +test "provider: walk/list/stat/open/read/write/create/remove/wstat/clunk and error mapping" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + const names = try x.h.listPath(&.{"prov"}); + defer TS.Harness.freeNames(names); + try testing.expectEqual(@as(usize, 3), names.len); + try testing.expect(TS.Harness.hasName(names, "hello") and TS.Harness.hasName(names, "dir") and TS.Harness.hasName(names, "locked")); + const hello = try x.h.readPath(&.{ "prov", "hello" }); + defer testing.allocator.free(hello); + try testing.expectEqualStrings("hello", hello); + try testing.expectEqual(@as(u32, 0), x.prov.total_refs); // every temp handle was clunked + // stat and qid scheme + try x.h.walkTo(1, &.{ "prov", "dir", "inner" }); + const st = try x.h.ok(.{ .stat = .{ .fid = 1 } }); + try testing.expectEqualStrings("inner", st.stat.name); + try testing.expectEqual(@as(u32, 0o600), st.stat.mode); + try testing.expectEqual(@as(u64, 3), st.stat.qid.path); // provider 0, handle 3 + try testing.expectEqualStrings("tester", st.stat.gid); + try x.h.walkTo(2, &.{"prov"}); + const root_st = try x.h.ok(.{ .stat = .{ .fid = 2 } }); + try testing.expectEqualStrings("prov", root_st.stat.name); + try testing.expect(root_st.stat.qid.type & cloud9.qtdir != 0); + try testing.expectEqual(@as(u64, 0), root_st.stat.qid.path); + try testing.expectEqual(@as(u32, 1), x.prov.total_refs); // fid 1 holds inner; fid 2 holds root (unref'd) + // write then read back; opens are tracked through close + _ = try x.h.ok(.{ .open = .{ .fid = 1, .mode = cloud9.ordwr } }); + try testing.expectEqual(@as(u32, 1), x.prov.nodes[3].opens); + _ = try x.h.ok(.{ .write = .{ .fid = 1, .offset = 0, .data = "abc" } }); + _ = try x.h.ok(.{ .write = .{ .fid = 1, .offset = 3, .data = "def" } }); + const r = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 1, .count = 100 } }); + try testing.expectEqualStrings("bcdef", r.read); + try x.h.expectFail(.{ .write = .{ .fid = 1, .offset = 100, .data = "z" } }, "no space left on device"); + _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); + try testing.expectEqual(@as(u32, 0), x.prov.nodes[3].opens); + try testing.expectEqual(@as(u32, 0), x.prov.nodes[3].refs); + // permission and kind errors come from the provider + try x.h.walkTo(3, &.{ "prov", "locked" }); + try x.h.expectFail(.{ .open = .{ .fid = 3, .mode = cloud9.oread } }, "permission denied"); + try x.h.walkTo(4, &.{ "prov", "hello" }); + try x.h.expectFail(.{ .walk = .{ .fid = 4, .newfid = 5, .names = &.{"x"} } }, "not a directory"); + try x.h.expectFail(.{ .walk = .{ .fid = 2, .newfid = 5, .names = &.{"missing"} } }, "file does not exist"); + try x.h.expectFail(.{ .open = .{ .fid = 2, .mode = cloud9.owrite } }, "is a directory"); + // create in a provider directory: the fid becomes the new open file + const cr = try x.h.ok(.{ .create = .{ .fid = 2, .name = "new", .perm = 0o644, .mode = cloud9.ordwr } }); + try testing.expectEqual(cloud9.qtfile, cr.create.qid.type); + _ = try x.h.ok(.{ .write = .{ .fid = 2, .offset = 0, .data = "fresh" } }); + const rr = try x.h.ok(.{ .read = .{ .fid = 2, .offset = 0, .count = 100 } }); + try testing.expectEqualStrings("fresh", rr.read); + try x.h.walkTo(6, &.{"prov"}); + try x.h.expectFail(.{ .create = .{ .fid = 6, .name = "new", .perm = 0o644, .mode = cloud9.oread } }, "file already exists"); + const long_name = [_]u8{'n'} ** (max_name + 1); + try x.h.expectFail(.{ .create = .{ .fid = 6, .name = &long_name, .perm = 0o644, .mode = cloud9.oread } }, "bad file name"); + try x.h.expectFail(.{ .create = .{ .fid = 6, .name = "d", .perm = cloud9.dmdir | 0o755, .mode = cloud9.owrite } }, "is a directory"); + const dr = try x.h.ok(.{ .create = .{ .fid = 6, .name = "d", .perm = cloud9.dmdir | 0o755, .mode = cloud9.oread } }); + try testing.expectEqual(cloud9.qtdir, dr.create.qid.type); + // wstat: rename, truncate, mode, mtime; immutable fields are refused + var ws = stat_dontcare; + ws.name = "renamed"; + ws.length = 2; + ws.mode = 0o600; + ws.mtime = 99; + _ = try x.h.ok(.{ .wstat = .{ .fid = 2, .stat = ws } }); + const st2 = try x.h.ok(.{ .stat = .{ .fid = 2 } }); + try testing.expectEqualStrings("renamed", st2.stat.name); + try testing.expectEqual(@as(u64, 2), st2.stat.length); + try testing.expectEqual(@as(u32, 0o600), st2.stat.mode); + try testing.expectEqual(@as(u32, 99), st2.stat.mtime); + ws = stat_dontcare; + ws.uid = "someone-else"; + try x.h.expectFail(.{ .wstat = .{ .fid = 2, .stat = ws } }, "permission denied"); + ws = stat_dontcare; + ws.mode = cloud9.dmdir | 0o755; + try x.h.expectFail(.{ .wstat = .{ .fid = 2, .stat = ws } }, "permission denied"); + ws = stat_dontcare; + ws.name = "bad/name"; + try x.h.expectFail(.{ .wstat = .{ .fid = 2, .stat = ws } }, "bad file name"); + ws.name = ".."; + try x.h.expectFail(.{ .wstat = .{ .fid = 2, .stat = ws } }, "bad file name"); + ws = stat_dontcare; + ws.length = 5; + try x.h.walkTo(7, &.{ "prov", "d" }); + try x.h.expectFail(.{ .wstat = .{ .fid = 7, .stat = ws } }, "is a directory"); + ws = stat_dontcare; + ws.name = "root2"; // the provider root cannot be renamed + try x.h.walkTo(14, &.{"prov"}); + try x.h.expectFail(.{ .wstat = .{ .fid = 14, .stat = ws } }, "permission denied"); + _ = try x.h.ok(.{ .clunk = .{ .fid = 14 } }); + // remove always clunks; a non-empty directory refuses + _ = try x.h.ok(.{ .clunk = .{ .fid = 6 } }); + try x.h.walkTo(8, &.{ "prov", "dir" }); + try x.h.expectFail(.{ .remove = .{ .fid = 8 } }, "directory not empty"); + try x.h.expectFail(.{ .clunk = .{ .fid = 8 } }, "unknown fid"); + _ = try x.h.ok(.{ .remove = .{ .fid = 2 } }); + try x.h.walkTo(9, &.{"prov"}); + try x.h.expectFail(.{ .walk = .{ .fid = 9, .newfid = 15, .names = &.{"renamed"} } }, "file does not exist"); + _ = try x.h.ok(.{ .clunk = .{ .fid = 9 } }); + // an i/o error from the provider maps to "i/o error" + try x.h.walkTo(9, &.{"prov"}); + x.prov.fail_io = true; + try x.h.expectFail(.{ .walk = .{ .fid = 9, .newfid = 15, .names = &.{"hello"} } }, "i/o error"); + x.prov.fail_io = false; + _ = try x.h.ok(.{ .clunk = .{ .fid = 9 } }); + // ORCLOSE removes on clunk + try x.h.walkTo(9, &.{"prov"}); + _ = try x.h.ok(.{ .create = .{ .fid = 9, .name = "tmp", .perm = 0o644, .mode = cloud9.owrite | cloud9.orclose } }); + _ = try x.h.ok(.{ .clunk = .{ .fid = 9 } }); + try x.h.walkTo(9, &.{"prov"}); + try x.h.expectFail(.{ .walk = .{ .fid = 9, .newfid = 15, .names = &.{"tmp"} } }, "file does not exist"); + _ = try x.h.ok(.{ .clunk = .{ .fid = 9 } }); + // walking .. out of the provider root and cloning provider fids keeps refs balanced + const up = try x.h.ok(.{ .walk = .{ .fid = 0, .newfid = 10, .names = &.{ "prov", "dir", "..", "..", "build" } } }); + try testing.expectEqual(@as(u16, 5), up.walk.nwqid); + try x.h.walkTo(11, &.{ "prov", "dir", "inner" }); + _ = try x.h.ok(.{ .walk = .{ .fid = 11, .newfid = 12, .names = &.{} } }); + try testing.expectEqual(@as(u32, 2), x.prov.nodes[3].refs); + _ = try x.h.ok(.{ .clunk = .{ .fid = 11 } }); + try testing.expectEqual(@as(u32, 1), x.prov.nodes[3].refs); + // a partial walk releases the handles it obtained + const part = try x.h.ok(.{ .walk = .{ .fid = 0, .newfid = 13, .names = &.{ "prov", "dir", "nope" } } }); + try testing.expectEqual(@as(u16, 2), part.walk.nwqid); + try x.h.expectFail(.{ .clunk = .{ .fid = 13 } }, "unknown fid"); + _ = try x.h.ok(.{ .clunk = .{ .fid = 12 } }); + for (x.h.conn.fids) |f| { + if (f.used) _ = try x.h.ok(.{ .clunk = .{ .fid = f.id } }); + } + try testing.expectEqual(@as(u32, 0), x.prov.total_refs); +} + +test "directory reads across offsets, bad offset, and records never split" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + try x.h.walkTo(1, &.{}); + _ = try x.h.ok(.{ .open = .{ .fid = 1, .mode = cloud9.oread } }); + const first = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = 4096 } }); + try testing.expect(first.read.len > 0); + try x.h.expectFail(.{ .read = .{ .fid = 1, .offset = 5, .count = 4096 } }, "bad offset"); + // offset 0 restarts; the same bytes come back + const again = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = 4096 } }); + try testing.expectEqual(first.read.len, again.read.len); + // small reads at consecutive offsets return every record exactly once + const names = try x.h.listDir(1, 80); + defer TS.Harness.freeNames(names); + try testing.expectEqual(@as(usize, 7), names.len); + // a count too small for even one record returns nothing rather than splitting it + const tiny = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = 10 } }); + try testing.expectEqual(@as(usize, 0), tiny.read.len); + // the same for provider and var directories + const pn = try x.h.listPath(&.{ "prov", "dir" }); + defer TS.Harness.freeNames(pn); + try testing.expectEqual(@as(usize, 1), pn.len); + try x.h.walkTo(2, &.{ "vars", "state", "f" }); + _ = try x.h.ok(.{ .open = .{ .fid = 2, .mode = cloud9.oread } }); + const vn = try x.h.listDir(2, 100); + defer TS.Harness.freeNames(vn); + try testing.expectEqual(@as(usize, 4), vn.len); + try x.h.expectFail(.{ .read = .{ .fid = 2, .offset = 1, .count = 100 } }, "bad offset"); +} + +test "Tversion mid-session resets fids and clunks every provider handle" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + try x.h.walkTo(1, &.{ "prov", "hello" }); + try x.h.walkTo(2, &.{ "prov", "dir", "inner" }); + _ = try x.h.ok(.{ .open = .{ .fid = 2, .mode = cloud9.oread } }); + try x.h.walkTo(3, &.{ "runtime", "fn", "fib30" }); + _ = try x.h.ok(.{ .open = .{ .fid = 3, .mode = cloud9.oread } }); + try testing.expectEqual(@as(u32, 2), x.prov.total_refs); + try testing.expectEqual(@as(u32, 1), x.prov.nodes[3].opens); + try testing.expectEqual(@as(usize, 4), x.h.conn.fidCount()); + const before = x.prov.clunks; + try x.h.version(4096); + try testing.expectEqual(@as(usize, 0), x.h.conn.fidCount()); + try testing.expectEqual(@as(u32, 0), x.prov.total_refs); + try testing.expectEqual(@as(u32, 0), x.prov.nodes[3].opens); + try testing.expectEqual(before + 2, x.prov.clunks); + try testing.expect(!x.h.conn.slot_used[0] and !x.h.conn.slot_used[1]); + try x.h.expectFail(.{ .clunk = .{ .fid = 1 } }, "unknown fid"); + _ = try x.h.ok(.{ .attach = .{ .fid = 0, .uname = "tester" } }); + try x.h.walkTo(1, &.{ "prov", "hello" }); + // hangup does the same + x.h.conn.hangup(); + try testing.expectEqual(@as(u32, 0), x.prov.total_refs); + try testing.expectEqual(@as(usize, 0), x.h.conn.fidCount()); +} + +test "a reply that does not fit msize is an Rerror, not a dead connection" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + try x.h.version(64); + _ = try x.h.ok(.{ .attach = .{ .fid = 0, .uname = "t" } }); + // Rstat of the root is ~70 bytes. + try x.h.expectFail(.{ .stat = .{ .fid = 0 } }, "reply too large for msize"); + // Rwalk with 5 qids is 74 bytes; the walk must not bind newfid. + try x.h.expectFail(.{ .walk = .{ .fid = 0, .newfid = 1, .names = &.{ ".", ".", ".", ".", "." } } }, "reply too large for msize"); + try x.h.expectFail(.{ .clunk = .{ .fid = 1 } }, "unknown fid"); + try x.h.expectFail(.{ .walk = .{ .fid = 0, .newfid = 0, .names = &.{ ".", ".", ".", ".", "." } } }, "reply too large for msize"); + const r = try x.h.ok(.{ .walk = .{ .fid = 0, .newfid = 1, .names = &.{"README"} } }); + try testing.expectEqual(@as(u16, 1), r.walk.nwqid); + _ = try x.h.ok(.{ .open = .{ .fid = 1, .mode = cloud9.oread } }); + const rd = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = 40 } }); + try testing.expect(rd.read.len > 0 and rd.read.len <= 64 - cloud9.iohdrsz); + _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); +} + +test "fid table is bounded per connection" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + var i: u32 = 1; + while (x.h.conn.fidCount() < test_cfg.max_fids) : (i += 1) { + _ = try x.h.ok(.{ .walk = .{ .fid = 0, .newfid = i, .names = &.{} } }); + } + try x.h.expectFail(.{ .walk = .{ .fid = 0, .newfid = i, .names = &.{} } }, "too many fids"); + try x.h.expectFail(.{ .attach = .{ .fid = i, .uname = "tester" } }, "too many fids"); + // self-walks and clunks still work at the limit + _ = try x.h.ok(.{ .walk = .{ .fid = 0, .newfid = 0, .names = &.{"build"} } }); + _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); + _ = try x.h.ok(.{ .walk = .{ .fid = 0, .newfid = i, .names = &.{} } }); + try x.h.expectFail(.{ .walk = .{ .fid = 0, .newfid = 2, .names = &.{"README"} } }, "fid in use"); + try x.h.expectFail(.{ .walk = .{ .fid = 1234, .newfid = 2, .names = &.{} } }, "unknown fid"); + // a walk into a provider at the limit must not leak the handle + _ = try x.h.ok(.{ .clunk = .{ .fid = 2 } }); + _ = try x.h.ok(.{ .walk = .{ .fid = 0, .newfid = 2, .names = &.{ "..", "prov", "hello" } } }); + try x.h.expectFail(.{ .walk = .{ .fid = 2, .newfid = i + 1, .names = &.{} } }, "too many fids"); + try testing.expectEqual(@as(u32, 1), x.prov.total_refs); +} + +test "Shared refuses more providers or vars than configured" { + var ctx: TestCtx = .{}; + var shared: TS.Shared = .init(&ctx); + var p1 = TestProv.init(); + var p2 = TestProv.init(); + var p3 = TestProv.init(); + try shared.addProvider(.{ .name = "a", .ctx = &p1, .vtable = &TestProv.vtable }); + try shared.addProvider(.{ .name = "b", .ctx = &p2, .vtable = &TestProv.vtable }); + try testing.expectError(error.Full, shared.addProvider(.{ .name = "c", .ctx = &p3, .vtable = &TestProv.vtable })); + var v: [5]u32 = @splat(0); + try shared.expose("v0", &v[0]); + try shared.expose("v1", &v[1]); + try shared.expose("v2", &v[2]); + try shared.expose("v3", &v[3]); + try testing.expectError(error.Full, shared.expose("v4", &v[4])); +} + +/// A server with a large fid table for the index tests. +const big_cfg: Config = .{ + .name = "big", + .msize = 8192, + .max_fids = 4096, + .max_providers = 1, + .max_vars = 1, + .snapshot_slots = 1, + .snapshot_bytes = 256, +}; +const BigS = Server(big_cfg); + +/// Fid numbers chosen to stress the index: dense low ids, ids with only high +/// bits set, and ids counting down from 2^32-1 (all distinct for i < 2^20). +fn adversarialId(i: u32) u32 { + return switch (i % 3) { + 0 => i * 8192 + 1, + 1 => 0x8000_0000 | i, + else => 0xFFFF_FFFF - i, + }; +} + +/// Every index bucket points at a used fid that finds itself, and every used +/// fid is found: the invariant the hostile fid tests check after each phase. +fn checkFidIndex(c: *BigS.Conn) !void { + var indexed: usize = 0; + for (c.index) |slot| { + if (slot == BigS.no_slot) continue; + indexed += 1; + try testing.expect(c.fids[slot].used); + try testing.expectEqual(&c.fids[slot], c.findFid(c.fids[slot].id).?); + } + var used: usize = 0; + for (c.fids[0..c.high_water]) |*f| if (f.used) { + used += 1; + try testing.expectEqual(f, c.findFid(f.id).?); + }; + for (c.fids[c.high_water..]) |*f| try testing.expect(!f.used); + try testing.expectEqual(indexed, used); + try testing.expectEqual(used, c.nfids); +} + +test "fid index: thousands of fids, clunk in hostile orders, reuse, Tversion" { + var ctx: TestCtx = .{}; + var shared: BigS.Shared = .init(&ctx); + var prov = TestProv.init(); + try shared.addProvider(prov.provider()); + const storage = try testing.allocator.create(BigS.Storage); + defer testing.allocator.destroy(storage); + var h: BigS.Harness = undefined; + try h.init(&shared, storage); + defer h.deinit(); + const n: u32 = big_cfg.max_fids - 1; // fid 0 is the attach + var i: u32 = 0; + while (i < n) : (i += 1) { + _ = try h.ok(.{ .walk = .{ .fid = 0, .newfid = adversarialId(i), .names = &.{ "prov", "hello" } } }); + } + try testing.expectEqual(@as(usize, n + 1), h.conn.fidCount()); + try testing.expectEqual(n, prov.total_refs); + try h.expectFail(.{ .walk = .{ .fid = 0, .newfid = 0x7FFF_FFFF, .names = &.{} } }, "too many fids"); + try h.expectFail(.{ .walk = .{ .fid = 0, .newfid = adversarialId(5), .names = &.{} } }, "fid in use"); + try h.expectFail(.{ .attach = .{ .fid = adversarialId(7), .uname = "t" } }, "fid in use"); + try testing.expect(h.conn.findFid(0x7FFF_FFFF) == null); + try testing.expect(h.conn.findFid(adversarialId(n)) == null); + try checkFidIndex(&h.conn); + // clunk every third fid, then the rest from the top: backward-shift deletion under churn + i = 0; + while (i < n) : (i += 3) _ = try h.ok(.{ .clunk = .{ .fid = adversarialId(i) } }); + try checkFidIndex(&h.conn); + i = n; + while (i > 0) { + i -= 1; + if (i % 3 == 0) { + try h.expectFail(.{ .clunk = .{ .fid = adversarialId(i) } }, "unknown fid"); + } else { + _ = try h.ok(.{ .clunk = .{ .fid = adversarialId(i) } }); + } + } + try testing.expectEqual(@as(usize, 1), h.conn.fidCount()); + try testing.expectEqual(@as(u32, 0), prov.total_refs); + try checkFidIndex(&h.conn); + // the whole table is reusable after the churn, through the free list + i = 0; + while (i < n) : (i += 1) _ = try h.ok(.{ .walk = .{ .fid = 0, .newfid = n - i, .names = &.{} } }); + try h.expectFail(.{ .walk = .{ .fid = 0, .newfid = n + 1, .names = &.{} } }, "too many fids"); + try checkFidIndex(&h.conn); + // pseudo-random alloc/free storm with verification + var prng = std.Random.DefaultPrng.init(0x9a11); + const rnd = prng.random(); + var live: [n + 1]bool = @splat(true); + live[0] = false; // never touch the attach fid + var round: usize = 0; + while (round < 20_000) : (round += 1) { + const id = 1 + rnd.uintLessThan(u32, n); + if (live[id]) { + _ = try h.ok(.{ .clunk = .{ .fid = id } }); + } else { + _ = try h.ok(.{ .walk = .{ .fid = 0, .newfid = id, .names = &.{"prov"} } }); + } + live[id] = !live[id]; + if (round % 997 == 0) try checkFidIndex(&h.conn); + } + try checkFidIndex(&h.conn); + // Tversion drops everything and the table starts over, provider refs balanced + try h.version(big_cfg.msize); + try testing.expectEqual(@as(usize, 0), h.conn.fidCount()); + try testing.expectEqual(@as(u32, 0), prov.total_refs); + try testing.expectEqual(@as(u16, 0), h.conn.high_water); + try checkFidIndex(&h.conn); + _ = try h.ok(.{ .attach = .{ .fid = 0xFFFF_FFFE, .uname = "t" } }); + _ = try h.ok(.{ .walk = .{ .fid = 0xFFFF_FFFE, .newfid = 0, .names = &.{} } }); + try checkFidIndex(&h.conn); +} + +test "open: a provider stat failure after a successful open closes the file again" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + try x.h.walkTo(1, &.{ "prov", "hello" }); + x.prov.fail_stat = true; + try x.h.expectFail(.{ .open = .{ .fid = 1, .mode = cloud9.oread } }, "i/o error"); + x.prov.fail_stat = false; + try testing.expectEqual(@as(u32, 0), x.prov.nodes[1].opens); + try x.h.expectFail(.{ .read = .{ .fid = 1, .offset = 0, .count = 10 } }, "file not open"); + _ = try x.h.ok(.{ .open = .{ .fid = 1, .mode = cloud9.oread } }); + try testing.expectEqual(@as(u32, 1), x.prov.nodes[1].opens); + // the same for create: a stat failure after the provider created the node releases it + try x.h.walkTo(2, &.{"prov"}); + x.prov.fail_stat = true; + try x.h.expectFail(.{ .create = .{ .fid = 2, .name = "born", .perm = 0o644, .mode = cloud9.owrite } }, "i/o error"); + x.prov.fail_stat = false; + for (x.prov.nodes) |e| if (e.used and std.mem.eql(u8, e.nameSlice(), "born")) { + try testing.expectEqual(@as(u32, 0), e.opens); + try testing.expectEqual(@as(u32, 0), e.refs); + }; + try testing.expect(!x.h.conn.findFid(2).?.open); + try testing.expectEqual(@as(u32, 1), x.prov.total_refs); // fid 1 only +} + +test "fid state machine: open twice, walk from open, remove/clunk of open provider fids" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + try x.h.walkTo(1, &.{ "prov", "dir", "inner" }); + _ = try x.h.ok(.{ .open = .{ .fid = 1, .mode = cloud9.ordwr } }); + try x.h.expectFail(.{ .open = .{ .fid = 1, .mode = cloud9.oread } }, "file already open"); + try x.h.expectFail(.{ .walk = .{ .fid = 1, .newfid = 2, .names = &.{"."} } }, "file already open"); + try x.h.expectFail(.{ .create = .{ .fid = 1, .name = "z", .perm = 0o644, .mode = cloud9.oread } }, "file already open"); + // a clone of an open fid is a fresh, unopened reference + _ = try x.h.ok(.{ .walk = .{ .fid = 1, .newfid = 2, .names = &.{} } }); + try testing.expectEqual(@as(u32, 2), x.prov.nodes[3].refs); + try testing.expectEqual(@as(u32, 1), x.prov.nodes[3].opens); + // walking newfid == fid with names on an unopened provider fid swaps the handle, refs balanced + try x.h.walkTo(7, &.{ "prov", "dir" }); + try testing.expectEqual(@as(u32, 1), x.prov.nodes[2].refs); + _ = try x.h.ok(.{ .walk = .{ .fid = 7, .newfid = 7, .names = &.{ "..", "dir", "inner", "..", "..", "dir" } } }); + try testing.expectEqual(@as(u32, 1), x.prov.nodes[2].refs); + try testing.expectEqual(@as(u32, 2), x.prov.nodes[3].refs); + _ = try x.h.ok(.{ .clunk = .{ .fid = 7 } }); + try testing.expectEqual(@as(u32, 0), x.prov.nodes[2].refs); + // remove of an open fid: close, then remove, then clunk; refs and opens return to zero + _ = try x.h.ok(.{ .remove = .{ .fid = 1 } }); + try testing.expectEqual(@as(u32, 0), x.prov.nodes[3].opens); + try testing.expectEqual(@as(u32, 1), x.prov.nodes[3].refs); + try x.h.expectFail(.{ .open = .{ .fid = 1, .mode = cloud9.oread } }, "unknown fid"); + _ = try x.h.ok(.{ .clunk = .{ .fid = 2 } }); + try testing.expectEqual(@as(u32, 0), x.prov.total_refs); + // walking "." on a file fid is "not a directory" at the protocol level, without a provider walk + try x.h.walkTo(3, &.{ "prov", "hello" }); + const before = x.prov.clunks; + try x.h.expectFail(.{ .walk = .{ .fid = 3, .newfid = 4, .names = &.{"."} } }, "not a directory"); + try testing.expectEqual(before, x.prov.clunks); + try testing.expectEqual(@as(u32, 1), x.prov.total_refs); + // a partial walk through a file releases the handles it took + const part = try x.h.ok(.{ .walk = .{ .fid = 0, .newfid = 5, .names = &.{ "prov", "hello", "x", "y" } } }); + try testing.expectEqual(@as(u16, 2), part.walk.nwqid); + try testing.expectEqual(@as(u32, 1), x.prov.total_refs); + try x.h.expectFail(.{ .clunk = .{ .fid = 5 } }, "unknown fid"); + // Tremove is always a clunk, even of a static node or when the provider refuses + try x.h.walkTo(6, &.{"README"}); + try x.h.expectFail(.{ .remove = .{ .fid = 6 } }, "permission denied"); + try x.h.expectFail(.{ .clunk = .{ .fid = 6 } }, "unknown fid"); + _ = try x.h.ok(.{ .clunk = .{ .fid = 3 } }); + try testing.expectEqual(@as(u32, 0), x.prov.total_refs); +} + +test "snapshot slots: exhaust, hold, Tversion frees; reads past the end and at huge offsets" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + try x.h.walkTo(1, &.{ "runtime", "fn", "fib30" }); + try x.h.walkTo(2, &.{ "vars", "state", "value" }); + try x.h.walkTo(3, &.{ "vars", "state", "addr" }); + _ = try x.h.ok(.{ .open = .{ .fid = 1, .mode = cloud9.oread } }); + _ = try x.h.ok(.{ .open = .{ .fid = 2, .mode = cloud9.oread } }); + try x.h.expectFail(.{ .open = .{ .fid = 3, .mode = cloud9.oread } }, "too many open dynamic files"); + try testing.expect(!x.h.conn.findFid(3).?.open); + // reads at offsets near 2^64 never trap (counts above msize are a raw-9P + // case: the cloud9 client refuses to send them; test/adv_core_hostile.py covers it) + const max_count = test_cfg.msize - cloud9.iohdrsz; + const r = try x.h.ok(.{ .read = .{ .fid = 1, .offset = std.math.maxInt(u64), .count = max_count } }); + try testing.expectEqualStrings("", r.read); + const r2 = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 1 << 63, .count = 0 } }); + try testing.expectEqualStrings("", r2.read); + const r3 = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = max_count } }); + try testing.expectEqualStrings("832040", r3.read); + // raw beyond @sizeOf is empty; a partial raw read at the tail is bounded + try x.h.walkTo(4, &.{ "vars", "state", "raw" }); + _ = try x.h.ok(.{ .open = .{ .fid = 4, .mode = cloud9.oread } }); + const raw_end = try x.h.ok(.{ .read = .{ .fid = 4, .offset = @sizeOf(Exposed), .count = 100 } }); + try testing.expectEqualStrings("", raw_end.read); + const raw_tail = try x.h.ok(.{ .read = .{ .fid = 4, .offset = @sizeOf(Exposed) - 1, .count = 100 } }); + try testing.expectEqual(@as(usize, 1), raw_tail.read.len); + const raw_huge = try x.h.ok(.{ .read = .{ .fid = 4, .offset = std.math.maxInt(u64) - 1, .count = 100 } }); + try testing.expectEqualStrings("", raw_huge.read); + // Tversion releases the held slots + try x.h.version(test_cfg.msize); + try testing.expect(!x.h.conn.slot_used[0] and !x.h.conn.slot_used[1]); + _ = try x.h.ok(.{ .attach = .{ .fid = 0, .uname = "tester" } }); + try x.h.walkTo(3, &.{ "vars", "state", "addr" }); + _ = try x.h.ok(.{ .open = .{ .fid = 3, .mode = cloud9.oread } }); +} + +test "static and var nodes refuse create, remove and wstat; directories refuse writes" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + const dirs = [_][]const []const u8{ &.{}, &.{"build"}, &.{"comptime"}, &.{ "comptime", "types" }, &.{ "comptime", "types", "Layout" }, &.{"runtime"}, &.{ "runtime", "fn" }, &.{"vars"}, &.{ "vars", "state" }, &.{ "vars", "state", "f" }, &.{ "vars", "state", "f", "inner" } }; + for (dirs, 0..) |d, k| { + const fid: u32 = @intCast(10 + k); + try x.h.walkTo(fid, d); + try x.h.expectFail(.{ .create = .{ .fid = fid, .name = "x", .perm = 0o644, .mode = cloud9.owrite } }, "permission denied"); + try x.h.expectFail(.{ .wstat = .{ .fid = fid, .stat = stat_dontcare } }, "permission denied"); + try x.h.expectFail(.{ .open = .{ .fid = fid, .mode = cloud9.owrite } }, "is a directory"); + try x.h.expectFail(.{ .open = .{ .fid = fid, .mode = cloud9.oread | cloud9.otrunc } }, "is a directory"); + try x.h.expectFail(.{ .remove = .{ .fid = fid } }, "permission denied"); + try x.h.expectFail(.{ .clunk = .{ .fid = fid } }, "unknown fid"); + } + const files = [_][]const []const u8{ &.{"README"}, &.{ "build", "time" }, &.{ "comptime", "decls" }, &.{ "runtime", "pid" }, &.{ "runtime", "fn", "fib30" }, &.{"ctl"}, &.{ "vars", "state", "value" }, &.{ "vars", "state", "raw" }, &.{ "vars", "state", "f", "a", "value" }, &.{ "vars", "counter", "type" } }; + for (files, 0..) |f, k| { + const fid: u32 = @intCast(30 + k); + try x.h.walkTo(fid, f); + try x.h.expectFail(.{ .wstat = .{ .fid = fid, .stat = stat_dontcare } }, "permission denied"); + try x.h.expectFail(.{ .walk = .{ .fid = fid, .newfid = 99, .names = &.{".."} } }, "not a directory"); + try x.h.expectFail(.{ .remove = .{ .fid = fid } }, "permission denied"); + } + // writes to a var value at a non-zero offset and with an empty payload + try x.h.walkTo(1, &.{ "vars", "state", "f", "a", "value" }); + _ = try x.h.ok(.{ .open = .{ .fid = 1, .mode = cloud9.owrite | cloud9.otrunc } }); + try x.h.expectFail(.{ .write = .{ .fid = 1, .offset = 0, .data = "" } }, "bad value"); + try x.h.expectFail(.{ .write = .{ .fid = 1, .offset = 0, .data = "-1" } }, "bad value"); + try x.h.expectFail(.{ .write = .{ .fid = 1, .offset = 0, .data = "1e3" } }, "bad value"); + try x.h.expectFail(.{ .write = .{ .fid = 1, .offset = 0, .data = "99999999999999999999" } }, "bad value"); + try testing.expectEqual(@as(u32, 1), x.exposed.a); + _ = try x.h.ok(.{ .write = .{ .fid = 1, .offset = std.math.maxInt(u64), .data = "77\n" } }); + try testing.expectEqual(@as(u32, 77), x.exposed.a); + // reads of a write-only fid are refused; OEXEC reads like OREAD + try x.h.expectFail(.{ .read = .{ .fid = 1, .offset = 0, .count = 10 } }, "file not open"); + try x.h.walkTo(2, &.{"README"}); + _ = try x.h.ok(.{ .open = .{ .fid = 2, .mode = cloud9.oexec } }); + try testing.expect((try x.h.ok(.{ .read = .{ .fid = 2, .offset = 0, .count = 10 } })).read.len == 10); +} + +test "msize 24: every request that fits is answered, every reply that cannot fit is an Rerror" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + try x.h.version(24); + _ = try x.h.ok(.{ .attach = .{ .fid = 0, .uname = "u" } }); // Tattach 20, Rattach 20 + try x.h.expectFail(.{ .stat = .{ .fid = 0 } }, "reply too large"); // Rerror truncated to fit 24 bytes + const w = try x.h.ok(.{ .walk = .{ .fid = 0, .newfid = 1, .names = &.{"ctl"} } }); // Rwalk 22 + try testing.expectEqual(@as(u16, 1), w.walk.nwqid); + try x.h.expectFail(.{ .walk = .{ .fid = 0, .newfid = 2, .names = &.{ ".", "." } } }, "reply too large"); + try x.h.expectFail(.{ .clunk = .{ .fid = 2 } }, "unknown fid"); + _ = try x.h.ok(.{ .open = .{ .fid = 1, .mode = cloud9.ordwr } }); // Ropen 24 + try x.h.expectFail(.{ .write = .{ .fid = 1, .offset = 0, .data = "e" } }, "bad command"); // Twrite 24 + // the largest read the client may ask for is msize - iohdrsz = 0 bytes + const r = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = 0 } }); + try testing.expectEqual(@as(usize, 0), r.read.len); + _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); + try x.h.walkTo(3, &.{"build"}); + _ = try x.h.ok(.{ .open = .{ .fid = 3, .mode = cloud9.oread } }); + const d = try x.h.ok(.{ .read = .{ .fid = 3, .offset = 0, .count = 0 } }); + try testing.expectEqual(@as(usize, 0), d.read.len); // no record fits in 0 bytes, nothing is split + try x.h.expectFail(.{ .read = .{ .fid = 3, .offset = 1, .count = 0 } }, "bad offset"); +} + +test "Conn.init clamps the msize cap to [msize_min, cfg.msize]" { + var ctx: TestCtx = .{}; + var shared: TS.Shared = .init(&ctx); + var storage: TS.Storage = undefined; + const lo: TS.Conn = .init(&shared, &storage, 0); + try testing.expectEqual(cloud9.Server.msize_min, lo.msize_cap); + const hi: TS.Conn = .init(&shared, &storage, std.math.maxInt(u32)); + try testing.expectEqual(test_cfg.msize, hi.msize_cap); + const mid: TS.Conn = .init(&shared, &storage, 4096); + try testing.expectEqual(@as(u32, 4096), mid.msize_cap); +} + +test "parseIso8601 rejects malformed stamps and never traps" { + try testing.expectEqual(@as(?u32, null), parseIso8601("")); + try testing.expectEqual(@as(?u32, null), parseIso8601("2023-11-14T22:13:20")); + try testing.expectEqual(@as(?u32, null), parseIso8601("2023-13-14T22:13:20Z")); + try testing.expectEqual(@as(?u32, null), parseIso8601("2023-11-32T22:13:20Z")); + try testing.expectEqual(@as(?u32, null), parseIso8601("2023-11-14T24:13:20Z")); + try testing.expectEqual(@as(?u32, null), parseIso8601("2023-11-14T22:60:20Z")); + try testing.expectEqual(@as(?u32, null), parseIso8601("1969-12-31T23:59:59Z")); + try testing.expectEqual(@as(?u32, null), parseIso8601("9999-12-31T23:59:59Z")); + try testing.expectEqual(@as(?u32, null), parseIso8601("20x3-11-14T22:13:20Z")); + try testing.expectEqual(@as(?u32, null), parseIso8601("0000-01-01T00:00:00Z")); + try testing.expectEqual(@as(u32, 1_700_000_000), parseIso8601("2023-11-14T22:13:20Z").?); + try testing.expectEqual(@as(u32, 951_782_400), parseIso8601("2000-02-29T00:00:00Z").?); + try testing.expectEqual(@as(u32, 4_102_444_799), parseIso8601("2099-12-31T23:59:59Z").?); + try testing.expectEqual(@as(u32, std.math.maxInt(u32)), parseIso8601("2106-02-07T06:28:15Z").?); + try testing.expectEqual(@as(?u32, null), parseIso8601("2106-02-07T06:28:16Z")); +} + +/// Multiplicative inverse of an odd 32-bit constant (Newton iteration). +fn inverseMod32(a: u32) u32 { + var x: u32 = a; + for (0..5) |_| x *%= 2 -% a *% x; + return x; +} + +test "fid index: fid numbers crafted to collide under the public hash do not cluster a seeded connection" { + var ctx: TestCtx = .{}; + var shared: BigS.Shared = .init(&ctx); + const storage = try testing.allocator.create(BigS.Storage); + defer testing.allocator.destroy(storage); + var h: BigS.Harness = undefined; + try h.init(&shared, storage); + defer h.deinit(); + // two connections on the same Shared never share a seed + const other: BigS.Conn = .init(&shared, storage, big_cfg.msize); + try testing.expect(other.hash_seed != h.conn.hash_seed); + // ids whose products with the golden ratio share their top bits: all one bucket when unseeded + const inv = inverseMod32(0x9E37_79B1); + try testing.expectEqual(@as(u32, 1), inv *% 0x9E37_79B1); + const n: u32 = big_cfg.max_fids - 1; + const base: u32 = 0x4242_0000; + var i: u32 = 0; + while (i < n) : (i += 1) { + const id = (base + i) *% inv; + try testing.expectEqual(@as(usize, base >> BigS.index_shift), @as(usize, @intCast((id *% 0x9E37_79B1) >> BigS.index_shift))); + _ = try h.ok(.{ .walk = .{ .fid = 0, .newfid = id, .names = &.{} } }); + } + try checkFidIndex(&h.conn); + // the longest probe sequence in the seeded table is short; unseeded it would be ~n + var worst: usize = 0; + i = 0; + while (i < n) : (i += 1) { + const id = (base + i) *% inv; + var pos = h.conn.fidHome(id); + var steps: usize = 0; + while (h.conn.fids[h.conn.index[pos]].id != id) : (pos = (pos + 1) & BigS.index_mask) steps += 1; + worst = @max(worst, steps); + } + try testing.expect(worst < 64); +} diff --git a/9proc/src/freestanding_check.zig b/9proc/src/freestanding_check.zig new file mode 100644 index 0000000..2a0c12f --- /dev/null +++ b/9proc/src/freestanding_check.zig @@ -0,0 +1,71 @@ +//! A tiny freestanding root proving that `core` and `vars` compile without an +//! OS: `zig build 9proc-check-freestanding` builds this for riscv32-freestanding-none. +//! It instantiates `Server(cfg)` with static Storage/Shared, exposes one +//! variable, and runs one push/step over a canned Tversion frame. It must not +//! import scratch.zig (allocator) or anything OS-specific. +const std = @import("std"); +const core = @import("core.zig"); +const Writer = std.Io.Writer; + +const Build = struct { + pub const zig_version: []const u8 = @import("builtin").zig_version_string; + pub const target: []const u8 = "riscv32-freestanding-none"; + pub const optimize: []const u8 = "check"; + pub const time: []const u8 = "1970-01-01T00:00:00Z"; + pub const change: []const u8 = "none"; +}; + +const State = struct { ticks: u32, phase: enum { idle, busy }, inner: struct { x: f32 } }; + +const Fns = struct { + pub fn ticks(ctx: *anyopaque, w: *Writer) anyerror!void { + const s: *State = @ptrCast(@alignCast(ctx)); + try w.print("{d}", .{s.ticks}); + } +}; + +fn ctl(ctx: *anyopaque, cmd: []const u8, out: *Writer) anyerror!void { + const s: *State = @ptrCast(@alignCast(ctx)); + if (std.mem.startsWith(u8, cmd, "reset")) s.ticks = 0; + try out.writeAll("ok"); +} + +const cfg: core.Config = .{ + .name = "fw", + .build = Build, + .types = &.{ State, core.NodeStat }, + .decls_of = Fns, + .fns = Fns, + .ctl = &ctl, + .msize = 2048, + .max_fids = 16, + .snapshot_slots = 2, + .snapshot_bytes = 1024, +}; + +const S = core.Server(cfg); + +var state: State = .{ .ticks = 0, .phase = .idle, .inner = .{ .x = 0 } }; +var storage: S.Storage = undefined; +var shared: S.Shared = undefined; +var conn: S.Conn = undefined; + +/// Tversion msize=2048 version="9P2000". +const tversion = [_]u8{ 19, 0, 0, 0, 100, 0xFF, 0xFF, 0, 8, 0, 0, 6, 0, '9', 'P', '2', '0', '0', '0' }; + +/// Runs one Tversion through the engine; returns the number of reply bytes. +pub export fn proc9_check() u32 { + shared = .init(&state); + shared.expose("state", &state) catch unreachable; + conn = .init(&shared, &storage, cfg.msize); + _ = conn.push(&tversion); + _ = conn.step() catch return 0; + const out = conn.output(); + conn.wrote(out.len); + return @intCast(out.len); +} + +pub export fn _start() noreturn { + _ = proc9_check(); + while (true) {} +} diff --git a/9proc/src/linux/debug.zig b/9proc/src/linux/debug.zig new file mode 100644 index 0000000..a344380 --- /dev/null +++ b/9proc/src/linux/debug.zig @@ -0,0 +1,1458 @@ +//! Linux debug facilities for the 9proc server: threads, stacks, +//! registers, address → source, memory, breakpoints and panics. +//! +//! This file is a pure API; a later adapter turns it into a core `Provider`. +//! All text is written to a `*std.Io.Writer`. Nothing here allocates after +//! `init` except from the caller-provided `text_buf`, which is used as a fixed +//! arena for symbol text and reset before every query. +//! +//! Only one `Debug` may exist per process: the signal handlers and the panic +//! hook find their state through the global `current` pointer set by `init`. +//! +//! Mechanics +//! +//! * Capturing another thread's stack or registers: the calling (server) +//! thread sends `capture_signal` with `tgkill`. The SA_SIGINFO handler copies +//! the interrupted register state (`cpu_context.fromPosixSignalContext`) into +//! the single capture slot and parks on a futex. The server unwinds the +//! parked thread's stack from that context, releases the target, then +//! symbolizes. The handler is async-signal-safe: no allocation, no +//! `std.debug`, no locks other than the futex. A target that does not run +//! the handler within `capture_timeout_ns` (signal masked, thread in D +//! state, ...) yields `error.Timeout`; a late-arriving handler run cannot +//! corrupt a reused slot because it must match the requested tid and win a +//! compare-and-swap from `armed` on the slot state (that pair plays the role +//! of a generation counter: a stale run finds the slot idle, armed for +//! another tid, or armed for itself, in which case its capture is simply the +//! valid answer to the new request). +//! * Breakpoints: `@breakpoint()` raises SIGTRAP on the executing thread only. +//! The handler claims a pause slot, saves the context and parks on a futex +//! until `resumeThread`. On x86_64 the saved PC is already past `int3`; on +//! aarch64 the handler advances PC by 4 in the ucontext before returning +//! (only for a real `brk`, i.e. a kernel-generated si_code; a SIGTRAP sent +//! with kill/tgkill parks the thread where it was). With no free slot the +//! thread steps over the breakpoint and keeps running (`traps_skipped` +//! counts them): the debug layer never kills the process. The server thread +//! itself (`server_tid`) is never parked, a breakpoint there is stepped +//! over, because nobody could resume it. Only a stale handler run after +//! `deinit` (no `current`) falls back to the default disposition. +//! * Panics: `panicHook` records the message and a stack capture, then, if +//! `hold_on_panic` and a `Debug` exists, parks until `panicContinue`; then +//! `std.debug.defaultPanic` runs. A nested or second panic, or a panic on +//! the server thread itself (which could never be continued), goes +//! straight to the default handler. +//! * std.debug's `SelfInfo` guards its state with an `Io.RwLock`. A target +//! parked while holding it (a thread inside a stack-trace dump, say) would +//! deadlock the unwind, so after parking a thread the lock is probed with +//! `tryLock`; a held lock yields `error.Busy` and the target is released. +//! * Known-module guard: `std.debug.SelfInfo` (Zig 0.16) rebuilds its module +//! list whenever it is asked about an address outside every known module, +//! freeing the CIE lists its unwind cache still points into; later unwinds +//! then read freed memory. `init` records the PT_LOAD ranges of the +//! executable (the same source std uses) and every lookup or unwind is +//! first checked against them; addresses outside (unmapped, vDSO, ...) +//! render as "?" and are never handed to std. + +const std = @import("std"); +const builtin = @import("builtin"); +const linux = std.os.linux; +const cpu_context = std.debug.cpu_context; +const Writer = std.Io.Writer; +const Native = cpu_context.Native; +const arch = builtin.cpu.arch; + +pub const Options = struct { + /// Used for `std.debug` symbolization (reading debug info from disk). + io: std.Io, + /// Fixed arena for symbol text. A `FixedBufferAllocator` is placed over it + /// and reset before every query. 16 KiB is plenty; 4 KiB is a sane floor. + text_buf: []u8, + /// Real-time signal used to snapshot other threads. SIGRTMIN is 32 on + /// Linux without libc; the default is SIGRTMIN+3. + capture_signal: u8 = default_capture_signal, + /// How long to wait for a target thread to run the capture handler. + capture_timeout_ns: u64 = 250 * std.time.ns_per_ms, + /// How many threads may be parked in `@breakpoint()` at once (≤ 32). + max_paused: u8 = 16, +}; + +pub const default_capture_signal: u8 = 32 + 3; + +/// Hard upper bound of `Options.max_paused` (slot storage is static). +pub const max_paused_cap = 32; +/// Maximum number of frames written by any stack function. +pub const max_frames = 64; +/// Maximum number of tids enumerated from /proc/self/task. +pub const max_threads = 512; +/// Upper bound of the recorded panic message. +pub const panic_msg_cap = 1024; +/// Maximum number of PT_LOAD ranges recorded by the known-module guard. +pub const max_ranges = 64; + +/// Consulted by `panicHook`: when true and a `Debug` is initialized, the +/// panicking thread is held until `panicContinue`. +pub var hold_on_panic: bool = true; + +/// The one live instance, set by `init`, cleared by `deinit`. +pub var current: ?*Debug = null; + +/// The tid of the thread serving requests (0 = none). That thread is never +/// parked by a breakpoint or held by a panic, since nobody could release it. +pub var server_tid: std.atomic.Value(u32) = .init(0); + +/// Breakpoints stepped over because no pause slot was free, or because they +/// were hit on the server thread. +pub var traps_skipped: std.atomic.Value(u32) = .init(0); + +pub const Error = error{ + /// The target thread did not run the capture handler in time. + Timeout, + /// No thread with that tid exists in this process. + NoThread, + /// The address is not mapped (EFAULT from process_vm_readv/writev). + Unmapped, + /// The thread is not parked in a breakpoint. + NotPaused, + /// No panic has been recorded / is being held. + NoPanic, + /// Another `Debug` already exists in this process. + AlreadyInitialized, + /// The operation is not available on this architecture / kernel. + Unsupported, + /// The target thread is parked inside std.debug (holding its lock); its + /// stack cannot be unwound without deadlocking. Retry later. + Busy, + /// Invalid option value. + InvalidOptions, + /// A syscall or /proc read failed unexpectedly. + Unexpected, + /// The writer failed. + WriteFailed, +}; + +// Capture slot states. +const cap_idle: u32 = 0; +const cap_armed: u32 = 1; +const cap_capturing: u32 = 2; +const cap_captured: u32 = 3; +const cap_failed: u32 = 4; + +// Pause slot states. +const pause_free: u32 = 0; +const pause_claimed: u32 = 1; +const pause_paused: u32 = 2; +const pause_resuming: u32 = 3; + +const CaptureSlot = struct { + state: std.atomic.Value(u32) = .init(cap_idle), + target_tid: std.atomic.Value(u32) = .init(0), + ctx: Native = undefined, +}; + +const PauseSlot = struct { + state: std.atomic.Value(u32) = .init(pause_free), + tid: std.atomic.Value(u32) = .init(0), + ctx: Native = undefined, +}; + +pub const Debug = struct { + io: std.Io, + text_buf: []u8, + capture_signal: linux.SIG, + capture_timeout_ns: u64, + max_paused: u8, + + capture: CaptureSlot = .{}, + paused: [max_paused_cap]PauseSlot = [_]PauseSlot{.{}} ** max_paused_cap, + + old_capture_action: linux.Sigaction = undefined, + old_trap_action: linux.Sigaction = undefined, + breakpoints_enabled: bool = false, + + tids: [max_threads]u32 = undefined, + tid_count: usize = 0, + + ranges: [max_ranges]Range = undefined, + range_count: usize = 0, + + const Range = struct { start: usize, len: usize }; + + /// Installs the capture handler (not the SIGTRAP handler) and publishes + /// `d` as `current`. + pub fn init(d: *Debug, opts: Options) Error!void { + if (current != null) return error.AlreadyInitialized; + if (opts.capture_signal < 32 or opts.capture_signal >= linux.NSIG) return error.InvalidOptions; + if (opts.max_paused == 0 or opts.max_paused > max_paused_cap) return error.InvalidOptions; + if (Native == noreturn) return error.Unsupported; + d.* = .{ + .io = opts.io, + .text_buf = opts.text_buf, + .capture_signal = @enumFromInt(opts.capture_signal), + .capture_timeout_ns = opts.capture_timeout_ns, + .max_paused = opts.max_paused, + }; + d.scanModules(); + const act: linux.Sigaction = .{ + .handler = .{ .sigaction = captureHandler }, + .mask = linux.sigemptyset(), + .flags = linux.SA.SIGINFO | linux.SA.RESTART, + }; + current = d; + if (linux.errno(linux.sigaction(d.capture_signal, &act, &d.old_capture_action)) != .SUCCESS) { + current = null; + return error.Unexpected; + } + } + + /// Restores the signal dispositions and clears `current`. Threads parked + /// in a breakpoint are resumed first. + pub fn deinit(d: *Debug) void { + d.disableBreakpoints(); + _ = linux.sigaction(d.capture_signal, &d.old_capture_action, null); + if (current == d) current = null; + } + + /// Installs the SIGTRAP handler so that `@breakpoint()` parks the thread. + pub fn enableBreakpoints(d: *Debug) Error!void { + if (d.breakpoints_enabled) return; + if (arch != .x86_64 and !arch.isAARCH64()) return error.Unsupported; + const act: linux.Sigaction = .{ + .handler = .{ .sigaction = trapHandler }, + .mask = linux.sigemptyset(), + .flags = linux.SA.SIGINFO | linux.SA.RESTART, + }; + if (linux.errno(linux.sigaction(.TRAP, &act, &d.old_trap_action)) != .SUCCESS) return error.Unexpected; + d.breakpoints_enabled = true; + } + + /// Restores the previous SIGTRAP disposition and resumes every parked thread. + pub fn disableBreakpoints(d: *Debug) void { + if (!d.breakpoints_enabled) return; + _ = linux.sigaction(.TRAP, &d.old_trap_action, null); + d.breakpoints_enabled = false; + for (&d.paused) |*slot| { + if (slot.state.cmpxchgStrong(pause_paused, pause_resuming, .acq_rel, .acquire) == null) + futexWake(&slot.state); + } + } + + // ---------------------------------------------------------------- threads + + /// The nth tid of this process, numerically sorted; null past the end. + /// Index 0 rescans /proc/self/task; higher indices reuse that scan. + pub fn threadAt(d: *Debug, index: usize) ?u32 { + if (index == 0 or d.tid_count == 0) d.scanThreads(); + if (index >= d.tid_count) return null; + return d.tids[index]; + } + + pub fn threadExists(d: *Debug, tid: u32) bool { + _ = d; + var path_buf: [64]u8 = undefined; + const path = std.fmt.bufPrintZ(&path_buf, "/proc/self/task/{d}/comm", .{tid}) catch return false; + var buf: [32]u8 = undefined; + _ = readFile(path, &buf) catch return false; + return true; + } + + /// The thread's comm (without the trailing newline). + pub fn threadName(d: *Debug, tid: u32, w: *Writer) Error!void { + _ = d; + var path_buf: [64]u8 = undefined; + const path = std.fmt.bufPrintZ(&path_buf, "/proc/self/task/{d}/comm", .{tid}) catch return error.Unexpected; + var buf: [64]u8 = undefined; + const text = readFile(path, &buf) catch |err| switch (err) { + error.NotFound => return error.NoThread, + else => return error.Unexpected, + }; + w.writeAll(std.mem.trimEnd(u8, text, "\n")) catch return error.WriteFailed; + } + + /// A few fields of /proc/self/task//stat, one "name value" per line: + /// state, utime, stime, minflt, majflt, priority, nice, processor. + pub fn threadStat(d: *Debug, tid: u32, w: *Writer) Error!void { + _ = d; + var path_buf: [64]u8 = undefined; + const path = std.fmt.bufPrintZ(&path_buf, "/proc/self/task/{d}/stat", .{tid}) catch return error.Unexpected; + var buf: [1024]u8 = undefined; + const text = readFile(path, &buf) catch |err| switch (err) { + error.NotFound => return error.NoThread, + else => return error.Unexpected, + }; + // " () S "; comm may contain spaces and parens. + const close = std.mem.lastIndexOfScalar(u8, text, ')') orelse return error.Unexpected; + var it = std.mem.tokenizeScalar(u8, text[close + 1 ..], ' '); + // Field numbers below are 0-based from `state`. + const wanted = [_]struct { idx: usize, name: []const u8 }{ + .{ .idx = 0, .name = "state" }, + .{ .idx = 11, .name = "utime" }, + .{ .idx = 12, .name = "stime" }, + .{ .idx = 7, .name = "minflt" }, + .{ .idx = 9, .name = "majflt" }, + .{ .idx = 15, .name = "priority" }, + .{ .idx = 16, .name = "nice" }, + .{ .idx = 36, .name = "processor" }, + }; + var fields: [40][]const u8 = undefined; + var n: usize = 0; + while (it.next()) |f| : (n += 1) { + if (n == fields.len) break; + fields[n] = f; + } + for (wanted) |want| { + const value = if (want.idx < n) fields[want.idx] else "?"; + w.print("{s} {s}\n", .{ want.name, value }) catch return error.WriteFailed; + } + } + + /// "#n 0x in (::)" per frame. The calling + /// thread unwinds itself directly; any other thread is captured with the + /// capture signal. + pub fn threadStack(d: *Debug, tid: u32, w: *Writer) Error!void { + var addrs: [max_frames]usize = undefined; + var trace: std.debug.StackTrace = undefined; + if (tid == selfTid()) { + trace = std.debug.captureCurrentStackTrace(.{}, &addrs); + } else { + try d.captureThread(tid); + if (!d.selfInfoFree()) { + d.releaseCapture(); + return error.Busy; + } + trace = d.unwindContext(&d.capture.ctx, &addrs); + d.releaseCapture(); + } + try d.writeFrames(trace.return_addresses, w); + } + + /// " 0x" per general register, plus pc/sp/fp aliases. + pub fn threadRegs(d: *Debug, tid: u32, w: *Writer) Error!void { + if (tid == selfTid()) { + const ctx = Native.current(); + return writeRegs(&ctx, w); + } + try d.captureThread(tid); + const ctx = d.capture.ctx; + d.releaseCapture(); + return writeRegs(&ctx, w); + } + + // ------------------------------------------------------ addresses & memory + + /// "fn\nfile:line:col\nmodule\n", unknown parts as "?". + pub fn resolveAddr(d: *Debug, addr: usize, w: *Writer) Error!void { + if (!d.knownCode(addr)) return w.writeAll("?\n?\n?\n") catch error.WriteFailed; + var fba = std.heap.FixedBufferAllocator.init(d.text_buf); + const alloc = fba.allocator(); + const di = std.debug.getSelfDebugInfo() catch return error.Unsupported; + var sym = std.debug.Symbol.unknown; + var symbols: std.ArrayList(std.debug.Symbol) = .empty; + if (di.getSymbols(d.io, alloc, alloc, addr, true, &symbols)) { + if (symbols.items.len > 0) sym = symbols.items[0]; + } else |_| {} + w.print("{s}\n", .{sym.name orelse "?"}) catch return error.WriteFailed; + if (sym.source_location) |sl| { + w.print("{s}:{d}:{d}\n", .{ sl.file_name, sl.line, sl.column }) catch return error.WriteFailed; + } else { + w.writeAll("?\n") catch return error.WriteFailed; + } + const module = di.getModuleName(d.io, addr) catch "?"; + w.print("{s}\n", .{module}) catch return error.WriteFailed; + } + + /// Reads `buf.len` bytes at `addr` via process_vm_readv on the own + /// process. Never faults. Returns the number of bytes read (short when the + /// range crosses into an unmapped page); `error.Unmapped` when nothing + /// could be read. + pub fn readMem(d: *Debug, addr: usize, buf: []u8) Error!usize { + _ = d; + if (buf.len == 0) return 0; + // Page 0 is never mapped (mmap_min_addr) and a null `iovec.base` is a + // safety-checked cast; the same answer without the trap. + if (addr == 0) return error.Unmapped; + const local = [_]std.posix.iovec{.{ .base = buf.ptr, .len = buf.len }}; + const remote = [_]std.posix.iovec_const{.{ .base = @ptrFromInt(addr), .len = buf.len }}; + const rc = linux.process_vm_readv(linux.getpid(), &local, &remote, 0); + switch (linux.errno(rc)) { + .SUCCESS => return rc, + .FAULT => return error.Unmapped, + .NOSYS, .PERM => return error.Unsupported, + else => return error.Unexpected, + } + } + + /// Writes `data` at `addr` via process_vm_writev. Read-only mappings also + /// report `error.Unmapped` (the kernel says EFAULT for both). + pub fn writeMem(d: *Debug, addr: usize, data: []const u8) Error!usize { + _ = d; + if (data.len == 0) return 0; + if (addr == 0) return error.Unmapped; + const local = [_]std.posix.iovec_const{.{ .base = data.ptr, .len = data.len }}; + const remote = [_]std.posix.iovec_const{.{ .base = @ptrFromInt(addr), .len = data.len }}; + const rc = linux.process_vm_writev(linux.getpid(), &local, &remote, 0); + switch (linux.errno(rc)) { + .SUCCESS => return rc, + .FAULT => return error.Unmapped, + .NOSYS, .PERM => return error.Unsupported, + else => return error.Unexpected, + } + } + + /// Hexdump of `len` bytes at `addr` in the shape of `std.debug.dumpHex` + /// (16 bytes per line, address column, bytes in two groups, ASCII column). + /// Stops early at the first unmapped byte; `error.Unmapped` only when the + /// very first chunk is unreadable. + pub fn hexdump(d: *Debug, addr: usize, len: usize, w: *Writer) Error!void { + var chunk: [256]u8 = undefined; + var done: usize = 0; + while (done < len) { + const want = @min(chunk.len, len - done); + const got = d.readMem(addr +% done, chunk[0..want]) catch |err| switch (err) { + error.Unmapped => if (done == 0) return error.Unmapped else break, + else => return err, + }; + if (got == 0) break; + try writeHexLines(addr +% done, chunk[0..got], w); + done += got; + if (got < want) break; + } + } + + /// Copies /proc/self/maps to `w`. + pub fn maps(d: *Debug, w: *Writer) Error!void { + _ = d; + return streamFile("/proc/self/maps", w); + } + + /// Reads `buf.len` bytes of /proc/self/maps at `offset` (0 at the end). + /// Not a consistent snapshot across reads; a map appearing between two + /// reads shifts the text, like `cat` on /proc itself. + pub fn readMaps(d: *Debug, offset: u64, buf: []u8) Error!usize { + _ = d; + if (offset > std.math.maxInt(i64)) return 0; + return preadFile("/proc/self/maps", offset, buf); + } + + // ------------------------------------------------------------ breakpoints + + /// The nth tid currently parked in `@breakpoint()`. + pub fn pausedAt(d: *Debug, index: usize) ?u32 { + var n: usize = 0; + for (d.paused[0..d.max_paused]) |*slot| { + if (slot.state.load(.acquire) != pause_paused) continue; + if (n == index) return slot.tid.load(.acquire); + n += 1; + } + return null; + } + + pub fn isPaused(d: *Debug, tid: u32) bool { + return d.pausedSlot(tid) != null; + } + + pub fn pausedStack(d: *Debug, tid: u32, w: *Writer) Error!void { + const slot = d.pausedSlot(tid) orelse return error.NotPaused; + if (!d.selfInfoFree()) return error.Busy; + var addrs: [max_frames]usize = undefined; + const trace = d.unwindContext(&slot.ctx, &addrs); + try d.writeFrames(trace.return_addresses, w); + } + + pub fn pausedRegs(d: *Debug, tid: u32, w: *Writer) Error!void { + const slot = d.pausedSlot(tid) orelse return error.NotPaused; + return writeRegs(&slot.ctx, w); + } + + /// Lets a parked thread continue past its breakpoint. + pub fn resumeThread(d: *Debug, tid: u32) Error!void { + const slot = d.pausedSlot(tid) orelse return error.NotPaused; + if (slot.state.cmpxchgStrong(pause_paused, pause_resuming, .acq_rel, .acquire) != null) return error.NotPaused; + futexWake(&slot.state); + } + + fn pausedSlot(d: *Debug, tid: u32) ?*PauseSlot { + for (d.paused[0..d.max_paused]) |*slot| { + if (slot.state.load(.acquire) == pause_paused and slot.tid.load(.acquire) == tid) return slot; + } + return null; + } + + // ------------------------------------------------------------------ panic + + /// The recorded panic message; nothing before any panic. + pub fn panicMessage(d: *Debug, w: *Writer) Error!void { + _ = d; + if (panic_state.load(.acquire) == panic_none) return; + w.writeAll(panic_msg[0..panic_msg_len]) catch return error.WriteFailed; + } + + /// Frames of the panicking thread, symbolized lazily. + pub fn panicStack(d: *Debug, w: *Writer) Error!void { + if (panic_state.load(.acquire) == panic_none) return; + try d.writeFrames(panic_addrs[0..panic_addr_count], w); + } + + /// True while a panicking thread is parked waiting for `panicContinue`. + pub fn panicHeld(d: *Debug) bool { + _ = d; + return panic_state.load(.acquire) == panic_held; + } + + /// Releases the held panicking thread into `std.debug.defaultPanic`. + pub fn panicContinue(d: *Debug) Error!void { + _ = d; + if (panic_state.cmpxchgStrong(panic_held, panic_continued, .acq_rel, .acquire) != null) return error.NoPanic; + futexWake(&panic_state); + } + + // -------------------------------------------------------------- internals + + fn scanThreads(d: *Debug) void { + d.tid_count = 0; + const fd_rc = linux.open("/proc/self/task", .{ .ACCMODE = .RDONLY, .DIRECTORY = true, .CLOEXEC = true }, 0); + if (linux.errno(fd_rc) != .SUCCESS) return; + const fd: i32 = @intCast(fd_rc); + defer _ = linux.close(fd); + var buf: [4096]u8 align(@alignOf(linux.dirent64)) = undefined; + while (true) { + const rc = linux.getdents64(fd, &buf, buf.len); + if (linux.errno(rc) != .SUCCESS or rc == 0) break; + var off: usize = 0; + while (off < rc) { + const ent: *align(1) const linux.dirent64 = @ptrCast(&buf[off]); + const name_ptr: [*:0]const u8 = @ptrCast(&buf[off + @offsetOf(linux.dirent64, "name")]); + const name = std.mem.span(name_ptr); + if (std.fmt.parseInt(u32, name, 10)) |tid| { + if (d.tid_count < max_threads) { + d.tids[d.tid_count] = tid; + d.tid_count += 1; + } + } else |_| {} + off += ent.reclen; + } + } + std.mem.sort(u32, d.tids[0..d.tid_count], {}, std.sort.asc(u32)); + } + + /// Arms the capture slot for `tid`, signals it and waits until the handler + /// has parked with its context copied. On success the caller owns the + /// slot until `releaseCapture`. + fn captureThread(d: *Debug, tid: u32) Error!void { + const slot = &d.capture; + slot.target_tid.store(tid, .release); + slot.state.store(cap_armed, .release); + const rc = linux.tgkill(linux.getpid(), @intCast(tid), d.capture_signal); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .SRCH => { + slot.state.store(cap_idle, .release); + return error.NoThread; + }, + else => { + slot.state.store(cap_idle, .release); + return error.Unexpected; + }, + } + const deadline = monotonicNs() + d.capture_timeout_ns; + while (true) { + const s = slot.state.load(.acquire); + switch (s) { + cap_captured => return, + cap_failed => { + slot.state.store(cap_idle, .release); + return error.Unsupported; + }, + cap_armed => { + const now = monotonicNs(); + if (now >= deadline) { + // Disarm; if the handler raced us it has moved on to + // `capturing` and we simply keep waiting for it. + if (slot.state.cmpxchgStrong(cap_armed, cap_idle, .acq_rel, .acquire) == null) return error.Timeout; + continue; + } + futexWaitNs(&slot.state, cap_armed, deadline - now); + }, + // The handler is copying registers; it finishes promptly. + cap_capturing => futexWaitNs(&slot.state, cap_capturing, 1 * std.time.ns_per_ms), + else => unreachable, + } + } + } + + /// Records the PT_LOAD ranges of every module `dl_iterate_phdr` reports + /// (for a static executable: the executable itself, not the vDSO). + fn scanModules(d: *Debug) void { + d.range_count = 0; + std.posix.dl_iterate_phdr(d, error{}, struct { + fn cb(info: *std.posix.dl_phdr_info, _: usize, ctx: *Debug) error{}!void { + for (info.phdr[0..info.phnum]) |phdr| { + if (phdr.type != .LOAD) continue; + if (ctx.range_count == max_ranges) return; + ctx.ranges[ctx.range_count] = .{ .start = info.addr +% phdr.vaddr, .len = phdr.memsz }; + ctx.range_count += 1; + } + } + }.cb) catch {}; + } + + /// True when `addr` lies in a module `std.debug` already knows about, so + /// that asking it about `addr` cannot trigger a module rescan. + fn knownCode(d: *const Debug, addr: usize) bool { + for (d.ranges[0..d.range_count]) |r| { + if (addr >= r.start and addr - r.start < r.len) return true; + } + return false; + } + + /// Unwinds from a saved context. A pc outside every known module (e.g. a + /// thread inside the vDSO) is reported as a single frame and not unwound, + /// because std would otherwise rescan its module list (see the header). + fn unwindContext(d: *const Debug, ctx: *const Native, addrs: *[max_frames]usize) std.debug.StackTrace { + if (!d.knownCode(ctx.getPc())) { + addrs[0] = ctx.getPc() +| 1; + return .{ .return_addresses = addrs[0..1], .skipped = .unknown }; + } + return std.debug.captureCurrentStackTrace(.{ .context = ctx }, addrs); + } + + /// True when nobody holds std.debug's `SelfInfo` lock right now. Called + /// with the target parked, so a held lock means the *target* (or another + /// live thread, which will let go) holds it; only the former deadlocks, + /// and the caller cannot tell them apart, so both yield `error.Busy`. + fn selfInfoFree(d: *const Debug) bool { + if (comptime !@hasField(std.debug.SelfInfo, "rwlock")) return true; + const di = std.debug.getSelfDebugInfo() catch return true; + if (!di.rwlock.tryLock(d.io)) return false; + di.rwlock.unlock(d.io); + return true; + } + + fn releaseCapture(d: *Debug) void { + d.capture.state.store(cap_idle, .release); + futexWake(&d.capture.state); + } + + fn writeFrames(d: *Debug, addrs: []const usize, w: *Writer) Error!void { + var fba = std.heap.FixedBufferAllocator.init(d.text_buf); + const alloc = fba.allocator(); + const di = std.debug.getSelfDebugInfo() catch return error.Unsupported; + for (addrs, 0..) |ret_addr, i| { + // Return addresses point after the call; the first frame of a + // context capture is stored as pc+1 by std for the same reason. + const addr = ret_addr -| 1; + fba.reset(); + var symbols: std.ArrayList(std.debug.Symbol) = .empty; + var sym = std.debug.Symbol.unknown; + if (d.knownCode(addr)) { + if (di.getSymbols(d.io, alloc, alloc, addr, true, &symbols)) { + if (symbols.items.len > 0) sym = symbols.items[0]; + } else |_| {} + } + w.print("#{d} 0x{x} in {s} (", .{ i, addr, sym.name orelse "?" }) catch return error.WriteFailed; + if (sym.source_location) |sl| { + w.print("{s}:{d}:{d})\n", .{ sl.file_name, sl.line, sl.column }) catch return error.WriteFailed; + } else { + w.writeAll("?)\n") catch return error.WriteFailed; + } + } + } +}; + +// ------------------------------------------------------------------ handlers + +fn selfTid() u32 { + return @intCast(linux.gettid()); +} + +fn captureHandler(_: linux.SIG, _: *const linux.siginfo_t, ctx_ptr: ?*anyopaque) callconv(.c) void { + const d = current orelse return; + const slot = &d.capture; + const me = selfTid(); + if (slot.target_tid.load(.acquire) != me) return; + if (slot.state.cmpxchgStrong(cap_armed, cap_capturing, .acq_rel, .acquire) != null) return; + // The tid check and the swap are not one atomic step: a stale run (a + // signal that stayed pending while its request timed out) may have read + // the old tid and then won the swap of a request re-armed for another + // thread. `target_tid` is fixed while the slot is armed, so re-checking + // after the swap closes the window; hand the slot back untouched. + if (slot.target_tid.load(.acquire) != me) { + slot.state.store(cap_armed, .release); + futexWake(&slot.state); + return; + } + if (cpu_context.fromPosixSignalContext(ctx_ptr)) |ctx| { + slot.ctx = ctx; + slot.state.store(cap_captured, .release); + futexWake(&slot.state); + while (slot.state.load(.acquire) == cap_captured) futexWaitNs(&slot.state, cap_captured, null); + } else { + slot.state.store(cap_failed, .release); + futexWake(&slot.state); + } +} + +/// aarch64 Linux ucontext_t, only as far as `mcontext.pc` (see +/// std.debug.cpu_context's signal_ucontext_t). +const UcontextAarch64 = extern struct { + flags: usize, + link: ?*UcontextAarch64, + stack: linux.stack_t, + sigmask: linux.sigset_t, + unused: [120]u8, + mcontext: extern struct { + fault_address: u64 align(16), + x: [30]u64, + lr: u64, + sp: u64, + pc: u64, + }, +}; + +fn trapHandler(_: linux.SIG, info: *const linux.siginfo_t, ctx_ptr: ?*anyopaque) callconv(.c) void { + const d = current orelse return trapFallback(); + const ctx = cpu_context.fromPosixSignalContext(ctx_ptr) orelse return trapFallback(); + // si_code > 0 is kernel-generated (TRAP_BRKPT for int3/brk); <= 0 is + // kill/tgkill/sigqueue from user space, where PC points at the + // interrupted instruction and must not be touched. + const from_instruction = info.code > 0; + if (comptime arch.isAARCH64()) { + // `brk #imm` does not advance PC; step over it so returning from the + // handler does not re-trap. + if (from_instruction) { + const uc: *UcontextAarch64 = @ptrCast(@alignCast(ctx_ptr.?)); + uc.mcontext.pc += 4; + } + } else if (comptime arch != .x86_64) { + return trapFallback(); + } + const tid = selfTid(); + if (tid == server_tid.load(.acquire)) { + // Nobody could resume the thread that serves /breakpoints: step over. + _ = traps_skipped.fetchAdd(1, .acq_rel); + return; + } + const slot: *PauseSlot = for (d.paused[0..d.max_paused]) |*slot| { + if (slot.state.cmpxchgStrong(pause_free, pause_claimed, .acq_rel, .acquire) == null) break slot; + } else { + _ = traps_skipped.fetchAdd(1, .acq_rel); + return; + }; + slot.ctx = ctx; + slot.tid.store(tid, .release); + slot.state.store(pause_paused, .release); + while (slot.state.load(.acquire) == pause_paused) futexWaitNs(&slot.state, pause_paused, null); + slot.state.store(pause_free, .release); +} + +/// Restores the default SIGTRAP disposition and re-raises it: the signal is +/// blocked while the handler runs, so it is delivered (fatally) on return. +/// Only for a handler run with no `Debug` (a trap in flight during `deinit`) +/// or on an architecture whose context cannot be read. +fn trapFallback() void { + const act: linux.Sigaction = .{ + .handler = .{ .handler = linux.SIG.DFL }, + .mask = linux.sigemptyset(), + .flags = 0, + }; + _ = linux.sigaction(.TRAP, &act, null); + _ = linux.tkill(linux.gettid(), .TRAP); +} + +// --------------------------------------------------------------------- panic + +const panic_none: u32 = 0; +const panic_recording: u32 = 1; +const panic_recorded: u32 = 2; +const panic_held: u32 = 3; +const panic_continued: u32 = 4; + +var panic_state: std.atomic.Value(u32) = .init(panic_none); +var panic_msg: [panic_msg_cap]u8 = undefined; +var panic_msg_len: usize = 0; +var panic_addrs: [max_frames]usize = undefined; +var panic_addr_count: usize = 0; +/// The tid of the panicking thread (0 before any panic). +pub var panic_tid: u32 = 0; + +/// Records the first panic: message (bounded copy) and stack addresses. +/// Returns false if a panic was already recorded (nested or second panic). +pub fn recordPanic(msg: []const u8, first_trace_addr: ?usize) bool { + if (panic_state.cmpxchgStrong(panic_none, panic_recording, .acq_rel, .acquire) != null) return false; + panic_tid = selfTid(); + panic_msg_len = @min(msg.len, panic_msg.len); + @memcpy(panic_msg[0..panic_msg_len], msg[0..panic_msg_len]); + const trace = std.debug.captureCurrentStackTrace(.{ .first_address = first_trace_addr }, &panic_addrs); + panic_addr_count = trace.return_addresses.len; + panic_state.store(panic_recorded, .release); + return true; +} + +/// Parks the panicking thread until `Debug.panicContinue` when holding is +/// enabled and a `Debug` exists; then hands over to `std.debug.defaultPanic`. +pub fn panicHook(msg: []const u8, first_trace_addr: ?usize) noreturn { + @branchHint(.cold); + if (recordPanic(msg, first_trace_addr)) { + // The server thread cannot be held: it is the one that would have to + // serve /panic/ctl. + if (hold_on_panic and current != null and panic_tid != server_tid.load(.acquire)) { + if (panic_state.cmpxchgStrong(panic_recorded, panic_held, .acq_rel, .acquire) == null) { + while (panic_state.load(.acquire) == panic_held) futexWaitNs(&panic_state, panic_held, null); + } + } + } + std.debug.defaultPanic(msg, first_trace_addr); +} + +/// Clears the recorded panic. Only meaningful in tests of the record path. +pub fn resetPanicRecord() void { + panic_msg_len = 0; + panic_addr_count = 0; + panic_tid = 0; + panic_state.store(panic_none, .release); +} + +// ------------------------------------------------------------------- helpers + +fn futexWake(word: *std.atomic.Value(u32)) void { + _ = linux.futex_3arg(&word.raw, .{ .cmd = .WAKE, .private = true }, std.math.maxInt(u32)); +} + +/// Waits while `*word == expect`, at most `timeout_ns` (forever when null). +/// Returns on wake, timeout, value change or EINTR; callers loop. +fn futexWaitNs(word: *std.atomic.Value(u32), expect: u32, timeout_ns: ?u64) void { + var ts: linux.timespec = undefined; + const ts_ptr: ?*const linux.timespec = if (timeout_ns) |ns| blk: { + ts = .{ .sec = @intCast(ns / std.time.ns_per_s), .nsec = @intCast(ns % std.time.ns_per_s) }; + break :blk &ts; + } else null; + _ = linux.futex_4arg(&word.raw, .{ .cmd = .WAIT, .private = true }, expect, ts_ptr); +} + +fn monotonicNs() u64 { + var ts: linux.timespec = undefined; + _ = linux.clock_gettime(.MONOTONIC, &ts); + return @as(u64, @intCast(ts.sec)) * std.time.ns_per_s + @as(u64, @intCast(ts.nsec)); +} + +const FileError = error{ NotFound, Unexpected, TooBig }; + +/// Reads a whole (small) file with raw syscalls. +fn readFile(path: [*:0]const u8, buf: []u8) FileError![]u8 { + const fd_rc = linux.open(path, .{ .ACCMODE = .RDONLY, .CLOEXEC = true }, 0); + switch (linux.errno(fd_rc)) { + .SUCCESS => {}, + .NOENT, .SRCH => return error.NotFound, + else => return error.Unexpected, + } + const fd: i32 = @intCast(fd_rc); + defer _ = linux.close(fd); + var len: usize = 0; + while (len < buf.len) { + const rc = linux.read(fd, buf[len..].ptr, buf.len - len); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .INTR => continue, + .SRCH, .NOENT => return error.NotFound, + else => return error.Unexpected, + } + if (rc == 0) return buf[0..len]; + len += rc; + } + return error.TooBig; +} + +/// One pread of `buf.len` bytes at `offset`; 0 at the end of the file. +fn preadFile(path: [*:0]const u8, offset: u64, buf: []u8) Error!usize { + const fd_rc = linux.open(path, .{ .ACCMODE = .RDONLY, .CLOEXEC = true }, 0); + if (linux.errno(fd_rc) != .SUCCESS) return error.Unexpected; + const fd: i32 = @intCast(fd_rc); + defer _ = linux.close(fd); + var len: usize = 0; + while (len < buf.len) { + const rc = linux.pread(fd, buf[len..].ptr, buf.len - len, @intCast(offset + len)); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .INTR => continue, + else => return error.Unexpected, + } + if (rc == 0) break; + len += rc; + } + return len; +} + +/// Streams a file of any size to `w`. +fn streamFile(path: [*:0]const u8, w: *Writer) Error!void { + const fd_rc = linux.open(path, .{ .ACCMODE = .RDONLY, .CLOEXEC = true }, 0); + if (linux.errno(fd_rc) != .SUCCESS) return error.Unexpected; + const fd: i32 = @intCast(fd_rc); + defer _ = linux.close(fd); + var buf: [4096]u8 = undefined; + while (true) { + const rc = linux.read(fd, &buf, buf.len); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .INTR => continue, + else => return error.Unexpected, + } + if (rc == 0) return; + w.writeAll(buf[0..rc]) catch return error.WriteFailed; + } +} + +fn writeHexLines(base: usize, bytes: []const u8, w: *Writer) Error!void { + var offset: usize = 0; + while (offset < bytes.len) : (offset += 16) { + const line = bytes[offset..@min(offset + 16, bytes.len)]; + w.print("{x:0>[1]} ", .{ base +% offset, @sizeOf(usize) * 2 }) catch return error.WriteFailed; + for (line, 0..) |byte, i| { + w.print("{X:0>2} ", .{byte}) catch return error.WriteFailed; + if (i == 7) w.writeByte(' ') catch return error.WriteFailed; + } + w.writeByte(' ') catch return error.WriteFailed; + if (line.len < 16) { + var missing = (16 - line.len) * 3; + if (line.len < 8) missing += 1; + w.splatByteAll(' ', missing) catch return error.WriteFailed; + } + for (line) |byte| { + w.writeByte(if (std.ascii.isPrint(byte)) byte else '.') catch return error.WriteFailed; + } + w.writeByte('\n') catch return error.WriteFailed; + } +} + +fn writeRegs(ctx: *const Native, w: *Writer) Error!void { + if (comptime arch == .x86_64) { + inline for (@typeInfo(Native.Gpr).@"enum".fields) |f| { + w.print("{s} 0x{x}\n", .{ f.name, ctx.gprs.get(@field(Native.Gpr, f.name)) }) catch return error.WriteFailed; + } + w.print("pc 0x{x}\nsp 0x{x}\nfp 0x{x}\n", .{ + ctx.gprs.get(.rip), ctx.gprs.get(.rsp), ctx.gprs.get(.rbp), + }) catch return error.WriteFailed; + } else if (comptime arch.isAARCH64()) { + for (ctx.x, 0..) |x, i| w.print("x{d} 0x{x}\n", .{ i, x }) catch return error.WriteFailed; + w.print("sp 0x{x}\npc 0x{x}\nfp 0x{x}\nlr 0x{x}\n", .{ + ctx.sp, ctx.pc, ctx.x[29], ctx.x[30], + }) catch return error.WriteFailed; + } else { + w.print("pc 0x{x}\nfp 0x{x}\n", .{ ctx.getPc(), ctx.getFp() }) catch return error.WriteFailed; + } +} + +// --------------------------------------------------------------------- tests + +const testing = std.testing; + +fn testOptions(text_buf: []u8) Options { + return .{ .io = testing.io, .text_buf = text_buf }; +} + +noinline fn sleepMs(ms: u64) void { + var ts: linux.timespec = .{ .sec = @intCast(ms / 1000), .nsec = @intCast((ms % 1000) * std.time.ns_per_ms) }; + _ = linux.nanosleep(&ts, null); +} + +// The test threads use atomic builtins rather than `std.atomic.Value` methods +// so that, in release modes, their pc is never inside an inlined callee: the +// DWARF symbolizer names the innermost inlined function at an address (see +// the notes on `writeFrames`). +const SpinState = struct { + tid: std.atomic.Value(u32) = .init(0), + stop: bool = false, + counter: u32 = 0, + done: bool = false, +}; + +noinline fn spinHere(st: *SpinState) void { + while (!@atomicLoad(bool, &st.stop, .acquire)) { + _ = @atomicRmw(u32, &st.counter, .Add, 1, .monotonic); + } +} + +fn spinThreadMain(st: *SpinState) void { + st.tid.store(selfTid(), .release); + spinHere(st); + @atomicStore(bool, &st.done, true, .release); // keeps the call above from becoming a tail call +} + +fn waitForTid(st: *SpinState) u32 { + var tries: usize = 0; + while (st.tid.load(.acquire) == 0) : (tries += 1) { + if (tries > 2000) return 0; + sleepMs(1); + } + return st.tid.load(.acquire); +} + +test "capture own stack" { + var text_buf: [16 * 1024]u8 = undefined; + var d: Debug = undefined; + try d.init(testOptions(&text_buf)); + defer d.deinit(); + try testing.expect(current == &d); + + var out: Writer.Allocating = .init(testing.allocator); + defer out.deinit(); + try d.threadStack(selfTid(), &out.writer); + const text = out.written(); + try testing.expect(std.mem.indexOf(u8, text, "#0 0x") != null); + try testing.expect(std.mem.indexOf(u8, text, "debug.zig:") != null); + try testing.expect(std.mem.indexOf(u8, text, "test.capture own stack") != null); + + out.clearRetainingCapacity(); + try d.threadRegs(selfTid(), &out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "pc 0x") != null); + try testing.expect(std.mem.indexOf(u8, out.written(), "pc 0x0\n") == null); +} + +test "capture another thread: stack, regs, name, stat" { + var text_buf: [16 * 1024]u8 = undefined; + var d: Debug = undefined; + try d.init(testOptions(&text_buf)); + defer d.deinit(); + + var st: SpinState = .{}; + const th = try std.Thread.spawn(.{}, spinThreadMain, .{&st}); + const tid = waitForTid(&st); + try testing.expect(tid != 0); + + var out: Writer.Allocating = .init(testing.allocator); + defer out.deinit(); + try d.threadStack(tid, &out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "spinHere") != null); + try testing.expect(std.mem.indexOf(u8, out.written(), "spinThreadMain") != null); + + out.clearRetainingCapacity(); + try d.threadRegs(tid, &out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "pc 0x") != null); + try testing.expect(std.mem.indexOf(u8, out.written(), "pc 0x0\n") == null); + + out.clearRetainingCapacity(); + try d.threadName(tid, &out.writer); + try testing.expect(out.written().len > 0); + try testing.expect(std.mem.indexOfScalar(u8, out.written(), '\n') == null); + + out.clearRetainingCapacity(); + try d.threadStat(tid, &out.writer); + try testing.expect(std.mem.startsWith(u8, out.written(), "state ")); + try testing.expect(std.mem.indexOf(u8, out.written(), "\nutime ") != null); + + // Enumeration lists both threads and nothing bogus. + try testing.expect(d.threadExists(tid)); + try testing.expect(d.threadExists(selfTid())); + var found_self = false; + var found_other = false; + var i: usize = 0; + var prev: u32 = 0; + while (d.threadAt(i)) |t| : (i += 1) { + try testing.expect(t > prev); + prev = t; + if (t == tid) found_other = true; + if (t == selfTid()) found_self = true; + } + try testing.expect(found_self and found_other); + + // Repeated captures of the same thread keep working. + var k: usize = 0; + while (k < 5) : (k += 1) { + out.clearRetainingCapacity(); + try d.threadStack(tid, &out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "spinHere") != null); + } + const before = @atomicLoad(u32, &st.counter, .acquire); + sleepMs(2); + try testing.expect(@atomicLoad(u32, &st.counter, .acquire) != before); // the thread is running again + + @atomicStore(bool, &st.stop, true, .release); + th.join(); + try testing.expect(!d.threadExists(tid)); + try testing.expectError(error.NoThread, d.threadStack(tid, &out.writer)); + try testing.expectError(error.NoThread, d.threadName(tid, &out.writer)); +} + +/// The address of the call site in the caller, i.e. inside this file's test. +noinline fn callerAddress() usize { + return @returnAddress() - 1; +} + +test "resolveAddr names this file" { + var text_buf: [16 * 1024]u8 = undefined; + var d: Debug = undefined; + try d.init(testOptions(&text_buf)); + defer d.deinit(); + var out: Writer.Allocating = .init(testing.allocator); + defer out.deinit(); + try d.resolveAddr(callerAddress(), &out.writer); + const text = out.written(); + var lines = std.mem.splitScalar(u8, text, '\n'); + const fn_name = lines.next().?; + const loc = lines.next().?; + const module = lines.next().?; + try testing.expect(fn_name.len > 0 and !std.mem.eql(u8, fn_name, "?")); + try testing.expect(std.mem.indexOf(u8, loc, "debug.zig:") != null); + try testing.expect(module.len > 0); + + out.clearRetainingCapacity(); + try d.resolveAddr(8, &out.writer); + try testing.expectEqualStrings("?\n?\n?\n", out.written()); + + // Regression: an unmapped lookup must not poison std's unwind cache (see + // the header); unwinding afterwards still works. + out.clearRetainingCapacity(); + try d.threadStack(selfTid(), &out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "test.resolveAddr names this file") != null); +} + +test "readMem, writeMem, hexdump" { + var text_buf: [16 * 1024]u8 = undefined; + var d: Debug = undefined; + try d.init(testOptions(&text_buf)); + defer d.deinit(); + + var value: [8]u8 = .{ 1, 2, 3, 4, 5, 6, 7, 8 }; + var got: [8]u8 = undefined; + try testing.expectEqual(@as(usize, 8), try d.readMem(@intFromPtr(&value), &got)); + try testing.expectEqualSlices(u8, &value, &got); + try testing.expectError(error.Unmapped, d.readMem(8, &got)); + + const new = [_]u8{ 0xaa, 0xbb, 0xcc }; + try testing.expectEqual(@as(usize, 3), try d.writeMem(@intFromPtr(&value) + 2, &new)); + try testing.expectEqualSlices(u8, &.{ 1, 2, 0xaa, 0xbb, 0xcc, 6, 7, 8 }, &value); + try testing.expectError(error.Unmapped, d.writeMem(8, &new)); + + var bytes: [19]u8 = .{ 0x00, 0x11, 0x22, 0x33, 0x44, 0x55, 0x66, 0x77, 0x88, 0x99, 0xaa, 0xbb, 0xcc, 0xdd, 0xee, 0xff, 0x01, 0x12, 0x13 }; + var out: Writer.Allocating = .init(testing.allocator); + defer out.deinit(); + try d.hexdump(@intFromPtr(&bytes), bytes.len, &out.writer); + const expected = try std.fmt.allocPrint(testing.allocator, + \\{x:0>[2]} 00 11 22 33 44 55 66 77 88 99 AA BB CC DD EE FF .."3DUfw........ + \\{x:0>[2]} 01 12 13 ... + \\ + , .{ @intFromPtr(&bytes), @intFromPtr(&bytes) + 16, @sizeOf(usize) * 2 }); + defer testing.allocator.free(expected); + try testing.expectEqualStrings(expected, out.written()); + try testing.expectError(error.Unmapped, d.hexdump(8, 16, &out.writer)); + + // Address 0 (also reached by an offset that wraps) must be an error, not a + // safety-checked null pointer cast on the server thread. + try testing.expectError(error.Unmapped, d.readMem(0, &got)); + try testing.expectError(error.Unmapped, d.writeMem(0, &new)); + try testing.expectError(error.Unmapped, d.hexdump(0, 16, &out.writer)); + try testing.expectError(error.Unmapped, d.readMem(std.math.maxInt(usize) - 3, &got)); + try testing.expectError(error.Unmapped, d.hexdump(std.math.maxInt(usize) - 3, 16, &out.writer)); + + out.clearRetainingCapacity(); + try d.maps(&out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "[stack]") != null); + + // readMaps serves the file piecewise at any offset and ends with 0. + var piece: [4096]u8 = undefined; + var total: usize = 0; + while (true) { + const n = try d.readMaps(total, &piece); + if (n == 0) break; + total += n; + } + try testing.expect(total >= out.written().len / 2); + try testing.expectEqual(@as(usize, 0), try d.readMaps(std.math.maxInt(u64), &piece)); +} + +test "breakpoint on the server thread and past the slot table steps over; tgkill SIGTRAP parks" { + if (arch != .x86_64 and !arch.isAARCH64()) return error.SkipZigTest; + var text_buf: [16 * 1024]u8 = undefined; + var d: Debug = undefined; + var opts = testOptions(&text_buf); + opts.max_paused = 1; + try d.init(opts); + defer d.deinit(); + try d.enableBreakpoints(); + defer d.disableBreakpoints(); + var out: Writer.Allocating = .init(testing.allocator); + defer out.deinit(); + + // The "server" thread (this one, for the test) hits a breakpoint: it keeps running. + const skipped0 = traps_skipped.load(.acquire); + server_tid.store(selfTid(), .release); + defer server_tid.store(0, .release); + @breakpoint(); + try testing.expectEqual(skipped0 + 1, traps_skipped.load(.acquire)); + try testing.expect(!d.isPaused(selfTid())); + + // One slot: the first trapping thread parks, the second steps over. + var a: TrapState = .{}; + const ta = try std.Thread.spawn(.{}, trapThreadMain, .{&a}); + var tries: usize = 0; + while (a.tid.load(.acquire) == 0 or !d.isPaused(a.tid.load(.acquire))) : (tries += 1) { + try testing.expect(tries < 5000); + sleepMs(1); + } + var b: TrapState = .{}; + const tb = try std.Thread.spawn(.{}, trapThreadMain, .{&b}); + tb.join(); + try testing.expectEqual(@as(u32, 1), @atomicLoad(u32, &b.counter, .acquire)); + try testing.expectEqual(skipped0 + 2, traps_skipped.load(.acquire)); + try testing.expectEqual(@as(u32, 0), @atomicLoad(u32, &a.counter, .acquire)); + try d.resumeThread(a.tid.load(.acquire)); + ta.join(); + try testing.expectEqual(@as(u32, 1), @atomicLoad(u32, &a.counter, .acquire)); + + // A SIGTRAP sent with tgkill (not an int3/brk) parks the thread where it + // was; resuming it must not skip an instruction: the spinner keeps counting. + var st: SpinState = .{}; + const th = try std.Thread.spawn(.{}, spinThreadMain, .{&st}); + const tid = waitForTid(&st); + try testing.expect(tid != 0); + try testing.expectEqual(linux.E.SUCCESS, linux.errno(linux.tgkill(linux.getpid(), @intCast(tid), .TRAP))); + tries = 0; + while (!d.isPaused(tid)) : (tries += 1) { + try testing.expect(tries < 5000); + sleepMs(1); + } + const frozen = @atomicLoad(u32, &st.counter, .acquire); + sleepMs(5); + try testing.expectEqual(frozen, @atomicLoad(u32, &st.counter, .acquire)); + out.clearRetainingCapacity(); + try d.pausedStack(tid, &out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "spinHere") != null); + try d.resumeThread(tid); + sleepMs(5); + try testing.expect(@atomicLoad(u32, &st.counter, .acquire) != frozen); + @atomicStore(bool, &st.stop, true, .release); + th.join(); +} + +const LockState = struct { + tid: std.atomic.Value(u32) = .init(0), + release: std.atomic.Value(bool) = .init(false), + unlocked: std.atomic.Value(bool) = .init(false), + stop: std.atomic.Value(bool) = .init(false), + io: std.Io, +}; + +fn lockHolderMain(st: *LockState) void { + const di = std.debug.getSelfDebugInfo() catch return; + di.rwlock.lockUncancelable(st.io); + st.tid.store(selfTid(), .release); + while (!st.release.load(.acquire)) sleepMs(1); + di.rwlock.unlock(st.io); + st.unlocked.store(true, .release); + while (!st.stop.load(.acquire)) sleepMs(1); +} + +test "a target parked while holding std.debug's lock is Busy, not a deadlock" { + if (comptime !@hasField(std.debug.SelfInfo, "rwlock")) return error.SkipZigTest; + var text_buf: [16 * 1024]u8 = undefined; + var d: Debug = undefined; + try d.init(testOptions(&text_buf)); + defer d.deinit(); + var st: LockState = .{ .io = testing.io }; + const th = try std.Thread.spawn(.{}, lockHolderMain, .{&st}); + var tries: usize = 0; + while (st.tid.load(.acquire) == 0) : (tries += 1) { + try testing.expect(tries < 2000); + sleepMs(1); + } + const tid = st.tid.load(.acquire); + // No allocation while the holder has the lock: `testing.allocator` + // records a stack trace per allocation, which needs that same lock. + var buf: [16 * 1024]u8 = undefined; + var w: Writer = .fixed(&buf); + try testing.expectError(error.Busy, d.threadStack(tid, &w)); + try testing.expectEqual(cap_idle, d.capture.state.load(.acquire)); + // Registers need no unwind and are still available. + try d.threadRegs(tid, &w); + try testing.expect(std.mem.indexOf(u8, w.buffered(), "pc 0x") != null); + // Handshake, not a sleep: a slow holder would otherwise still hold the + // lock and the next capture would legitimately be Busy again. + st.release.store(true, .release); + tries = 0; + while (!st.unlocked.load(.acquire)) : (tries += 1) { + try testing.expect(tries < 5000); + sleepMs(1); + } + w = .fixed(&buf); + try d.threadStack(tid, &w); + try testing.expect(std.mem.indexOf(u8, w.buffered(), "lockHolderMain") != null); + st.stop.store(true, .release); + th.join(); +} + +const TrapState = struct { + tid: std.atomic.Value(u32) = .init(0), + counter: u32 = 0, +}; + +noinline fn trapThreadMain(st: *TrapState) void { + st.tid.store(selfTid(), .release); + @breakpoint(); + _ = @atomicRmw(u32, &st.counter, .Add, 1, .acq_rel); +} + +test "breakpoint: pause, inspect, resume" { + if (arch != .x86_64 and !arch.isAARCH64()) return error.SkipZigTest; + var text_buf: [16 * 1024]u8 = undefined; + var d: Debug = undefined; + try d.init(testOptions(&text_buf)); + defer d.deinit(); + try d.enableBreakpoints(); + + var st: TrapState = .{}; + const th = try std.Thread.spawn(.{}, trapThreadMain, .{&st}); + var tries: usize = 0; + while (st.tid.load(.acquire) == 0 or !d.isPaused(st.tid.load(.acquire))) : (tries += 1) { + try testing.expect(tries < 5000); + sleepMs(1); + } + const tid = st.tid.load(.acquire); + try testing.expectEqual(@as(?u32, tid), d.pausedAt(0)); + try testing.expectEqual(@as(?u32, null), d.pausedAt(1)); + try testing.expectEqual(@as(u32, 0), @atomicLoad(u32, &st.counter, .acquire)); + + var out: Writer.Allocating = .init(testing.allocator); + defer out.deinit(); + try d.pausedStack(tid, &out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "trapThreadMain") != null); + out.clearRetainingCapacity(); + try d.pausedRegs(tid, &out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "pc 0x") != null); + + // A paused thread can also be captured through the signal path. + out.clearRetainingCapacity(); + try d.threadStack(tid, &out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "#0 0x") != null); + + sleepMs(5); + try testing.expectEqual(@as(u32, 0), @atomicLoad(u32, &st.counter, .acquire)); + try d.resumeThread(tid); + th.join(); + try testing.expectEqual(@as(u32, 1), @atomicLoad(u32, &st.counter, .acquire)); + try testing.expect(!d.isPaused(tid)); + try testing.expectEqual(@as(?u32, null), d.pausedAt(0)); + try testing.expectError(error.NotPaused, d.resumeThread(tid)); + try testing.expectError(error.NotPaused, d.pausedStack(tid, &out.writer)); + d.disableBreakpoints(); +} + +/// Stands in for `FullPanic`'s call: the first trace address is the return +/// address into the panicking function. +noinline fn panicLike(msg: []const u8) bool { + return recordPanic(msg, @returnAddress()); +} + +test "panic record path" { + var text_buf: [16 * 1024]u8 = undefined; + var d: Debug = undefined; + try d.init(testOptions(&text_buf)); + defer d.deinit(); + defer resetPanicRecord(); + + var out: Writer.Allocating = .init(testing.allocator); + defer out.deinit(); + try d.panicMessage(&out.writer); + try testing.expectEqualStrings("", out.written()); + try testing.expect(!d.panicHeld()); + try testing.expectError(error.NoPanic, d.panicContinue()); + + try testing.expect(panicLike("something broke")); + try testing.expect(!recordPanic("nested", null)); + try testing.expectEqual(selfTid(), panic_tid); + + try d.panicMessage(&out.writer); + try testing.expectEqualStrings("something broke", out.written()); + out.clearRetainingCapacity(); + try d.panicStack(&out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "#0 0x") != null); + try testing.expect(std.mem.indexOf(u8, out.written(), "test.panic record path") != null); + try testing.expect(!d.panicHeld()); + try testing.expectError(error.NoPanic, d.panicContinue()); + + // A long message is truncated, not overflowed. + resetPanicRecord(); + const long = [_]u8{'x'} ** (panic_msg_cap + 100); + try testing.expect(recordPanic(&long, null)); + out.clearRetainingCapacity(); + try d.panicMessage(&out.writer); + try testing.expectEqual(@as(usize, panic_msg_cap), out.written().len); +} + +const MaskState = struct { + tid: std.atomic.Value(u32) = .init(0), + unblock: std.atomic.Value(bool) = .init(false), + stop: std.atomic.Value(bool) = .init(false), + signal: linux.SIG, +}; + +fn maskedThreadMain(st: *MaskState) void { + var set = linux.sigemptyset(); + linux.sigaddset(&set, st.signal); + _ = linux.sigprocmask(linux.SIG.BLOCK, &set, null); + st.tid.store(selfTid(), .release); + while (!st.unblock.load(.acquire)) sleepMs(1); + _ = linux.sigprocmask(linux.SIG.UNBLOCK, &set, null); + while (!st.stop.load(.acquire)) sleepMs(1); +} + +test "capture timeout on a thread with the signal masked" { + var text_buf: [16 * 1024]u8 = undefined; + var d: Debug = undefined; + var opts = testOptions(&text_buf); + opts.capture_timeout_ns = 50 * std.time.ns_per_ms; + try d.init(opts); + defer d.deinit(); + + var st: MaskState = .{ .signal = d.capture_signal }; + const th = try std.Thread.spawn(.{}, maskedThreadMain, .{&st}); + var tries: usize = 0; + while (st.tid.load(.acquire) == 0) : (tries += 1) { + try testing.expect(tries < 2000); + sleepMs(1); + } + const masked_tid = st.tid.load(.acquire); + + var out: Writer.Allocating = .init(testing.allocator); + defer out.deinit(); + const t0 = monotonicNs(); + try testing.expectError(error.Timeout, d.threadStack(masked_tid, &out.writer)); + try testing.expect(monotonicNs() - t0 >= 50 * std.time.ns_per_ms); + try testing.expectEqual(cap_idle, d.capture.state.load(.acquire)); + + // The process is healthy: another thread can still be captured... + var spin: SpinState = .{}; + const spinner = try std.Thread.spawn(.{}, spinThreadMain, .{&spin}); + const spin_tid = waitForTid(&spin); + try testing.expect(spin_tid != 0); + out.clearRetainingCapacity(); + try d.threadStack(spin_tid, &out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "spinHere") != null); + + // ...and the late delivery of the pending signal is harmless. + st.unblock.store(true, .release); + sleepMs(20); + out.clearRetainingCapacity(); + try d.threadStack(spin_tid, &out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "spinHere") != null); + out.clearRetainingCapacity(); + try d.threadStack(masked_tid, &out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "maskedThreadMain") != null); + + @atomicStore(bool, &spin.stop, true, .release); + spinner.join(); + st.stop.store(true, .release); + th.join(); +} + +test "options validation and single instance" { + var text_buf: [4096]u8 = undefined; + var d: Debug = undefined; + var opts = testOptions(&text_buf); + opts.capture_signal = 5; + try testing.expectError(error.InvalidOptions, d.init(opts)); + opts = testOptions(&text_buf); + opts.max_paused = max_paused_cap + 1; + try testing.expectError(error.InvalidOptions, d.init(opts)); + try d.init(testOptions(&text_buf)); + defer d.deinit(); + var d2: Debug = undefined; + try testing.expectError(error.AlreadyInitialized, d2.init(testOptions(&text_buf))); +} diff --git a/9proc/src/linux/probe.zig b/9proc/src/linux/probe.zig new file mode 100644 index 0000000..559d981 --- /dev/null +++ b/9proc/src/linux/probe.zig @@ -0,0 +1,829 @@ +//! The Linux platform layer: one background thread runs a `poll()` loop over +//! a listener and every client connection, feeding each connection's core +//! `Conn` with `push`/`step`/`output`/`wrote`. No per-connection threads, no +//! allocation after `init`; every buffer lives in a caller-placed `Storage`. +//! +//! Also home of the debug facilities (`debug`, `provider`) and the /runtime +//! generators (`runtime`). See docs/LIBRARY.md. +//! +//! Client admission: a new connection takes a free slot. When every slot is +//! taken, the connection that has held a slot without any fid (never +//! attached, or fully clunked) for longer than `evict_idle_ms` is dropped in +//! its favour; if there is none, the new connection is closed ("refused"). +//! Nothing that holds a fid is ever evicted. +//! +//! `sleepServing(ms)` lets a request handler (a ctl command, say) wait +//! without stalling the other clients: called on the probe thread from inside +//! a request it keeps running the poll loop for every client whose request +//! is not in progress until the time is up. Requests served from inside such +//! a wait may wait themselves, up to `max_nested_sleeps` deep (each level is a +//! different client, so the depth is bounded by the client table anyway); the +//! level past that, and any call off the probe thread, is a plain sleep. +const std = @import("std"); +const builtin = @import("builtin"); +const linux = std.os.linux; +const cloud9 = @import("cloud9"); +const core = @import("../core.zig"); + +pub const debug = @import("debug.zig"); +pub const provider = @import("provider.zig"); +pub const runtime = @import("runtime.zig"); +pub const DebugProvider = provider.DebugProvider; + +pub const Listen = union(enum) { + /// A unix socket path (< 108 bytes); a stale socket file is unlinked first. + unix: []const u8, + /// An IPv4 literal "a.b.c.d:port". + tcp: []const u8, + /// An already listening socket, owned by the caller. + fd: i32, + /// One pre-connected client on these descriptors (stdio: 0 and 1). Nothing + /// is accepted and the loop ends when the client hangs up. + client: struct { in: i32, out: i32 }, +}; + +pub const Options = struct { + /// For `std.debug` symbolization. + io: std.Io, + listen: Listen, + /// Largest msize offered to clients (clamped to the server's `cfg.msize`). + msize: u32 = 64 * 1024, + /// Hold a panicking thread until /panic/ctl says "continue". + hold_on_panic: bool = true, + /// Real-time signal used to snapshot other threads. + capture_signal: u8 = debug.default_capture_signal, + /// Install the SIGTRAP handler so `@breakpoint()` parks the thread. + breakpoints: bool = true, + /// Mount /threads, /addr, /mem, /hex, /breakpoints, /panic (six provider slots). + mount_debug: bool = true, +}; + +pub const Error = error{ + /// Another `Debug` (another probe) exists in this process. + AlreadyInitialized, + /// No register capture on this architecture. + Unsupported, + /// `Shared` has fewer than six free provider slots. + TooManyProviders, + PathTooLong, + BadAddress, + /// A syscall failed; `last_errno` says which error. + Syscall, +}; + +/// Idle time without fids after which a slot holder may be evicted. +pub const evict_idle_ms: i64 = 500; +/// Largest number of connections accepted per poll wakeup. +const accept_burst = 64; +/// How deep `sleepServing` may nest (each level keeps a poll round on the stack). +pub const max_nested_sleeps = 8; + +/// Static per-client storage: `max_clients` core `Storage`s and `Conn`s, the +/// poll table, the debug text arena and the debug provider's snapshot pool. +pub fn Storage(comptime max_clients: u8, comptime Srv: type) type { + return Probe(Srv).Storage(max_clients); +} + +pub fn Probe(comptime Srv: type) type { + return struct { + const Self = @This(); + + pub fn Storage(comptime max_clients: u8) type { + comptime std.debug.assert(max_clients > 0); + return struct { + pub const capacity = max_clients; + conns: [max_clients]Srv.Storage, + clients: [max_clients]Client, + /// [0] wake eventfd, [1] listener, [2..] one per client slot. + pollfds: [max_clients + 2]linux.pollfd, + text_buf: [16 * 1024]u8, + dp: DebugProvider, + }; + } + + pub const Client = struct { + conn: Srv.Conn, + in: i32 = -1, + out: i32 = -1, + used: bool = false, + /// Descriptors we opened (accepted) are closed on drop; borrowed ones are not. + owned: bool = false, + /// Send with MSG_NOSIGNAL; falls back to write(2) on ENOTSOCK. + is_socket: bool = true, + /// Monotonic ms of the last byte received. + last_active: i64 = 0, + }; + + shared: *Srv.Shared, + clients: []Client, + conns: []Srv.Storage, + pollfds: []linux.pollfd, + dbg: debug.Debug, + dp: *DebugProvider, + msize: u32, + listen_fd: i32 = -1, + own_listener: bool = false, + is_tcp: bool = false, + single: bool = false, + wake_fd: i32 = -1, + unix_path: [108]u8 = undefined, + unix_len: usize = 0, + thread: ?std.Thread = null, + thread_tid: std.atomic.Value(u32) = .init(0), + nclients: std.atomic.Value(u32) = .init(0), + /// Connections closed because no slot was free. + refused: u64 = 0, + stopping: std.atomic.Value(bool) = .init(false), + /// Slots whose request is being handled (excluded from nested servicing and eviction). + serving: std.StaticBitSet(256) = .initEmpty(), + /// Current `sleepServing` nesting depth. + nested: u8 = 0, + debug_ready: bool = false, + last_errno: linux.E = .SUCCESS, + + // -- lifecycle ------------------------------------------------------- + + /// Installs the debug facilities, mounts the debug providers into + /// `shared` and opens the listener. `storage` is a `*Storage(n)`. On + /// failure `shared` may already hold the debug providers and must be + /// discarded. + pub fn init(p: *Self, shared: *Srv.Shared, storage: anytype, opts: Options) Error!void { + p.* = .{ + .shared = shared, + .clients = &storage.clients, + .conns = &storage.conns, + .pollfds = &storage.pollfds, + .dbg = undefined, + .dp = &storage.dp, + .msize = opts.msize, + }; + for (p.clients) |*c| c.used = false; + shared.hash_seed = randomSeed(); + debug.hold_on_panic = opts.hold_on_panic; + p.dbg.init(.{ .io = opts.io, .text_buf = &storage.text_buf, .capture_signal = opts.capture_signal }) catch |e| return switch (e) { + error.AlreadyInitialized => error.AlreadyInitialized, + error.Unsupported => error.Unsupported, + else => error.Syscall, + }; + p.debug_ready = true; + errdefer { + p.dbg.deinit(); + p.debug_ready = false; + } + if (opts.breakpoints) p.dbg.enableBreakpoints() catch |e| switch (e) { + // No breakpoint support on this architecture: everything else still works. + error.Unsupported => {}, + else => return error.Syscall, + }; + if (opts.mount_debug) { + p.dp.init(&p.dbg); + p.dp.mountAll(shared) catch return error.TooManyProviders; + } + const efd = linux.eventfd(0, linux.EFD.CLOEXEC | linux.EFD.NONBLOCK); + try p.check(efd); + p.wake_fd = @intCast(efd); + errdefer { + _ = linux.close(p.wake_fd); + p.wake_fd = -1; + } + switch (opts.listen) { + .unix => |path| try p.listenUnix(path), + .tcp => |text| try p.listenTcp(text), + .fd => |fd| { + try p.setNonblock(fd); + p.listen_fd = fd; + }, + .client => |c| { + p.single = true; + try p.setNonblock(c.in); + if (c.out != c.in) try p.setNonblock(c.out); + _ = p.addClient(c.in, c.out, false); + }, + } + } + + /// Spawns the poll thread. + pub fn start(p: *Self) std.Thread.SpawnError!void { + std.debug.assert(p.thread == null); + p.stopping.store(false, .release); + p.thread = try std.Thread.spawn(.{}, run, .{p}); + } + + /// Waits for the poll thread to end (only happens by itself in + /// `.client` mode, when the client hangs up). + pub fn wait(p: *Self) void { + if (p.thread) |t| { + t.join(); + p.thread = null; + } + } + + /// Stops the poll thread, drops every client, closes what `init` + /// opened and restores the signal dispositions. + pub fn stop(p: *Self) void { + // Joining the poll thread from itself would hang forever; a + // request handler that wants the server gone uses `requestStop`. + std.debug.assert(p.thread_tid.load(.acquire) != @as(u32, @intCast(linux.gettid()))); + p.stopping.store(true, .release); + p.wakeLoop(); + p.wait(); + for (p.clients, 0..) |*c, i| if (c.used) p.dropClient(i); + if (p.listen_fd >= 0) { + if (p.own_listener) _ = linux.close(p.listen_fd); + p.listen_fd = -1; + } + if (p.unix_len > 0) { + _ = linux.unlink(@ptrCast(&p.unix_path)); + p.unix_len = 0; + } + if (p.wake_fd >= 0) { + _ = linux.close(p.wake_fd); + p.wake_fd = -1; + } + if (p.debug_ready) { + p.dbg.deinit(); + p.debug_ready = false; + } + } + + /// Live client count (for /runtime/clients). + pub fn clientCount(p: *const Self) u32 { + return p.nclients.load(.acquire); + } + + pub fn clientCounter(p: *const Self) *const std.atomic.Value(u32) { + return &p.nclients; + } + + /// Waits `ms` while keeping the other clients served (see the file comment). + pub fn sleepServing(p: *Self, ms: u64) void { + const on_thread = p.thread_tid.load(.acquire) == @as(u32, @intCast(linux.gettid())); + if (!on_thread or p.serving.count() == 0 or p.nested >= max_nested_sleeps) return sleepMs(ms); + p.nested += 1; + defer p.nested -= 1; + const deadline = monotonicMs() + @as(i64, @intCast(@min(ms, std.math.maxInt(i32)))); + while (!p.stopping.load(.acquire)) { + const now = monotonicMs(); + if (now >= deadline) break; + p.pollOnce(@intCast(deadline - now)); + } + } + + /// Asks the poll thread to stop; safe to call from a signal handler + /// (an atomic store and one write to the wake eventfd). `stop` (or + /// `wait`) still has to run afterwards to release everything. + pub fn requestStop(p: *Self) void { + p.stopping.store(true, .release); + p.wakeLoop(); + } + + // -- the loop -------------------------------------------------------- + + fn run(p: *Self) void { + const tid: u32 = @intCast(linux.gettid()); + p.thread_tid.store(tid, .release); + // A breakpoint or panic on this thread must never park it (see debug.zig). + debug.server_tid.store(tid, .release); + setThreadName("9proc"); + while (!p.stopping.load(.acquire)) { + if (p.single and p.clientCount() == 0) break; + p.pollOnce(-1); + } + debug.server_tid.store(0, .release); + p.thread_tid.store(0, .release); + } + + fn wakeLoop(p: *Self) void { + if (p.wake_fd < 0) return; + const one: u64 = 1; + _ = linux.write(p.wake_fd, @ptrCast(&one), 8); + } + + /// One `poll()` round: accept, read, step, write. Slots whose request + /// is in progress (`serving`, only inside `sleepServing`) are left untouched. + fn pollOnce(p: *Self, timeout_ms: i32) void { + p.pollfds[0] = .{ .fd = p.wake_fd, .events = linux.POLL.IN, .revents = 0 }; + p.pollfds[1] = .{ .fd = p.listen_fd, .events = linux.POLL.IN, .revents = 0 }; + for (p.clients, 0..) |*c, i| { + var fd: i32 = -1; + var events: i16 = 0; + if (c.used and !p.serving.isSet(i)) { + fd = c.in; + if (c.conn.output().len > 0) { + fd = c.out; + events = linux.POLL.OUT; + } else if (inputRoom(&c.conn) > 0) { + events = linux.POLL.IN; + } + } + p.pollfds[2 + i] = .{ .fd = fd, .events = events, .revents = 0 }; + } + const rc = linux.poll(p.pollfds.ptr, p.pollfds.len, timeout_ms); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .INTR => return, + else => { + sleepMs(10); + return; + }, + } + if (p.pollfds[0].revents != 0) { + var v: u64 = 0; + _ = linux.read(p.wake_fd, @ptrCast(&v), 8); + } + if (p.stopping.load(.acquire)) return; + if (p.pollfds[1].revents != 0) p.acceptSome(); + for (p.clients, 0..) |*c, i| { + const re = p.pollfds[2 + i].revents; + if (re == 0 or !c.used or p.serving.isSet(i)) continue; + if (re & (linux.POLL.IN | linux.POLL.HUP | linux.POLL.ERR | linux.POLL.NVAL) != 0) { + p.readClient(i, re & linux.POLL.HUP != 0); + } else if (re & linux.POLL.OUT != 0) { + p.service(i); + } + if (p.stopping.load(.acquire)) return; + } + } + + fn acceptSome(p: *Self) void { + var n: usize = 0; + while (n < accept_burst) : (n += 1) { + const rc = linux.accept4(p.listen_fd, null, null, linux.SOCK.NONBLOCK | linux.SOCK.CLOEXEC); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .AGAIN => return, + .INTR, .CONNABORTED => continue, + // Descriptor/memory exhaustion is transient (clients hang up); + // back off instead of spinning on a readable listener. + .MFILE, .NFILE, .NOBUFS, .NOMEM, .PERM => { + sleepMs(100); + return; + }, + else => return, + } + const cfd: i32 = @intCast(rc); + if (p.is_tcp) { + const one: u32 = 1; + _ = linux.setsockopt(cfd, linux.IPPROTO.TCP, linux.TCP.NODELAY, @ptrCast(&one), @sizeOf(u32)); + } + if (p.addClient(cfd, cfd, true) != null) continue; + if (p.evictable()) |victim| { + p.dropClient(victim); + _ = p.addClient(cfd, cfd, true); + continue; + } + p.refused += 1; + // A flood must not flood stderr. + if (p.refused == 1 or p.refused % 1000 == 0) + std.debug.print("9proc: refused connection ({d} clients open, {d} refused so far)\n", .{ p.clients.len, p.refused }); + _ = linux.close(cfd); + } + } + + /// The longest-idle slot holder without fids, if idle long enough. + fn evictable(p: *Self) ?usize { + const now = monotonicMs(); + var best: ?usize = null; + for (p.clients, 0..) |*c, i| { + if (!c.used or !c.owned) continue; + if (p.serving.isSet(i)) continue; + if (c.conn.fidCount() != 0) continue; + if (now - c.last_active < evict_idle_ms) continue; + if (best == null or c.last_active < p.clients[best.?].last_active) best = i; + } + return best; + } + + fn addClient(p: *Self, in: i32, out: i32, owned: bool) ?usize { + for (p.clients, 0..) |*c, i| { + if (c.used) continue; + c.conn = Srv.Conn.init(p.shared, &p.conns[i], p.msize); + c.in = in; + c.out = out; + c.used = true; + c.owned = owned; + c.is_socket = true; + c.last_active = monotonicMs(); + _ = p.nclients.fetchAdd(1, .acq_rel); + return i; + } + return null; + } + + fn dropClient(p: *Self, i: usize) void { + const c = &p.clients[i]; + if (!c.used) return; + c.conn.hangup(); + if (c.owned) { + _ = linux.close(c.in); + if (c.out != c.in) _ = linux.close(c.out); + } + c.used = false; + c.in = -1; + c.out = -1; + _ = p.nclients.fetchSub(1, .acq_rel); + } + + /// Free space in the connection's input buffer (cloud9 keeps one + /// msize-sized frame; `push` copies at most this much). + fn inputRoom(conn: *const Srv.Conn) usize { + return conn.server.in.len - conn.server.in_len; + } + + fn readClient(p: *Self, i: usize, hup: bool) void { + const c = &p.clients[i]; + var buf: [64 * 1024]u8 = undefined; + const room = inputRoom(&c.conn); + if (room == 0) return p.service(i); + const want = @min(room, buf.len); + while (true) { + const rc = linux.read(c.in, &buf, want); + switch (linux.errno(rc)) { + .SUCCESS => { + if (rc == 0) return p.dropClient(i); + const taken = c.conn.push(buf[0..rc]); + std.debug.assert(taken == rc); + c.last_active = monotonicMs(); + return p.service(i); + }, + .INTR => continue, + .AGAIN => { + if (hup) p.dropClient(i); + return; + }, + else => return p.dropClient(i), + } + } + } + + /// Runs requests and drains output until nothing moves. + fn service(p: *Self, i: usize) void { + const c = &p.clients[i]; + std.debug.assert(!p.serving.isSet(i)); + p.serving.set(i); + defer p.serving.unset(i); + while (c.used) { + var moved = false; + while (true) { + const more = c.conn.step() catch return p.dropClient(i); + if (!more) break; + moved = true; + } + const before = c.conn.output().len; + p.flush(c) catch return p.dropClient(i); + if (c.conn.output().len != before) moved = true; + if (!moved) return; + } + } + + fn flush(p: *Self, c: *Client) error{Closed}!void { + _ = p; + while (c.conn.output().len > 0) { + const chunk = c.conn.output(); + const rc = if (c.is_socket) + linux.sendto(c.out, chunk.ptr, chunk.len, linux.MSG.NOSIGNAL, null, 0) + else + linux.write(c.out, chunk.ptr, chunk.len); + switch (linux.errno(rc)) { + .SUCCESS => { + if (rc == 0) return; + c.conn.wrote(rc); + }, + .INTR => continue, + .AGAIN => return, + .NOTSOCK => c.is_socket = false, + else => return error.Closed, + } + } + } + + // -- listeners ------------------------------------------------------- + + fn check(p: *Self, rc: usize) Error!void { + const e = linux.errno(rc); + if (e != .SUCCESS) { + p.last_errno = e; + return error.Syscall; + } + } + + fn setNonblock(p: *Self, fd: i32) Error!void { + const rc = linux.fcntl(fd, linux.F.GETFL, 0); + try p.check(rc); + const nonblock: u32 = @bitCast(linux.O{ .NONBLOCK = true }); + try p.check(linux.fcntl(fd, linux.F.SETFL, rc | nonblock)); + } + + fn listenUnix(p: *Self, path: []const u8) Error!void { + var sa: linux.sockaddr.un = .{ .path = @splat(0) }; + if (path.len == 0 or path.len >= sa.path.len) return error.PathTooLong; + @memcpy(sa.path[0..path.len], path); + const rc = linux.socket(linux.AF.UNIX, linux.SOCK.STREAM | linux.SOCK.CLOEXEC | linux.SOCK.NONBLOCK, 0); + try p.check(rc); + const lfd: i32 = @intCast(rc); + errdefer _ = linux.close(lfd); + // No libc, so no "is it still listening" probe: unlink a stale socket and bind. + _ = linux.unlink(@ptrCast(&sa.path)); + try p.check(linux.bind(lfd, @ptrCast(&sa), @sizeOf(linux.sockaddr.un))); + try p.check(linux.listen(lfd, 128)); + p.listen_fd = lfd; + p.own_listener = true; + p.unix_path = sa.path; + p.unix_len = path.len; + } + + fn listenTcp(p: *Self, text: []const u8) Error!void { + const sa = parseIpv4(text) orelse return error.BadAddress; + const rc = linux.socket(linux.AF.INET, linux.SOCK.STREAM | linux.SOCK.CLOEXEC | linux.SOCK.NONBLOCK, 0); + try p.check(rc); + const lfd: i32 = @intCast(rc); + errdefer _ = linux.close(lfd); + const one: u32 = 1; + _ = linux.setsockopt(lfd, linux.SOL.SOCKET, linux.SO.REUSEADDR, @ptrCast(&one), @sizeOf(u32)); + try p.check(linux.bind(lfd, @ptrCast(&sa), @sizeOf(linux.sockaddr.in))); + try p.check(linux.listen(lfd, 128)); + p.listen_fd = lfd; + p.own_listener = true; + p.is_tcp = true; + } + }; +} + +/// Entropy for the core's fid hash (so fid numbers cannot be chosen to +/// collide); falls back to the clock if getrandom fails. +fn randomSeed() u32 { + var b: [4]u8 = undefined; + if (linux.errno(linux.getrandom(&b, b.len, 0)) == .SUCCESS) return std.mem.readInt(u32, &b, .little); + var ts: linux.timespec = undefined; + _ = linux.clock_gettime(.MONOTONIC, &ts); + return @truncate(@as(u64, @bitCast(ts.nsec)) ^ (@as(u64, @bitCast(ts.sec)) << 20)); +} + +/// "a.b.c.d:port" as a socket address, or null. +pub fn parseIpv4(text: []const u8) ?linux.sockaddr.in { + const colon = std.mem.lastIndexOfScalar(u8, text, ':') orelse return null; + const port = std.fmt.parseInt(u16, text[colon + 1 ..], 10) catch return null; + var octets: [4]u8 = undefined; + var it = std.mem.splitScalar(u8, text[0..colon], '.'); + for (&octets) |*o| o.* = std.fmt.parseInt(u8, it.next() orelse return null, 10) catch return null; + if (it.next() != null) return null; + return .{ .port = std.mem.nativeToBig(u16, port), .addr = @bitCast(octets) }; +} + +/// Names the calling thread (comm, at most 15 bytes) via prctl. +pub fn setThreadName(name: []const u8) void { + var buf: [16]u8 = @splat(0); + const n = @min(name.len, 15); + @memcpy(buf[0..n], name[0..n]); + _ = linux.prctl(@intFromEnum(linux.PR.SET_NAME), @intFromPtr(&buf), 0, 0, 0); +} + +pub fn sleepMs(ms: u64) void { + var req: linux.timespec = .{ .sec = @intCast(ms / 1000), .nsec = @intCast((ms % 1000) * 1_000_000) }; + var rem: linux.timespec = undefined; + while (linux.errno(linux.nanosleep(&req, &rem)) == .INTR) req = rem; +} + +pub fn monotonicMs() i64 { + var ts: linux.timespec = undefined; + _ = linux.clock_gettime(.MONOTONIC, &ts); + return ts.sec * 1000 + @divTrunc(ts.nsec, 1_000_000); +} + +// --------------------------------------------------------------------------- +// Tests: a real unix socket, a cloud9.Client on the other end. +// --------------------------------------------------------------------------- + +const testing = std.testing; + +test { + _ = debug; + _ = provider; + _ = runtime; +} + +const TestBuild = struct { + pub const zig_version: []const u8 = builtin.zig_version_string; + pub const target: []const u8 = "test"; + pub const optimize: []const u8 = "Debug"; + pub const time: []const u8 = "2024-01-01T00:00:00Z"; + pub const change: []const u8 = "none"; +}; + +const test_cfg: core.Config = .{ + .name = "probetest", + .build = TestBuild, + .msize = 8192, + .max_fids = 16, + .max_providers = 6, + .snapshot_slots = 2, + .snapshot_bytes = 1024, +}; +const TS = core.Server(test_cfg); +const TP = Probe(TS); + +/// A blocking client over a connected socket. +const SockClient = struct { + fd: i32, + client: cloud9.Client, + cin: [8192]u8 = undefined, + cout: [8192]u8 = undefined, + + fn connect(sc: *SockClient, path: []const u8) !void { + var sa: linux.sockaddr.un = .{ .path = @splat(0) }; + @memcpy(sa.path[0..path.len], path); + const rc = linux.socket(linux.AF.UNIX, linux.SOCK.STREAM | linux.SOCK.CLOEXEC, 0); + if (linux.errno(rc) != .SUCCESS) return error.Socket; + sc.fd = @intCast(rc); + if (linux.errno(linux.connect(sc.fd, &sa, @sizeOf(linux.sockaddr.un))) != .SUCCESS) return error.Connect; + sc.client = .init(.{ .in = &sc.cin, .out = &sc.cout }); + } + + fn close(sc: *SockClient) void { + _ = linux.close(sc.fd); + } + + /// One round trip; null when the server closed the connection. + fn rpc(sc: *SockClient, req: cloud9.Client.Request) !?cloud9.Client.Result { + _ = try sc.client.submit(req); + while (sc.client.output().len > 0) { + const out = sc.client.output(); + const rc = linux.write(sc.fd, out.ptr, out.len); + switch (linux.errno(rc)) { + .SUCCESS => sc.client.wrote(rc), + .PIPE, .CONNRESET => return null, + else => return error.Write, + } + } + var buf: [8192]u8 = undefined; + while (true) { + if (sc.client.take()) |done| return done.result; + const rc = linux.read(sc.fd, &buf, buf.len); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .CONNRESET => return null, + else => return error.Read, + } + if (rc == 0) return null; + var rest: []const u8 = buf[0..rc]; + while (rest.len > 0) rest = rest[sc.client.push(rest)..]; + } + } + + fn session(sc: *SockClient) !void { + const v = (try sc.rpc(.{ .version = .{ .msize = 8192 } })) orelse return error.Closed; + try testing.expectEqual(@as(u32, 8192), v.version.msize); + const a = (try sc.rpc(.{ .attach = .{ .fid = 0, .uname = "t" } })) orelse return error.Closed; + try testing.expect(a == .attach); + } + + fn readFile(sc: *SockClient, names: []const []const u8, out: []u8) ![]u8 { + const w = (try sc.rpc(.{ .walk = .{ .fid = 0, .newfid = 1, .names = names } })) orelse return error.Closed; + try testing.expectEqual(@as(u16, @intCast(names.len)), w.walk.nwqid); + _ = (try sc.rpc(.{ .open = .{ .fid = 1, .mode = cloud9.oread } })) orelse return error.Closed; + const r = (try sc.rpc(.{ .read = .{ .fid = 1, .offset = 0, .count = @intCast(out.len) } })) orelse return error.Closed; + const n = r.read.len; + @memcpy(out[0..n], r.read); + _ = (try sc.rpc(.{ .clunk = .{ .fid = 1 } })) orelse return error.Closed; + return out[0..n]; + } +}; + +fn testSockPath(buf: []u8, tag: []const u8) ![]const u8 { + return std.fmt.bufPrint(buf, "/tmp/9proc-probe-{d}-{s}.sock", .{ linux.getpid(), tag }); +} + +const TestCtx = struct { info: runtime.Info }; + +test "probe: start, serve a client over a unix socket, stop" { + var ctx: TestCtx = .{ .info = .now() }; + var shared: TS.Shared = .init(&ctx); + const storage = try testing.allocator.create(TP.Storage(2)); + defer testing.allocator.destroy(storage); + var probe: TP = undefined; + var path_buf: [64]u8 = undefined; + const path = try testSockPath(&path_buf, "basic"); + try probe.init(&shared, storage, .{ .io = testing.io, .listen = .{ .unix = path } }); + defer probe.stop(); + try probe.start(); + ctx.info.clients = probe.clientCounter(); + + var sc: SockClient = undefined; + try sc.connect(path); + defer sc.close(); + try sc.session(); + var buf: [1024]u8 = undefined; + const zv = try sc.readFile(&.{ "build", "zig_version" }, &buf); + try testing.expectEqualStrings(builtin.zig_version_string, zv); + try testing.expectEqual(@as(u32, 1), probe.clientCount()); + + // The debug providers are mounted: /threads lists the probe thread by name. + const names = (try sc.rpc(.{ .walk = .{ .fid = 0, .newfid = 2, .names = &.{"threads"} } })) orelse return error.Closed; + try testing.expectEqual(@as(u16, 1), names.walk.nwqid); + _ = (try sc.rpc(.{ .open = .{ .fid = 2, .mode = cloud9.oread } })) orelse return error.Closed; + const dir = (try sc.rpc(.{ .read = .{ .fid = 2, .offset = 0, .count = 4096 } })) orelse return error.Closed; + try testing.expect(dir.read.len > 0); + _ = (try sc.rpc(.{ .clunk = .{ .fid = 2 } })) orelse return error.Closed; + + // A missing file is the Plan 9 error string. + const bad = (try sc.rpc(.{ .walk = .{ .fid = 0, .newfid = 3, .names = &.{"nope"} } })) orelse return error.Closed; + try testing.expect(bad == .fail); + try testing.expectEqualStrings("file does not exist", bad.fail); + + probe.stop(); + // stop() is idempotent and the socket file is gone. + probe.stop(); + var gone: SockClient = undefined; + try testing.expectError(error.Connect, gone.connect(path)); + // The debug facilities can be set up again after stop. + var probe2: TP = undefined; + var shared2: TS.Shared = .init(&ctx); + try probe2.init(&shared2, storage, .{ .io = testing.io, .listen = .{ .unix = path } }); + probe2.stop(); +} + +test "probe: max_clients refusal and idle eviction" { + var ctx: TestCtx = .{ .info = .now() }; + var shared: TS.Shared = .init(&ctx); + const storage = try testing.allocator.create(TP.Storage(2)); + defer testing.allocator.destroy(storage); + var probe: TP = undefined; + var path_buf: [64]u8 = undefined; + const path = try testSockPath(&path_buf, "limit"); + try probe.init(&shared, storage, .{ .io = testing.io, .listen = .{ .unix = path } }); + defer probe.stop(); + try probe.start(); + + // Two attached clients fill the table; a third is accepted then closed. + var a: SockClient = undefined; + try a.connect(path); + defer a.close(); + try a.session(); + var b: SockClient = undefined; + try b.connect(path); + defer b.close(); + try b.session(); + var c: SockClient = undefined; + try c.connect(path); + defer c.close(); + try testing.expectEqual(@as(?cloud9.Client.Result, null), try c.rpc(.{ .version = .{ .msize = 8192 } })); + try testing.expectEqual(@as(u64, 1), probe.refused); + // Attached clients are never evicted, even when idle for long. + sleepMs(evict_idle_ms + 100); + var d: SockClient = undefined; + try d.connect(path); + defer d.close(); + try testing.expectEqual(@as(?cloud9.Client.Result, null), try d.rpc(.{ .version = .{ .msize = 8192 } })); + var buf: [256]u8 = undefined; + _ = try a.readFile(&.{"README"}, &buf); + + // A client without fids that has been idle long enough gives way. + _ = (try b.rpc(.{ .clunk = .{ .fid = 0 } })) orelse return error.Closed; + sleepMs(evict_idle_ms + 100); + var e: SockClient = undefined; + try e.connect(path); + defer e.close(); + try e.session(); + try testing.expectEqual(@as(?cloud9.Client.Result, null), try b.rpc(.{ .version = .{ .msize = 8192 } })); + try testing.expectEqual(@as(u32, 2), probe.clientCount()); +} + +test "probe: single pre-connected client mode ends when the client hangs up" { + var ctx: TestCtx = .{ .info = .now() }; + var shared: TS.Shared = .init(&ctx); + const storage = try testing.allocator.create(TP.Storage(1)); + defer testing.allocator.destroy(storage); + var sv: [2]i32 = undefined; + try testing.expectEqual(linux.E.SUCCESS, linux.errno(linux.socketpair(linux.AF.UNIX, linux.SOCK.STREAM | linux.SOCK.CLOEXEC, 0, &sv))); + var probe: TP = undefined; + try probe.init(&shared, storage, .{ .io = testing.io, .listen = .{ .client = .{ .in = sv[1], .out = sv[1] } } }); + defer probe.stop(); + try probe.start(); + var sc: SockClient = .{ .fd = sv[0], .client = undefined }; + sc.client = .init(.{ .in = &sc.cin, .out = &sc.cout }); + try sc.session(); + var buf: [256]u8 = undefined; + try testing.expect((try sc.readFile(&.{"README"}, &buf)).len > 0); + try testing.expectEqual(@as(u32, 1), probe.clientCount()); + _ = linux.close(sv[0]); + probe.wait(); + try testing.expectEqual(@as(u32, 0), probe.clientCount()); + _ = linux.close(sv[1]); +} + +test "parseIpv4 and sleepServing off the probe thread" { + const sa = parseIpv4("127.0.0.1:564").?; + try testing.expectEqual(std.mem.nativeToBig(u16, 564), sa.port); + try testing.expectEqual(@as(u32, @bitCast([4]u8{ 127, 0, 0, 1 })), sa.addr); + try testing.expect(parseIpv4("localhost:1") == null); + try testing.expect(parseIpv4("1.2.3:1") == null); + try testing.expect(parseIpv4("1.2.3.4") == null); + try testing.expect(parseIpv4("1.2.3.4:70000") == null); + const t0 = monotonicMs(); + var probe: TP = undefined; + probe.thread_tid = .init(0); + probe.serving = .initEmpty(); + probe.nested = 0; + probe.sleepServing(20); + try testing.expect(monotonicMs() - t0 >= 20); +} diff --git a/9proc/src/linux/provider.zig b/9proc/src/linux/provider.zig new file mode 100644 index 0000000..9952398 --- /dev/null +++ b/9proc/src/linux/provider.zig @@ -0,0 +1,604 @@ +//! Adapts `debug.Debug` into core `Provider`s. The core mounts providers at +//! top level only, so one `DebugProvider` registers six of them, all sharing +//! the same `Debug` and the same snapshot pool: +//! +//! /threads//{name,stat,stack,regs} (lists only live tids) +//! /addr/ "fn\nfile:line:col\nmodule\n" +//! /mem/maps, /mem/ /proc/self/maps; raw bytes at address+offset (writable) +//! /hex/ hexdump of 256 bytes at the address +//! /breakpoints//{stack,regs,ctl} ctl accepts "continue" (lists only paused tids) +//! /panic/{message,stack,ctl} ctl accepts "continue" +//! +//! /addr, /mem, /hex and /panic carry a README; /threads and /breakpoints +//! list nothing but tids so that a shell glob over them sees only threads. +//! +//! Handles encode `(kind, tid-or-address)` in 56 bits (the core keeps the +//! low 56 bits of a handle for the qid path): kind in bits 48..55, value in +//! bits 0..47. Handles carry no reference count, so `clunk` is a no-op. +//! +//! The core hands providers a buffer-based `read`, not a writer, so every +//! text file is generated at `open` into one of `snapshot_slots` fixed slots +//! (keyed by handle, reference counted across fids) and served from there; a +//! read at offset 0 regenerates, like the core's own dynamic files. `/mem/` +//! is read and written directly at address+offset and never snapshotted, and +//! `/mem/maps` is read straight from /proc/self/maps at the requested offset +//! (a big process has more mappings than a snapshot slot holds). +const std = @import("std"); +const cloud9 = @import("cloud9"); +const core = @import("../core.zig"); +const debug = @import("debug.zig"); +const Writer = std.Io.Writer; +const Provider = core.Provider; +const Handle = Provider.Handle; +const Error = Provider.Error; +const NodeStat = core.NodeStat; + +pub const snapshot_slots = 8; +pub const snapshot_bytes = 32 * 1024; +/// Bytes shown by /hex/. +pub const hex_bytes = 256; + +pub const Tree = enum(u8) { threads, addr, mem, hex, breakpoints, panic }; +pub const tree_names = [_][]const u8{ "threads", "addr", "mem", "hex", "breakpoints", "panic" }; + +const Kind = enum(u8) { + root = 0, + readme, + thread_dir, + thread_name, + thread_stat, + thread_stack, + thread_regs, + addr_file, + maps, + mem_file, + hex_file, + bp_dir, + bp_stack, + bp_regs, + bp_ctl, + panic_message, + panic_stack, + panic_ctl, + + fn isDir(k: Kind) bool { + return k == .root or k == .thread_dir or k == .bp_dir; + } + + /// Text files generated into a snapshot slot at open. + fn isText(k: Kind) bool { + return switch (k) { + .readme, .thread_name, .thread_stat, .thread_stack, .thread_regs, .addr_file, .hex_file, .bp_stack, .bp_regs, .panic_message, .panic_stack => true, + else => false, + }; + } + + fn isCtl(k: Kind) bool { + return k == .bp_ctl or k == .panic_ctl; + } + + fn mode(k: Kind) u32 { + if (k.isDir()) return cloud9.dmdir | 0o555; + if (k.isCtl()) return 0o222; + if (k == .mem_file) return 0o666; + return 0o444; + } + + fn fixedName(k: Kind) ?[]const u8 { + return switch (k) { + .readme => "README", + .thread_name => "name", + .thread_stat => "stat", + .thread_stack, .bp_stack, .panic_stack => "stack", + .thread_regs, .bp_regs => "regs", + .maps => "maps", + .bp_ctl, .panic_ctl => "ctl", + .panic_message => "message", + else => null, + }; + } +}; + +const value_bits = 48; +const value_mask: u64 = (1 << value_bits) - 1; + +fn mk(kind: Kind, value: u64) Handle { + return (@as(u64, @intFromEnum(kind)) << value_bits) | (value & value_mask); +} + +fn kindOf(h: Handle) Kind { + return @enumFromInt(@as(u8, @truncate(h >> value_bits))); +} + +fn valueOf(h: Handle) u64 { + return h & value_mask; +} + +const readme_threads = + \\One directory per thread of this process, named by tid: + \\ name the thread's comm + \\ stat state and a few fields of /proc/self/task//stat + \\ stack "#n 0x in (::)" per frame + \\ regs general registers captured while the thread was stopped + \\ +; +const readme_addr = + \\Walk any hex address: /addr/ reads as "fn\nfile:line:col\nmodule\n". + \\ +; +const readme_mem = + \\maps /proc/self/maps + \\ raw process memory at that address (+ file offset); writable + \\ +; +const readme_hex = + \\Walk any hex address: /hex/ is a hexdump of the 256 bytes there. + \\ +; +const readme_breakpoints = + \\One directory per thread stopped in @breakpoint(), named by tid: + \\ stack, regs as under /threads + \\ ctl write "continue" to resume the thread + \\ +; +const readme_panic = + \\message the first panic's message (empty before any panic) + \\stack frames of the panicking thread + \\ctl write "continue" to let the default panic handler run + \\ +; + +fn readmeFor(tree: Tree) []const u8 { + return switch (tree) { + .threads => readme_threads, + .addr => readme_addr, + .mem => readme_mem, + .hex => readme_hex, + .breakpoints => readme_breakpoints, + .panic => readme_panic, + }; +} + +const Slot = struct { + handle: Handle = 0, + refs: u32 = 0, + len: u32 = 0, + buf: [snapshot_bytes]u8 = undefined, +}; + +pub const DebugProvider = struct { + d: *debug.Debug, + slots: [snapshot_slots]Slot = @splat(.{}), + /// Backs `NodeStat.name` until the next call. + name_buf: [32]u8 = undefined, + + pub fn init(dp: *DebugProvider, d: *debug.Debug) void { + dp.* = .{ .d = d }; + } + + /// The provider for one tree, to pass to `Shared.addProvider`. + pub fn provider(dp: *DebugProvider, comptime tree: Tree) Provider { + return .{ .name = tree_names[@intFromEnum(tree)], .ctx = dp, .vtable = vtableFor(tree) }; + } + + /// Mounts all six trees; `shared` is a `Server(cfg).Shared`. + pub fn mountAll(dp: *DebugProvider, shared: anytype) error{Full}!void { + inline for (comptime std.meta.tags(Tree)) |tree| try shared.addProvider(dp.provider(tree)); + } + + fn self(ctx: *anyopaque) *DebugProvider { + return @ptrCast(@alignCast(ctx)); + } + + fn vtableFor(comptime tree: Tree) *const Provider.VTable { + return &struct { + const vt: Provider.VTable = .{ + .walk = walkFn, + .stat = statFn, + .list = listFn, + .open = openFn, + .read = readFn, + .write = writeFn, + .close = closeFn, + .clunk = clunkFn, + }; + fn walkFn(ctx: *anyopaque, parent: Handle, name: []const u8) Error!Handle { + return self(ctx).walk(tree, parent, name); + } + fn statFn(ctx: *anyopaque, h: Handle, out: *NodeStat) Error!void { + return self(ctx).stat(tree, h, out); + } + fn listFn(ctx: *anyopaque, dir: Handle, index: usize, out: *NodeStat) Error!bool { + return self(ctx).list(tree, dir, index, out); + } + fn openFn(ctx: *anyopaque, h: Handle, mode: u8) Error!void { + return self(ctx).open(tree, h, mode); + } + fn readFn(ctx: *anyopaque, h: Handle, offset: u64, buf: []u8) Error!usize { + return self(ctx).read(tree, h, offset, buf); + } + fn writeFn(ctx: *anyopaque, h: Handle, offset: u64, data: []const u8) Error!usize { + return self(ctx).write(tree, h, offset, data); + } + fn closeFn(ctx: *anyopaque, h: Handle) void { + self(ctx).close(h); + } + fn clunkFn(_: *anyopaque, _: Handle) void {} + }.vt; + } + + // -- naming ------------------------------------------------------------ + + fn parseTid(name: []const u8) ?u32 { + if (name.len == 0 or name.len > 10) return null; + for (name) |ch| if (!std.ascii.isDigit(ch)) return null; + return std.fmt.parseInt(u32, name, 10) catch null; + } + + fn parseHex(name: []const u8) ?u64 { + const digits = if (std.mem.startsWith(u8, name, "0x")) name[2..] else name; + if (digits.len == 0 or digits.len > 12) return null; + for (digits) |ch| if (!std.ascii.isHex(ch)) return null; + const v = std.fmt.parseInt(u64, digits, 16) catch return null; + if (v > value_mask) return null; + return v; + } + + fn nodeName(dp: *DebugProvider, h: Handle) []const u8 { + const k = kindOf(h); + if (k.fixedName()) |n| return n; + return switch (k) { + .root => "", + .thread_dir, .bp_dir => std.fmt.bufPrint(&dp.name_buf, "{d}", .{valueOf(h)}) catch unreachable, + .addr_file, .mem_file, .hex_file => std.fmt.bufPrint(&dp.name_buf, "{x}", .{valueOf(h)}) catch unreachable, + else => unreachable, + }; + } + + // -- vtable ------------------------------------------------------------ + + fn walk(dp: *DebugProvider, tree: Tree, parent: Handle, name: []const u8) Error!Handle { + const k = kindOf(parent); + if (std.mem.eql(u8, name, ".")) return parent; + if (!k.isDir()) return error.NotDir; + if (std.mem.eql(u8, name, "..")) return Provider.root; + switch (k) { + .root => { + if (tree != .threads and tree != .breakpoints and std.mem.eql(u8, name, "README")) return mk(.readme, 0); + switch (tree) { + .threads => { + const tid = parseTid(name) orelse return error.NotFound; + if (!dp.d.threadExists(tid)) return error.NotFound; + return mk(.thread_dir, tid); + }, + .addr => return mk(.addr_file, parseHex(name) orelse return error.NotFound), + .mem => { + if (std.mem.eql(u8, name, "maps")) return mk(.maps, 0); + return mk(.mem_file, parseHex(name) orelse return error.NotFound); + }, + .hex => return mk(.hex_file, parseHex(name) orelse return error.NotFound), + .breakpoints => { + const tid = parseTid(name) orelse return error.NotFound; + if (!dp.d.isPaused(tid)) return error.NotFound; + return mk(.bp_dir, tid); + }, + .panic => { + if (std.mem.eql(u8, name, "message")) return mk(.panic_message, 0); + if (std.mem.eql(u8, name, "stack")) return mk(.panic_stack, 0); + if (std.mem.eql(u8, name, "ctl")) return mk(.panic_ctl, 0); + return error.NotFound; + }, + } + }, + .thread_dir => { + const tid = valueOf(parent); + if (std.mem.eql(u8, name, "name")) return mk(.thread_name, tid); + if (std.mem.eql(u8, name, "stat")) return mk(.thread_stat, tid); + if (std.mem.eql(u8, name, "stack")) return mk(.thread_stack, tid); + if (std.mem.eql(u8, name, "regs")) return mk(.thread_regs, tid); + return error.NotFound; + }, + .bp_dir => { + const tid = valueOf(parent); + if (std.mem.eql(u8, name, "stack")) return mk(.bp_stack, tid); + if (std.mem.eql(u8, name, "regs")) return mk(.bp_regs, tid); + if (std.mem.eql(u8, name, "ctl")) return mk(.bp_ctl, tid); + return error.NotFound; + }, + else => unreachable, + } + } + + fn fill(dp: *DebugProvider, tree: Tree, h: Handle, out: *NodeStat) void { + const k = kindOf(h); + out.* = .{ + .mode = k.mode(), + .length = if (k == .readme) readmeFor(tree).len else 0, + .name = dp.nodeName(h), + .handle = h, + }; + } + + fn stat(dp: *DebugProvider, tree: Tree, h: Handle, out: *NodeStat) Error!void { + dp.fill(tree, h, out); + } + + fn list(dp: *DebugProvider, tree: Tree, dir: Handle, index: usize, out: *NodeStat) Error!bool { + const k = kindOf(dir); + if (!k.isDir()) return error.NotDir; + const h: Handle = switch (k) { + .root => switch (tree) { + .threads => mk(.thread_dir, dp.d.threadAt(index) orelse return false), + .addr, .hex => if (index == 0) mk(.readme, 0) else return false, + .mem => switch (index) { + 0 => mk(.readme, 0), + 1 => mk(.maps, 0), + else => return false, + }, + .breakpoints => mk(.bp_dir, dp.d.pausedAt(index) orelse return false), + .panic => switch (index) { + 0 => mk(.readme, 0), + 1 => mk(.panic_message, 0), + 2 => mk(.panic_stack, 0), + 3 => mk(.panic_ctl, 0), + else => return false, + }, + }, + .thread_dir => switch (index) { + 0 => mk(.thread_name, valueOf(dir)), + 1 => mk(.thread_stat, valueOf(dir)), + 2 => mk(.thread_stack, valueOf(dir)), + 3 => mk(.thread_regs, valueOf(dir)), + else => return false, + }, + .bp_dir => switch (index) { + 0 => mk(.bp_stack, valueOf(dir)), + 1 => mk(.bp_regs, valueOf(dir)), + 2 => mk(.bp_ctl, valueOf(dir)), + else => return false, + }, + else => unreachable, + }; + dp.fill(tree, h, out); + return true; + } + + fn open(dp: *DebugProvider, tree: Tree, h: Handle, mode: u8) Error!void { + const k = kindOf(h); + const acc = mode & 3; + const wants_write = acc == cloud9.owrite or acc == cloud9.ordwr; + const wants_read = acc != cloud9.owrite; + if (k.isDir()) { + if (wants_write or mode & cloud9.otrunc != 0) return error.IsDir; + return; + } + if (k.isCtl()) { + if (wants_read) return error.Perm; + return; + } + if (k == .mem_file) return; + if (wants_write or mode & cloud9.otrunc != 0) return error.Perm; + if (k == .maps) return; + std.debug.assert(k.isText()); + const slot = dp.takeSlot(h) orelse return error.NoSpace; + errdefer dp.releaseSlot(slot); + try dp.generate(tree, h, slot); + } + + fn close(dp: *DebugProvider, h: Handle) void { + if (!kindOf(h).isText()) return; + if (dp.findSlot(h)) |s| dp.releaseSlot(s); + } + + fn read(dp: *DebugProvider, tree: Tree, h: Handle, offset: u64, buf: []u8) Error!usize { + const k = kindOf(h); + if (k.isDir()) return error.IsDir; + if (k.isCtl()) return error.Perm; + if (k == .mem_file) { + const addr = valueOf(h) +% offset; + return dp.d.readMem(addr, buf) catch |e| mapErr(e); + } + if (k == .maps) return dp.d.readMaps(offset, buf) catch |e| mapErr(e); + const slot = dp.findSlot(h) orelse return error.Io; + if (offset == 0) try dp.generate(tree, h, slot); + if (offset >= slot.len) return 0; + const off: usize = @intCast(offset); + const n = @min(buf.len, slot.len - off); + @memcpy(buf[0..n], slot.buf[off..][0..n]); + return n; + } + + fn write(dp: *DebugProvider, tree: Tree, h: Handle, offset: u64, data: []const u8) Error!usize { + _ = tree; + const k = kindOf(h); + if (k.isDir()) return error.IsDir; + switch (k) { + .mem_file => { + const addr = valueOf(h) +% offset; + return dp.d.writeMem(addr, data) catch |e| mapErr(e); + }, + .bp_ctl, .panic_ctl => { + const cmd = std.mem.trim(u8, data, " \t\r\n\x00"); + if (!std.mem.eql(u8, cmd, "continue")) return error.Unsupported; + if (k == .bp_ctl) { + dp.d.resumeThread(@intCast(valueOf(h))) catch |e| return mapErr(e); + } else { + dp.d.panicContinue() catch |e| return mapErr(e); + } + return data.len; + }, + else => return error.Perm, + } + } + + // -- snapshots --------------------------------------------------------- + + fn findSlot(dp: *DebugProvider, h: Handle) ?*Slot { + for (&dp.slots) |*s| if (s.refs > 0 and s.handle == h) return s; + return null; + } + + fn takeSlot(dp: *DebugProvider, h: Handle) ?*Slot { + if (dp.findSlot(h)) |s| { + s.refs += 1; + return s; + } + for (&dp.slots) |*s| if (s.refs == 0) { + s.* = .{ .handle = h, .refs = 1 }; + return s; + }; + return null; + } + + fn releaseSlot(_: *DebugProvider, s: *Slot) void { + s.refs -= 1; + } + + /// (Re)generates the text of `h` into `slot`. A text that does not fit is + /// truncated, not an error. + fn generate(dp: *DebugProvider, tree: Tree, h: Handle, slot: *Slot) Error!void { + var w: Writer = .fixed(&slot.buf); + slot.len = 0; + dp.render(tree, h, &w) catch |e| switch (e) { + error.WriteFailed => {}, + else => return mapErr(e), + }; + slot.len = @intCast(w.buffered().len); + } + + fn render(dp: *DebugProvider, tree: Tree, h: Handle, w: *Writer) debug.Error!void { + const d = dp.d; + const v = valueOf(h); + switch (kindOf(h)) { + .readme => w.writeAll(readmeFor(tree)) catch return error.WriteFailed, + .thread_name => try d.threadName(@intCast(v), w), + .thread_stat => try d.threadStat(@intCast(v), w), + .thread_stack => try d.threadStack(@intCast(v), w), + .thread_regs => try d.threadRegs(@intCast(v), w), + .addr_file => try d.resolveAddr(@intCast(v), w), + .hex_file => try d.hexdump(@intCast(v), hex_bytes, w), + .bp_stack => try d.pausedStack(@intCast(v), w), + .bp_regs => try d.pausedRegs(@intCast(v), w), + .panic_message => try d.panicMessage(w), + .panic_stack => try d.panicStack(w), + else => unreachable, + } + } + + fn mapErr(e: debug.Error) Error { + return switch (e) { + error.NoThread, error.NotPaused, error.NoPanic => error.NotFound, + error.Unsupported => error.Unsupported, + error.WriteFailed => error.NoSpace, + error.Timeout, error.Busy, error.Unmapped, error.Unexpected, error.AlreadyInitialized, error.InvalidOptions => error.Io, + }; + } +}; + +// --------------------------------------------------------------------------- +// Tests (through the core's in-memory harness) +// --------------------------------------------------------------------------- + +const testing = std.testing; + +const TestCfg: core.Config = .{ .name = "dbgtest", .msize = 8192, .max_fids = 16, .max_providers = 6, .snapshot_slots = 2, .snapshot_bytes = 512 }; +const TS = core.Server(TestCfg); + +test "debug provider: threads, addr, mem, hex, breakpoints, panic through the core" { + var text_buf: [16 * 1024]u8 = undefined; + var d: debug.Debug = undefined; + try d.init(.{ .io = testing.io, .text_buf = &text_buf }); + defer d.deinit(); + var dp: DebugProvider = undefined; + dp.init(&d); + + var dummy: u8 = 0; + var shared: TS.Shared = .init(&dummy); + try dp.mountAll(&shared); + var storage: TS.Storage = undefined; + var h: TS.Harness = undefined; + try h.init(&shared, &storage); + defer h.deinit(); + + // /threads lists tids only, among them this thread. + const names = try h.listPath(&.{"threads"}); + defer TS.Harness.freeNames(names); + try testing.expect(names.len >= 1); + try testing.expect(!TS.Harness.hasName(names, "README")); + const no_readme = try h.ok(.{ .walk = .{ .fid = 0, .newfid = 7, .names = &.{ "threads", "README" } } }); + try testing.expectEqual(@as(u16, 1), no_readme.walk.nwqid); + var tid_buf: [16]u8 = undefined; + const tid = try std.fmt.bufPrint(&tid_buf, "{d}", .{std.os.linux.gettid()}); + try testing.expect(TS.Harness.hasName(names, tid)); + + // Own stack names this test function's file. + const stack = try h.readPath(&.{ "threads", tid, "stack" }); + defer testing.allocator.free(stack); + try testing.expect(std.mem.indexOf(u8, stack, "#0 0x") != null); + + // /addr/ of a function here resolves to this file. + var addr_buf: [32]u8 = undefined; + const addr_name = try std.fmt.bufPrint(&addr_buf, "{x}", .{@intFromPtr(&DebugProvider.parseTid)}); + const resolved = try h.readPath(&.{ "addr", addr_name }); + defer testing.allocator.free(resolved); + try testing.expect(std.mem.indexOf(u8, resolved, "provider.zig") != null); + const partial = try h.ok(.{ .walk = .{ .fid = 0, .newfid = 5, .names = &.{ "addr", "zzz" } } }); + try testing.expectEqual(@as(u16, 1), partial.walk.nwqid); + try h.walkTo(5, &.{"addr"}); + try h.expectFail(.{ .walk = .{ .fid = 5, .newfid = 6, .names = &.{"zzz"} } }, "file does not exist"); + _ = try h.ok(.{ .clunk = .{ .fid = 5 } }); + + // /mem/ reads and writes live memory; unmapped is an error. + var cell: [8]u8 = "abcdefgh".*; + var mem_buf: [32]u8 = undefined; + const mem_name = try std.fmt.bufPrint(&mem_buf, "{x}", .{@intFromPtr(&cell)}); + try h.walkTo(1, &.{ "mem", mem_name }); + _ = try h.ok(.{ .open = .{ .fid = 1, .mode = cloud9.ordwr } }); + const r = try h.ok(.{ .read = .{ .fid = 1, .offset = 2, .count = 4 } }); + try testing.expectEqualStrings("cdef", r.read); + _ = try h.ok(.{ .write = .{ .fid = 1, .offset = 0, .data = "XY" } }); + try testing.expectEqualStrings("XYcdefgh", &cell); + _ = try h.ok(.{ .clunk = .{ .fid = 1 } }); + try h.walkTo(2, &.{ "mem", "8" }); + _ = try h.ok(.{ .open = .{ .fid = 2, .mode = cloud9.oread } }); + try h.expectFail(.{ .read = .{ .fid = 2, .offset = 0, .count = 4 } }, "i/o error"); + _ = try h.ok(.{ .clunk = .{ .fid = 2 } }); + const maps = try h.readPath(&.{ "mem", "maps" }); + defer testing.allocator.free(maps); + try testing.expect(std.mem.indexOf(u8, maps, "r-xp") != null or std.mem.indexOf(u8, maps, "r--p") != null); + + // /hex/ is a hexdump. + const hex = try h.readPath(&.{ "hex", mem_name }); + defer testing.allocator.free(hex); + try testing.expect(std.mem.indexOf(u8, hex, "XYcdefgh") != null); + + // Nothing paused, no panic. + const bps = try h.listPath(&.{"breakpoints"}); + defer TS.Harness.freeNames(bps); + try testing.expectEqual(@as(usize, 0), bps.len); + const msg = try h.readPath(&.{ "panic", "message" }); + defer testing.allocator.free(msg); + try testing.expectEqualStrings("", msg); + try h.walkTo(3, &.{ "panic", "ctl" }); + _ = try h.ok(.{ .open = .{ .fid = 3, .mode = cloud9.owrite } }); + try h.expectFail(.{ .write = .{ .fid = 3, .offset = 0, .data = "continue" } }, "file does not exist"); + try h.expectFail(.{ .write = .{ .fid = 3, .offset = 0, .data = "bogus" } }, "not supported"); + _ = try h.ok(.{ .clunk = .{ .fid = 3 } }); + + // Snapshot slots are released on clunk: open more files than slots, sequentially. + for (0..4) |_| { + const t = try h.readPath(&.{ "threads", tid, "name" }); + testing.allocator.free(t); + } + for (&dp.slots) |s| try testing.expectEqual(@as(u32, 0), s.refs); +} + +test "handle encoding round-trips" { + const h = mk(.mem_file, 0x7fff_dead_beef); + try testing.expectEqual(Kind.mem_file, kindOf(h)); + try testing.expectEqual(@as(u64, 0x7fff_dead_beef), valueOf(h)); + try testing.expect(h < (1 << 56)); + try testing.expectEqual(@as(?u64, null), DebugProvider.parseHex("1_0")); + try testing.expectEqual(@as(?u64, 0x10), DebugProvider.parseHex("0x10")); + try testing.expectEqual(@as(?u32, null), DebugProvider.parseTid("+5")); +} diff --git a/9proc/src/linux/runtime.zig b/9proc/src/linux/runtime.zig new file mode 100644 index 0000000..503d6c2 --- /dev/null +++ b/9proc/src/linux/runtime.zig @@ -0,0 +1,90 @@ +//! Generators for `Config.runtime`: /runtime/{pid,ppid,uptime,argv,cwd,env,clients}. +//! The core passes every generator `Shared.ctx`; `Fns(Ctx, field)` casts it +//! to `*Ctx` and reads the `Info` stored in `@field(ctx, field)`. +const std = @import("std"); +const linux = std.os.linux; +const Writer = std.Io.Writer; + +/// What the generators report. Texts are borrowed for the server's lifetime. +pub const Info = struct { + /// argv, one argument per line. + argv: []const u8 = "", + /// Environment, one KEY=VALUE per line. + env: []const u8 = "", + cwd: []const u8 = "", + /// Monotonic seconds at startup; /runtime/uptime is the difference. + start_mono: i64 = 0, + /// Live client count, published by the probe. + clients: ?*const std.atomic.Value(u32) = null, + + pub fn now() Info { + return .{ .start_mono = monotonicSecs() }; + } +}; + +pub fn monotonicSecs() i64 { + var ts: linux.timespec = undefined; + _ = linux.clock_gettime(.MONOTONIC, &ts); + return ts.sec; +} + +/// Seconds since the epoch, clamped to u32 (for atime/mtime and `fn/now`). +pub fn realtimeSecs() u32 { + var ts: linux.timespec = undefined; + _ = linux.clock_gettime(.REALTIME, &ts); + return @intCast(std.math.clamp(ts.sec, 0, std.math.maxInt(u32))); +} + +/// The `Config.runtime` type: `Ctx` is the type behind `Shared.ctx`, `field` +/// the name of its `Info` field. +pub fn Fns(comptime Ctx: type, comptime field: []const u8) type { + return struct { + fn info(ctx: *anyopaque) *const Info { + const c: *Ctx = @ptrCast(@alignCast(ctx)); + return &@field(c, field); + } + pub fn pid(_: *anyopaque, w: *Writer) anyerror!void { + try w.print("{d}", .{linux.getpid()}); + } + pub fn ppid(_: *anyopaque, w: *Writer) anyerror!void { + try w.print("{d}", .{linux.getppid()}); + } + pub fn uptime(ctx: *anyopaque, w: *Writer) anyerror!void { + try w.print("{d}", .{monotonicSecs() - info(ctx).start_mono}); + } + pub fn argv(ctx: *anyopaque, w: *Writer) anyerror!void { + try w.writeAll(info(ctx).argv); + } + pub fn cwd(ctx: *anyopaque, w: *Writer) anyerror!void { + try w.writeAll(info(ctx).cwd); + } + pub fn env(ctx: *anyopaque, w: *Writer) anyerror!void { + try w.writeAll(info(ctx).env); + } + pub fn clients(ctx: *anyopaque, w: *Writer) anyerror!void { + const n: u32 = if (info(ctx).clients) |c| c.load(.acquire) else 0; + try w.print("{d}", .{n}); + } + }; +} + +test "runtime generators read Info through the context" { + const Ctx = struct { x: u32, info: Info }; + var count: std.atomic.Value(u32) = .init(3); + var ctx: Ctx = .{ .x = 0, .info = .{ .argv = "a\nb\n", .cwd = "/tmp", .env = "K=V\n", .start_mono = monotonicSecs(), .clients = &count } }; + const F = Fns(Ctx, "info"); + var buf: [64]u8 = undefined; + var w: Writer = .fixed(&buf); + try F.clients(&ctx, &w); + try std.testing.expectEqualStrings("3", w.buffered()); + w = .fixed(&buf); + try F.argv(&ctx, &w); + try std.testing.expectEqualStrings("a\nb\n", w.buffered()); + w = .fixed(&buf); + try F.uptime(&ctx, &w); + try std.testing.expect(w.buffered().len >= 1); + w = .fixed(&buf); + try F.pid(&ctx, &w); + try std.testing.expectEqual(linux.getpid(), try std.fmt.parseInt(i32, w.buffered(), 10)); + try std.testing.expectEqual(@as(usize, 7), @typeInfo(F).@"struct".decls.len); +} diff --git a/9proc/src/root.zig b/9proc/src/root.zig new file mode 100644 index 0000000..5a8e135 --- /dev/null +++ b/9proc/src/root.zig @@ -0,0 +1,24 @@ +//! 9proc: a 9P2000 debug/introspection server as a library. See +//! docs/LIBRARY.md. `core` and `vars` are freestanding; `scratch` takes an +//! allocator; `linux` is the platform layer (only on Linux). +const std = @import("std"); +const builtin = @import("builtin"); + +pub const core = @import("core.zig"); +pub const vars = @import("vars.zig"); +pub const scratch = @import("scratch.zig"); +pub const linux = if (builtin.os.tag == .linux) @import("linux/probe.zig") else struct {}; + +pub const Config = core.Config; +pub const Server = core.Server; +pub const Provider = core.Provider; +pub const NodeStat = core.NodeStat; +pub const Scratch = scratch.Scratch; + +test { + std.testing.refAllDecls(@This()); + _ = core; + _ = vars; + _ = scratch; + if (builtin.os.tag == .linux) _ = linux; +} diff --git a/9proc/src/scratch.zig b/9proc/src/scratch.zig new file mode 100644 index 0000000..2163a28 --- /dev/null +++ b/9proc/src/scratch.zig @@ -0,0 +1,689 @@ +//! An in-memory read/write tree as a `Provider`: create, write, truncate, +//! rename, remove, mkdir, DMAPPEND, DMEXCL. The one core-level component that +//! takes an `Allocator` (nodes and file contents live on it); it is optional. +//! +//! Nodes are kept alive by `refs` (fids holding a handle) after removal, so a +//! handle stays valid until the core clunks it. Handles are node addresses; +//! the root is handle 0. Not internally synchronized (like `Shared`). +const std = @import("std"); +const cloud9 = @import("cloud9"); +const core = @import("core.zig"); +const Allocator = std.mem.Allocator; +const Provider = core.Provider; +const Handle = Provider.Handle; +const Error = Provider.Error; +const NodeStat = core.NodeStat; + +/// A node of the tree. +pub const Node = struct { + name: []u8, + path: u64, + version: u32 = 0, + mode: u32, + atime: u32, + mtime: u32, + data: std.ArrayList(u8) = .empty, + children: std.ArrayList(*Node) = .empty, + parent: ?*Node, + /// Handles held by the core. + refs: u32 = 0, + /// Fids currently open on this node (DMEXCL admits at most one). + opens: u32 = 0, + removed: bool = false, + + pub fn isDir(n: *const Node) bool { + return n.mode & cloud9.dmdir != 0; + } + + fn find(n: *const Node, name: []const u8) ?*Node { + for (n.children.items) |ch| if (std.mem.eql(u8, ch.name, name)) return ch; + return null; + } +}; + +/// Seconds since the epoch, for atime/mtime; the default clock reports 0. +pub const Clock = *const fn () u32; + +fn zeroClock() u32 { + return 0; +} + +pub const Scratch = struct { + gpa: Allocator, + root: *Node, + /// Qid paths are a counter, never reused: the root is 0 (the provider + /// root handle), so a removed-and-recreated file gets a fresh identity + /// even when the allocator hands back the same address. + next_path: u64 = 0, + /// Sum of all file lengths, bounded by `budget`. + bytes: usize = 0, + /// Largest total of file contents across all files. + budget: usize, + /// Largest single file; defaults to the budget. + max_file: usize, + /// The time source for atime/mtime (a platform layer sets it). + now: Clock = &zeroClock, + + /// The tree's only allocation policy: every node and every file's + /// contents come from `gpa`, and no file content ever exceeds `budget_bytes` + /// in total. + pub fn init(gpa: Allocator, budget_bytes: usize) Allocator.Error!Scratch { + var s: Scratch = .{ .gpa = gpa, .root = undefined, .budget = budget_bytes, .max_file = budget_bytes }; + s.root = try s.newNode("", cloud9.dmdir | 0o777, null); + return s; + } + + pub fn deinit(s: *Scratch) void { + s.destroyTree(s.root); + s.* = undefined; + } + + /// The provider to mount, at `/`. + pub fn provider(s: *Scratch, name: []const u8) Provider { + return .{ .name = name, .ctx = s, .vtable = &vtable }; + } + + pub const vtable: Provider.VTable = .{ + .walk = &walk, + .stat = &stat, + .list = &list, + .open = &open, + .read = &read, + .write = &write, + .create = &create, + .remove = &remove, + .wstat = &wstat, + .close = &close, + .clunk = &clunk, + }; + + // -- node management -- + + fn destroyTree(s: *Scratch, n: *Node) void { + for (n.children.items) |ch| s.destroyTree(ch); + n.children.clearRetainingCapacity(); + n.removed = true; + if (n.refs == 0 or n == s.root) s.destroyNode(n); + } + + fn destroyNode(s: *Scratch, n: *Node) void { + s.bytes -= n.data.items.len; + s.gpa.free(n.name); + n.data.deinit(s.gpa); + n.children.deinit(s.gpa); + s.gpa.destroy(n); + } + + fn newNode(s: *Scratch, name: []const u8, mode: u32, parent: ?*Node) Allocator.Error!*Node { + const n = try s.gpa.create(Node); + errdefer s.gpa.destroy(n); + const t = s.now(); + n.* = .{ + .name = try s.gpa.dupe(u8, name), + .path = s.next_path, + .mode = mode, + .atime = t, + .mtime = t, + .parent = parent, + }; + errdefer s.gpa.free(n.name); + if (parent) |p| try p.children.append(s.gpa, n); + s.next_path += 1; + return n; + } + + /// Sets a file's length, zero-filling growth and charging the budget. + /// Shrinking releases the memory so a truncated file costs nothing. + fn resizeData(s: *Scratch, n: *Node, new_len: usize) Error!void { + const old = n.data.items.len; + if (new_len > old) { + if (new_len > s.max_file) return error.NoSpace; + if (s.bytes + (new_len - old) > s.budget) return error.NoSpace; + n.data.resize(s.gpa, new_len) catch return error.NoSpace; + @memset(n.data.items[old..new_len], 0); + s.bytes += new_len - old; + } else if (new_len < old) { + n.data.shrinkAndFree(s.gpa, new_len); + s.bytes -= old - new_len; + } + } + + fn touch(s: *Scratch, n: *Node) void { + n.version +%= 1; + n.mtime = s.now(); + } + + fn self(ctx: *anyopaque) *Scratch { + return @ptrCast(@alignCast(ctx)); + } + + fn handle(s: *Scratch, n: *Node) Handle { + return if (n == s.root) Provider.root else @intFromPtr(n); + } + + fn node(s: *Scratch, h: Handle) *Node { + return if (h == Provider.root) s.root else @ptrFromInt(@as(usize, @intCast(h))); + } + + /// A handle the core will clunk exactly once. + fn retain(s: *Scratch, n: *Node) Handle { + if (n != s.root) n.refs += 1; + return s.handle(n); + } + + fn release(s: *Scratch, n: *Node) void { + if (n == s.root) return; + n.refs -= 1; + if (n.refs == 0 and n.removed) s.destroyNode(n); + } + + fn fillStat(n: *const Node, h: Handle, out: *NodeStat) void { + out.* = .{ + .mode = n.mode, + .length = if (n.isDir()) 0 else n.data.items.len, + .atime = n.atime, + .mtime = n.mtime, + .version = n.version, + .name = n.name, + .handle = h, + .path = n.path, + }; + } + + // -- the vtable -- + + fn walk(ctx: *anyopaque, parent: Handle, name: []const u8) Error!Handle { + const s = self(ctx); + const p = s.node(parent); + if (std.mem.eql(u8, name, ".")) return s.retain(p); + if (p.removed) return error.NotFound; + if (!p.isDir()) return error.NotDir; + if (std.mem.eql(u8, name, "..")) return s.retain(p.parent orelse s.root); + return s.retain(p.find(name) orelse return error.NotFound); + } + + fn stat(ctx: *anyopaque, h: Handle, out: *NodeStat) Error!void { + const s = self(ctx); + fillStat(s.node(h), h, out); + } + + fn list(ctx: *anyopaque, dir: Handle, index: usize, out: *NodeStat) Error!bool { + const s = self(ctx); + const d = s.node(dir); + if (!d.isDir()) return error.NotDir; + if (index >= d.children.items.len) return false; + const ch = d.children.items[index]; + fillStat(ch, s.handle(ch), out); + return true; + } + + fn open(ctx: *anyopaque, h: Handle, mode: u8) Error!void { + const s = self(ctx); + const n = s.node(h); + if (n.removed) return error.NotFound; + const acc = mode & 3; + const want_write = acc == cloud9.owrite or acc == cloud9.ordwr; + const want_read = !want_write or acc == cloud9.ordwr; + const trunc = mode & cloud9.otrunc != 0; + if (n.isDir()) { + if (want_write or trunc) return error.IsDir; + if (n.mode & 0o400 == 0) return error.Perm; + } else { + if (want_read and n.mode & 0o400 == 0) return error.Perm; + if ((want_write or trunc) and n.mode & 0o200 == 0) return error.Perm; + if (n.mode & cloud9.dmexcl != 0 and n.opens != 0) return error.Excl; + if (trunc and n.mode & cloud9.dmappend == 0) { + s.resizeData(n, 0) catch unreachable; // shrinking cannot fail + s.touch(n); + } + } + n.opens += 1; + } + + fn close(ctx: *anyopaque, h: Handle) void { + const s = self(ctx); + s.node(h).opens -= 1; + } + + fn read(ctx: *anyopaque, h: Handle, offset: u64, buf: []u8) Error!usize { + const s = self(ctx); + const n = s.node(h); + if (n.isDir()) return error.IsDir; + const src = n.data.items; + if (offset >= src.len) return 0; + const off: usize = @intCast(offset); + const len = @min(buf.len, src.len - off); + @memcpy(buf[0..len], src[off..][0..len]); + return len; + } + + fn write(ctx: *anyopaque, h: Handle, offset: u64, data: []const u8) Error!usize { + const s = self(ctx); + const n = s.node(h); + if (n.isDir()) return error.IsDir; + // A zero-length write changes nothing (and must not extend the file). + if (data.len == 0) return 0; + const off: usize = if (n.mode & cloud9.dmappend != 0) n.data.items.len else @intCast(@min(offset, s.max_file)); + const end = off + data.len; + if (end > s.max_file) return error.NoSpace; + if (end > n.data.items.len) try s.resizeData(n, end); + @memcpy(n.data.items[off..end], data); + s.touch(n); + return data.len; + } + + fn create(ctx: *anyopaque, dir: Handle, name: []const u8, perm: u32, mode: u8) Error!Handle { + const s = self(ctx); + const d = s.node(dir); + if (d.removed) return error.NotFound; + if (!d.isDir()) return error.NotDir; + if (d.mode & 0o200 == 0) return error.Perm; + if (d.find(name) != null) return error.Exists; + const is_dir = perm & cloud9.dmdir != 0; + const inherit: u32 = if (is_dir) d.mode & 0o777 else d.mode & 0o666; + const n = s.newNode(name, perm & (~@as(u32, 0o777) | inherit), d) catch return error.NoSpace; + s.touch(d); + n.opens += 1; + _ = mode; + return s.retain(n); + } + + fn remove(ctx: *anyopaque, h: Handle) Error!void { + const s = self(ctx); + const n = s.node(h); + if (n.removed) return error.NotFound; + const parent = n.parent orelse return error.Perm; + if (parent.mode & 0o200 == 0) return error.Perm; + if (n.isDir() and n.children.items.len != 0) return error.NotEmpty; + const i = std.mem.indexOfScalar(*Node, parent.children.items, n) orelse return error.NotFound; + _ = parent.children.orderedRemove(i); + s.touch(parent); + n.removed = true; + if (n.refs == 0) s.destroyNode(n); + } + + fn wstat(ctx: *anyopaque, h: Handle, st: *const cloud9.Stat) Error!void { + const s = self(ctx); + const n = s.node(h); + if (n.removed) return error.NotFound; + // Validate everything before changing anything. + const rename = st.name.len != 0 and !std.mem.eql(u8, st.name, n.name); + if (rename) { + const parent = n.parent orelse return error.Perm; + if (parent.find(st.name) != null) return error.Exists; + } + const cur_len: u64 = if (n.isDir()) 0 else n.data.items.len; + const set_len = st.length != 0xFFFF_FFFF_FFFF_FFFF and st.length != cur_len; + if (set_len) { + if (n.isDir()) return error.IsDir; + if (st.length > s.max_file) return error.NoSpace; + } + const set_mode = st.mode != 0xFFFF_FFFF and st.mode != n.mode; + if (set_mode and (st.mode & cloud9.dmdir) != (n.mode & cloud9.dmdir)) return error.Perm; + const set_mtime = st.mtime != 0xFFFF_FFFF and st.mtime != n.mtime; + if (!(rename or set_len or set_mode or set_mtime)) return; + const new_name: ?[]u8 = if (rename) s.gpa.dupe(u8, st.name) catch return error.NoSpace else null; + errdefer if (new_name) |nn| s.gpa.free(nn); + if (set_len) try s.resizeData(n, @intCast(st.length)); + // Nothing below can fail. + if (new_name) |nn| { + s.gpa.free(n.name); + n.name = nn; + s.touch(n.parent.?); + } + if (set_mode) n.mode = (n.mode & cloud9.dmdir) | (st.mode & ~cloud9.dmdir); + s.touch(n); + if (set_mtime) n.mtime = st.mtime; + } + + fn clunk(ctx: *anyopaque, h: Handle) void { + const s = self(ctx); + s.release(s.node(h)); + } +}; + +// --------------------------------------------------------------------------- +// Tests: the scratch tree mounted at /scratch of a core server. +// --------------------------------------------------------------------------- + +const testing = std.testing; + +const test_cfg: core.Config = .{ .name = "tester", .msize = 8192, .max_fids = 32 }; +const TS = core.Server(test_cfg); + +const budget: usize = 1 << 20; + +const Fixture = struct { + ctx: u8 = 0, + shared: TS.Shared = undefined, + storage: TS.Storage = undefined, + scratch: Scratch = undefined, + h: TS.Harness = undefined, + + fn init(x: *Fixture) !void { + x.shared = .init(&x.ctx); + x.scratch = try Scratch.init(testing.allocator, budget); + errdefer x.scratch.deinit(); + try x.shared.addProvider(x.scratch.provider("scratch")); + try x.h.init(&x.shared, &x.storage); + } + + fn deinit(x: *Fixture) void { + x.h.deinit(); + x.scratch.deinit(); + } + + fn nodeOf(x: *Fixture, fid: u32) *Node { + for (x.h.conn.fids) |f| if (f.used and f.id == fid) return x.scratch.node(f.node.prov.h); + unreachable; + } +}; + +const dontcare = core.stat_dontcare; + +test "scratch create/write/read/rename/truncate/remove" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + try x.h.walkTo(1, &.{"scratch"}); + // create + write + const cr = try x.h.ok(.{ .create = .{ .fid = 1, .name = "x", .perm = 0o644, .mode = cloud9.ordwr } }); + try testing.expectEqual(cloud9.qtfile, cr.create.qid.type); + _ = try x.h.ok(.{ .write = .{ .fid = 1, .offset = 0, .data = "hello" } }); + _ = try x.h.ok(.{ .write = .{ .fid = 1, .offset = 5, .data = " world" } }); + const r = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = 100 } }); + try testing.expectEqualStrings("hello world", r.read); + _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); + // rename x -> y + try x.h.walkTo(2, &.{ "scratch", "x" }); + var st = dontcare; + st.name = "y"; + _ = try x.h.ok(.{ .wstat = .{ .fid = 2, .stat = st } }); + try x.h.walkTo(3, &.{"scratch"}); + try x.h.expectFail(.{ .walk = .{ .fid = 3, .newfid = 30, .names = &.{"x"} } }, "file does not exist"); + _ = try x.h.ok(.{ .clunk = .{ .fid = 3 } }); + try x.h.walkTo(3, &.{ "scratch", "y" }); + // truncate then extend with zero fill + st = dontcare; + st.length = 2; + _ = try x.h.ok(.{ .wstat = .{ .fid = 3, .stat = st } }); + st.length = 4; + _ = try x.h.ok(.{ .wstat = .{ .fid = 3, .stat = st } }); + const text = try x.h.readAll(3); + defer testing.allocator.free(text); + try testing.expectEqualStrings("he\x00\x00", text); + const s3 = try x.h.ok(.{ .stat = .{ .fid = 3 } }); + try testing.expectEqualStrings("y", s3.stat.name); + try testing.expectEqual(@as(u64, 4), s3.stat.length); + try testing.expectEqualStrings("tester", s3.stat.uid); + _ = try x.h.ok(.{ .clunk = .{ .fid = 3 } }); + _ = try x.h.ok(.{ .clunk = .{ .fid = 2 } }); + // mkdir, nested create, remove rules + try x.h.walkTo(4, &.{"scratch"}); + const dr = try x.h.ok(.{ .create = .{ .fid = 4, .name = "d", .perm = cloud9.dmdir | 0o755, .mode = cloud9.oread } }); + try testing.expectEqual(cloud9.qtdir, dr.create.qid.type); + _ = try x.h.ok(.{ .clunk = .{ .fid = 4 } }); + try x.h.walkTo(5, &.{ "scratch", "d" }); + _ = try x.h.ok(.{ .create = .{ .fid = 5, .name = "inner", .perm = 0o600, .mode = cloud9.owrite } }); + _ = try x.h.ok(.{ .write = .{ .fid = 5, .offset = 0, .data = "z" } }); + _ = try x.h.ok(.{ .clunk = .{ .fid = 5 } }); + try x.h.walkTo(6, &.{ "scratch", "d" }); + try x.h.expectFail(.{ .remove = .{ .fid = 6 } }, "directory not empty"); + try x.h.expectFail(.{ .clunk = .{ .fid = 6 } }, "unknown fid"); // remove always clunks + try x.h.walkTo(7, &.{ "scratch", "d", "inner" }); + _ = try x.h.ok(.{ .remove = .{ .fid = 7 } }); + try x.h.walkTo(8, &.{ "scratch", "d" }); + _ = try x.h.ok(.{ .remove = .{ .fid = 8 } }); + try x.h.walkTo(9, &.{ "scratch", "y" }); + _ = try x.h.ok(.{ .remove = .{ .fid = 9 } }); + try x.h.walkTo(10, &.{"scratch"}); + _ = try x.h.ok(.{ .open = .{ .fid = 10, .mode = cloud9.oread } }); + const names = try x.h.listDir(10, 1024); + defer testing.allocator.free(names); + try testing.expectEqual(@as(usize, 0), names.len); + // append-only files ignore the offset + try x.h.walkTo(11, &.{"scratch"}); + _ = try x.h.ok(.{ .create = .{ .fid = 11, .name = "log", .perm = cloud9.dmappend | 0o644, .mode = cloud9.ordwr } }); + _ = try x.h.ok(.{ .write = .{ .fid = 11, .offset = 100, .data = "a" } }); + _ = try x.h.ok(.{ .write = .{ .fid = 11, .offset = 0, .data = "b" } }); + const lr = try x.h.ok(.{ .read = .{ .fid = 11, .offset = 0, .count = 10 } }); + try testing.expectEqualStrings("ab", lr.read); + try testing.expect(lr.read.len == 2); + const ls = try x.h.ok(.{ .stat = .{ .fid = 11 } }); + try testing.expect(ls.stat.qid.type & cloud9.qtappend != 0); + // the scratch root cannot be removed + try x.h.walkTo(12, &.{"scratch"}); + try x.h.expectFail(.{ .remove = .{ .fid = 12 } }, "permission denied"); +} + +test "walk of a missing name and walking a file" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + try x.h.walkTo(1, &.{"scratch"}); + try x.h.expectFail(.{ .walk = .{ .fid = 1, .newfid = 2, .names = &.{"nope"} } }, "file does not exist"); + _ = try x.h.ok(.{ .create = .{ .fid = 1, .name = "f", .perm = 0o644, .mode = cloud9.oread } }); + _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); + try x.h.walkTo(3, &.{ "scratch", "f" }); + try x.h.expectFail(.{ .walk = .{ .fid = 3, .newfid = 4, .names = &.{"x"} } }, "not a directory"); + // a walk that fails past the first element is a partial Rwalk that leaves newfid unused + const r = try x.h.ok(.{ .walk = .{ .fid = 0, .newfid = 4, .names = &.{ "scratch", "nope", "x" } } }); + try testing.expectEqual(@as(u16, 1), r.walk.nwqid); + try x.h.expectFail(.{ .clunk = .{ .fid = 4 } }, "unknown fid"); + // .. from a file is not a directory; .. from the scratch root reaches the server root + try x.h.expectFail(.{ .walk = .{ .fid = 3, .newfid = 5, .names = &.{".."} } }, "not a directory"); + try x.h.walkTo(5, &.{"scratch"}); + const up = try x.h.ok(.{ .walk = .{ .fid = 5, .newfid = 6, .names = &.{ "..", "scratch", "..", "README" } } }); + try testing.expectEqual(@as(u16, 4), up.walk.nwqid); + try testing.expectEqual(@as(u64, 0), up.walk.wqid[1].path); // provider 0, root + try testing.expect(up.walk.wqid[0].type & cloud9.qtdir != 0); +} + +test "directory read across consecutive offsets returns every record exactly once" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + const n = 40; + for (0..n) |i| { + try x.h.walkTo(1, &.{"scratch"}); + var name_buf: [64]u8 = undefined; + const name = try std.fmt.bufPrint(&name_buf, "file-with-a-long-name-{d:0>3}", .{i}); + _ = try x.h.ok(.{ .create = .{ .fid = 1, .name = name, .perm = 0o644, .mode = cloud9.oread } }); + _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); + } + try x.h.walkTo(2, &.{"scratch"}); + _ = try x.h.ok(.{ .open = .{ .fid = 2, .mode = cloud9.oread } }); + // 200 bytes fits two records, so this takes many reads. + const names = try x.h.listDir(2, 200); + defer TS.Harness.freeNames(names); + try testing.expectEqual(@as(usize, n), names.len); + var seen: [n]bool = @splat(false); + for (names) |nm| { + const idx = try std.fmt.parseInt(usize, nm[nm.len - 3 ..], 10); + try testing.expect(!seen[idx]); + seen[idx] = true; + } + for (seen) |s| try testing.expect(s); + try x.h.expectFail(.{ .read = .{ .fid = 2, .offset = 7, .count = 200 } }, "bad offset"); + // a read that cannot fit even one record returns nothing rather than splitting it + const tiny = try x.h.ok(.{ .read = .{ .fid = 2, .offset = 0, .count = 30 } }); + try testing.expectEqual(@as(usize, 0), tiny.read.len); + _ = try x.h.ok(.{ .clunk = .{ .fid = 2 } }); + // Tversion resets every fid and every reference + try x.h.version(8192); + try testing.expectEqual(@as(usize, 0), x.h.conn.fidCount()); + for (x.scratch.root.children.items) |ch| try testing.expectEqual(@as(u32, 0), ch.refs); + _ = try x.h.ok(.{ .attach = .{ .fid = 0, .uname = "tester" } }); +} + +test "DMEXCL admits one open fid at a time" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + try x.h.walkTo(1, &.{"scratch"}); + const cr = try x.h.ok(.{ .create = .{ .fid = 1, .name = "lock", .perm = cloud9.dmexcl | 0o644, .mode = cloud9.owrite } }); + try testing.expect(cr.create.qid.type & cloud9.qtexcl != 0); + try x.h.walkTo(2, &.{ "scratch", "lock" }); + try x.h.expectFail(.{ .open = .{ .fid = 2, .mode = cloud9.oread } }, "exclusive use file already open"); + _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); + _ = try x.h.ok(.{ .open = .{ .fid = 2, .mode = cloud9.oread } }); + try x.h.walkTo(3, &.{ "scratch", "lock" }); + try x.h.expectFail(.{ .open = .{ .fid = 3, .mode = cloud9.oread } }, "exclusive use file already open"); + // a Tversion reset drops the open and frees the file for the next session + try x.h.version(8192); + _ = try x.h.ok(.{ .attach = .{ .fid = 0, .uname = "tester" } }); + try x.h.walkTo(4, &.{ "scratch", "lock" }); + _ = try x.h.ok(.{ .open = .{ .fid = 4, .mode = cloud9.oread } }); + try testing.expectEqual(@as(u32, 1), x.nodeOf(4).opens); + _ = try x.h.ok(.{ .remove = .{ .fid = 4 } }); +} + +test "scratch memory: zero-length writes, truncation frees, global budget" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + x.scratch.max_file = 4096; + try x.h.walkTo(1, &.{"scratch"}); + _ = try x.h.ok(.{ .create = .{ .fid = 1, .name = "f", .perm = 0o644, .mode = cloud9.ordwr } }); + // a zero-length write at a huge offset must not extend the file + const w0 = try x.h.ok(.{ .write = .{ .fid = 1, .offset = std.math.maxInt(u64), .data = "" } }); + try testing.expectEqual(@as(u32, 0), w0.write); + var st = try x.h.ok(.{ .stat = .{ .fid = 1 } }); + try testing.expectEqual(@as(u64, 0), st.stat.length); + // growth is charged to the budget; truncation releases it (memory too) + _ = try x.h.ok(.{ .write = .{ .fid = 1, .offset = 1000, .data = "x" } }); + try testing.expectEqual(@as(usize, 1001), x.scratch.bytes); + var ws = dontcare; + ws.length = 10; + _ = try x.h.ok(.{ .wstat = .{ .fid = 1, .stat = ws } }); + try testing.expectEqual(@as(usize, 10), x.scratch.bytes); + try testing.expectEqual(@as(usize, 10), x.nodeOf(1).data.capacity); + // per-file cap and the global budget both answer "no space" + try x.h.expectFail(.{ .write = .{ .fid = 1, .offset = 4096, .data = "x" } }, "no space left on device"); + try x.h.expectFail(.{ .write = .{ .fid = 1, .offset = std.math.maxInt(u64), .data = "x" } }, "no space left on device"); + x.scratch.bytes = budget - 10; // pretend other files hold the rest + try x.h.expectFail(.{ .write = .{ .fid = 1, .offset = 10, .data = "0123456789A" } }, "no space left on device"); + _ = try x.h.ok(.{ .write = .{ .fid = 1, .offset = 10, .data = "0123456789" } }); + try testing.expectEqual(budget, x.scratch.bytes); + ws.length = 4096; + try x.h.expectFail(.{ .wstat = .{ .fid = 1, .stat = ws } }, "no space left on device"); + x.scratch.bytes -= budget - 20; + // OTRUNC releases too + try x.h.walkTo(2, &.{ "scratch", "f" }); + _ = try x.h.ok(.{ .open = .{ .fid = 2, .mode = cloud9.owrite | cloud9.otrunc } }); + try testing.expectEqual(@as(usize, 0), x.scratch.bytes); + st = try x.h.ok(.{ .stat = .{ .fid = 2 } }); + try testing.expectEqual(@as(u64, 0), st.stat.length); + // removing a file with content returns its bytes once the last fid lets go + _ = try x.h.ok(.{ .write = .{ .fid = 2, .offset = 0, .data = "abc" } }); + try testing.expectEqual(@as(usize, 3), x.scratch.bytes); + _ = try x.h.ok(.{ .remove = .{ .fid = 2 } }); + try testing.expectEqual(@as(usize, 3), x.scratch.bytes); // fid 1 still holds it + const r = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = 10 } }); + try testing.expectEqualStrings("abc", r.read); + _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); + try testing.expectEqual(@as(usize, 0), x.scratch.bytes); +} + +test "wstat with every field equal to the current stat changes nothing" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + try x.h.walkTo(1, &.{"scratch"}); + _ = try x.h.ok(.{ .create = .{ .fid = 1, .name = "same", .perm = 0o640, .mode = cloud9.oread } }); + const before = (try x.h.ok(.{ .stat = .{ .fid = 1 } })).stat; + var copy = before; + var name_buf: [core.max_name]u8 = undefined; + @memcpy(name_buf[0..before.name.len], before.name); + copy.name = name_buf[0..before.name.len]; + copy.uid = "tester"; + copy.gid = "tester"; + copy.muid = "tester"; + _ = try x.h.ok(.{ .wstat = .{ .fid = 1, .stat = copy } }); + const after = (try x.h.ok(.{ .stat = .{ .fid = 1 } })).stat; + try testing.expectEqual(before.qid, after.qid); + try testing.expectEqual(before.mtime, after.mtime); + try testing.expectEqual(before.mode, after.mode); + try testing.expectEqualStrings("same", after.name); + // and a rename to the very same name is also a no-op + var st = dontcare; + st.name = "same"; + _ = try x.h.ok(.{ .wstat = .{ .fid = 1, .stat = st } }); + try testing.expectEqual(before.qid, (try x.h.ok(.{ .stat = .{ .fid = 1 } })).stat.qid); + // renaming onto an existing sibling is refused + _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); + try x.h.walkTo(2, &.{"scratch"}); + _ = try x.h.ok(.{ .create = .{ .fid = 2, .name = "other", .perm = 0o640, .mode = cloud9.oread } }); + st.name = "same"; + try x.h.expectFail(.{ .wstat = .{ .fid = 2, .stat = st } }, "file already exists"); + // the mode's directory bit is immutable, mtime is settable + st = dontcare; + st.mode = cloud9.dmdir | 0o640; + try x.h.expectFail(.{ .wstat = .{ .fid = 2, .stat = st } }, "permission denied"); + st = dontcare; + st.mtime = 12345; + _ = try x.h.ok(.{ .wstat = .{ .fid = 2, .stat = st } }); + try testing.expectEqual(@as(u32, 12345), (try x.h.ok(.{ .stat = .{ .fid = 2 } })).stat.mtime); + _ = try x.h.ok(.{ .remove = .{ .fid = 2 } }); +} + +test "ORCLOSE removes on clunk and removed files stay readable through open fids" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + try x.h.walkTo(1, &.{"scratch"}); + _ = try x.h.ok(.{ .create = .{ .fid = 1, .name = "tmp", .perm = 0o644, .mode = cloud9.ordwr | cloud9.orclose } }); + _ = try x.h.ok(.{ .write = .{ .fid = 1, .offset = 0, .data = "gone" } }); + try x.h.walkTo(2, &.{ "scratch", "tmp" }); + _ = try x.h.ok(.{ .open = .{ .fid = 2, .mode = cloud9.oread } }); + _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); + try x.h.walkTo(3, &.{"scratch"}); + try x.h.expectFail(.{ .walk = .{ .fid = 3, .newfid = 4, .names = &.{"tmp"} } }, "file does not exist"); + const r = try x.h.ok(.{ .read = .{ .fid = 2, .offset = 0, .count = 10 } }); + try testing.expectEqualStrings("gone", r.read); + try testing.expectEqual(@as(usize, 4), x.scratch.bytes); + _ = try x.h.ok(.{ .clunk = .{ .fid = 2 } }); + try testing.expectEqual(@as(usize, 0), x.scratch.bytes); + // create inside a removed directory fails + _ = try x.h.ok(.{ .create = .{ .fid = 3, .name = "d", .perm = cloud9.dmdir | 0o755, .mode = cloud9.oread } }); + try x.h.walkTo(5, &.{ "scratch", "d" }); + _ = try x.h.ok(.{ .remove = .{ .fid = 5 } }); + _ = try x.h.ok(.{ .clunk = .{ .fid = 3 } }); + try x.h.walkTo(6, &.{"scratch"}); + try x.h.expectFail(.{ .walk = .{ .fid = 6, .newfid = 7, .names = &.{"d"} } }, "file does not exist"); + try x.h.expectFail(.{ .create = .{ .fid = 5, .name = "x", .perm = 0o644, .mode = cloud9.oread } }, "unknown fid"); // remove clunked it +} + +test "qid paths are stable identities, not addresses: remove + recreate differ" { + var x: Fixture = .{}; + try x.init(); + defer x.deinit(); + try x.h.walkTo(1, &.{"scratch"}); + const a = try x.h.ok(.{ .create = .{ .fid = 1, .name = "f", .perm = 0o644, .mode = cloud9.oread } }); + const path_a = a.create.qid.path; + try testing.expectEqual(x.nodeOf(1).path, path_a & ((1 << 56) - 1)); + try testing.expect(path_a != 0); + // a rename keeps the identity + var st = dontcare; + st.name = "g"; + _ = try x.h.ok(.{ .wstat = .{ .fid = 1, .stat = st } }); + try testing.expectEqual(path_a, (try x.h.ok(.{ .stat = .{ .fid = 1 } })).stat.qid.path); + _ = try x.h.ok(.{ .remove = .{ .fid = 1 } }); + // the allocator very likely reuses the freed node's address here + try x.h.walkTo(2, &.{"scratch"}); + const b = try x.h.ok(.{ .create = .{ .fid = 2, .name = "f", .perm = 0o644, .mode = cloud9.oread } }); + try testing.expect(b.create.qid.path != path_a); + _ = try x.h.ok(.{ .remove = .{ .fid = 2 } }); + // the scratch root keeps path 0 (its handle), like every provider root + try x.h.walkTo(3, &.{"scratch"}); + try testing.expectEqual(@as(u64, 0), (try x.h.ok(.{ .stat = .{ .fid = 3 } })).stat.qid.path); + // directory listing reports the same identities as walking + _ = try x.h.ok(.{ .create = .{ .fid = 3, .name = "listed", .perm = 0o644, .mode = cloud9.oread } }); + const via_create = (try x.h.ok(.{ .stat = .{ .fid = 3 } })).stat.qid; + try x.h.walkTo(4, &.{"scratch"}); + _ = try x.h.ok(.{ .open = .{ .fid = 4, .mode = cloud9.oread } }); + const r = try x.h.ok(.{ .read = .{ .fid = 4, .offset = 0, .count = 1024 } }); + const listed = try cloud9.Stat.decode(r.read[0 .. std.mem.readInt(u16, r.read[0..2], .little) + 2]); + try testing.expectEqual(via_create, listed.qid); + _ = try x.h.ok(.{ .remove = .{ .fid = 3 } }); +} diff --git a/9proc/src/vars.zig b/9proc/src/vars.zig new file mode 100644 index 0000000..20cadfc --- /dev/null +++ b/9proc/src/vars.zig @@ -0,0 +1,824 @@ +//! Comptime value renderers for /vars. For a type T, `vtableFor(T)` builds (at +//! comptime) a flat table of the files and directories that describe a value of +//! that type: +//! +//! /value rendered text /type @typeName +//! /size @sizeOf /addr 0x… +//! /raw the bytes /f//... recursively (depth <= max_depth) +//! +//! The core serves a variable by walking this table; a node index is the whole +//! state it needs. Rendering writes into a `*std.Io.Writer` and never allocates. +//! Writes to scalar `value` files are plain stores (not atomic). +const std = @import("std"); +const Writer = std.Io.Writer; + +/// Deepest `f/` nesting: /vars/x/f/a/f/b/f/c/f/d/value is depth 4. +pub const max_depth: u8 = 4; +/// Longest string rendered from a `[]const u8` / `[*:0]const u8` before "…". +pub const max_string: usize = 256; +/// Most array/slice elements rendered before "…". +pub const max_elems: usize = 64; + +pub const Kind = enum(u8) { + /// The directory of a value: value, type, size, addr, raw, [f]. + dir, + /// Rendered text; writable when `set` is non-null. + value, + /// @typeName, static content. + type_name, + /// @sizeOf, static content. + size, + /// "0x…" of the value's address. + addr, + /// The bytes of the value, length = size. + raw, + /// The `f` directory: one `dir` per struct field. + fields, +}; + +pub const RenderFn = *const fn (base: [*]const u8, w: *Writer) Writer.Error!void; +pub const SetFn = *const fn (base: [*]u8, text: []const u8) SetError!void; +pub const SetError = error{ Invalid, Unsupported }; + +/// One node of a type's tree. Children occupy `first..first+count`. +pub const Node = struct { + name: []const u8, + kind: Kind, + parent: u32, + first: u32 = 0, + count: u32 = 0, + /// Byte offset of the described value from the variable's base address. + offset: usize, + /// @sizeOf the described value. + size: usize, + /// Static text for `type_name` and `size` leaves. + content: []const u8 = "", + render: ?RenderFn = null, + set: ?SetFn = null, + + pub fn isDir(n: Node) bool { + return n.kind == .dir or n.kind == .fields; + } + + pub fn writable(n: Node) bool { + return n.kind == .value and n.set != null; + } +}; + +pub const VTable = struct { + nodes: []const Node, + type_name: []const u8, + size: usize, + + /// The child of `dir` named `name`, if any. + pub fn child(vt: *const VTable, dir: u32, name: []const u8) ?u32 { + const d = vt.nodes[dir]; + for (d.first..d.first + d.count) |i| { + if (std.mem.eql(u8, vt.nodes[i].name, name)) return @intCast(i); + } + return null; + } +}; + +/// The comptime-generated table for `T`; the same pointer for the same `T`. +pub fn vtableFor(comptime T: type) *const VTable { + const S = struct { + const nodes = buildTable(T); + const vt: VTable = .{ .nodes = &nodes, .type_name = @typeName(T), .size = @sizeOf(T) }; + }; + return &S.vt; +} + +/// Whether a struct's fields get an `f/` directory at this depth. +fn hasFields(comptime T: type, depth: u8) bool { + if (depth >= max_depth) return false; + return switch (@typeInfo(T)) { + .@"struct" => |s| s.layout != .@"packed" and fieldCount(T) > 0, + else => false, + }; +} + +fn fieldCount(comptime T: type) usize { + var n: usize = 0; + for (@typeInfo(T).@"struct".fields) |f| { + if (!f.is_comptime and @sizeOf(f.type) != 0) n += 1; + } + return n; +} + +fn countNodes(comptime T: type, depth: u8) usize { + var n: usize = 6; // dir + value, type, size, addr, raw + if (hasFields(T, depth)) { + n += 1; // f + for (@typeInfo(T).@"struct".fields) |f| { + if (f.is_comptime or @sizeOf(f.type) == 0) continue; + n += countNodes(f.type, depth + 1); + } + } + return n; +} + +fn fill(nodes: []Node, next: *usize, idx: usize, comptime T: type, name: []const u8, offset: usize, depth: u8, parent: u32) void { + const with_fields = hasFields(T, depth); + const count: u32 = if (with_fields) 6 else 5; + const first = next.*; + next.* += count; + nodes[idx] = .{ .name = name, .kind = .dir, .parent = parent, .first = @intCast(first), .count = count, .offset = offset, .size = @sizeOf(T) }; + const me: u32 = @intCast(idx); + nodes[first + 0] = .{ .name = "value", .kind = .value, .parent = me, .offset = offset, .size = @sizeOf(T), .render = renderFor(T), .set = setFor(T) }; + nodes[first + 1] = .{ .name = "type", .kind = .type_name, .parent = me, .offset = offset, .size = @sizeOf(T), .content = @typeName(T) }; + nodes[first + 2] = .{ .name = "size", .kind = .size, .parent = me, .offset = offset, .size = @sizeOf(T), .content = std.fmt.comptimePrint("{d}", .{@sizeOf(T)}) }; + nodes[first + 3] = .{ .name = "addr", .kind = .addr, .parent = me, .offset = offset, .size = @sizeOf(T) }; + nodes[first + 4] = .{ .name = "raw", .kind = .raw, .parent = me, .offset = offset, .size = @sizeOf(T) }; + if (with_fields) { + const fdir: u32 = @intCast(first + 5); + const nf = fieldCount(T); + const ffirst = next.*; + next.* += nf; + nodes[fdir] = .{ .name = "f", .kind = .fields, .parent = me, .first = @intCast(ffirst), .count = @intCast(nf), .offset = offset, .size = @sizeOf(T) }; + var i: usize = 0; + for (@typeInfo(T).@"struct".fields) |f| { + if (f.is_comptime or @sizeOf(f.type) == 0) continue; + fill(nodes, next, ffirst + i, f.type, f.name, offset + @offsetOf(T, f.name), depth + 1, fdir); + i += 1; + } + } +} + +fn buildTable(comptime T: type) [countNodes(T, 0)]Node { + @setEvalBranchQuota(1_000_000); + var nodes: [countNodes(T, 0)]Node = undefined; + var next: usize = 1; + fill(&nodes, &next, 0, T, "", 0, 0, 0); + std.debug.assert(next == nodes.len); + return nodes; +} + +fn renderFor(comptime T: type) RenderFn { + return &struct { + fn f(base: [*]const u8, w: *Writer) Writer.Error!void { + const p: *const T = @ptrCast(@alignCast(base)); + try render(p, w, max_depth); + } + }.f; +} + +fn setFor(comptime T: type) ?SetFn { + if (!settable(T)) return null; + return &struct { + fn f(base: [*]u8, text: []const u8) SetError!void { + const p: *T = @ptrCast(@alignCast(base)); + try set(p, text); + } + }.f; +} + +fn settable(comptime T: type) bool { + return switch (@typeInfo(T)) { + .int, .float, .bool, .@"enum" => true, + else => false, + }; +} + +// --------------------------------------------------------------------------- +// Rendering +// --------------------------------------------------------------------------- + +/// Renders `ptr.*`. Structs become "field: value" lines (nested structs +/// indented); everything else is a single line without a trailing newline. +/// `depth` bounds struct/union/optional nesting; deeper values print as "…". +pub fn render(ptr: anytype, w: *Writer, depth: usize) Writer.Error!void { + const T = @TypeOf(ptr.*); + if (comptime isPlainStruct(T)) { + try renderStruct(T, ptr, w, depth, 0); + } else { + try renderValue(T, ptr, w, depth); + } +} + +fn isPlainStruct(comptime T: type) bool { + return switch (@typeInfo(T)) { + .@"struct" => |s| !s.is_tuple and s.fields.len > 0, + else => false, + }; +} + +/// The multi-line form: each field on its own line, nested structs indented. +fn renderStruct(comptime T: type, ptr: *const T, w: *Writer, depth: usize, indent: usize) Writer.Error!void { + if (depth == 0) { + try w.splatByteAll(' ', indent); + try w.writeAll("…\n"); + return; + } + const packed_layout = @typeInfo(T).@"struct".layout == .@"packed"; + inline for (@typeInfo(T).@"struct".fields) |f| { + try w.splatByteAll(' ', indent); + try w.writeAll(f.name); + try w.writeByte(':'); + if (comptime f.is_comptime) { + try w.writeAll(" (comptime)\n"); + } else if (comptime packed_layout) { + // Fields of a packed struct have no byte address: render a copy. + const v = @field(ptr.*, f.name); + try w.writeByte(' '); + try renderValue(f.type, &v, w, depth - 1); + try w.writeByte('\n'); + } else if (comptime isPlainStruct(f.type)) { + try w.writeByte('\n'); + try renderStruct(f.type, &@field(ptr.*, f.name), w, depth - 1, indent + 2); + } else { + try w.writeByte(' '); + try renderValue(f.type, &@field(ptr.*, f.name), w, depth - 1); + try w.writeByte('\n'); + } + } +} + +/// The single-line form of any value. +fn renderValue(comptime T: type, ptr: *const T, w: *Writer, depth: usize) Writer.Error!void { + switch (@typeInfo(T)) { + .int, .comptime_int => try w.print("{d}", .{ptr.*}), + .float, .comptime_float => try w.print("{d}", .{ptr.*}), + .bool => { + // The variable is live memory that anything (a debugger's /mem write, + // a torn update) may have corrupted: judge the byte, not the bool. + const b = @as(*const u8, @ptrCast(ptr)).*; + switch (b) { + 0 => try w.writeAll("false"), + 1 => try w.writeAll("true"), + else => try w.print("{d}", .{b}), + } + }, + .void => try w.writeAll("{}"), + .@"enum" => |e| { + // Read the storage bytes as one integer: @tagName/switch on a corrupt + // value is a safety panic, and the value is caller memory we do not + // control. A load through the tag type would truncate to its bit + // width (a u2 tag in a byte), so the full storage width is read. + if (@sizeOf(T) == 0) return w.writeAll(e.fields[0].name); + const Raw = std.meta.Int(.unsigned, @sizeOf(T) * 8); + const raw = @as(*align(@alignOf(T)) const Raw, @ptrCast(ptr)).*; + const TagU = std.meta.Int(.unsigned, @bitSizeOf(e.tag_type)); + const padding: Raw = if (@bitSizeOf(TagU) == @bitSizeOf(Raw)) 0 else ~@as(Raw, std.math.maxInt(TagU)); + if (raw & padding == 0) { + const low: TagU = @truncate(raw); + inline for (e.fields) |f| { + if (low == @as(TagU, @bitCast(@as(e.tag_type, f.value)))) return w.writeAll(f.name); + } + } + try w.print("{d}", .{raw}); + }, + .error_set => try w.print("error.{s}", .{@errorName(ptr.*)}), + .error_union => |eu| if (ptr.*) |v| { + try renderValue(eu.payload, &v, w, depth); + } else |e| { + try w.print("error.{s}", .{@errorName(e)}); + }, + .optional => |o| if (ptr.*) |v| { + try renderValue(o.child, &v, w, depth); + } else { + try w.writeAll("null"); + }, + .pointer => |p| switch (p.size) { + .slice => if (p.child == u8) { + try renderString(ptr.*, w); + } else { + try renderElems(p.child, ptr.*, w, depth); + }, + .many => if (p.child == u8 and p.sentinel() == 0) { + try renderCString(ptr.*, w); + } else { + try w.print("0x{x}", .{@intFromPtr(ptr.*)}); + }, + .one, .c => try w.print("0x{x}", .{@intFromPtr(ptr.*)}), + }, + .array => |a| if (a.child == u8) { + try renderString(ptr.*[0..], w); + } else { + try renderElems(a.child, ptr.*[0..], w, depth); + }, + .vector => |v| { + const arr: [v.len]v.child = ptr.*; + try renderElems(v.child, &arr, w, depth); + }, + .@"struct" => |s| { + if (depth == 0) { + try w.writeAll("…"); + return; + } + if (s.fields.len == 0) { + try w.writeAll("{}"); + return; + } + try w.writeAll("{ "); + inline for (s.fields, 0..) |f, i| { + if (i != 0) try w.writeAll(", "); + if (!s.is_tuple) { + try w.writeAll(f.name); + try w.writeAll(": "); + } + if (comptime f.is_comptime) { + try w.writeAll("(comptime)"); + } else if (comptime s.layout == .@"packed") { + const v = @field(ptr.*, f.name); + try renderValue(f.type, &v, w, depth - 1); + } else { + try renderValue(f.type, &@field(ptr.*, f.name), w, depth - 1); + } + } + try w.writeAll(" }"); + }, + .@"union" => |u| { + const Tag = u.tag_type orelse { + try w.print("(untagged union, {d} bytes)", .{@sizeOf(T)}); + return; + }; + if (depth == 0) { + try w.writeAll("…"); + return; + } + // A switch on a corrupt tag is a safety panic: match the integer first. + const raw = @intFromEnum(@as(Tag, ptr.*)); + inline for (u.fields) |f| { + if (raw == @intFromEnum(@field(Tag, f.name))) { + try w.writeAll(f.name); + if (f.type != void) { + try w.writeAll(": "); + try renderValue(f.type, &@field(ptr.*, f.name), w, depth - 1); + } + return; + } + } + try w.print("(invalid tag {d})", .{raw}); + }, + .@"fn" => try w.print("0x{x}", .{@intFromPtr(ptr)}), + else => try w.print("<{s}>", .{@typeName(T)}), + } +} + +fn renderElems(comptime E: type, items: []const E, w: *Writer, depth: usize) Writer.Error!void { + try w.writeByte('['); + for (items, 0..) |*item, i| { + if (i == max_elems) { + try w.writeAll(", …"); + break; + } + if (i != 0) try w.writeAll(", "); + try renderValue(E, item, w, depth); + } + try w.writeByte(']'); +} + +/// A NUL-terminated string, scanning at most `max_string` + 1 bytes for the +/// terminator so that a missing one cannot walk off the end of the mapping. +fn renderCString(s: [*:0]const u8, w: *Writer) Writer.Error!void { + var n: usize = 0; + while (n <= max_string and s[n] != 0) n += 1; + try renderString(s[0..n], w); +} + +/// A double-quoted string with C-style escapes, truncated to `max_string` bytes. +fn renderString(s: []const u8, w: *Writer) Writer.Error!void { + try w.writeByte('"'); + for (s[0..@min(s.len, max_string)]) |b| switch (b) { + '\n' => try w.writeAll("\\n"), + '\r' => try w.writeAll("\\r"), + '\t' => try w.writeAll("\\t"), + '\\' => try w.writeAll("\\\\"), + '"' => try w.writeAll("\\\""), + ' '...'!', '#'...'[', ']'...'~' => try w.writeByte(b), + else => { + const hex = "0123456789abcdef"; + try w.writeAll("\\x"); + try w.writeByte(hex[b >> 4]); + try w.writeByte(hex[b & 15]); + }, + }; + try w.writeByte('"'); + if (s.len > max_string) try w.writeAll("…"); +} + +// --------------------------------------------------------------------------- +// Setting +// --------------------------------------------------------------------------- + +/// Parses `text` and stores it into `ptr.*`: ints in decimal or 0x/0o/0b, +/// floats, bools (true/false/1/0), enums by tag name (or by integer value for +/// non-exhaustive enums). Other types are `error.Unsupported`. +pub fn set(ptr: anytype, text: []const u8) SetError!void { + const T = @TypeOf(ptr.*); + const s = std.mem.trim(u8, text, " \t\r\n\x00"); + switch (@typeInfo(T)) { + .int => ptr.* = std.fmt.parseInt(T, s, 0) catch return error.Invalid, + .float => ptr.* = std.fmt.parseFloat(T, s) catch return error.Invalid, + .bool => { + if (std.mem.eql(u8, s, "true") or std.mem.eql(u8, s, "1")) { + ptr.* = true; + } else if (std.mem.eql(u8, s, "false") or std.mem.eql(u8, s, "0")) { + ptr.* = false; + } else return error.Invalid; + }, + .@"enum" => |e| { + if (std.meta.stringToEnum(T, s)) |v| { + ptr.* = v; + } else if (!e.is_exhaustive) { + const raw = std.fmt.parseInt(e.tag_type, s, 0) catch return error.Invalid; + ptr.* = @enumFromInt(raw); + } else return error.Invalid; + }, + else => return error.Unsupported, + } +} + +// --------------------------------------------------------------------------- +// Tests +// --------------------------------------------------------------------------- + +const testing = std.testing; + +fn renderToBuf(buf: []u8, ptr: anytype) ![]const u8 { + var w: Writer = .fixed(buf); + try render(ptr, &w, max_depth); + return w.buffered(); +} + +test "render scalars, strings, pointers, optionals, enums, arrays" { + var buf: [512]u8 = undefined; + const i: i32 = -42; + try testing.expectEqualStrings("-42", try renderToBuf(&buf, &i)); + const f: f32 = 1.5; + try testing.expectEqualStrings("1.5", try renderToBuf(&buf, &f)); + const b: bool = true; + try testing.expectEqualStrings("true", try renderToBuf(&buf, &b)); + const s: []const u8 = "hi \"there\"\n"; + try testing.expectEqualStrings("\"hi \\\"there\\\"\\n\"", try renderToBuf(&buf, &s)); + const z: [*:0]const u8 = "zed"; + try testing.expectEqualStrings("\"zed\"", try renderToBuf(&buf, &z)); + const p: *const i32 = &i; + var expect_buf: [32]u8 = undefined; + const expect = try std.fmt.bufPrint(&expect_buf, "0x{x}", .{@intFromPtr(&i)}); + try testing.expectEqualStrings(expect, try renderToBuf(&buf, &p)); + const o: ?u8 = null; + try testing.expectEqualStrings("null", try renderToBuf(&buf, &o)); + const o2: ?u8 = 7; + try testing.expectEqualStrings("7", try renderToBuf(&buf, &o2)); + const E = enum { red, green }; + const e: E = .green; + try testing.expectEqualStrings("green", try renderToBuf(&buf, &e)); + const NE = enum(u8) { a, _ }; + const ne: NE = @enumFromInt(9); + try testing.expectEqualStrings("9", try renderToBuf(&buf, &ne)); + const arr = [_]u16{ 1, 2, 3 }; + try testing.expectEqualStrings("[1, 2, 3]", try renderToBuf(&buf, &arr)); + const bytes = [_]u8{ 0, 'a', 0xff }; + try testing.expectEqualStrings("\"\\x00a\\xff\"", try renderToBuf(&buf, &bytes)); + const U = union(enum) { none, some: u32 }; + const u: U = .{ .some = 5 }; + try testing.expectEqualStrings("some: 5", try renderToBuf(&buf, &u)); + const un: U = .none; + try testing.expectEqualStrings("none", try renderToBuf(&buf, &un)); +} + +test "render structs multi-line with nested indentation and depth limit" { + const Inner = struct { x: f32, flags: [2]bool }; + const Outer = struct { a: u32, b: bool, name: []const u8, inner: Inner, items: []const Inner }; + const v: Outer = .{ .a = 1, .b = false, .name = "n", .inner = .{ .x = 2.5, .flags = .{ true, false } }, .items = &.{.{ .x = 0, .flags = .{ false, false } }} }; + var buf: [512]u8 = undefined; + try testing.expectEqualStrings( + \\a: 1 + \\b: false + \\name: "n" + \\inner: + \\ x: 2.5 + \\ flags: [true, false] + \\items: [{ x: 0, flags: [false, false] }] + \\ + , try renderToBuf(&buf, &v)); + var w: Writer = .fixed(&buf); + try render(&v, &w, 1); + try testing.expectEqualStrings( + \\a: 1 + \\b: false + \\name: "n" + \\inner: + \\ … + \\items: […] + \\ + , w.buffered()); +} + +test "long strings and arrays are truncated" { + const long = [_]u8{'x'} ** 300; + var buf: [1024]u8 = undefined; + const s: []const u8 = &long; + const out = try renderToBuf(&buf, &s); + try testing.expectEqual(@as(usize, 1 + max_string + 1 + "…".len), out.len); + try testing.expect(std.mem.endsWith(u8, out, "\"…")); + const nums: [100]u32 = @splat(1); + const out2 = try renderToBuf(&buf, &nums); + try testing.expect(std.mem.endsWith(u8, out2, ", …]")); + try testing.expectEqual(@as(usize, max_elems), std.mem.count(u8, out2, "1")); +} + +test "set parses ints, floats, bools and enums" { + var i: u32 = 0; + try set(&i, "42\n"); + try testing.expectEqual(@as(u32, 42), i); + try set(&i, "0x10"); + try testing.expectEqual(@as(u32, 16), i); + try testing.expectError(error.Invalid, set(&i, "-1")); + try testing.expectError(error.Invalid, set(&i, "abc")); + var si: i8 = 0; + try set(&si, " -7 "); + try testing.expectEqual(@as(i8, -7), si); + try testing.expectError(error.Invalid, set(&si, "200")); + var f: f64 = 0; + try set(&f, "2.25"); + try testing.expectEqual(@as(f64, 2.25), f); + var b: bool = false; + try set(&b, "true"); + try testing.expect(b); + try set(&b, "0"); + try testing.expect(!b); + try testing.expectError(error.Invalid, set(&b, "maybe")); + const E = enum { off, on }; + var e: E = .off; + try set(&e, "on"); + try testing.expectEqual(E.on, e); + try testing.expectError(error.Invalid, set(&e, "blue")); + var s: []const u8 = "x"; + try testing.expectError(error.Unsupported, set(&s, "y")); +} + +test "vtable table layout for a nested struct" { + const Inner = struct { x: f32 }; + const T = struct { a: u32, b: bool, name: []const u8, inner: Inner }; + const vt = vtableFor(T); + try testing.expectEqual(vt, vtableFor(T)); + try testing.expectEqualStrings(@typeName(T), vt.type_name); + const root = vt.nodes[0]; + try testing.expect(root.isDir()); + try testing.expectEqual(@as(u32, 6), root.count); + const value = vt.child(0, "value").?; + try testing.expect(!vt.nodes[value].writable()); // a struct is not settable + try testing.expectEqualStrings(std.fmt.comptimePrint("{d}", .{@sizeOf(T)}), vt.nodes[vt.child(0, "size").?].content); + const f = vt.child(0, "f").?; + try testing.expectEqual(Kind.fields, vt.nodes[f].kind); + try testing.expectEqual(@as(u32, 4), vt.nodes[f].count); + const a = vt.child(f, "a").?; + try testing.expectEqual(@offsetOf(T, "a"), vt.nodes[a].offset); + const a_value = vt.child(a, "value").?; + try testing.expect(vt.nodes[a_value].writable()); + try testing.expectEqual(@as(usize, 4), vt.nodes[a_value].size); + try testing.expectEqual(a, vt.nodes[a_value].parent); + const inner = vt.child(f, "inner").?; + const inner_f = vt.child(inner, "f").?; + const x = vt.child(inner_f, "x").?; + try testing.expectEqual(@offsetOf(T, "inner") + @offsetOf(Inner, "x"), vt.nodes[x].offset); + try testing.expectEqualStrings("f32", vt.nodes[vt.child(x, "type").?].content); + try testing.expect(vt.child(f, "nope") == null); + // rendering and setting through the table + var v: T = .{ .a = 1, .b = true, .name = "n", .inner = .{ .x = 0.5 } }; + const base: [*]u8 = @ptrCast(&v); + var buf: [256]u8 = undefined; + var w: Writer = .fixed(&buf); + const x_value = vt.child(x, "value").?; + try vt.nodes[x_value].render.?(base + vt.nodes[x_value].offset, &w); + try testing.expectEqualStrings("0.5", w.buffered()); + try vt.nodes[a_value].set.?(base + vt.nodes[a_value].offset, "42"); + try testing.expectEqual(@as(u32, 42), v.a); + w = .fixed(&buf); + try vt.nodes[value].render.?(base, &w); + try testing.expect(std.mem.startsWith(u8, w.buffered(), "a: 42\nb: true\n")); +} + +test "depth limit stops the f/ tree at max_depth" { + const L4 = struct { v: u8 }; + const L3 = struct { l4: L4 }; + const L2 = struct { l3: L3 }; + const L1 = struct { l2: L2 }; + const L0 = struct { l1: L1 }; + const vt = vtableFor(L0); + var node: u32 = 0; + var depth: usize = 0; + while (vt.child(node, "f")) |f| : (depth += 1) { + node = vt.nodes[f].first; // the single field + } + try testing.expectEqual(@as(usize, max_depth), depth); + try testing.expect(vt.child(node, "value") != null); +} + +test "every @typeInfo category renders without dereferencing anything unbounded" { + var buf: [2048]u8 = undefined; + // packed and extern structs (packed fields have no address: rendered by copy) + const Packed = packed struct { a: u3, b: bool, c: u12, e: enum(u2) { p, q, r } }; + const pk: Packed = .{ .a = 5, .b = true, .c = 300, .e = .r }; + try testing.expectEqualStrings("a: 5\nb: true\nc: 300\ne: r\n", try renderToBuf(&buf, &pk)); + const Ext = extern struct { x: u16, y: f32, inner: extern struct { z: u8 } }; + const ex: Ext = .{ .x = 1, .y = 0.5, .inner = .{ .z = 9 } }; + try testing.expectEqualStrings("x: 1\ny: 0.5\ninner:\n z: 9\n", try renderToBuf(&buf, &ex)); + const Holder = struct { p: Packed, list: [2]Packed }; + const ho: Holder = .{ .p = pk, .list = .{ pk, pk } }; + try testing.expect(std.mem.startsWith(u8, try renderToBuf(&buf, &ho), "p:\n a: 5\n")); + // the f/ tree has no entries for a packed struct and works through a table + const vt = vtableFor(Packed); + try testing.expect(vt.child(0, "f") == null); + var w: Writer = .fixed(&buf); + try vt.nodes[vt.child(0, "value").?].render.?(@ptrCast(&pk), &w); + try testing.expect(std.mem.startsWith(u8, w.buffered(), "a: 5\n")); + // optionals of pointers are printed, never followed + var target: u32 = 7; + const op: ?*u32 = ⌖ + var expect_buf: [32]u8 = undefined; + try testing.expectEqualStrings(try std.fmt.bufPrint(&expect_buf, "0x{x}", .{@intFromPtr(&target)}), try renderToBuf(&buf, &op)); + const np: ?*u32 = null; + try testing.expectEqualStrings("null", try renderToBuf(&buf, &np)); + const dangling: *const u32 = @ptrFromInt(0x1000); + try testing.expectEqualStrings("0x1000", try renderToBuf(&buf, &dangling)); + const cptr: [*c]const u8 = @ptrFromInt(0x2000); + try testing.expectEqualStrings("0x2000", try renderToBuf(&buf, &cptr)); + const manyp: [*]const u32 = @ptrFromInt(0x3000); + try testing.expectEqualStrings("0x3000", try renderToBuf(&buf, &manyp)); + // untagged and tagged unions, error unions, error sets + const Untagged = union { a: u32, b: f32 }; + const un: Untagged = .{ .a = 1 }; + try testing.expectEqualStrings(std.fmt.comptimePrint("(untagged union, {d} bytes)", .{@sizeOf(Untagged)}), try renderToBuf(&buf, &un)); + const Tagged = union(enum(u8)) { none, some: u32, pair: struct { l: u8, r: u8 } }; + const tg: Tagged = .{ .pair = .{ .l = 1, .r = 2 } }; + try testing.expectEqualStrings("pair: { l: 1, r: 2 }", try renderToBuf(&buf, &tg)); + const eu: anyerror!u8 = error.Boom; + try testing.expectEqualStrings("error.Boom", try renderToBuf(&buf, &eu)); + const eu2: error{X}!u8 = 4; + try testing.expectEqualStrings("4", try renderToBuf(&buf, &eu2)); + const es: anyerror = error.Zap; + try testing.expectEqualStrings("error.Zap", try renderToBuf(&buf, &es)); + // wide ints and floats, vectors, sentinel arrays, slices of slices, void, comptime fields + const big: u128 = std.math.maxInt(u128); + try testing.expectEqualStrings("340282366920938463463374607431768211455", try renderToBuf(&buf, &big)); + const neg: i128 = std.math.minInt(i128); + try testing.expectEqualStrings("-170141183460469231731687303715884105728", try renderToBuf(&buf, &neg)); + const h: f16 = 1.5; + try testing.expectEqualStrings("1.5", try renderToBuf(&buf, &h)); + const ld: f80 = 2.25; + try testing.expectEqualStrings("2.25", try renderToBuf(&buf, &ld)); + const quad: f128 = 3.125; + try testing.expectEqualStrings("3.125", try renderToBuf(&buf, &quad)); + const vec: @Vector(4, i16) = .{ 1, -2, 3, -4 }; + try testing.expectEqualStrings("[1, -2, 3, -4]", try renderToBuf(&buf, &vec)); + const sarr: [3:0]u8 = .{ 'a', 'b', 'c' }; + try testing.expectEqualStrings("\"abc\"", try renderToBuf(&buf, &sarr)); + const rows: []const []const u8 = &.{ "ab", "cd" }; + try testing.expectEqualStrings("[\"ab\", \"cd\"]", try renderToBuf(&buf, &rows)); + const Odd = struct { v: void, comptime k: u8 = 3, n: u8 }; + const odd: Odd = .{ .v = {}, .n = 1 }; + try testing.expectEqualStrings("v: {}\nk: (comptime)\nn: 1\n", try renderToBuf(&buf, &odd)); + try testing.expectEqual(@as(u32, 1), vtableFor(Odd).nodes[vtableFor(Odd).child(0, "f").?].count); + // self-referential through a pointer: rendered as an address, table stays finite + const Link = struct { next: ?*const @This(), v: u8 }; + var a: Link = .{ .next = null, .v = 1 }; + const b: Link = .{ .next = &a, .v = 2 }; + a.next = &b; + try testing.expectEqualStrings(try std.fmt.bufPrint(&expect_buf, "next: 0x{x}\nv: 2\n", .{@intFromPtr(&a)}), try renderToBuf(&buf, &b)); + try testing.expect(vtableFor(Link).nodes.len < 32); + // tuples + const tup: struct { u8, []const u8 } = .{ 1, "x" }; + try testing.expectEqualStrings("{ 1, \"x\" }", try renderToBuf(&buf, &tup)); +} + +test "corrupt live memory renders instead of trapping: enums, unions, bools" { + var buf: [128]u8 = undefined; + const E = enum(u8) { a, b }; + var raw_e: u8 = 7; + try testing.expectEqualStrings("7", try renderToBuf(&buf, @as(*const E, @ptrCast(&raw_e)))); + raw_e = 1; + try testing.expectEqualStrings("b", try renderToBuf(&buf, @as(*const E, @ptrCast(&raw_e)))); + // a u2 tag in a byte: the whole byte is judged, not the truncated tag (ReleaseSafe would say "c") + const E3 = enum { a, b, c }; + var raw3: u8 = 0xEE; + try testing.expectEqualStrings("238", try renderToBuf(&buf, @as(*const E3, @ptrCast(&raw3)))); + raw3 = 3; + try testing.expectEqualStrings("3", try renderToBuf(&buf, @as(*const E3, @ptrCast(&raw3)))); + raw3 = 2; + try testing.expectEqualStrings("c", try renderToBuf(&buf, @as(*const E3, @ptrCast(&raw3)))); + const E12 = enum(u12) { p = 5, q = 4095 }; + var raw12: u16 = 0xF005; + try testing.expectEqualStrings("61445", try renderToBuf(&buf, @as(*const E12, @ptrCast(&raw12)))); + raw12 = 4095; + try testing.expectEqualStrings("q", try renderToBuf(&buf, @as(*const E12, @ptrCast(&raw12)))); + const ES = enum(i8) { neg = -3, pos = 7 }; + var raws: u8 = 0xFD; + try testing.expectEqualStrings("neg", try renderToBuf(&buf, @as(*const ES, @ptrCast(&raws)))); + raws = 0x80; + try testing.expectEqualStrings("128", try renderToBuf(&buf, @as(*const ES, @ptrCast(&raws)))); + const E1 = enum { only }; + const e1: E1 = .only; + try testing.expectEqualStrings("only", try renderToBuf(&buf, &e1)); + const NE = enum(u16) { x = 5, _ }; + var raw_ne: u16 = 5; + try testing.expectEqualStrings("x", try renderToBuf(&buf, @as(*const NE, @ptrCast(&raw_ne)))); + raw_ne = 6; + try testing.expectEqualStrings("6", try renderToBuf(&buf, @as(*const NE, @ptrCast(&raw_ne)))); + var raw_b: u8 = 2; + try testing.expectEqualStrings("2", try renderToBuf(&buf, @as(*const bool, @ptrCast(&raw_b)))); + const U = union(enum(u8)) { x: u32, y: bool }; + var raw_u: [@sizeOf(U)]u8 align(@alignOf(U)) = @splat(0x55); + const out = try renderToBuf(&buf, @as(*const U, @ptrCast(&raw_u))); + try testing.expectEqualStrings("(invalid tag 85)", out); + const S = struct { e: E, u: U, b: bool }; + var raw_s: [@sizeOf(S)]u8 align(@alignOf(S)) = @splat(0xEE); + const ps: *const S = @ptrCast(&raw_s); + _ = try renderToBuf(&buf, ps); // no trap + try testing.expect(std.mem.indexOf(u8, try renderToBuf(&buf, ps), "238") != null); +} + +test "a [*:0]const u8 without a terminator is read at most max_string + 1 bytes" { + // Only the first max_string + 1 bytes exist; anything beyond is the + // testing allocator's guard, which a wider scan would touch. + const mem = try testing.allocator.alloc(u8, max_string + 1); + defer testing.allocator.free(mem); + @memset(mem, 'x'); + const z: [*:0]const u8 = @ptrCast(mem.ptr); + var buf: [1024]u8 = undefined; + const out = try renderToBuf(&buf, &z); + try testing.expectEqual(@as(usize, 1 + max_string + 1 + "…".len), out.len); + try testing.expect(std.mem.endsWith(u8, out, "\"…")); + // exactly max_string bytes then NUL: no ellipsis + const mem2 = try testing.allocator.alloc(u8, max_string + 1); + defer testing.allocator.free(mem2); + @memset(mem2, 'y'); + mem2[max_string] = 0; + const z2: [*:0]const u8 = @ptrCast(mem2.ptr); + const out2 = try renderToBuf(&buf, &z2); + try testing.expectEqual(@as(usize, 1 + max_string + 1), out2.len); + // a garbage-length []const u8 still reads at most max_string bytes + const garbage: []const u8 = mem[0..max_string]; + _ = try renderToBuf(&buf, &garbage); +} + +test "set rejects hostile input without partial writes" { + var u: u8 = 200; + for ([_][]const u8{ "-1", "256", "1e3", "0x", "", " ", "1.5", "+", "0b2", "١", "12abc", "0x100", "\x00", "1 2" }) |bad| { + try testing.expectError(error.Invalid, set(&u, bad)); + try testing.expectEqual(@as(u8, 200), u); + } + try set(&u, "0b1111_1111"); + try testing.expectEqual(@as(u8, 255), u); + try set(&u, "+0o17"); + try testing.expectEqual(@as(u8, 15), u); + var i: i64 = 1; + try set(&i, "-9223372036854775808"); + try testing.expectEqual(std.math.minInt(i64), i); + try testing.expectError(error.Invalid, set(&i, "9223372036854775808")); + var w: u128 = 0; + try set(&w, "340282366920938463463374607431768211455"); + try testing.expectEqual(std.math.maxInt(u128), w); + try testing.expectError(error.Invalid, set(&w, "340282366920938463463374607431768211456")); + // floats: exponents, hex floats, inf/nan spellings, and junk + var f: f32 = 1; + try set(&f, "1.5e3"); + try testing.expectEqual(@as(f32, 1500), f); + try set(&f, "-0x1p-2"); + try testing.expectEqual(@as(f32, -0.25), f); + try set(&f, "1e999"); + try testing.expect(std.math.isInf(f)); + try testing.expectError(error.Invalid, set(&f, "1.5.5")); + try testing.expectError(error.Invalid, set(&f, "e5")); + try testing.expectError(error.Invalid, set(&f, "")); + var h: f16 = 0; + try set(&h, "65504"); + try testing.expectEqual(@as(f16, 65504), h); + var q: f128 = 0; + try set(&q, "2.5"); + try testing.expectEqual(@as(f128, 2.5), q); + // enums: NULs inside the tag, case, trailing junk; non-exhaustive by integer only when out of names + const E = enum(u8) { off, on }; + var e: E = .off; + for ([_][]const u8{ "on\x00x", "On", "on x", "1", "0x1", "" }) |bad| { + try testing.expectError(error.Invalid, set(&e, bad)); + try testing.expectEqual(E.off, e); + } + try set(&e, "\x00on\n"); + try testing.expectEqual(E.on, e); + const NE = enum(u8) { a, _ }; + var ne: NE = .a; + try set(&ne, "200"); + try testing.expectEqual(@as(u8, 200), @intFromEnum(ne)); + try testing.expectError(error.Invalid, set(&ne, "256")); + try testing.expectError(error.Invalid, set(&ne, "-1")); + try set(&ne, "a"); + try testing.expectEqual(NE.a, ne); + // bools + var b: bool = true; + for ([_][]const u8{ "yes", "TRUE", "2", "", "01" }) |bad| { + try testing.expectError(error.Invalid, set(&b, bad)); + try testing.expect(b); + } + // unsupported types are refused without touching memory + var opt: ?u8 = 3; + try testing.expectError(error.Unsupported, set(&opt, "4")); + try testing.expectEqual(@as(?u8, 3), opt); + var arr: [2]u8 = .{ 1, 2 }; + try testing.expectError(error.Unsupported, set(&arr, "x")); + var un: union(enum) { a: u8 } = .{ .a = 1 }; + try testing.expectError(error.Unsupported, set(&un, "a")); +} diff --git a/9proc/test/adv_9proc_hostile.py b/9proc/test/adv_9proc_hostile.py new file mode 100755 index 0000000..934e757 --- /dev/null +++ b/9proc/test/adv_9proc_hostile.py @@ -0,0 +1,999 @@ +#!/usr/bin/env python3 +"""Hostile raw-9P2000 client for the 9proc-demo server (stdlib only). + +Usage: + adv_9proc_hostile.py --server zig-out/bin/9proc-demo # spawns it on a temp unix socket + adv_9proc_hostile.py --socket PATH # attacks a running server + +Every attack is followed by a "server still healthy" probe on a fresh connection. +Exit status is non-zero if any check fails, the server dies, or a probe hangs. +""" +import argparse +import os +import signal +import socket +import struct +import subprocess +import sys +import tempfile +import threading +import time + +NOTAG = 0xFFFF +NOFID = 0xFFFFFFFF +Tversion, Rversion, Tauth, Rauth, Tattach, Rattach, Rerror = 100, 101, 102, 103, 104, 105, 107 +Tflush, Rflush, Twalk, Rwalk, Topen, Ropen, Tcreate, Rcreate = 108, 109, 110, 111, 112, 113, 114, 115 +Tread, Rread, Twrite, Rwrite, Tclunk, Rclunk, Tremove, Rremove = 116, 117, 118, 119, 120, 121, 122, 123 +Tstat, Rstat, Twstat, Rwstat = 124, 125, 126, 127 +OREAD, OWRITE, ORDWR, OEXEC, OTRUNC, ORCLOSE = 0, 1, 2, 3, 0x10, 0x40 +DMDIR, DMAPPEND, DMEXCL = 0x80000000, 0x40000000, 0x20000000 +NAMES = {v: k for k, v in globals().items() if k[:1] in "TR" and isinstance(v, int) and 100 <= v <= 127} + +FAILS = [] +PASSES = 0 + + +def ok(name, cond, detail=""): + global PASSES + if cond: + PASSES += 1 + print(f"ok - {name}") + else: + FAILS.append(name) + print(f"FAIL - {name} {detail}") + + +def s16(b): + return struct.pack(" connection closed + c.raw(frame(Twalk, 1, struct.pack("0 + ok("walk to file", c.walk_ok(0, 1, [b"build", b"target"]) == 2) + ok("walk from file fails 'not a directory'", c.err(Twalk, struct.pack(" 0) + ok("read dir at bad offset", c.err(Tread, struct.pack(" 6: + return + if c.walk_ok(0, 9, names) != len(names): + ok("walk " + b"/".join(names).decode(), False) + return + rt, st = c.stat(9) + if st["mode"] & DMDIR: + c.open(9, OREAD) + d = c.read_all(9, 512) + c.clunk(9) + while d: + n, = struct.unpack_from(" 0: + ok("server process still running", proc.poll() is None, proc.poll()) + finally: + if proc is not None: + proc.send_signal(signal.SIGTERM) + try: + _, err = proc.communicate(timeout=5) + except subprocess.TimeoutExpired: + proc.kill() + _, err = proc.communicate() + lines = [ln for ln in err.decode("utf-8", "replace").splitlines() if "connection ended" not in ln and "read: " not in ln] + if lines: + print("# server stderr (filtered):") + for ln in lines[:40]: + print(" " + ln) + if tmp: + try: + os.unlink(path) + os.rmdir(tmp) + except OSError: + pass + print(f"# {PASSES} passed, {len(FAILS)} failed") + for f in FAILS: + print("# FAIL " + f) + sys.exit(1 if FAILS else 0) + + +if __name__ == "__main__": + main() diff --git a/9proc/test/adv_9proc_hostile.sh b/9proc/test/adv_9proc_hostile.sh new file mode 100755 index 0000000..80d3342 --- /dev/null +++ b/9proc/test/adv_9proc_hostile.sh @@ -0,0 +1,10 @@ +#!/usr/bin/env bash +# Adversarial raw-9P2000 client tests for the 9proc-demo server. +# Usage: bash 9proc/test/adv_9proc_hostile.sh <9proc-demo> [--fast] (part of zig build 9proc-adv) +# Spawns the server on a temporary unix socket and attacks it with +# 9proc/test/adv_9proc_hostile.py (Python 3 stdlib). Exit 1 on any failure. +set -u +PROC=$(realpath "${1:?path to 9proc-demo}") +shift +command -v python3 >/dev/null || { echo "SKIP: python3 missing"; exit 0; } +exec python3 "$(dirname "$0")/adv_9proc_hostile.py" --server "$PROC" "$@" diff --git a/9proc/test/adv_core_hostile.py b/9proc/test/adv_core_hostile.py new file mode 100755 index 0000000..464876f --- /dev/null +++ b/9proc/test/adv_core_hostile.py @@ -0,0 +1,1018 @@ +#!/usr/bin/env python3 +"""Hostile raw-9P2000 client aimed at the 9proc *core* (stdlib only). + +Complements adv_9proc_hostile.py in this directory (framing, tags, scratch, floods) +with attacks on the freestanding engine's own paths: the /vars tree and its +comptime renderers, snapshot slots, the static tree, the fid table at its +configured maximum, directory-read offsets, msize 24, the ctl staging rule, +and the demo's debug providers driven as black boxes. + +Usage: + adv_core_hostile.py --server zig-out/bin/9proc-demo # spawns it on a temp unix socket + adv_core_hostile.py --socket PATH # attacks a running server + +Exit status is non-zero if any check fails or the server dies. +""" +import argparse +import os +import signal +import struct +import subprocess +import sys +import tempfile +import threading +import time + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +import adv_9proc_hostile as base # noqa: E402 +from adv_9proc_hostile import ( # noqa: E402 + NOTAG, Tversion, Tflush, Rflush, Twalk, Rwalk, Topen, Ropen, Rcreate, + Tread, Rread, Twrite, Rwrite, Tclunk, Rclunk, Tremove, Rremove, Tstat, Rstat, Twstat, Rwstat, + Rerror, OREAD, OWRITE, ORDWR, OEXEC, OTRUNC, ORCLOSE, DMDIR, + Nine, frame, s16, mkstat, parse_stat, ok, healthy, expect_dead, +) + +MAX_FIDS = 32768 # demo/main.zig cfg.max_fids +SNAPSHOT_SLOTS = 8 # demo/main.zig cfg.snapshot_slots (per connection) +SCRATCH_BUDGET = 512 << 20 +SCRATCH_MAX_FILE = 64 << 20 + + +def records(d): + """Splits a directory read into (name, raw-record) pairs.""" + out = [] + while d: + n, = struct.unpack_from("/name.""" + c.walk_ok(0, 40, [b"threads"]) + c.open(40, OREAD) + d = c.read_all(40) + c.clunk(40) + for name, _ in records(d): + if c.path_read([b"threads", name, b"name"], fid=41) == b"worker": + return name + return None + + +# --------------------------------------------------------------------------- /vars + + +def attack_vars(path): + print("# /vars: deep walks, hostile names, renderer edge cases, hostile writes") + c = Nine(path) + c.session(1 << 20) + deep = [b"vars", b"state", b"f", b"last_job", b"f", b"id", b".", b"..", b"id", b".", b"..", b"id", b".", b"..", b"id", b"value"] + assert len(deep) == 16 + ok("16-element walk deep into /vars/state/f/... succeeds", c.walk_ok(0, 1, deep) == 16) + rt, _, _ = c.open(1, OREAD) + ok("deep walk lands on a readable value file", rt == Ropen, rt) + c.clunk(1) + up = [b"vars", b"state", b"f", b"inner"] if False else [b"vars", b"state", b"f", b"last_job"] + [b".."] * 12 + n = c.walk_ok(0, 1, up) + ok("12 x '..' from inside /vars climbs to the root and stays there", n == 16, n) + rt, st = c.stat(1) + ok("fid after the climb is the root directory", rt == Rstat and st["qid"][2] == 0xFF << 56, st) + c.clunk(1) + # names that are hex/decimal edge cases or otherwise hostile: never anything but Rerror/partial walk + for nm in (b"0", b"-1", b"0x", b"0x0", b"state\x00", b"State", b" state", b"state ", b"a" * 255, b"a" * 65535, b"\xff\xfe", b"..\x00", b"f", b"value"): + n = c.walk_ok(0, 1, [b"vars", nm]) + ok(f"walk /vars/{nm[:12]!r}{'...' if len(nm) > 12 else ''} is a partial walk (1)", n == 1, n) + ok(" and newfid stays unbound", c.err(Tclunk, struct.pack(" state -> /vars -> /", len(q) == 4 and q[3][2] == 0xFF << 56 and (q[1][0] & 0x80), q) + c.clunk(1) + c.clunk(2) + # every file under /vars/state reads; raw reads beyond @sizeOf are empty + size = int(c.path_read([b"vars", b"state", b"size"])) + ok("/vars/state/size is a number", size > 0, size) + raw = c.path_read([b"vars", b"state", b"raw"]) + ok("/vars/state/raw has exactly @sizeOf bytes", raw is not None and len(raw) == size, (len(raw) if raw else raw, size)) + c.walk_ok(0, 1, [b"vars", b"state", b"raw"]) + c.open(1, OREAD) + rt, d = c.read(1, size, 100) + ok("raw read at offset @sizeOf is empty", rt == Rread and d == b"", (rt, d)) + rt, d = c.read(1, size - 1, 100) + ok("raw read at @sizeOf-1 returns one byte", rt == Rread and len(d) == 1, (rt, d)) + rt, d = c.read(1, (1 << 64) - 1, 100) + ok("raw read at 2^64-1 is empty", rt == Rread and d == b"") + rt, d = c.read(1, 0, 0xFFFFFFFF) + ok("raw read with count 2^32-1 is clamped", rt == Rread and len(d) == size, (rt, len(d) if d else d)) + rt, st = c.stat(1) + ok("raw stat length is @sizeOf and mode 0444", rt == Rstat and st["length"] == size and st["mode"] == 0o444, st) + ok("raw is read-only", c.err(Twrite, struct.pack(" the refused one now opens; clunk via Tremove (denied) also frees the slot + c.clunk(100) + rt, _, _ = c.open(100 + opened, OREAD) + ok("after one clunk the refused open succeeds", rt == Ropen, rt) + ok("remove of an open dynamic file is denied", c.err(Tremove, struct.pack("/stack/x is 'not a directory'", c.walk_ok(0, 1, [b"threads", tid, b"stack"]) == 3 and c.err(Twalk, struct.pack("/stack is 'not a directory'", c.err(Twalk, struct.pack("/../..//name walks", c.walk_ok(0, 1, [b"threads", tid, b"..", b"..", b"threads", tid, b"name"]) == 7) + c.clunk(1) + c.walk_ok(0, 1, [b"threads", tid]) + ok("create under /threads/ is denied", c.err(base.Tcreate, struct.pack(" is denied", c.err(Twstat, struct.pack(" is denied", c.err(Tremove, struct.pack(" reads @sizeOf bytes", rt == Rread and len(d) == size, (rt, len(d) if d else d)) + rt, d = c.read(1, 0, 0xFFFFFFFF) + ok("/mem read with count 2^32-1 is clamped and answered", rt in (Rread, Rerror), rt) + c.clunk(1) + hexd = c.path_read([b"hex", hx]) + ok("/hex/ is a hexdump", hexd is not None and len(hexd) > 64, hexd[:40] if hexd else hexd) + # a value written through /mem must render, not trap: corrupt the phase enum and read /vars/state/value + phase_addr = int(c.path_read([b"vars", b"state", b"f", b"phase", b"addr"]), 16) + c.walk_ok(0, 1, [b"mem", b"%x" % phase_addr]) + c.open(1, OWRITE) + rt, _, _ = c.write(1, 0, b"\xee") + ok("write a corrupt enum byte through /mem", rt == Rwrite, rt) + c.clunk(1) + v = c.path_read([b"vars", b"state", b"value"]) + ok("/vars/state/value renders the corrupt enum as a number instead of trapping", v is not None and b"phase: 238" in v, v) + pv = c.path_read([b"vars", b"state", b"f", b"phase", b"value"]) + ok("/vars/state/f/phase/value renders 238", pv == b"238", pv) + c.walk_ok(0, 1, [b"vars", b"state", b"f", b"phase", b"value"]) + c.open(1, OWRITE) + rt, _, _ = c.write(1, 0, b"idle") + ok("the enum can be repaired through /vars", rt == Rwrite, rt) + c.clunk(1) + # /panic: ctl refuses reads and garbage; message/stack read + ok("/panic/message reads (empty, no panic)", c.path_read([b"panic", b"message"]) == b"") + ok("/panic/stack reads", c.path_read([b"panic", b"stack"]) is not None) + c.walk_ok(0, 1, [b"panic", b"ctl"]) + ok("/panic/ctl refuses OREAD", c.err(Topen, struct.pack("= 1, (len(held), err)) + for f in held: + c.clunk(f) + ok("after clunking, /addr opens again", c.path_read([b"addr", b"1000"]) is not None) + # Tversion with debug files open (hexdumps of the exposed state: mapped memory) + base_addr = int(addr, 16) if addr else 0 + opened = 0 + for i in range(4): + c.walk_ok(0, 300 + i, [b"hex", b"%x" % (base_addr + i)]) + opened += c.open(300 + i, OREAD)[0] == Ropen + ok("four /hex snapshots open", opened == 4, opened) + rt, _, _ = c.version(65536) + ok("Tversion with debug snapshots open", rt == base.Rversion) + c.attach() + ok("/hex still opens after the reset", c.path_read([b"hex", b"%x" % base_addr]) is not None) + ok("/hex of unmapped memory is an Rerror at open, not a crash", c.walk_ok(0, 1, [b"hex", b"3000"]) == 2 and c.open(1, OREAD)[0] == Rerror) + c.clunk(1) + c.close() + ok("server healthy after debug provider attacks", healthy(path)) + + +# --------------------------------------------------------------------------- fids at the maximum + + +def flood(c, ids, names): + """Pipelines one Twalk per id and counts the Rwalk replies; returns (ok_count, error_count, seconds).""" + got = [0, 0] + dead = [False] + + def reader(): + try: + for _ in ids: + rt, _, _ = c.recv_frame() + if rt == Rwalk: + got[0] += 1 + else: + got[1] += 1 + except (EOFError, OSError): + dead[0] = True + + t = threading.Thread(target=reader) + t.start() + t0 = time.time() + body = b"".join(frame(Twalk, i & 0xFFFE, struct.pack(" the handler's writer fails + rt2, d = c.read(1, 0, 100) + rt3, st5 = c.stat(1) + ok("an over-long result is an Rerror and the previous result survives", rt == Rerror and d == b"y" * 100 and st5["length"] == 60000, (rt, d[:10] if d else d, st5["length"])) + c.clunk(1) + c.close() + ok("server healthy after ctl attacks", healthy(path)) + + +# --------------------------------------------------------------------------- fid state machine on scratch + + +def attack_fid_states(path): + print("# fid state machine on /scratch") + c = Nine(path) + c.session() + tag = os.urandom(3).hex().encode() + root = b"fs-" + tag + c.walk_ok(0, 1, [b"scratch"]) + c.create(1, root, DMDIR | 0o755, OREAD) + c.clunk(1) + S = [b"scratch", root] + c.walk_ok(0, 1, S) + rt, _, _ = c.create(1, b"f", 0o644, ORDWR) + ok("create f", rt == Rcreate) + ok("open of an open fid is 'file already open'", c.err(Topen, struct.pack("> 20} MiB fill the {SCRATCH_BUDGET >> 20} MiB budget exactly", made == count, made) + print(f" filled the budget in {time.time() - t0:.1f}s") + c.walk_ok(0, 1, S) + c.create(1, b"one-more", 0o644, OWRITE) + ok("one more byte is 'no space left on device'", c.err(Twrite, struct.pack(" [--fast] (part of zig build 9proc-adv) +# Spawns the server on a temporary unix socket and attacks it with +# 9proc/test/adv_core_hostile.py (Python 3 stdlib). Exit 1 on any failure. +set -u +PROC=$(realpath "${1:?path to 9proc-demo}") +shift +command -v python3 >/dev/null || { echo "SKIP: python3 missing"; exit 0; } +exec python3 "$(dirname "$0")/adv_core_hostile.py" --server "$PROC" "$@" diff --git a/9proc/test/adv_linux_probe.py b/9proc/test/adv_linux_probe.py new file mode 100755 index 0000000..aa28cf9 --- /dev/null +++ b/9proc/test/adv_linux_probe.py @@ -0,0 +1,421 @@ +#!/usr/bin/env python3 +"""Adversarial tests of the 9proc Linux layer: probe loop, debug provider, +signal machinery. Raw 9P2000 over a unix socket, plus one 9ns mount. +Usage: adv_linux_probe.py --ns <9ns> --server <9proc-demo> +Reuses the client of adv_9proc_hostile.py. Exit 1 on any failure. +""" +import argparse +import ctypes +import os +import signal +import socket +import struct +import subprocess +import sys +import tempfile +import threading +import time + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +from adv_9proc_hostile import ( # noqa: E402 + NOFID, NOTAG, OREAD, OWRITE, Nine, Rerror, Ropen, Rread, Rversion, Rwalk, Rwrite, + Tattach, Tread, Tversion, Twrite, frame, healthy, ok, parse_stat, s16) +import adv_9proc_hostile as hostile # noqa: E402 + +libc = ctypes.CDLL(None, use_errno=True) +SYS_tgkill = 234 if os.uname().machine == "x86_64" else 131 # aarch64: 131 + + +def tgkill(pid, tid, sig): + return libc.syscall(SYS_tgkill, pid, tid, sig) + + +class Srv: + def __init__(self, server, extra=()): + self.tmp = tempfile.mkdtemp(prefix="advlin.") + self.path = os.path.join(self.tmp, "sock") + self.proc = subprocess.Popen([server, "--unix", self.path, *extra], stderr=subprocess.PIPE) + for _ in range(200): + if os.path.exists(self.path): + break + time.sleep(0.02) + self.pid = self.proc.pid + + def alive(self): + return self.proc.poll() is None + + def stop(self): + if self.proc.poll() is None: + self.proc.send_signal(signal.SIGTERM) + try: + self.proc.wait(timeout=5) + except subprocess.TimeoutExpired: + self.proc.kill() + self.proc.wait() + err = self.proc.stderr.read().decode("utf-8", "replace") + try: + os.unlink(self.path) + except OSError: + pass + try: + os.rmdir(self.tmp) + except OSError: + pass + return err + + +def client(path, timeout=10): + c = Nine(path, timeout=timeout) + c.session() + return c + + +def rd(c, names, fid=50, offset=0, count=8192): + """walk+open+one read; returns (rtype-or-tag, data-or-error-string).""" + if c.walk_ok(0, fid, names) != len(names): + c.clunk(fid) + return "walkfail", None + rt, _, rb = c.open(fid, OREAD) + if rt != Ropen: + c.clunk(fid) + return "openfail", rb + rt, d = c.read(fid, offset, count) + c.clunk(fid) + if rt == Rerror: + n, = struct.unpack_from("/maps (read in 4 KiB pieces)", via == real, f"{len(via)} vs {len(real)}") + maps = real.decode() + for tag in ("[stack]", "[vdso]", "[heap]"): + m = [ln for ln in maps.splitlines() if ln.endswith(tag)] + if not m: + continue + lo = int(m[0].split("-")[0], 16) + 0x100 + ok(f"/addr of {tag} renders ? (never handed to std)", rd(c, [b"addr", b"%x" % lo])[1] == b"?\n?\n?\n") + ok(f"/hex of {tag} dumps", rd(c, [b"hex", b"%x" % lo])[0] == Rread) + m = [ln for ln in maps.splitlines() if ln.endswith("[stack]")][0] + hi = int(m.split()[0].split("-")[1], 16) + rt, d = rd(c, [b"mem", b"%x" % (hi - 16)], count=4096) + ok("read across the end of a mapping is a short read of 16 bytes", rt == Rread and len(d) == 16, (rt, d and len(d))) + rt, d = rd(c, [b"hex", b"%x" % (hi - 16)]) + ok("hexdump across the end of a mapping stops at the boundary", rt == Rread and d.count(b"\n") == 1, (rt, d)) + code = [ln for ln in maps.splitlines() if "r-xp" in ln and "9proc-demo" in ln][0] + clo = int(code.split("-")[0], 16) + ok("write into read-only code is an error, not a fault", wr(c, [b"mem", b"%x" % (clo + 0x100)], b"\xcc") == ("err", "i/o error")) + ok("server alive after memory attacks", s.alive() and healthy(s.path)) + # a big write over the demo's own globals (state onwards) must not fault the server path + addr = rd(c, [b"vars", b"state", b"addr"])[1].decode().strip()[2:] + # (it zeroes the demo's own globals, this connection's state included, so + # the reply may never come; the server as a whole must keep working) + wr(c, [b"mem", addr.encode()], b"\x00" * 65536) + time.sleep(0.3) + ok("server alive and serving after a 64 KiB overwrite of its own globals", s.alive() and healthy(s.path)) + s.stop() + + +def attack_signals(server): + print("# signal machinery") + s = Srv(server) + c = client(s.path, timeout=8) + w = thread_by_name(c, s.pid, b"worker") + probe = thread_by_name(c, s.pid, b"9proc") + ok("worker and probe threads found by name", bool(w and probe), (w, probe)) + names = sorted(open(f"/proc/{s.pid}/task/{t}/comm").read().strip() for t in os.listdir(f"/proc/{s.pid}/task")) + ok("thread names are 9proc, 9proc-demo, worker", names == ["9proc", "9proc-demo", "worker"], names) + + # 4 clients hammer stacks of every thread while trap/continue interleave + errs, count = [], [0] + stop = threading.Event() + + def hammer(k): + try: + cc = client(s.path, timeout=8) + while not stop.is_set(): + for t in (w, probe, str(s.pid).encode()): + rt, d = rd(cc, [b"threads", t, b"stack"], fid=10 + k) + count[0] += 1 + if rt in ("walkfail", "openfail") or (rt == "err" and "i/o" not in d): + errs.append((t, rt, d)) + except Exception as e: # noqa: BLE001 + errs.append(repr(e)) + + ths = [threading.Thread(target=hammer, args=(k,)) for k in range(4)] + for t in ths: + t.start() + rounds_ok = True + for _ in range(5): + wr(c, [b"runtime", b"ctl"], b"trap") + time.sleep(0.15) + rounds_ok &= ls(c, [b"breakpoints"]) == [w] + rounds_ok &= b"workerLoop" in (rd(c, [b"breakpoints", w, b"stack"])[1] or b"") + rounds_ok &= rd(c, [b"threads", w, b"stack"])[0] == Rread # capture of a paused thread + rounds_ok &= wr(c, [b"breakpoints", w, b"ctl"], b"continue")[0] == Rwrite + rounds_ok &= wr(c, [b"breakpoints", w, b"ctl"], b"continue")[0] == "walkfail" # twice: gone + stop.set() + for t in ths: + t.join() + ok("trap/inspect/continue rounds while 4 clients capture stacks", rounds_ok) + ok(f"{count[0]} concurrent captures without a wrong answer", count[0] > 50 and not errs, errs[:3]) + + # SIGTRAP from outside (tgkill, not int3): parks without corrupting the thread + ok("tgkill SIGTRAP to the worker", tgkill(s.pid, int(w), signal.SIGTRAP) == 0) + time.sleep(0.3) + ok("worker listed under /breakpoints after tgkill", ls(c, [b"breakpoints"]) == [w]) + ok("its stack names workerLoop", b"workerLoop" in (rd(c, [b"breakpoints", w, b"stack"])[1] or b"")) + t1 = ticks(c) + time.sleep(0.3) + ok("ticks frozen while parked", ticks(c) == t1) + ok("continue after tgkill", wr(c, [b"breakpoints", w, b"ctl"], b"continue")[0] == Rwrite) + time.sleep(0.4) + ok("ticks advance after continue (no instruction skipped)", ticks(c) > t1) + + # SIGTRAP on the probe thread itself: stepped over, the server keeps serving + ok("tgkill SIGTRAP to the probe thread", tgkill(s.pid, int(probe), signal.SIGTRAP) == 0) + time.sleep(0.2) + ok("server serves after a SIGTRAP on its own thread", healthy(s.path)) + ok("probe thread not parked", ls(c, [b"breakpoints"]) == []) + # process-directed SIGTRAP lands on some thread; whichever it is, it is resumable + os.kill(s.pid, signal.SIGTRAP) + time.sleep(0.3) + ok("alive after kill -TRAP ", s.alive() and healthy(s.path)) + for t in ls(c, [b"breakpoints"]): + ok(f"thread {t.decode()} parked by kill -TRAP resumes", wr(c, [b"breakpoints", t, b"ctl"], b"continue")[0] == Rwrite) + ok("continue on a never-paused tid does not walk", wr(c, [b"breakpoints", w, b"ctl"], b"continue")[0] == "walkfail") + ok("a bogus tid does not walk", c.walk_ok(0, 31, [b"threads", b"999999"]) == 1) + + # panic: held, inspectable, capture of the held thread works, trap meanwhile harmless, continue aborts + wr(c, [b"runtime", b"ctl"], b"panic") + time.sleep(0.3) + ok("panic message published", rd(c, [b"panic", b"message"])[1] == b"demo panic requested over 9p") + ok("panic stack names workerLoop", b"workerLoop" in rd(c, [b"panic", b"stack"])[1]) + ok("capture of the held panicking thread answers", rd(c, [b"threads", w, b"stack"])[0] == Rread) + ok("trap request while a panic is held is harmless", wr(c, [b"runtime", b"ctl"], b"trap")[0] == Rwrite and s.alive()) + ok("panic continue", wr(c, [b"panic", b"ctl"], b"continue")[0] == Rwrite) + ok("second panic continue is an error", wr(c, [b"panic", b"ctl"], b"continue") == ("err", "file does not exist")) + time.sleep(1.0) + ok("process aborted after continue", not s.alive() and s.proc.poll() not in (0, None), s.proc.poll()) + s.stop() + + s = Srv(server, ["--no-hold"]) + c = client(s.path, timeout=5) + wr(c, [b"runtime", b"ctl"], b"panic") + time.sleep(1.0) + ok("--no-hold: panic aborts at once", not s.alive() and s.proc.poll() not in (0, None), s.proc.poll()) + s.stop() + + s = Srv(server) + c = client(s.path, timeout=5) + wr(c, [b"runtime", b"ctl"], b"trap") + time.sleep(0.3) + ok("worker parked", ls(c, [b"breakpoints"]) != []) + path = s.path + s.proc.send_signal(signal.SIGTERM) + try: + rc = s.proc.wait(timeout=5) + except subprocess.TimeoutExpired: + rc = None + ok("SIGTERM with a parked thread exits promptly", rc is not None, rc) + ok("SIGTERM unlinks the unix socket (clean stop path)", not os.path.exists(path)) + s.stop() + + +def attack_probe(ns, server): + print("# probe loop and admission") + s = Srv(server) + c0 = cpu_ticks(s.pid) + time.sleep(5.0) + ok("0 CPU ticks over 5 s idle", cpu_ticks(s.pid) - c0 == 0, cpu_ticks(s.pid) - c0) + f0 = fds(s.pid) + for i in range(1000): + so = socket.socket(socket.AF_UNIX, socket.SOCK_STREAM) + so.connect(s.path) + if i % 3 == 0: + so.sendall(frame(Tversion, NOTAG, struct.pack(" <9proc-demo> (part of zig build 9proc-adv) +# Exit 1 on any failure; SKIP (exit 0) without python3. +set -u +NS=$(realpath "${1:?path to 9ns}") +PROC=$(realpath "${2:?path to 9proc-demo}") +command -v python3 >/dev/null || { echo "SKIP: python3 missing"; exit 0; } +exec python3 "$(dirname "$0")/adv_linux_probe.py" --ns "$NS" --server "$PROC" diff --git a/9proc/test/adversarial.sh b/9proc/test/adversarial.sh new file mode 100755 index 0000000..f41c276 --- /dev/null +++ b/9proc/test/adversarial.sh @@ -0,0 +1,19 @@ +#!/usr/bin/env bash +# Runs every 9proc adversarial suite in sequence: hostile raw-9P clients +# against the demo (framing, tags, floods; the core's /vars, snapshots, fids) +# and the Linux layer through one 9ns mount (memory, signals, poll loop). +# Usage: bash 9proc/test/adversarial.sh <9ns> <9proc-demo> [--fast] +# `zig build 9proc-adv` runs the same suites as separate steps. +set -u +NS=${1:?path to 9ns} +PROC=${2:?path to 9proc-demo} +shift 2 +HERE=$(cd "$(dirname "$0")" && pwd) +status=0 +for suite in adv_9proc_hostile adv_core_hostile; do + echo "### $suite" + if bash "$HERE/$suite.sh" "$PROC" "$@"; then echo "### $suite: ok"; else echo "### $suite: FAILED"; status=1; fi +done +echo "### adv_linux_probe" +if bash "$HERE/adv_linux_probe.sh" "$NS" "$PROC"; then echo "### adv_linux_probe: ok"; else echo "### adv_linux_probe: FAILED"; status=1; fi +exit $status diff --git a/9proc/test/debug.sh b/9proc/test/debug.sh new file mode 100755 index 0000000..1334dff --- /dev/null +++ b/9proc/test/debug.sh @@ -0,0 +1,84 @@ +#!/usr/bin/env bash +# End-to-end test of the 9proc debug facilities through 9ns: +# threads and stacks, address resolution, memory, exposed values, breakpoints, panics. +# Usage: bash 9proc/test/debug.sh <9ns> <9proc-demo> (zig build 9proc-debug-itest) +set -u +NS=$(realpath "${1:?path to 9ns}") +PROC=$(realpath "${2:?path to 9proc-demo}") +TMP=$(mktemp -d /tmp/9pdbg.XXXXXX) +M=/mnt/9p +FAILED=0; PASSED=0 +SRV= + +cleanup() { [ -n "$SRV" ] && kill "$SRV" 2>/dev/null; rm -rf "$TMP"; } +trap cleanup EXIT +if ! unshare -Urm true 2>/dev/null || [ ! -c /dev/fuse ]; then echo "SKIP: namespaces or /dev/fuse unavailable"; exit 0; fi + +pass() { PASSED=$((PASSED + 1)); echo "ok - $1"; } +fail() { FAILED=$((FAILED + 1)); echo "FAIL - $1"; shift; [ $# -gt 0 ] && printf ' %s\n' "$@"; } +expect_eq() { if [ "$2" = "$3" ]; then pass "$1"; else fail "$1" "expected: $(printf %q "$2")" "actual: $(printf %q "$3")"; fi; } +expect_contains() { case "$3" in *"$2"*) pass "$1" ;; *) fail "$1" "missing: $(printf %q "$2")" "in: $(printf %q "$3")" ;; esac; } +run_in() { timeout 60 "$NS" --unix "$SOCK" -- sh -c "$1" 2>"$TMP/stderr"; } + +SOCK=$TMP/dbg.sock +"$PROC" --unix "$SOCK" >"$TMP/server.log" 2>&1 & +SRV=$! +for _ in $(seq 1 100); do [ -S "$SOCK" ] && break; sleep 0.05; done +[ -S "$SOCK" ] || { echo "server did not start"; cat "$TMP/server.log"; exit 1; } + +echo "# threads" +expect_eq "threads listed" "yes" "$(run_in "[ \$(ls $M/threads | wc -l) -ge 2 ] && echo yes")" +WORKER=$(run_in "for t in $M/threads/*; do if grep -q '^worker' \$t/name 2>/dev/null; then basename \$t; fi; done | head -1") +expect_eq "worker thread found by name" "yes" "$([ -n "$WORKER" ] && echo yes)" +STACK=$(run_in "cat $M/threads/$WORKER/stack") +expect_contains "worker stack names workerLoop" "workerLoop" "$STACK" +expect_contains "worker stack has source locations" "demo/main.zig:" "$STACK" +expect_contains "worker regs" "0x" "$(run_in "cat $M/threads/$WORKER/regs | head -3")" +expect_eq "own (server) thread stack works" "yes" "$(run_in "for t in $M/threads/*; do cat \$t/stack >/dev/null 2>&1 || echo bad; done; echo yes")" + +echo "# addresses and memory" +FRAME=$(printf '%s\n' "$STACK" | grep -oE '0x[0-9a-f]+' | head -1) +expect_contains "addr resolves a stack frame to the demo source" "demo/main.zig" "$(run_in "cat $M/addr/${FRAME#0x}")" +expect_contains "addr of garbage is an error, not a crash" "No such file" "$(run_in "cat $M/addr/zzz 2>&1")" +expect_contains "mem/maps readable" "r-xp" "$(run_in "head -c 4000 $M/mem/maps")" +STATE_ADDR=$(run_in "cat $M/vars/state/addr") +expect_contains "hexdump of the exposed state" " " "$(run_in "head -2 $M/hex/${STATE_ADDR#0x}")" +expect_eq "raw bytes of the state match its size" "$(run_in "cat $M/vars/state/size")" "$(run_in "cat $M/mem/${STATE_ADDR#0x} | head -c \$(cat $M/vars/state/size) | wc -c")" +expect_eq "reading unmapped memory is an error, not a crash" "no" "$(run_in "cat $M/mem/8 >/dev/null 2>&1 && echo yes || echo no")" + +echo "# exposed values" +T1=$(run_in "cat $M/vars/state/f/ticks/value"); sleep 0.4; T2=$(run_in "cat $M/vars/state/f/ticks/value") +expect_eq "ticks is numeric" "num" "$(printf '%s' "$T1" | grep -Eq '^[0-9]+$' && echo num)" +expect_eq "ticks advance" "yes" "$([ "$T2" -gt "$T1" ] 2>/dev/null && echo yes)" +expect_contains "rendered struct value" "ticks" "$(run_in "cat $M/vars/state/value")" +expect_contains "type name" "State" "$(run_in "cat $M/vars/state/type")" +T3=$(run_in "echo 5 > $M/vars/state/f/ticks/value && cat $M/vars/state/f/ticks/value") +expect_eq "writing a scalar changes the live variable" "yes" "$([ "$T3" -lt "$T2" ] 2>/dev/null && echo yes)" + +echo "# breakpoints" +expect_eq "no breakpoints initially" "" "$(run_in "ls $M/breakpoints")" +run_in "echo trap > $M/runtime/ctl" >/dev/null; sleep 0.6 +PAUSED=$(run_in "ls $M/breakpoints | head -1") +expect_eq "worker paused at @breakpoint()" "$WORKER" "$PAUSED" +expect_contains "paused stack names workerLoop" "workerLoop" "$(run_in "cat $M/breakpoints/$WORKER/stack 2>&1")" +P1=$(run_in "cat $M/vars/state/f/ticks/value"); sleep 0.4; P2=$(run_in "cat $M/vars/state/f/ticks/value") +expect_eq "ticks frozen while paused" "$P1" "$P2" +run_in "echo continue > $M/breakpoints/$WORKER/ctl" >/dev/null; sleep 0.4 +expect_eq "breakpoint list empty after continue" "" "$(run_in "ls $M/breakpoints")" +P3=$(run_in "cat $M/vars/state/f/ticks/value") +expect_eq "ticks advance after continue" "yes" "$([ "$P3" -gt "$P2" ] 2>/dev/null && echo yes)" + +echo "# panic" +expect_eq "no panic recorded" "" "$(run_in "cat $M/panic/message")" +run_in "echo panic > $M/runtime/ctl" >/dev/null; sleep 0.6 +expect_contains "panic message published" "demo panic" "$(run_in "cat $M/panic/message")" +expect_contains "panic stack names the worker" "workerLoop" "$(run_in "cat $M/panic/stack")" +expect_eq "server still alive while holding the panic" "yes" "$(kill -0 $SRV 2>/dev/null && echo yes)" +run_in "echo continue > $M/panic/ctl" >/dev/null 2>&1 +for _ in $(seq 1 50); do kill -0 $SRV 2>/dev/null || break; sleep 0.1; done +if kill -0 $SRV 2>/dev/null; then fail "server exits after panic continue"; else wait $SRV; RC=$?; SRV=; expect_eq "server exit status is non-zero after the panic" "yes" "$([ $RC -ne 0 ] && echo yes)"; fi +expect_contains "default panic output reached stderr" "demo panic" "$(cat "$TMP/server.log")" + +echo +echo "passed=$PASSED failed=$FAILED" +[ "$FAILED" -eq 0 ] diff --git a/README.md b/README.md index d1e52f9..79d044f 100644 --- a/README.md +++ b/README.md @@ -79,31 +79,36 @@ tags, counts, negotiated frame sizes, and flush completion. ## Programs -Programs built on cloud9 live in sibling directories of `src/`, each with its +Related programs live in `cloud9//`, one directory per program beside +`src/`, and carry `9`-prefixed names (`9web`, `9proc`, `9ns`). Each has its own sources, tests, README and a `build.zig` fragment that the root -`build.zig` imports and enables with a toggle (`zig build --help` lists the -steps). New related programs follow the same layout. - -* [`introspect/`](introspect/README.md) — a 9P debug/introspection server as - a library (module `introspect`, exported next to `cloud9`; freestanding - core, Linux debug layer) and its demo binary `introspect`. - `-Dintrospect=[bool]`; steps `introspect`, `introspect-test`, - `introspect-check-freestanding`, `introspect-debug-test`, - `introspect-debug-itest`, `introspect-adv`. -* [`9player/`](9player/README.md) — mount a 9P tree into a fresh user+mount +`build.zig` imports and enables with a `-D` toggle (`zig build --help` +lists the steps). New related programs follow the same layout. + +* [`web/`](docs/http.md) — the HTTP/WebSocket gateway `9web` and its + WebAssembly browser client (`web/main.zig`, `web/client.zig`, assets under + `web/static/`). Always built; steps `serve`, `web`, `http-test`, `e2e`. +* [`9proc/`](9proc/README.md) — a 9P debug/introspection server as + a library (module `9proc`, exported next to `cloud9`; freestanding + core, Linux debug layer) and its demo binary `9proc-demo`. + `-D9proc=[bool]`; steps `9proc`, `9proc-test`, + `9proc-check-freestanding`, `9proc-debug-test`, + `9proc-debug-itest`, `9proc-adv`. +* [`9ns/`](9ns/README.md) — mount a 9P tree into a fresh user+mount namespace via FUSE and run a program in it, without root (Linux, no libc). - `-D9player=[bool]`; steps `9player`, `9player-test`, `9player-itest`, - `9player-adv`. + `-D9ns=[bool]`; steps `9ns`, `9ns-test`, `9ns-itest`, + `9ns-adv`. -Both toggles default to on for Linux targets and off elsewhere; enabled -programs are installed by the plain `zig build`, and `programs-test` / -`programs-itest` run every enabled program's unit and integration steps. A -dependent package gets both modules from the one dependency: +The `9proc` and `9ns` toggles default to on for Linux targets and off +elsewhere; enabled programs are installed by the plain `zig build` next to +`9web` and `cloud9-probe`, and `programs-test` / `programs-itest` run every +enabled program's unit and integration steps. A dependent package gets both +modules from the one dependency: ```zig const cloud9_dep = b.dependency("cloud9", .{ .target = target, .optimize = optimize }); app.root_module.addImport("cloud9", cloud9_dep.module("cloud9")); -app.root_module.addImport("introspect", cloud9_dep.module("introspect")); +app.root_module.addImport("9proc", cloud9_dep.module("9proc")); ``` See [design and ownership contracts](docs/design.md), diff --git a/app/client.zig b/app/client.zig deleted file mode 100644 index 2ec9163..0000000 --- a/app/client.zig +++ /dev/null @@ -1,176 +0,0 @@ -//! Browser ABI. Protocol state, encoding, reply validation, and directory decoding -//! all use the same cloud9 sources as native clients. No allocator or host imports. -const std = @import("std"); -const c9 = @import("cloud9"); -const capacity = 65536; -var input: [capacity]u8 = undefined; -var output: [capacity]u8 = undefined; -var staging: [capacity]u8 = undefined; -var json: [capacity * 2]u8 = undefined; -var client: c9.Client = undefined; -var data: []const u8 = ""; -var number: u32 = 0; -var directory: bool = false; - -export fn init() void { - client = .init(.{ .in = &input, .out = &output }); - data = ""; - number = 0; - directory = false; -} -export fn input_ptr() [*]u8 { - return &staging; -} -export fn input_capacity() u32 { - return capacity; -} -export fn output_ptr() [*]const u8 { - return client.output().ptr; -} -export fn output_len() u32 { - return @intCast(client.output().len); -} -export fn sent() void { - client.wrote(client.output().len); -} -export fn data_ptr() [*]const u8 { - return data.ptr; -} -export fn data_len() u32 { - return @intCast(data.len); -} -export fn result_number() u32 { - return number; -} -export fn result_directory() bool { - return directory; -} -export fn is_dead() bool { - return client.dead; -} -export fn max_read() u32 { - return client.maxRead(); -} -export fn max_write() u32 { - return client.maxWrite(); -} - -fn fail(message: []const u8) i32 { - data = message; - return -1; -} -fn submit(request: c9.Client.Request) i32 { - _ = client.submit(request) catch |err| return fail(@errorName(err)); - return 0; -} -export fn version() i32 { - return submit(.{ .version = .{} }); -} -export fn attach(user_len: u32, tree_len: u32) i32 { - if (user_len > capacity or tree_len > capacity - user_len) return fail("InputTooLarge"); - return submit(.{ .attach = .{ .fid = 0, .uname = staging[0..user_len], .aname = staging[user_len..][0..tree_len] } }); -} -export fn walk(fid: u32, newfid: u32, len: u32) i32 { - if (len > capacity) return fail("InputTooLarge"); - var names: [c9.max_welem][]const u8 = undefined; - var count: usize = 0; - var it = std.mem.splitScalar(u8, staging[0..len], '/'); - while (it.next()) |name| { - if (name.len == 0) continue; - if (count == names.len) return fail("TooManyNames"); - names[count] = name; - count += 1; - } - return submit(.{ .walk = .{ .fid = fid, .newfid = newfid, .names = names[0..count] } }); -} -export fn open(fid: u32, mode: u8) i32 { - return submit(.{ .open = .{ .fid = fid, .mode = mode } }); -} -export fn read(fid: u32, offset: u64, count: u32) i32 { - return submit(.{ .read = .{ .fid = fid, .offset = offset, .count = count } }); -} -export fn write(fid: u32, offset: u64, len: u32) i32 { - if (len > capacity) return fail("InputTooLarge"); - return submit(.{ .write = .{ .fid = fid, .offset = offset, .data = staging[0..len] } }); -} -export fn clunk(fid: u32) i32 { - return submit(.{ .clunk = .{ .fid = fid } }); -} -export fn stat(fid: u32) i32 { - return submit(.{ .stat = .{ .fid = fid } }); -} - -fn writeStat(s: c9.Stat, writer: *std.Io.Writer) !void { - try std.json.Stringify.value(.{ .name = s.name, .directory = s.qid.type & c9.qtdir != 0, .length = s.length, .mode = s.mode }, .{}, writer); -} - -/// Result codes: 0 incomplete, -1 error, 1 version, 2 attach, 3 walk, -/// 4 open, 5 read, 6 write, 7 clunk, 8 stat. Data is borrowed until next call. -export fn receive(len: u32) i32 { - if (len > capacity) return fail("InputTooLarge"); - if (client.push(staging[0..len]) != len) return fail("InputFull"); - const done = client.take() orelse return if (client.dead) fail("ProtocolError") else 0; - data = ""; - number = 0; - directory = false; - return switch (done.result) { - .fail => |message| fail(message), - .version => |v| blk: { - data = v.version; - number = v.msize; - break :blk 1; - }, - .attach => |qid| blk: { - directory = qid.type & c9.qtdir != 0; - break :blk 2; - }, - .walk => |w| blk: { - number = w.nwqid; - if (w.nwqid != 0) directory = w.wqid[w.nwqid - 1].type & c9.qtdir != 0; - break :blk 3; - }, - .open => |o| blk: { - number = o.iounit; - directory = o.qid.type & c9.qtdir != 0; - break :blk 4; - }, - .read => |bytes| blk: { - data = bytes; - break :blk 5; - }, - .write => |count| blk: { - number = count; - break :blk 6; - }, - .clunk => 7, - .stat => |s| blk: { - var writer: std.Io.Writer = .fixed(&json); - writeStat(s, &writer) catch return fail("OutputTooLarge"); - data = writer.buffered(); - break :blk 8; - }, - else => fail("UnexpectedReply"), - }; -} - -/// Parse complete directory stat records from one Rread, using cloud9.Stat. -export fn decode_directory(len: u32) i32 { - if (len > capacity) return fail("InputTooLarge"); - var remaining: []const u8 = staging[0..len]; - var writer: std.Io.Writer = .fixed(&json); - writer.writeByte('[') catch unreachable; - var first = true; - while (remaining.len != 0) { - if (remaining.len < 2) return fail("TruncatedDirectory"); - const size: usize = @as(usize, std.mem.readInt(u16, remaining[0..2], .little)) + 2; - if (size > remaining.len) return fail("TruncatedDirectory"); - const s = c9.Stat.decode(remaining[0..size]) catch return fail("InvalidDirectory"); - if (!first) writer.writeByte(',') catch return fail("OutputTooLarge"); - writeStat(s, &writer) catch return fail("OutputTooLarge"); - first = false; - remaining = remaining[size..]; - } - writer.writeByte(']') catch return fail("OutputTooLarge"); - data = writer.buffered(); - return 0; -} diff --git a/app/main.zig b/app/main.zig deleted file mode 100644 index 720503f..0000000 --- a/app/main.zig +++ /dev/null @@ -1,204 +0,0 @@ -//! HTTP application; no web or namespace policy is added to the cloud9 library. -const std = @import("std"); -const c9 = @import("cloud9"); -const Io = std.Io; -const max_frame = 65536; -const max_connections = 32; -var serial_busy: std.atomic.Value(bool) = .init(false); -const Upstream = union(enum) { network: c9.transport.Address, file: []const u8 }; -var connections: std.atomic.Value(u32) = .init(0); - -const Config = struct { - upstream: Upstream, - host: []const u8, - origin: []const u8, - public_host: []const u8, - timeout_ms: u32, - browser_config: []const u8, -}; - -pub fn main(init: std.process.Init) !void { - const allocator = init.arena.allocator(); - const args = try init.minimal.args.toSlice(allocator); - var listen_text: []const u8 = "127.0.0.1:8080"; - var upstream_text: []const u8 = "tcp:127.0.0.1:564"; - var timeout_ms: u32 = 300000; - var user: []const u8 = "user"; - var tree: []const u8 = ""; - var public_origin: ?[]const u8 = null; - var i: usize = 1; - while (i < args.len) : (i += 1) { - if (std.mem.eql(u8, args[i], "--help")) { - std.debug.print("Usage: cloud9-http [--listen IP:PORT] [--upstream tcp:IP:PORT|unix:PATH|file:PATH] [--origin https://HOST:PORT] [--timeout-ms 300000] [--user NAME] [--tree NAME]\nDefault: http://127.0.0.1:8080 -> tcp:127.0.0.1:564\n", .{}); - return; - } - if (i + 1 == args.len) return error.MissingArgument; - if (std.mem.eql(u8, args[i], "--listen")) { - i += 1; - listen_text = args[i]; - } else if (std.mem.eql(u8, args[i], "--upstream")) { - i += 1; - upstream_text = args[i]; - } else if (std.mem.eql(u8, args[i], "--origin")) { - i += 1; - public_origin = args[i]; - } else if (std.mem.eql(u8, args[i], "--timeout-ms")) { - i += 1; - timeout_ms = try std.fmt.parseInt(u32, args[i], 10); - } else if (std.mem.eql(u8, args[i], "--user")) { - i += 1; - user = args[i]; - } else if (std.mem.eql(u8, args[i], "--tree")) { - i += 1; - tree = args[i]; - } else return error.UnknownArgument; - } - const address = try Io.net.IpAddress.parseLiteral(listen_text); - const upstream: Upstream = if (std.mem.startsWith(u8, upstream_text, "tcp:")) - .{ .network = .{ .tcp = try Io.net.IpAddress.parseLiteral(upstream_text[4..]) } } - else if (std.mem.startsWith(u8, upstream_text, "unix:")) - .{ .network = .{ .unix = try allocator.dupeZ(u8, upstream_text[5..]) } } - else if (std.mem.startsWith(u8, upstream_text, "file:")) - .{ .file = upstream_text[5..] } - else - return error.InvalidUpstream; - const io = init.io; - var listener = c9.transport.listen(io, .{ .tcp = address }, max_connections) catch |err| { - if (err == error.AddressInUse) { - std.debug.print("cloud9-http: cannot listen on {s}: address already in use.\nUse --listen 127.0.0.1:0 to select a free port; the URL is printed at startup.\n", .{listen_text}); - } else { - std.debug.print("cloud9-http: cannot listen on {s}: {s}\n", .{ listen_text, @errorName(err) }); - } - return err; - }; - defer listener.deinit(io); - const host = try std.fmt.allocPrint(allocator, "{f}", .{listener.socket.address}); - const origin_text = public_origin orelse try std.fmt.allocPrint(allocator, "http://{s}", .{host}); - const origin_uri = try std.Uri.parse(origin_text); - if ((!std.mem.eql(u8, origin_uri.scheme, "https") and !std.mem.eql(u8, origin_uri.scheme, "http")) or - origin_uri.host == null or origin_uri.user != null or origin_uri.password != null or - origin_uri.path.percent_encoded.len != 0 or origin_uri.query != null or origin_uri.fragment != null) return error.InvalidOrigin; - if (user.len > 256 or tree.len > 256) return error.AttachNameTooLong; - const browser_config = try std.json.Stringify.valueAlloc(allocator, .{ .user = user, .tree = tree }, .{}); - const config: Config = .{ .upstream = upstream, .host = host, .origin = origin_text, .public_host = origin_text[origin_uri.scheme.len + 3 ..], .timeout_ms = timeout_ms, .browser_config = browser_config }; - std.debug.print("cloud9-http {s} -> {s}\n", .{ config.origin, upstream_text }); - var group: Io.Group = .init; - defer group.cancel(io); - while (true) { - const stream = try listener.accept(io); - if (connections.fetchAdd(1, .monotonic) >= max_connections) { - _ = connections.fetchSub(1, .monotonic); - stream.close(io); - continue; - } - group.concurrent(io, handle, .{ io, stream, config }) catch |err| { - _ = connections.fetchSub(1, .monotonic); - stream.close(io); - return err; - }; - } -} - -fn handle(io: Io, stream: Io.net.Stream, config: Config) void { - defer _ = connections.fetchSub(1, .monotonic); - defer stream.close(io); - var work: Connection = .{ .io = io, .stream = stream, .config = config }; - var group: Io.Group = .init; - defer group.cancel(io); - group.concurrent(io, Connection.run, .{&work}) catch return; - const timeout: Io.Timeout = if (config.timeout_ms == 0) .none else .{ .duration = .{ .raw = .fromMilliseconds(config.timeout_ms), .clock = .awake } }; - work.done.waitTimeout(io, timeout) catch {}; -} - -const Connection = struct { - io: Io, - stream: Io.net.Stream, - config: Config, - done: Io.Event = .unset, - fn run(connection: *Connection) void { - defer connection.done.set(connection.io); - serve(connection.io, connection.stream, connection.config) catch {}; - } -}; - -fn respond(request: *std.http.Server.Request, content: []const u8, mime: []const u8, status: std.http.Status) !void { - try request.respond(content, .{ .status = status, .keep_alive = false, .extra_headers = &.{ - .{ .name = "content-type", .value = mime }, - .{ .name = "cache-control", .value = "no-store" }, - .{ .name = "x-content-type-options", .value = "nosniff" }, - .{ .name = "content-security-policy", .value = "default-src 'self'; script-src 'self' 'wasm-unsafe-eval'; style-src 'self'; connect-src 'self'; frame-ancestors 'none'; base-uri 'none'" }, - } }); -} - -fn serve(io: Io, stream: Io.net.Stream, config: Config) !void { - var input: [max_frame + 14]u8 = undefined; - var output: [8192]u8 = undefined; - var reader = stream.reader(io, &input); - var writer = stream.writer(io, &output); - var http: std.http.Server = .init(&reader.interface, &writer.interface); - http.reader.max_head_len = 8192; - var request = try http.receiveHead(); - var host: ?[]const u8 = null; - var origin: ?[]const u8 = null; - var version: ?[]const u8 = null; - var headers = request.iterateHeaders(); - while (headers.next()) |header| { - if (std.ascii.eqlIgnoreCase(header.name, "host")) host = header.value; - if (std.ascii.eqlIgnoreCase(header.name, "origin")) origin = header.value; - if (std.ascii.eqlIgnoreCase(header.name, "sec-websocket-version")) version = header.value; - } - if (!std.mem.eql(u8, host orelse "", config.host) and !std.mem.eql(u8, host orelse "", config.public_host)) return respond(&request, "Unexpected Host\n", "text/plain", .forbidden); - if (request.head.method != .GET) return respond(&request, "Use GET\n", "text/plain", .method_not_allowed); - const path = std.mem.sliceTo(request.head.target, '?'); - if (std.mem.eql(u8, path, "/_cloud9/config.json")) return respond(&request, config.browser_config, "application/json", .ok); - if (std.mem.eql(u8, path, "/_cloud9/app.mjs")) return respond(&request, @embedFile("web/app.mjs"), "text/javascript; charset=utf-8", .ok); - if (std.mem.eql(u8, path, "/_cloud9/client.mjs")) return respond(&request, @embedFile("web/client.mjs"), "text/javascript; charset=utf-8", .ok); - if (std.mem.eql(u8, path, "/_cloud9/style.css")) return respond(&request, @embedFile("web/style.css"), "text/css; charset=utf-8", .ok); - if (std.mem.eql(u8, path, "/_cloud9/cloud9.wasm")) return respond(&request, @embedFile("client.wasm"), "application/wasm", .ok); - if (!std.mem.eql(u8, path, "/_cloud9/9p")) { - if (std.mem.eql(u8, path, "/_cloud9") or std.mem.startsWith(u8, path, "/_cloud9/")) - return respond(&request, "Not found\n", "text/plain", .not_found); - // File URLs boot the same browser client. The client resolves the path - // in the upstream 9P namespace, never in this machine's filesystem. - return respond(&request, @embedFile("web/index.html"), "text/html; charset=utf-8", .ok); - } - if (!std.mem.eql(u8, origin orelse "", config.origin)) return respond(&request, "Unexpected Origin\n", "text/plain", .forbidden); - if (!std.mem.eql(u8, version orelse "", "13")) return respond(&request, "WebSocket version 13 required\n", "text/plain", .bad_request); - switch (config.upstream) { - .network => |address| { - const upstream = c9.transport.connect(io, address) catch return respond(&request, "9P upstream unavailable\n", "text/plain", .bad_gateway); - defer upstream.close(io); - var in_buffer: [8192]u8 = undefined; - var out_buffer: [8192]u8 = undefined; - var upstream_reader = upstream.reader(io, &in_buffer); - var upstream_writer = upstream.writer(io, &out_buffer); - try bridge(io, &request, &upstream_reader.interface, &upstream_writer.interface); - }, - .file => |path_name| { - if (serial_busy.swap(true, .acquire)) return respond(&request, "Device already in use\n", "text/plain", .service_unavailable); - defer serial_busy.store(false, .release); - const file = Io.Dir.cwd().openFile(io, path_name, .{ .mode = .read_write }) catch return respond(&request, "9P device unavailable\n", "text/plain", .bad_gateway); - defer file.close(io); - var in_buffer: [8192]u8 = undefined; - var out_buffer: [8192]u8 = undefined; - var upstream_reader = file.readerStreaming(io, &in_buffer); - var upstream_writer = file.writerStreaming(io, &out_buffer); - try bridge(io, &request, &upstream_reader.interface, &upstream_writer.interface); - }, - } -} - -fn bridge(io: Io, request: *std.http.Server.Request, reader: *Io.Reader, writer: *Io.Writer) !void { - var ws = try c9.http.accept(request); - var requests: [max_frame]u8 = undefined; - var replies: [max_frame]u8 = undefined; - var relay: c9.http.Bridge = .{ - .socket = &ws, - .upstream_reader = reader, - .upstream_writer = writer, - .request_buffer = &requests, - .reply_buffer = &replies, - .frame_limit = max_frame, - }; - try relay.run(io, .none); -} diff --git a/app/web/app.mjs b/app/web/app.mjs deleted file mode 100644 index a6985f0..0000000 --- a/app/web/app.mjs +++ /dev/null @@ -1,183 +0,0 @@ -import { Client } from './client.mjs'; -const $ = id => document.getElementById(id); -const text = new TextDecoder('utf-8', { fatal: true }); -let client = null, current = null, currentPath = '/', busy = false, dirty = false; -let configuration = null; - -function status(message, error = false) { - $('status').textContent = message; - $('status').classList.toggle('error', error); -} -function controls() { - for (const id of ['go', 'reload']) $(id).disabled = busy; - $('save').disabled = busy || !current || current.stat.directory || $('editor').hidden || !dirty; - $('download').disabled = busy || !current; - $('editor').readOnly = busy; - $('edit-status').textContent = dirty ? 'Unsaved changes.' : 'No unsaved changes.'; - document.querySelector('main').setAttribute('aria-busy', String(busy)); -} -async function action(operation) { - if (busy) return; - busy = true; - controls(); - try { await operation(); } - catch (error) { status(error.message, true); } - finally { busy = false; controls(); } -} -async function connectedClient() { - if (client && !client.closed) return client; - if (!configuration) { - const response = await fetch('/_cloud9/config.json'); - if (!response.ok) throw new Error('Unable to load server settings. Reload to try again.'); - configuration = await response.json(); - } - try { client = await Client.connect('/_cloud9/9p', configuration.user, configuration.tree); } - catch { throw new Error('Unable to reach the file server. Reload to try again.'); } - return client; -} -async function readPath(path) { - const active = await connectedClient(); - try { return await active.readPath(path); } - catch (error) { - // A timed-out/closed transport has lost its fids. Retry reads once on a - // fresh session. Writes are never replayed after an uncertain result. - if (!active.closed) throw error; - return (await connectedClient()).readPath(path); - } -} -function normalizePath(path) { - const parts = []; - for (const name of path.split('/')) { - if (!name || name === '.') continue; - if (name === '..') parts.pop(); else parts.push(name); - } - return '/' + parts.join('/'); -} -function pathURL(path) { - const parts = path.split('/').map(encodeURIComponent); - // Preserve access to an upstream directory named _cloud9 without colliding - // with the literal gateway prefix. Decode each component exactly once. - if (parts[1] === '_cloud9') parts[1] = '%5Fcloud9'; - return parts.join('/'); -} -function locationPath() { - const parts = location.pathname.split('/').map(part => { - let name; - try { name = decodeURIComponent(part); } - catch { throw new Error('The URL contains invalid path encoding.'); } - if (name.includes('/') || name.includes('\0')) throw new Error('The URL contains an invalid path component.'); - return name; - }); - return normalizePath(parts.join('/')); -} -function child(path, name) { return `${path.replace(/\/$/, '')}/${name}`; } -function mode(stat) { - let result = stat.directory ? 'd' : '-'; - for (let bit = 8; bit >= 0; bit--) result += stat.mode & (1 << bit) ? 'rwx'[(8 - bit) % 3] : '-'; - return result; -} -function pathLink(label, path, className = '') { - const link = document.createElement('a'); - link.textContent = label; - link.href = pathURL(path); - link.dataset.path = path; - link.className = className; - return link; -} -function breadcrumbs(path) { - const node = $('breadcrumbs'); - node.replaceChildren(pathLink('root', '/')); - let walked = ''; - for (const name of path.split('/').filter(Boolean)) { - walked += '/' + name; - const separator = document.createElement('span'); - separator.textContent = '/'; separator.className = 'separator'; - node.append(separator, pathLink(name, walked)); - } - node.lastElementChild.setAttribute('aria-current', 'page'); - const parent = path.replace(/\/[^/]+$/, '') || '/'; - $('up').hidden = path === '/'; - $('up').href = pathURL(parent); - $('up').dataset.path = parent; -} -async function navigate(path, historyMode = 'push') { - path = normalizePath(path); - status('Loading…'); - const result = await readPath(path); - current = result; currentPath = path; dirty = false; - $('path').value = path; - const url = pathURL(path); - if (historyMode === 'push' && location.pathname + location.search + location.hash !== url) history.pushState(null, '', url); - document.title = `${path} — cloud9`; - breadcrumbs(path); - $('directory').hidden = !result.stat.directory; - $('file').hidden = result.stat.directory; - if (result.stat.directory) { - const rows = document.createDocumentFragment(); - const sorted = result.entries.sort((a, b) => Number(b.directory) - Number(a.directory) || a.name.localeCompare(b.name)); - for (const entry of sorted) { - const row = document.createElement('tr'); - const permissions = document.createElement('td'), name = document.createElement('td'), size = document.createElement('td'); - permissions.className = 'mode'; permissions.textContent = mode(entry); - name.append(pathLink(entry.name + (entry.directory ? '/' : ''), child(path, entry.name), 'entry' + (entry.directory ? ' directory' : ''))); - size.className = 'size'; size.textContent = entry.directory ? '—' : entry.length.toLocaleString(); - row.append(permissions, name, size); rows.append(row); - } - $('entries').replaceChildren(rows); - $('empty').hidden = sorted.length !== 0; - status(`${sorted.length} ${sorted.length === 1 ? 'entry' : 'entries'}`); - } else { - $('filename').textContent = result.stat.name; - $('file-mode').textContent = mode(result.stat); - $('file-size').textContent = `${result.bytes.length.toLocaleString()} bytes`; - let editable = true; - try { - if (result.bytes.includes(0)) throw new Error('Binary'); - $('editor').value = text.decode(result.bytes); - } catch { editable = false; $('editor').value = ''; } - $('editor').hidden = !editable; - $('binary').hidden = editable; - $('edit-actions').hidden = !editable; - status('File loaded.'); - } -} -function mayLeave() { return !dirty || window.confirm('Discard unsaved changes?'); } -document.addEventListener('click', event => { - const link = event.target.closest('a[data-path]'); - if (!link || event.button !== 0 || event.ctrlKey || event.metaKey || event.shiftKey || event.altKey) return; - event.preventDefault(); - if (!busy && mayLeave()) action(() => navigate(link.dataset.path)); -}); -$('path-form').onsubmit = event => { - event.preventDefault(); - if (mayLeave()) action(() => navigate($('path').value)); -}; -$('reload').onclick = () => { if (mayLeave()) action(() => navigate(current ? currentPath : locationPath(), 'none')); }; -$('editor').oninput = () => { dirty = true; controls(); }; -$('save').onclick = () => action(async () => { - const bytes = new TextEncoder().encode($('editor').value); - const active = await connectedClient(); - let count; - try { count = await active.writePath(currentPath, bytes); } - catch (error) { - if (active.closed) throw new Error('Save interrupted. The file may be partially written; your edits are still here.'); - throw error; - } - // Keep the editor intact if refreshing after a successful write fails. - dirty = false; - await navigate(currentPath, 'none'); - status(`Saved ${count.toLocaleString()} bytes.`); -}); -$('download').onclick = () => { - const url = URL.createObjectURL(new Blob([current.bytes])); - const link = document.createElement('a'); link.href = url; link.download = current.stat.name; link.click(); - setTimeout(() => URL.revokeObjectURL(url), 1000); -}; -window.addEventListener('popstate', () => { - if (busy || !mayLeave()) { history.pushState(null, '', pathURL(currentPath)); return; } - action(() => navigate(locationPath(), 'none')); -}); -window.addEventListener('beforeunload', event => { - if (dirty) { event.preventDefault(); event.returnValue = ''; } -}); -action(() => navigate(locationPath(), 'none')); diff --git a/app/web/client.mjs b/app/web/client.mjs deleted file mode 100644 index 69f44d2..0000000 --- a/app/web/client.mjs +++ /dev/null @@ -1,138 +0,0 @@ -// Browser transport and file operations. All 9P serialization lives in WASM. -const encoder = new TextEncoder(); -const decoder = new TextDecoder(); -export class Client { - static async connect(url = '/_cloud9/9p', user = 'user', tree = '') { - const { instance } = await WebAssembly.instantiateStreaming(fetch(new URL('./cloud9.wasm', import.meta.url)), {}); - const wsURL = new URL(url, location.href); - wsURL.protocol = location.protocol === 'https:' ? 'wss:' : 'ws:'; - const socket = new WebSocket(wsURL); - socket.binaryType = 'arraybuffer'; - const client = new Client(instance.exports, socket); - try { - await new Promise((resolve, reject) => { - const timer = setTimeout(() => reject(new Error('Connection timed out')), 10000); - socket.onopen = () => { clearTimeout(timer); resolve(); }; - socket.onerror = () => { clearTimeout(timer); reject(new Error('Cannot connect to 9P bridge')); }; - }); - await client.ask('version'); - const u = encoder.encode(user), a = encoder.encode(tree); - client.stage(new Uint8Array([...u, ...a])); - await client.ask('attach', u.length, a.length); - return client; - } catch (error) { client.close(); throw error; } - } - constructor(wasm, socket) { - this.wasm = wasm; - this.socket = socket; - this.pending = null; - this.queue = Promise.resolve(); - this.closed = false; - wasm.init(); - socket.onmessage = ({ data }) => { - if (!this.pending) { this.close(); return; } - try { - this.stage(new Uint8Array(data)); - const kind = wasm.receive(data.byteLength); - if (!kind) return; - const result = { kind, data: this.data(), number: wasm.result_number(), directory: Boolean(wasm.result_directory()) }; - const pending = this.pending; - this.pending = null; - clearTimeout(pending.timer); - if (kind < 0) { - const error = new Error(decoder.decode(result.data)); - pending.reject(error); - if (wasm.is_dead()) this.abort(error); - } - else pending.resolve(result); - } catch (error) { this.abort(error); } - }; - socket.onclose = () => this.abort(new Error('Connection closed')); - } - data() { return new Uint8Array(this.wasm.memory.buffer, this.wasm.data_ptr(), this.wasm.data_len()).slice(); } - stage(bytes) { - if (bytes.length > this.wasm.input_capacity()) throw new Error('Input exceeds 64 KiB'); - new Uint8Array(this.wasm.memory.buffer, this.wasm.input_ptr(), bytes.length).set(bytes); - return bytes.length; - } - ask(operation, ...args) { - if (this.closed) return Promise.reject(new Error('Connection closed')); - if (this.pending) return Promise.reject(new Error('Request already in flight')); - if (this.wasm[operation](...args) < 0) return Promise.reject(new Error(decoder.decode(this.data()))); - return new Promise((resolve, reject) => { - const timer = setTimeout(() => this.abort(new Error('9P request timed out')), 15000); - this.pending = { resolve, reject, timer }; - try { - this.socket.send(new Uint8Array(this.wasm.memory.buffer, this.wasm.output_ptr(), this.wasm.output_len())); - this.wasm.sent(); - } catch (error) { this.abort(error); } - }); - } - abort(error) { - this.closed = true; - if (this.pending) { clearTimeout(this.pending.timer); this.pending.reject(error); this.pending = null; } - this.socket.close(); - } - close() { this.abort(new Error('Disconnected')); } - serial(operation) { - const result = this.queue.then(operation); - this.queue = result.catch(() => {}); - return result; - } - async withFile(path, operation) { - const names = path.split('/').filter(Boolean); - let live = false; - try { - // Walk in spec-sized batches, retaining an unopened root fid. - for (let index = 0; index < Math.max(1, names.length); index += 16) { - const batch = names.slice(index, index + 16); - const len = this.stage(encoder.encode(batch.join('/'))); - const result = await this.ask('walk', index === 0 ? 0 : 1, 1, len); - live = live || result.number > 0 || batch.length === 0; - if (result.number !== batch.length) throw new Error('Path does not exist'); - } - return await operation(); - } finally { - if (live && !this.closed) await this.ask('clunk', 1); - } - } - readPath(path) { - return this.serial(() => this.withFile(path, async () => { - const stat = JSON.parse(decoder.decode((await this.ask('stat', 1)).data)); - const opened = await this.ask('open', 1, 0); - const count = Math.min(this.wasm.max_read(), opened.number || Infinity); - let offset = 0; - const chunks = [], entries = []; - while (true) { - const { data } = await this.ask('read', 1, BigInt(offset), count); - if (!data.length) break; - offset += data.length; - if (offset > 8 * 1024 * 1024) throw new Error('Browser view is limited to 8 MiB'); - if (stat.directory) { - this.stage(data); - if (this.wasm.decode_directory(data.length) < 0) throw new Error(decoder.decode(this.data())); - entries.push(...JSON.parse(decoder.decode(this.data()))); - } else chunks.push(data); - } - const bytes = new Uint8Array(stat.directory ? 0 : offset); - let position = 0; - for (const chunk of chunks) { bytes.set(chunk, position); position += chunk.length; } - return { stat, entries, bytes }; - })); - } - writePath(path, bytes) { - if (!(bytes instanceof Uint8Array) || bytes.length > 8 * 1024 * 1024) return Promise.reject(new Error('Write is limited to 8 MiB')); - return this.serial(() => this.withFile(path, async () => { - const opened = await this.ask('open', 1, 1 | 16); // OWRITE | OTRUNC - const count = Math.min(this.wasm.max_write(), opened.number || Infinity); - let offset = 0; - while (offset < bytes.length) { - const len = this.stage(bytes.subarray(offset, offset + count)); - const result = await this.ask('write', 1, BigInt(offset), len); - if (!result.number) throw new Error('Server made no write progress'); - offset += result.number; - } - return offset; - })); - } -} diff --git a/app/web/index.html b/app/web/index.html deleted file mode 100644 index aeca5d3..0000000 --- a/app/web/index.html +++ /dev/null @@ -1,53 +0,0 @@ - - - - - - cloud9 — files - - - - -
- cloud9 -
A web interface to your file server.
-
- -
- -
-

Loading files…

- -
-
- - - -
ModeNameSize
- -
- -
-
served by cloud9
- - diff --git a/app/web/style.css b/app/web/style.css deleted file mode 100644 index a8605f5..0000000 --- a/app/web/style.css +++ /dev/null @@ -1,87 +0,0 @@ -:root { - font: 14px/1.45 sans-serif; - color: #333; - background: white; - font-synthesis: none; -} -* { box-sizing: border-box; } -body { margin: 0; padding: 22px 3%; } -a { color: #07539b; text-decoration: none; } -a:hover { text-decoration: underline; } -header { padding: 0 8px 17px; } -.brand { font-size: 30px; font-weight: bold; color: #222; letter-spacing: -1px; } -.description { color: #777; margin-top: 1px; } -.toolbar { - display: flex; - align-items: center; - gap: 2px; - border-bottom: 3px solid #ccc; - padding: 0 7px; -} -.toolbar > a, .toolbar > button { - padding: 5px 14px; - color: #555; - background: none; - border: 0; - font-size: 15px; -} -.toolbar .selected { background: #ccc; color: #111; } -button, input, textarea { font: inherit; color: inherit; } -button { - cursor: pointer; - border: 1px solid #aaa; - padding: 2px 9px; - background: #f5f5f5; - border-radius: 0; -} -button:hover:enabled { background: #e8e8e8; } -button:disabled { color: #999; cursor: default; } -#path-form { margin-left: auto; display: flex; align-items: center; gap: 6px; padding-bottom: 5px; } -#path-form label { color: #777; font-size: 12px; } -#path { width: 250px; border: 1px solid #bbb; padding: 2px 5px; font: 13px/1.5 monospace; } -main { padding: 0 8px; } -#breadcrumbs { display: flex; flex-wrap: wrap; gap: 7px; padding: 13px 0 5px; font-family: monospace; overflow-wrap: anywhere; } -#breadcrumbs .separator { color: #999; } -#breadcrumbs [aria-current] { color: #333; font-weight: bold; } -.summary { display: flex; align-items: baseline; justify-content: space-between; gap: 12px; margin: 2px 0 13px; font-size: 12px; } -#status { color: #777; margin: 0; min-height: 18px; } -#status.error { color: #a22; } -#up { white-space: nowrap; } -table { width: 100%; border-collapse: collapse; text-align: left; } -th { background: #eee; padding: 4px 8px; font-weight: bold; border-bottom: 1px solid #ccc; } -td { padding: 3px 8px; vertical-align: top; } -tbody tr:nth-child(even) { background: #f7f7f7; } -tbody tr:hover { background: #edf3f8; } -.mode { width: 120px; color: #777; font: 12px/1.7 monospace; white-space: nowrap; } -.entry { font-family: monospace; overflow-wrap: anywhere; } -.entry.directory { font-weight: bold; } -.size { width: 110px; text-align: right; font-variant-numeric: tabular-nums; white-space: nowrap; } -td.size { color: #666; font-size: 12px; } -#empty { color: #777; padding: 4px 8px; } -.file-heading { display: flex; flex-wrap: wrap; align-items: baseline; gap: 16px; padding: 6px 8px; background: #eee; border-bottom: 1px solid #ccc; } -h1 { font: bold 14px monospace; margin: 0; overflow-wrap: anywhere; } -#file-mode, #file-size { color: #666; font: 12px monospace; } -#download { margin-left: auto; font-size: 12px; } -textarea { display: block; width: 100%; min-height: 420px; resize: vertical; border: 1px solid #ddd; border-top: 0; padding: 10px; font: 13px/1.6 monospace; tab-size: 4; white-space: pre; overflow: auto; } -#binary { color: #777; padding: 16px 8px; } -.actions { display: flex; align-items: center; gap: 12px; margin-top: 10px; font-size: 12px; } -#edit-status { color: #777; } -footer { margin: 30px 8px 0; padding-top: 7px; border-top: 1px solid #ddd; text-align: right; color: #999; font-size: 11px; } -a:focus-visible, button:focus-visible { outline: 2px solid #07539b; outline-offset: 2px; } -[hidden] { display: none !important; } -@media (max-width: 600px) { - body { padding: 14px 10px; } - header { padding-left: 3px; } - .brand { font-size: 26px; } - .toolbar { flex-wrap: wrap; padding: 0; } - .toolbar > a, .toolbar > button { padding: 5px 10px; } - #path-form { order: -1; width: 100%; margin: 0 0 6px; } - #path { width: auto; min-width: 0; flex: 1; } - main { padding: 0; } - th, td { padding-left: 5px; padding-right: 5px; } - .mode { width: 86px; font-size: 10px; } - .size { width: 65px; } - .summary { align-items: start; } - .file-heading { gap: 7px 12px; } - textarea { min-height: 350px; } -} diff --git a/build.zig b/build.zig index 63d9735..13e54ec 100644 --- a/build.zig +++ b/build.zig @@ -16,7 +16,7 @@ pub fn build(b: *std.Build) void { .optimize = optimize, }); const wasm = b.addExecutable(.{ .name = "cloud9", .root_module = b.createModule(.{ - .root_source_file = b.path("app/client.zig"), + .root_source_file = b.path("web/client.zig"), .target = wasm_target, .optimize = optimize, .imports = &.{.{ .name = "cloud9", .module = wasm_core }}, @@ -30,10 +30,10 @@ pub fn build(b: *std.Build) void { const web = b.step("web", "Install the standalone browser client into zig-out/web"); web.dependOn(&b.addInstallFileWithDir(wasm.getEmittedBin(), .{ .custom = "web/_cloud9" }, "cloud9.wasm").step); for ([_][]const u8{ "index.html", "app.mjs", "client.mjs", "style.css" }) |file| { - web.dependOn(&b.addInstallFileWithDir(b.path(b.fmt("app/web/{s}", .{file})), .{ .custom = if (std.mem.eql(u8, file, "index.html")) "web" else "web/_cloud9" }, file).step); + web.dependOn(&b.addInstallFileWithDir(b.path(b.fmt("web/static/{s}", .{file})), .{ .custom = if (std.mem.eql(u8, file, "index.html")) "web" else "web/_cloud9" }, file).step); } - const bridge = b.addExecutable(.{ .name = "cloud9-http", .root_module = b.createModule(.{ - .root_source_file = b.path("app/main.zig"), + const bridge = b.addExecutable(.{ .name = "9web", .root_module = b.createModule(.{ + .root_source_file = b.path("web/main.zig"), .target = target, .optimize = optimize, .link_libc = true, @@ -44,7 +44,7 @@ pub fn build(b: *std.Build) void { const run_bridge = b.addRunArtifact(bridge); if (b.args) |args| run_bridge.addArgs(args); b.step("serve", "Run the HTTP/WebSocket bridge (--upstream tcp:127.0.0.1:564)").dependOn(&run_bridge.step); - const native_http = b.addExecutable(.{ .name = "cloud9-http-test", .root_module = b.createModule(.{ + const native_http = b.addExecutable(.{ .name = "9web-test", .root_module = b.createModule(.{ .root_source_file = b.path("test/web/native.zig"), .target = target, .optimize = optimize, @@ -127,58 +127,58 @@ pub fn build(b: *std.Build) void { // ------------------------------------------------------------------ // Programs related to cloud9. Each lives in its own directory beside - // src/ (introspect/, 9player/) with a build.zig *fragment* exposing + // src/ (9proc/, 9ns/) with a build.zig *fragment* exposing // `add(b, ctx)`; it is imported here and runs with this builder, so its // b.path() calls are relative to this root and its steps are namespaced // by the program's name. New related programs follow the same pattern: // add a directory, a fragment, a toggle here, and the path to // build.zig.zon. // - // -Dintrospect=[bool] library module (any target) + freestanding check; - // demo server and Linux suites when the target is Linux - // -D9player=[bool] the FUSE mount CLI (Linux only) + // -D9proc=[bool] library module (any target) + freestanding check; + // demo server and Linux suites when the target is Linux + // -D9ns=[bool] the FUSE mount CLI (Linux only) // // Both default to true on a Linux target and false elsewhere. Enabled - // programs are installed by the plain `zig build` next to cloud9-http + // programs are installed by the plain `zig build` next to 9web // and cloud9-probe. const is_linux = target.result.os.tag == .linux; - const want_introspect = b.option(bool, "introspect", "Build the introspect library module and demo (default: target is Linux)") orelse is_linux; - const want_9player = b.option(bool, "9player", "Build the 9player FUSE mount CLI (default: target is Linux; Linux only)") orelse is_linux; - if (want_9player and !is_linux) { - std.debug.print("error: -D9player=true needs a Linux target (got {s})\n", .{@tagName(target.result.os.tag)}); + const want_9proc = b.option(bool, "9proc", "Build the 9proc library module and demo (default: target is Linux)") orelse is_linux; + const want_9ns = b.option(bool, "9ns", "Build the 9ns FUSE mount CLI (default: target is Linux; Linux only)") orelse is_linux; + if (want_9ns and !is_linux) { + std.debug.print("error: -D9ns=true needs a Linux target (got {s})\n", .{@tagName(target.result.os.tag)}); std.process.exit(1); } const programs_test = b.step("programs-test", "Run every enabled program's unit tests and checks"); const programs_itest = b.step("programs-itest", "Run every enabled program's integration suites"); - const introspect_build = @import("introspect/build.zig"); - const nineplayer_build = @import("9player/build.zig"); + const proc_build = @import("9proc/build.zig"); + const ns_build = @import("9ns/build.zig"); - const introspect: ?introspect_build.Artifacts = if (want_introspect) introspect_build.add(b, .{ + const proc: ?proc_build.Artifacts = if (want_9proc) proc_build.add(b, .{ .target = target, .optimize = optimize, .cloud9 = module, }) else null; - if (introspect) |i| { + if (proc) |i| { programs_test.dependOn(i.test_step); programs_test.dependOn(i.check_step); if (i.debug_test_step) |s| programs_test.dependOn(s); } - const nineplayer: ?nineplayer_build.Artifacts = if (want_9player) nineplayer_build.add(b, .{ + const ns: ?ns_build.Artifacts = if (want_9ns) ns_build.add(b, .{ .target = target, .optimize = optimize, .cloud9 = module, - .introspect_demo = if (introspect) |i| i.demo else null, + .proc_demo = if (proc) |i| i.demo else null, }) else null; - if (nineplayer) |p| { + if (ns) |p| { programs_test.dependOn(p.test_step); programs_itest.dependOn(p.itest_step); } - // introspect's end-to-end suites drive the demo through a 9player mount, + // 9proc's end-to-end suites drive the demo through a 9ns mount, // so they are wired once both fragments have run. - if (introspect) |i| { - if (introspect_build.addPlayerTests(b, i, if (nineplayer) |p| p.exe else null)) |s| programs_itest.dependOn(s); + if (proc) |i| { + if (proc_build.addNsTests(b, i, if (ns) |p| p.exe else null)) |s| programs_itest.dependOn(s); } } diff --git a/build.zig.zon b/build.zig.zon index c170b9d..a7a4e3b 100644 --- a/build.zig.zon +++ b/build.zig.zon @@ -3,5 +3,5 @@ .fingerprint = 0xc8b2d5eaa3adfca, .version = "0.1.0", .minimum_zig_version = "0.16.0", - .paths = .{ "build.zig", "build.zig.zon", "src", "app", "test", "docs", "README.md", "9player", "introspect" }, + .paths = .{ "build.zig", "build.zig.zon", "src", "web", "test", "docs", "README.md", "9ns", "9proc" }, } diff --git a/docs/design.md b/docs/design.md index d12b92c..2013b3e 100644 --- a/docs/design.md +++ b/docs/design.md @@ -57,7 +57,7 @@ or buffers. `http.Client` negotiates HTTP/1.1 upgrades using `std.http.Client`, including its certificate-verified HTTPS support. `http.Bridge` relays between an accepted WebSocket and arbitrary standard readers/writers, including UART adapters. The runnable application's web UI, device leases, deadlines, and -origin policy remain in `app/`. See [HTTP ownership and usage](http.md). +origin policy remain in `web/`. See [HTTP ownership and usage](http.md). `Quic(OpenSSL, alpn)` is an optional OpenSSL 3.6+ adapter with a single ordered bidirectional stream per connection. It preserves the previous Pardes transport: @@ -68,15 +68,16 @@ protocol core. Accepted connections must close before their shared listener. # Related programs -Programs built on the library ship from this repository in sibling -directories of `src/` (`introspect/`, `9player/`), never inside it: each has +Programs built on the library ship from this repository as `cloud9//`, +sibling directories of `src/` with `9`-prefixed names (`web/` builds `9web`; +`9proc/` and `9ns/` build `9proc-demo` and `9ns`), never inside it: each has its own `src/`, `test/`, `docs/`, README and a `build.zig` fragment (`pub fn add(b, ctx)`) that the root `build.zig` imports, passes the resolved target, optimize mode and the `cloud9` module to, and enables with a `-D` toggle. Fragments register namespaced steps (``, `-test`, ...), never call `standardTargetOptions`, and use root-relative `b.path("/...")`. The library keeps its contract: mounting, namespaces, -threads, allocation and process policy stay in the program. `introspect` is +threads, allocation and process policy stay in the program. `9proc` is also exported as a module next to `cloud9` for dependents. New related programs follow this layout. diff --git a/docs/http.md b/docs/http.md index f486d52..4e92e5e 100644 --- a/docs/http.md +++ b/docs/http.md @@ -16,7 +16,7 @@ flowchart LR ```sh zig build -Doptimize=ReleaseSafe -./zig-out/bin/cloud9-http --upstream tcp:127.0.0.1:564 +./zig-out/bin/9web --upstream tcp:127.0.0.1:564 # Or run directly: zig build serve -Doptimize=ReleaseSafe -- --upstream unix:/path/to/9p.sock ``` @@ -91,7 +91,7 @@ For a configured serial device: ```sh # Set raw mode, baud rate, and flow control with your platform's serial tools. -./zig-out/bin/cloud9-http --upstream file:/dev/ttyUSB0 +./zig-out/bin/9web --upstream file:/dev/ttyUSB0 ``` The serial path transports 9P bytes; it does not interpret ESP32 boot logs or @@ -174,9 +174,9 @@ checks; it does not allocate fids, mount filesystems, or implement a filesystem. Each Bridge value runs once. Timeout and cancellation stop both pumps before returning. The caller owns and closes the underlying streams. -`app/client.zig` is a browser ABI around the same `cloud9.Client` and `Stat` +`web/client.zig` is a browser ABI around the same `cloud9.Client` and `Stat` implementations. Its WASM memory is fixed at 1 MiB and it imports no host -functions. `app/web/client.mjs` supplies asynchronous WebSocket I/O and file +functions. `web/static/client.mjs` supplies asynchronous WebSocket I/O and file operations. Each browser Client owns a separate WASM instance. The JavaScript contains no 9P encoder or decoder. diff --git a/docs/validation.md b/docs/validation.md index 98b73d6..5b34b82 100644 --- a/docs/validation.md +++ b/docs/validation.md @@ -86,7 +86,7 @@ HTTP/3 is not implemented or tested. The [HTTP E2E report](results/http-e2e.json) records the scenarios and byte counts. The run's screenshots and TLS fixtures remain in `.zig-cache/e2e-3HAs6G/`. -`zig build -Doptimize=ReleaseSafe` installs the standalone `cloud9-http` binary; +`zig build -Doptimize=ReleaseSafe` installs the standalone `9web` binary; `zig build web -Doptimize=ReleaseSafe` also installs the separate browser assets. See [HTTP usage and ownership](http.md) for the reusable API and runnable gateway. diff --git a/introspect/README.md b/introspect/README.md deleted file mode 100644 index ba7120b..0000000 --- a/introspect/README.md +++ /dev/null @@ -1,130 +0,0 @@ -# introspect - -A 9P2000 debug/introspection server as a Zig 0.16 library, built on -[cloud9](../): a debugger-shaped interface where the protocol is just files. -Anything that can read a filesystem (a shell, an agent, an editor, `9p`, -[9player](../9player)) can inspect a running program: build facts, comptime -type layouts, live values, threads and their stacks, memory, breakpoints, -panics. - -The core (`core`, `vars`) is freestanding: no allocator, no OS, no threads, -caller-owned static `Storage`, fixed-capacity tables sized at comptime. It -compiles for `riscv32-freestanding-none`. `scratch` (an in-memory read/write -tree) takes an allocator; `linux` is the platform layer (listeners, a poll -loop on one background thread, threads/stacks/registers, memory, breakpoints -and panics via `std.debug`). [docs/LIBRARY.md](docs/LIBRARY.md) has the full -contract. - -``` -introspect/ - build.zig fragment imported by cloud9's root build.zig (steps below) - src/root.zig pub const core, vars, scratch, linux; Config, Server(cfg), Provider - src/core.zig Tree/Server engine on cloud9.Server: fids, walks, dir reads, providers - src/vars.zig comptime value renderers (@typeInfo) for /vars - src/scratch.zig in-memory read/write tree provider (takes an Allocator) - src/freestanding_check.zig root for the riscv32-freestanding-none compile check - src/linux/probe.zig background thread + poll loop + unix/tcp/fd listeners - src/linux/debug.zig threads, stacks, registers, addr→source, memory, breakpoints, panic - src/linux/provider.zig the debug provider (/threads, /addr, /mem, /hex, /breakpoints, /panic) - src/linux/runtime.zig /runtime generators (pid, uptime, argv, cwd, env, clients) - demo/main.zig the `introspect` binary (below) - test/ debug.sh, adv_introspect_hostile, adv_core_hostile, adv_linux_probe, adversarial.sh - docs/LIBRARY.md design rules and the module contracts -``` - -## Using the library - -cloud9's `build.zig` exports two modules: `cloud9` (the protocol) and -`introspect` (this library, which imports `cloud9` itself). A package that -depends on cloud9 takes both from the one dependency: - -```zig -const cloud9_dep = b.dependency("cloud9", .{ .target = target, .optimize = optimize }); -exe.root_module.addImport("cloud9", cloud9_dep.module("cloud9")); -exe.root_module.addImport("introspect", cloud9_dep.module("introspect")); -``` - -Embedding the core is three static objects and a push/step/output loop, the -same shape as cloud9's `Server`: - -```zig -const introspect = @import("introspect"); - -const State = struct { ticks: u32, phase: enum { idle, busy } }; -const cfg: introspect.Config = .{ .name = "fw", .types = &.{State}, .msize = 2048, .max_fids = 16 }; -const S = introspect.Server(cfg); - -var state: State = .{ .ticks = 0, .phase = .idle }; -var storage: S.Storage = undefined; // per connection: in/out frames + snapshot slots -var shared: S.Shared = undefined; // once: providers and exposed variables - -pub fn main() void { - shared = .init(&state); - shared.expose("state", &state) catch unreachable; // /vars/state/{value,type,size,addr,raw,f/...} - var conn: S.Conn = .init(&shared, &storage, cfg.msize); - // transport loop: conn.push(bytes) ... while (try conn.step()) {} ... send conn.output(), conn.wrote(n) -} -``` - -On Linux, `introspect.linux.Probe` runs that loop for you on one background -thread over a Unix, TCP or inherited listener, and adds the debug provider; -`pub const panic = std.debug.FullPanic(introspect.linux.debug.panicHook);` -in the root module publishes panics under `/panic`. `demo/main.zig` shows -every piece together. - -## The demo (`zig build introspect`, binary `introspect`) - -A single-binary 9P2000 server whose file tree is the binary itself: build-time -facts, `comptime` reflection, live runtime state, a worker thread whose state -is exposed under `/vars`, and the Linux debug layer. - -``` -/README -/build/{zig_version,target,optimize,time,change} captured by introspect/build.zig (jj change id, UTC time) -/comptime/types//{name,size,align,fields} @sizeOf/@alignOf/@typeInfo, generated at comptime -/comptime/decls pub declarations of the server module -/runtime/{pid,ppid,uptime,argv,cwd,env,clients} -/runtime/fn/ reading calls a Zig function (hostname, now, random, uname, fib30) -/runtime/ctl write "fib N" | "add A B" | "echo TEXT" | "sleep-ms N" | "trap" | "panic" -/scratch/ in-memory read/write tree -/vars/state/... the worker's State (readable and writable leaves) -/threads//{name,stat,stack,regs} /addr/ /mem/{maps,} /hex/ -/breakpoints//{stack,regs,ctl} /panic/{message,stack,ctl} -``` - -`/runtime/env` exposes the server's whole environment, so serve it on a Unix -socket or loopback only. - -```sh -zig-out/bin/introspect --unix /tmp/intro.sock & # or --tcp IP:PORT, --stdio, --no-hold -zig-out/bin/9player --unix /tmp/intro.sock -- sh -c ' - cat $NINEPLAYER_MOUNT/build/zig_version; echo - cat $NINEPLAYER_MOUNT/comptime/types/Qid/fields - echo "fib 20" > $NINEPLAYER_MOUNT/runtime/ctl; cat $NINEPLAYER_MOUNT/runtime/ctl - cat $NINEPLAYER_MOUNT/threads/*/stack' -``` - -## Building and testing - -introspect lives in the cloud9 repository as `cloud9/introspect/` and is -wired into cloud9's `build.zig` through the fragment `introspect/build.zig`. -Everything is run from the cloud9 root: - -```sh -zig build # installs zig-out/bin/introspect with the other binaries -zig build introspect # build and install only the demo -zig build introspect-test # library unit tests (core, vars, scratch, linux) and the demo's -zig build introspect-check-freestanding # compile the core for riscv32-freestanding-none -zig build introspect-debug-test # src/linux/debug.zig unit tests -zig build introspect-debug-itest # test/debug.sh: threads, stacks, breakpoints, panic through 9player -zig build introspect-adv # hostile raw-9P clients against the demo and the core, - # the Linux layer through a 9player mount (several minutes) -zig build -Dintrospect=false # leave introspect out -zig build introspect-check-freestanding -Dtarget=riscv32-freestanding-none -Dintrospect=true -``` - -`-Dintrospect` (default: on for Linux targets) enables the module and the -freestanding check on any target; the demo and the Linux suites are added -only when the target OS is Linux. The end-to-end suites also need 9player -(`-D9player=true`, the Linux default), unprivileged user namespaces, -`/dev/fuse` and Python 3, and skip themselves otherwise. diff --git a/introspect/build.zig b/introspect/build.zig deleted file mode 100644 index 6662151..0000000 --- a/introspect/build.zig +++ /dev/null @@ -1,169 +0,0 @@ -//! Build fragment for introspect: the 9P debug/introspection library (module -//! `introspect`), its freestanding check, the `introspect` demo server and the -//! test suites under introspect/test. It is `@import`ed by the root build.zig -//! and called with the root builder, so every `b.path(...)` here is relative -//! to the cloud9 root (hence the `introspect/` prefix), every option is -//! defined by the root (no `standardTargetOptions` here) and every step it -//! registers lands in the root's step list under the `introspect` prefix. -//! -//! Steps: introspect, introspect-test, introspect-check-freestanding, -//! introspect-debug-test, introspect-debug-itest, introspect-adv. -const std = @import("std"); - -/// What the root passes in. The root owns target/optimize resolution and the -/// cloud9 module; this fragment derives everything else from them. -pub const Context = struct { - target: std.Build.ResolvedTarget, - optimize: std.builtin.OptimizeMode, - /// The cloud9 library module for `target`. Its `root_source_file` is also - /// used to instantiate cloud9 for the freestanding check target. - cloud9: *std.Build.Module, -}; - -pub const Artifacts = struct { - /// The `introspect` module, exported with `b.addModule` so dependents of - /// cloud9 can `.module("introspect")`. Built for any target; the Linux - /// layer is compiled in only when the target OS is Linux. - module: *std.Build.Module, - /// The demo 9P2000 server (binary `introspect`); null when the target is - /// not Linux. 9player's integration tests use it as their server. - demo: ?*std.Build.Step.Compile, - /// `introspect-test`: library (and demo) unit tests. - test_step: *std.Build.Step, - /// `introspect-check-freestanding`: the core compiled for riscv32-freestanding-none. - check_step: *std.Build.Step, - /// `introspect-debug-test`: linux/debug.zig unit tests; null when not Linux. - debug_test_step: ?*std.Build.Step, - /// `introspect-adv`: the hostile-client suites are attached by `add`, the - /// Linux-layer suite (which needs a 9player mount) by `addPlayerTests`. - adv_step: *std.Build.Step, -}; - -pub fn add(b: *std.Build, ctx: Context) Artifacts { - const target = ctx.target; - const optimize = ctx.optimize; - const is_linux = target.result.os.tag == .linux; - - // Build-time facts embedded into the demo (/build/*): jj change id, UTC - // time, optimize mode, target triple. `jj` is pointed at the build root - // (the cloud9 checkout) explicitly, so the result does not depend on the - // directory `zig build` was invoked from. - const build_options = b.addOptions(); - var code: u8 = 0; - const repo = b.build_root.path orelse "."; - const change_id = b.runAllowFail(&.{ "jj", "-R", repo, "log", "--no-graph", "-r", "@", "-T", "change_id.short()", "--ignore-working-copy" }, &code, .ignore) catch "unknown"; - build_options.addOption([]const u8, "change_id", std.mem.trim(u8, change_id, " \n\r\t")); - const stamp = b.runAllowFail(&.{ "date", "-u", "+%Y-%m-%dT%H:%M:%SZ" }, &code, .ignore) catch "unknown"; - build_options.addOption([]const u8, "build_time", std.mem.trim(u8, stamp, " \n\r\t")); - build_options.addOption([]const u8, "optimize", @tagName(optimize)); - build_options.addOption([]const u8, "target", target.result.zigTriple(b.allocator) catch "unknown"); - - // The library: freestanding core + vars, scratch (allocator), Linux layer. - // Exported under the name `introspect` for packages that depend on cloud9. - const lib_mod = b.addModule("introspect", .{ - .root_source_file = b.path("introspect/src/root.zig"), - .target = target, - .optimize = optimize, - .imports = &.{.{ .name = "cloud9", .module = ctx.cloud9 }}, - }); - - const test_step = b.step("introspect-test", "Run the introspect library's unit tests (and the demo's)"); - test_step.dependOn(&b.addRunArtifact(b.addTest(.{ .root_module = lib_mod })).step); - - // The core must compile without an OS (rule 1 of introspect/docs/LIBRARY.md). - // cloud9 is re-instantiated for that target from the same root source. - const fs_target = b.resolveTargetQuery(.{ .cpu_arch = .riscv32, .os_tag = .freestanding, .abi = .none }); - const fs_cloud9 = b.createModule(.{ - .root_source_file = ctx.cloud9.root_source_file.?, - .target = fs_target, - .optimize = optimize, - }); - const fs_check = b.addObject(.{ - .name = "introspect-freestanding", - .root_module = b.createModule(.{ - .root_source_file = b.path("introspect/src/freestanding_check.zig"), - .target = fs_target, - .optimize = optimize, - .imports = &.{.{ .name = "cloud9", .module = fs_cloud9 }}, - }), - }); - const check_step = b.step("introspect-check-freestanding", "Compile the introspect core for riscv32-freestanding-none"); - check_step.dependOn(&fs_check.step); - - // `introspect-adv` exists on every target so the step list is stable; its - // suites are attached below (Linux only) and by addPlayerTests. - const adv_step = b.step("introspect-adv", "Run introspect's adversarial suites (hostile clients, Linux layer; several minutes)"); - - if (!is_linux) { - adv_step.dependOn(&b.addFail("introspect-adv needs a Linux target (the demo server is Linux-only)").step); - return .{ .module = lib_mod, .demo = null, .test_step = test_step, .check_step = check_step, .debug_test_step = null, .adv_step = adv_step }; - } - - // The demo: a 9P2000 server whose tree is the binary itself (build-time, - // comptime and runtime facts, a worker thread, the Linux debug layer). - const demo_mod = b.createModule(.{ - .root_source_file = b.path("introspect/demo/main.zig"), - .target = target, - .optimize = optimize, - .imports = &.{ - .{ .name = "cloud9", .module = ctx.cloud9 }, - .{ .name = "build_options", .module = build_options.createModule() }, - .{ .name = "introspect", .module = lib_mod }, - }, - }); - const demo = b.addExecutable(.{ .name = "introspect", .root_module = demo_mod }); - const install_demo = b.addInstallArtifact(demo, .{}); - b.getInstallStep().dependOn(&install_demo.step); - b.step("introspect", "Build and install only the introspect demo server").dependOn(&install_demo.step); - test_step.dependOn(&b.addRunArtifact(b.addTest(.{ .root_module = demo_mod })).step); - - // Linux debug facilities (threads, stacks, breakpoints, panic): self-contained unit tests. - const debug_mod = b.createModule(.{ - .root_source_file = b.path("introspect/src/linux/debug.zig"), - .target = target, - .optimize = optimize, - }); - const debug_test_step = b.step("introspect-debug-test", "Run the introspect/src/linux/debug.zig unit tests"); - debug_test_step.dependOn(&b.addRunArtifact(b.addTest(.{ .root_module = debug_mod })).step); - - // Hostile raw-9P clients against the demo (framing, tags, floods; the core's - // /vars tree, snapshots, fid table). `--fast` as in the umbrella script. - inline for (.{ "adv_introspect_hostile", "adv_core_hostile" }) |suite| { - const run = b.addSystemCommand(&.{"bash"}); - run.addFileArg(b.path("introspect/test/" ++ suite ++ ".sh")); - run.addArtifactArg(demo); - run.addArg("--fast"); - adv_step.dependOn(&run.step); - } - - return .{ .module = lib_mod, .demo = demo, .test_step = test_step, .check_step = check_step, .debug_test_step = debug_test_step, .adv_step = adv_step }; -} - -/// The suites that drive the demo through a 9player mount: test/debug.sh -/// (threads, stacks, breakpoints, panic end to end) and -/// test/adv_linux_probe.sh (memory endpoints, signal machinery, poll loop). -/// Called by the root after the 9player fragment; `player` is null when -/// 9player is disabled, in which case the steps exist but fail with a notice. -/// Returns the `introspect-debug-itest` step (null when the target is not Linux). -pub fn addPlayerTests(b: *std.Build, arts: Artifacts, player: ?*std.Build.Step.Compile) ?*std.Build.Step { - const demo = arts.demo orelse return null; // not Linux: nothing to drive - const debug_itest = b.step("introspect-debug-itest", "Run introspect/test/debug.sh (threads, stacks, breakpoints, panic through 9player)"); - const exe = player orelse { - const fail = b.addFail("introspect-debug-itest and the Linux-layer adversarial suite need 9player (build with -D9player=true)"); - debug_itest.dependOn(&fail.step); - arts.adv_step.dependOn(&fail.step); - return debug_itest; - }; - const dbg = b.addSystemCommand(&.{"bash"}); - dbg.addFileArg(b.path("introspect/test/debug.sh")); - dbg.addArtifactArg(exe); - dbg.addArtifactArg(demo); - debug_itest.dependOn(&dbg.step); - - const adv_linux = b.addSystemCommand(&.{"bash"}); - adv_linux.addFileArg(b.path("introspect/test/adv_linux_probe.sh")); - adv_linux.addArtifactArg(exe); - adv_linux.addArtifactArg(demo); - arts.adv_step.dependOn(&adv_linux.step); - return debug_itest; -} diff --git a/introspect/demo/main.zig b/introspect/demo/main.zig deleted file mode 100644 index b1c2c57..0000000 --- a/introspect/demo/main.zig +++ /dev/null @@ -1,423 +0,0 @@ -//! introspect: the demo 9P2000 server, built on the introspect library. -//! -//! /README, /build/*, /comptime/{types,decls}, /runtime/{pid,ppid,uptime,argv,cwd,env,clients}, -//! /runtime/fn/{fib30,hostname,now,random,uname}, /runtime/ctl (echo|fib|sleep-ms|add|trap|panic), -//! /scratch (in-memory tree), /vars/state (the worker's exposed state), -//! /threads, /addr, /mem, /hex, /breakpoints, /panic (the Linux debug layer). -//! -//! A worker thread ("worker") runs `workerLoop`, incrementing `state.ticks` -//! every ~10 ms; `trap` makes it execute `@breakpoint()` on its next tick and -//! `panic` makes it panic from inside `workerLoop`. Panics go through the -//! library's hook, so the message and stack are published under /panic and -//! the process is held there until /panic/ctl says "continue" (`--no-hold` -//! disables the hold). -//! -//! Static memory: every buffer is a global; the only heap user is /scratch -//! (init.gpa, 512 MiB budget). With `max_clients` = 16 and msize = 1 MiB the -//! per-client core Storage is 3 MiB + 8 x 64 KiB snapshots and the Conn's fid -//! table ~1.88 MiB (32768 fids plus their hash index, needed for the 20000-fid -//! adversarial test), so `probe_storage` is ~86 MiB of BSS; untouched pages -//! cost nothing (an idle server has an RSS of ~7 MiB). -const std = @import("std"); -const builtin = @import("builtin"); -const cloud9 = @import("cloud9"); -const build_options = @import("build_options"); -const introspect = @import("introspect"); -const linux = std.os.linux; -const Writer = std.Io.Writer; -const plinux = introspect.linux; -const runtime = plinux.runtime; - -pub const panic = std.debug.FullPanic(plinux.debug.panicHook); - -/// Simultaneous 9P clients; further connections are closed (see probe.zig for -/// the idle-eviction rule). -pub const max_clients = 16; -pub const max_msize: u32 = 1 << 20; - -pub const State = struct { - ticks: u64, - phase: enum { idle, working, trapped }, - last_job: struct { id: u32, cost: f32 }, -}; - -const Build = struct { - pub const zig_version: []const u8 = builtin.zig_version_string; - pub const target: []const u8 = build_options.target; - pub const optimize: []const u8 = build_options.optimize; - pub const time: []const u8 = build_options.build_time; - pub const change: []const u8 = build_options.change_id; -}; - -/// What every generator and the ctl handler see (`Shared.ctx`). -const App = struct { - info: runtime.Info, - probe: *ProbeT, - shared: *S.Shared, -}; - -/// /runtime/fn/: reading the file calls the function. -pub const Fns = struct { - pub fn hostname(_: *anyopaque, w: *Writer) anyerror!void { - var u: linux.utsname = undefined; - if (linux.errno(linux.uname(&u)) != .SUCCESS) return error.Uname; - try w.writeAll(std.mem.sliceTo(&u.nodename, 0)); - } - - pub fn now(_: *anyopaque, w: *Writer) anyerror!void { - try w.print("{d}", .{runtime.realtimeSecs()}); - } - - pub fn random(_: *anyopaque, w: *Writer) anyerror!void { - var b: [8]u8 = undefined; - var got: usize = 0; - while (got < b.len) { - const rc = linux.getrandom(b[got..].ptr, b.len - got, 0); - switch (linux.errno(rc)) { - .SUCCESS => got += rc, - .INTR => continue, - else => return error.Random, - } - } - try w.print("{x:0>16}", .{std.mem.readInt(u64, &b, .little)}); - } - - pub fn uname(_: *anyopaque, w: *Writer) anyerror!void { - var u: linux.utsname = undefined; - if (linux.errno(linux.uname(&u)) != .SUCCESS) return error.Uname; - try w.writeAll(std.mem.sliceTo(&u.release, 0)); - } - - pub fn fib30(_: *anyopaque, w: *Writer) anyerror!void { - try w.print("{d}", .{fib(30)}); - } -}; - -const cfg: introspect.Config = .{ - .name = "introspect", - .build = Build, - .types = &.{ cloud9.Qid, cloud9.Stat, cloud9.Msg, introspect.core.Node, linux.Statx }, - .decls_of = @This(), - .fns = Fns, - .runtime = runtime.Fns(App, "info"), - .ctl = &ctl, - .ctl_dir = .runtime, - .ctl_bytes = 64 * 1024, - .msize = max_msize, - .max_fids = 32768, - .max_providers = 8, - .snapshot_slots = 8, - .snapshot_bytes = 64 * 1024, -}; -const S = introspect.Server(cfg); -const ProbeT = plinux.Probe(S); -const ProbeStorage = ProbeT.Storage(max_clients); - -// -- static state ------------------------------------------------------------- - -pub var state: State = .{ .ticks = 0, .phase = .idle, .last_job = .{ .id = 0, .cost = 0 } }; -var trap_requested: std.atomic.Value(bool) = .init(false); -var panic_requested: std.atomic.Value(bool) = .init(false); - -/// Zero-filled static memory for `T`. (An `= undefined` global is emitted as -/// 0xAA-filled .data in Debug builds, which would make the binary 120 MiB; -/// zeros go to .bss and cost nothing until touched.) -fn Bss(comptime T: type) type { - return struct { - bytes: [@sizeOf(T)]u8 align(@alignOf(T)) = @splat(0), - fn get(b: *@This()) *T { - return @ptrCast(&b.bytes); - } - }; -} -var app_mem: Bss(App) = .{}; -var shared_mem: Bss(S.Shared) = .{}; -var probe_storage_mem: Bss(ProbeStorage) = .{}; -var probe_mem: Bss(ProbeT) = .{}; -var scratch_mem: Bss(introspect.Scratch) = .{}; - -/// Sizes of the static pieces, for the report and `--help`. -pub const static_bytes = @sizeOf(ProbeStorage) + @sizeOf(S.Shared) + @sizeOf(ProbeT); - -// -- the worker --------------------------------------------------------------- - -/// Ticks every ~10 ms; honours `trap` and `panic` requests from /runtime/ctl. -pub noinline fn workerLoop() void { - plinux.setThreadName("worker"); - var job: u32 = 0; - while (true) { - napMs(10); - state.ticks +%= 1; - if (panic_requested.swap(false, .acq_rel)) { - state.phase = .working; - @panic("demo panic requested over 9p"); - } - if (trap_requested.swap(false, .acq_rel)) { - state.phase = .trapped; - @breakpoint(); - state.phase = .idle; - } - if (state.ticks % 100 == 0) { - job +%= 1; - state.phase = .working; - state.last_job = .{ .id = job, .cost = @as(f32, @floatFromInt(job % 7)) * 0.5 }; - state.phase = .idle; - } - } -} - -/// The worker's sleep, issued as a raw syscall from this file so that the -/// thread's innermost frame (the first line of /threads//stack, which -/// introspect/test/debug.sh resolves through /addr) is in demo/main.zig rather than in std. -/// EINTR (a capture signal) just ends the nap early. -inline fn napMs(ms: u64) void { - var req: linux.timespec = .{ .sec = @intCast(ms / 1000), .nsec = @intCast((ms % 1000) * 1_000_000) }; - switch (builtin.cpu.arch) { - .x86_64 => _ = asm volatile ("syscall" - : [ret] "={rax}" (-> usize), - : [number] "{rax}" (@intFromEnum(linux.SYS.nanosleep)), - [arg1] "{rdi}" (@intFromPtr(&req)), - [arg2] "{rsi}" (@as(usize, 0)), - : .{ .rcx = true, .r11 = true, .memory = true }), - .aarch64 => _ = asm volatile ("svc #0" - : [ret] "={x0}" (-> usize), - : [number] "{x8}" (@intFromEnum(linux.SYS.nanosleep)), - [arg1] "{x0}" (@intFromPtr(&req)), - [arg2] "{x1}" (@as(usize, 0)), - : .{ .memory = true }), - else => plinux.sleepMs(ms), - } -} - -// -- /runtime/ctl ------------------------------------------------------------- - -/// The /runtime/ctl handler. The core stages the output and commits it only -/// on success, so a failed command leaves the previous result in place -/// (test/adv_introspect_hostile.py checks that). -fn ctl(ctx: *anyopaque, cmd: []const u8, out: *Writer) anyerror!void { - const a: *App = @ptrCast(@alignCast(ctx)); - const line = std.mem.trim(u8, cmd, " \t\r\n\x00"); - var it = std.mem.tokenizeScalar(u8, line, ' '); - const verb = it.next() orelse return error.BadCommand; - if (std.mem.eql(u8, verb, "echo")) { - try out.writeAll(std.mem.trimStart(u8, line[verb.len..], " \t")); - } else if (std.mem.eql(u8, verb, "fib")) { - const n = std.fmt.parseInt(u32, it.next() orelse return error.BadCommand, 10) catch return error.BadCommand; - if (n > 93) return error.BadCommand; // fib(94) overflows u64 - try out.print("{d}", .{fib(n)}); - } else if (std.mem.eql(u8, verb, "sleep-ms")) { - const n = std.fmt.parseInt(u64, it.next() orelse return error.BadCommand, 10) catch return error.BadCommand; - const ms = @min(n, 10_000); - a.probe.sleepServing(ms); - try out.print("slept {d} ms", .{ms}); - } else if (std.mem.eql(u8, verb, "add")) { - const x = std.fmt.parseInt(i64, it.next() orelse return error.BadCommand, 10) catch return error.BadCommand; - const y = std.fmt.parseInt(i64, it.next() orelse return error.BadCommand, 10) catch return error.BadCommand; - try out.print("{d}", .{x +% y}); - } else if (std.mem.eql(u8, verb, "trap")) { - trap_requested.store(true, .release); - try out.writeAll("trap armed: the worker stops in @breakpoint() on its next tick"); - } else if (std.mem.eql(u8, verb, "panic")) { - panic_requested.store(true, .release); - try out.writeAll("panic armed: the worker panics on its next tick"); - } else return error.BadCommand; -} - -/// fib(n) for n <= 93 (fib(93) is the largest that fits u64). -fn fib(n: u32) u64 { - std.debug.assert(n <= 93); - if (n == 0) return 0; - var a: u64 = 0; - var b: u64 = 1; - for (1..n) |_| { - const c = a + b; - a = b; - b = c; - } - return b; -} - -// -- main --------------------------------------------------------------------- - -const usage_text = - \\usage: introspect [--unix PATH | --tcp IP:PORT | --stdio] [--no-hold] - \\ - \\A demo 9P2000 file server exposing this binary's build-time, comptime and - \\runtime facts, plus a debugger-shaped view of the process (threads, stacks, - \\memory, breakpoints, panics). Default is --stdio (9P on fd 0/1). - \\--no-hold lets a panic abort at once instead of waiting for /panic/ctl. - \\ -; - -pub fn main(init: std.process.Init) !void { - run(init) catch |e| switch (e) { - // Already reported on stderr; no stack trace wanted. - error.Usage, error.Syscall => std.process.exit(1), - else => return e, - }; -} - -fn run(init: std.process.Init) !void { - // Transparent huge pages would back the first touched page of every - // client buffer with 2 MiB. Best effort: ignore failure. - _ = linux.prctl(@intFromEnum(linux.PR.SET_THP_DISABLE), 1, 0, 0, 0); - const arena = init.arena.allocator(); - const args = try init.minimal.args.toSlice(arena); - - const Mode = enum { stdio, unix, tcp }; - var mode: Mode = .stdio; - var address: []const u8 = ""; - var hold = true; - var i: usize = 1; - while (i < args.len) : (i += 1) { - const a = args[i]; - if (std.mem.eql(u8, a, "--stdio")) { - mode = .stdio; - } else if (std.mem.eql(u8, a, "--unix") or std.mem.eql(u8, a, "--tcp")) { - i += 1; - if (i >= args.len) { - std.debug.print("introspect: {s} needs an argument\n{s}", .{ a, usage_text }); - return error.Usage; - } - mode = if (a[2] == 'u') .unix else .tcp; - address = args[i]; - } else if (std.mem.eql(u8, a, "--no-hold")) { - hold = false; - } else if (std.mem.eql(u8, a, "--help") or std.mem.eql(u8, a, "-h")) { - std.debug.print("{s}\nstatic memory: {d} bytes ({d} clients, msize {d})\n", .{ usage_text, static_bytes, max_clients, max_msize }); - return; - } else { - std.debug.print("introspect: unknown argument {s}\n{s}", .{ a, usage_text }); - return error.Usage; - } - } - - // argv, env and cwd are gathered once, into the arena. - var argv_text: std.ArrayList(u8) = .empty; - for (args) |a| { - try argv_text.appendSlice(arena, a); - try argv_text.append(arena, '\n'); - } - var env_text: std.ArrayList(u8) = .empty; - for (init.minimal.environ.block.view().slice) |entry| { - try env_text.appendSlice(arena, std.mem.span(entry)); - try env_text.append(arena, '\n'); - } - var cwd_buf: [4096]u8 = undefined; - const cwd_rc = linux.getcwd(&cwd_buf, cwd_buf.len); - const cwd_text: []const u8 = if (linux.errno(cwd_rc) == .SUCCESS) - try arena.dupe(u8, std.mem.sliceTo(cwd_buf[0..cwd_rc], 0)) - else - ""; - - const app = app_mem.get(); - const shared = shared_mem.get(); - const probe = probe_mem.get(); - const scratch = scratch_mem.get(); - app.* = .{ - .info = .{ .argv = argv_text.items, .env = env_text.items, .cwd = cwd_text, .start_mono = runtime.monotonicSecs() }, - .probe = probe, - .shared = shared, - }; - shared.* = .init(app); - try shared.expose("state", &state); - scratch.* = try introspect.Scratch.init(init.gpa, 512 << 20); - // Freed on the way out so a Debug build's allocator does not report the - // tree as leaked (with a stack trace on stderr) after a clean --stdio EOF. - defer scratch.deinit(); - scratch.max_file = 64 << 20; - scratch.now = &runtime.realtimeSecs; - try shared.addProvider(scratch.provider("scratch")); - - const listen: plinux.Listen = switch (mode) { - .stdio => .{ .client = .{ .in = 0, .out = 1 } }, - .unix => .{ .unix = address }, - .tcp => .{ .tcp = address }, - }; - probe.init(shared, probe_storage_mem.get(), .{ .io = init.io, .listen = listen, .msize = max_msize, .hold_on_panic = hold }) catch |e| { - switch (e) { - error.PathTooLong => std.debug.print("introspect: unix socket path too long\n", .{}), - error.BadAddress => std.debug.print("introspect: --tcp wants an IPv4 literal a.b.c.d:port\n", .{}), - error.Syscall => std.debug.print("introspect: {t} ({t})\n", .{ e, probe.last_errno }), - else => std.debug.print("introspect: {t}\n", .{e}), - } - return if (e == error.PathTooLong or e == error.BadAddress) error.Usage else error.Syscall; - }; - // Failure past this point (thread spawn) still closes the listener and - // restores the signal dispositions. - errdefer probe.stop(); - app.info.clients = probe.clientCounter(); - - const worker = std.Thread.spawn(.{}, workerLoop, .{}) catch |e| { - std.debug.print("introspect: worker thread: {t}\n", .{e}); - return error.Syscall; - }; - worker.detach(); - - switch (mode) { - .unix => std.debug.print("introspect: listening on unix!{s}\n", .{address}), - .tcp => std.debug.print("introspect: listening on tcp!{s}\n", .{address}), - .stdio => {}, - } - probe.start() catch |e| { - std.debug.print("introspect: thread spawn failed: {t}\n", .{e}); - return error.Syscall; - }; - // SIGTERM/SIGINT end the loop cleanly: the socket file is unlinked, the - // signal dispositions restored and the scratch tree freed. - // A disposition of SIG_IGN inherited from the parent is left alone (Unix - // convention): 9player runs a --spawn server with SIGINT ignored so that - // Ctrl-C on the terminal reaches only the program, not its file server. - const term: linux.Sigaction = .{ .handler = .{ .handler = onTerm }, .mask = linux.sigemptyset(), .flags = 0 }; - for ([_]linux.SIG{ .TERM, .INT }) |sig| { - var old: linux.Sigaction = undefined; - _ = linux.sigaction(sig, null, &old); - if (old.handler.handler != linux.SIG.IGN) _ = linux.sigaction(sig, &term, null); - } - probe.wait(); - probe.stop(); -} - -/// Async-signal-safe: an atomic store and one eventfd write. -fn onTerm(_: linux.SIG) callconv(.c) void { - probe_mem.get().requestStop(); -} - -// -- tests -------------------------------------------------------------------- - -test "ctl commands" { - const probe = probe_mem.get(); - const shared = shared_mem.get(); - var a: App = .{ .info = .{}, .probe = probe, .shared = shared }; - probe.thread_tid = .init(0); - probe.serving = .initEmpty(); - probe.nested = 0; - var buf: [128]u8 = undefined; - var w: Writer = .fixed(&buf); - try ctl(&a, "add 2 3\n", &w); - try std.testing.expectEqualStrings("5", w.buffered()); - try std.testing.expectError(error.BadCommand, ctl(&a, "nope", &w)); - w = .fixed(&buf); - try ctl(&a, "fib 93", &w); - try std.testing.expectEqualStrings("12200160415121876738", w.buffered()); - w = .fixed(&buf); - try std.testing.expectError(error.BadCommand, ctl(&a, "fib 94", &w)); - try std.testing.expectError(error.BadCommand, ctl(&a, "frobnicate", &w)); - w = .fixed(&buf); - try ctl(&a, "echo hi there ", &w); - try std.testing.expectEqualStrings("hi there", w.buffered()); - w = .fixed(&buf); - try ctl(&a, "sleep-ms 1", &w); - try std.testing.expectEqualStrings("slept 1 ms", w.buffered()); - w = .fixed(&buf); - try ctl(&a, "trap", &w); - try std.testing.expect(trap_requested.swap(false, .acq_rel)); - w = .fixed(&buf); - try ctl(&a, "panic", &w); - try std.testing.expect(panic_requested.swap(false, .acq_rel)); -} - -test "static footprint is what the file comment says" { - try std.testing.expect(@sizeOf(ProbeStorage) > 16 * (3 << 20)); - try std.testing.expect(@sizeOf(ProbeStorage) < 100 << 20); -} diff --git a/introspect/docs/LIBRARY.md b/introspect/docs/LIBRARY.md deleted file mode 100644 index 22840dc..0000000 --- a/introspect/docs/LIBRARY.md +++ /dev/null @@ -1,287 +0,0 @@ -# introspect: a 9P debug/introspection server as a library - -The demo server that 9player's tests use grows into a library any Zig program -can embed: a debugger-shaped interface where the protocol is just files. -Anything that can read a filesystem (a shell, an agent, an editor, `9p`, -9player) can inspect a running process: build facts, comptime type layouts, -live values, threads and their stacks, memory, breakpoints, panics. - -Design rules (non-negotiable, they mirror cloud9): - -1. **The core is freestanding.** No allocator, no OS, no threads, no `std.Io`. - Caller-owned buffers, fixed-capacity tables sized at comptime. It must - compile for `riscv32-freestanding-none` (the ESP32-P4 firmware target, - `../05-zig-p4`), and `zig build introspect-check-freestanding` proves it. -2. **Every dependency on a runtime is an explicit argument.** Features that - truly need an `Allocator` or an `std.Io` take them in their `init`; nothing - reaches for `std.heap.page_allocator` or a global `Io`. Where memory is - needed it is preferably a caller-provided `[]u8` or a comptime-sized - `Storage` struct the caller places in static memory. -3. **All allocation happens up front**, at init, from what the caller passed. - Steady-state operation does not allocate. -4. **Platform layers are separate modules** (`introspect.linux`) and are the - only places that touch sockets, threads, signals, `/proc` or `std.debug`. - -``` - introspect/src/root.zig pub const core, vars, scratch, linux (linux only), Server(cfg) - introspect/src/core.zig Tree/Server engine on cloud9.Server: fids, walks, dir reads, providers - introspect/src/vars.zig comptime value renderers (@typeInfo) for /vars - introspect/src/scratch.zig in-memory read/write tree provider (takes an Allocator) - introspect/src/linux/probe.zig background thread + poll loop + unix/tcp/fd listeners - introspect/src/linux/debug.zig threads, stacks, registers, addr→source, memory, breakpoints, panic - introspect/demo/main.zig the `introspect` binary: embeds everything, worker thread, exposed vars - -(paths from the cloud9 root; the library is the module `introspect` that -cloud9's `build.zig` exports next to `cloud9`, wired by `introspect/build.zig`) -``` - -## Core (`core.zig`) - -```zig -pub const Config = struct { - name: []const u8 = "introspect", // /README and Stat uid/gid - types: []const type = &.{}, // /comptime/types//... - decls_of: ?type = null, // /comptime/decls lists this type's pub decls - fns: type = struct {}, // /runtime/fn/: pub fn (ctx: *anyopaque, w: *std.Io.Writer) anyerror!void - ctl: ?*const fn (ctx: *anyopaque, cmd: []const u8, out: *std.Io.Writer) anyerror!void = null, // /ctl - max_fids: u16 = 64, - max_providers: u8 = 8, - max_vars: u8 = 32, - /// Dynamic file contents are generated at open time into per-fid snapshot - /// slots so that reads at arbitrary offsets are consistent. - snapshot_slots: u8 = 8, - snapshot_bytes: u32 = 16 * 1024, -}; - -pub fn Server(comptime cfg: Config) type { - return struct { - pub const Storage = struct { // caller places this in static memory - in: [msize]u8, out: [msize]u8, snapshots: [cfg.snapshot_slots][cfg.snapshot_bytes]u8, - }; - pub const Shared = struct { // state common to all connections (providers, vars) - pub fn init(name_ctx: *anyopaque) Shared; - pub fn addProvider(s: *Shared, p: Provider) error{Full}!void; - pub fn expose(s: *Shared, name: []const u8, ptr: anytype) error{Full}!void; // typed value → /vars/ - }; - pub const Conn = struct { // one 9P connection, push/step/output like cloud9 - pub fn init(shared: *Shared, storage: *Storage, msize: u32) Conn; - pub fn push(c: *Conn, bytes: []const u8) usize; // feed transport bytes - pub fn step(c: *Conn) error{Protocol}!bool; // handle ≤ 1 request; false = nothing to do - pub fn output(c: *const Conn) []const u8; // bytes to send - pub fn wrote(c: *Conn, n: usize) void; - pub fn hangup(c: *Conn) void; // drop fids, tell providers - }; - }; -} -``` - -`step` drives `cloud9.Server.receive/reply/negotiate` and the backend: the -static tree (comptime-generated from `cfg`: `/README`, `/build/*` via a -`build_options`-like struct passed in `cfg.build`, `/comptime/types/*`, -`/comptime/decls`, `/runtime/fn/*`, `/ctl`, `/vars/*`) plus **providers**. - -A provider is a runtime vtable mounted at a top-level name. It owns a subtree -with its own naming (dynamic directories such as `/threads/` or -`/addr/` cannot be enumerated at comptime): - -```zig -pub const Provider = struct { - name: []const u8, - ctx: *anyopaque, - vtable: *const VTable, - pub const Handle = u64; // provider-defined node id; 0 = provider root - pub const VTable = struct { - walk: *const fn (ctx, parent: Handle, name: []const u8) Error!Handle, - stat: *const fn (ctx, h: Handle, out: *NodeStat) Error!void, // kind (dir/file), mode, length, mtime - list: *const fn (ctx, dir: Handle, index: usize, out: *NodeStat) Error!bool, // nth entry; false when done - open: *const fn (ctx, h: Handle, mode: u8) Error!void, - read: *const fn (ctx, h: Handle, offset: u64, buf: []u8) Error!usize, - write: *const fn (ctx, h: Handle, offset: u64, data: []const u8) Error!usize, - create: ?*const fn (ctx, dir: Handle, name: []const u8, perm: u32, mode: u8) Error!Handle, - remove: ?*const fn (ctx, h: Handle) Error!void, - wstat: ?*const fn (ctx, h: Handle, st: *const cloud9.Stat) Error!void, - clunk: *const fn (ctx, h: Handle) void, // fid released (also on hangup) - }; - pub const Error = error{ NotFound, Exists, Perm, NotDir, IsDir, NotEmpty, BadOffset, NoSpace, Io, Unsupported }; -}; -``` - -Error → Rerror text mapping lives in one place in the core, using the Plan 9 -strings 9player's bridge already understands (`file does not exist`, -`permission denied`, `file already exists`, `directory not empty`, -`not a directory`, `is a directory`, `bad offset`, `no space`, `i/o error`, -`not supported`). - -Directory reads follow the 9P rule (offset 0 or previous offset+count, never -split a record). Dynamic file reads: on open the content is generated once -into a snapshot slot (`open` runs the generator; `read` serves the slot; a -read at offset 0 regenerates); no free slot → Rerror `too many open dynamic -files`. Stats of dynamic files report length 0. - -Qids: static nodes get comptime paths; provider nodes get -`(provider index << 56) | handle`. - -Static memory: `Server(cfg).Storage` per connection, `Shared` once. No heap. -The core has unit tests driven through `cloud9.Client` in memory (like today). - -## Value renderers (`vars.zig`) - -`expose(name, ptr: anytype)` builds at comptime a `VTable` for -`@TypeOf(ptr.*)`: - -``` -/vars//value rendered text (structs: "field: value" lines, nested indented; unions: tag + payload; - optionals: "null" or the value; enums: tag; ints/floats/bools; []const u8 and [*:0]const u8 - as quoted strings (≤ 256 bytes); other pointers as 0x… never followed; arrays/slices ≤ 64 elements) -/vars//type @typeName -/vars//size @sizeOf -/vars//addr 0x… -/vars//raw the bytes (length = @sizeOf) -/vars//f//... same layout recursively for struct fields (depth ≤ 4), leaves writable: - writing text to a scalar's `value` parses and stores it (ints: decimal/0x, bools, floats, enums by tag) -``` - -Rendering is by a comptime-generated function table; no allocation. -Writes to scalars are plain stores (not atomic; documented). - -## Scratch provider (`scratch.zig`) - -The in-memory read/write tree from the current server, as a provider, with -`init(allocator, budget_bytes)`; the only core-level component that takes an -allocator, and it is optional. - -## Linux layer (`linux/probe.zig`) - -```zig -pub const Probe = struct { - pub const Options = struct { - io: std.Io, // for std.debug symbolization - listen: union(enum) { unix: []const u8, tcp: []const u8, fd: i32 }, - max_clients: u8 = 8, - msize: u32 = 64 * 1024, - hold_on_panic: bool = true, - capture_signal: u8 = SIGRTMIN + 3, // used to snapshot other threads - breakpoints: bool = true, // install the SIGTRAP handler - }; - pub fn Storage(comptime max_clients: u8) type; // static: per-client Server.Storage + poll table - pub fn init(p: *Probe, shared: *Server.Shared, storage: *Storage, opts: Options) !void; // listens, registers the debug provider - pub fn start(p: *Probe) !void; // spawns ONE background thread running a poll loop over listener + clients - pub fn stop(p: *Probe) void; // closes, joins -}; -``` - -One thread, `poll()` over the listener and every connection; each connection -is a core `Conn` fed with `push`/`step`/`output`. No per-connection threads. -Symbolization uses `std.debug.getSelfDebugInfo()` with the `io` passed in and -a caller-provided fixed buffer as the text arena. - -## Debug provider (`linux/debug.zig`) - -Mounted as `/threads`, `/addr`, `/mem`, `/hex`, `/breakpoints`, `/panic`. - -``` -/threads/ one directory per tid, enumerated from /proc/self/task at list time -/threads//name comm -/threads//stat state letter + a few fields from /proc/self/task//stat -/threads//stack "#n 0x in (::)" per frame -/threads//regs " 0x" per general register, from the captured cpu context -/addr/ dynamic dir: walk of any hex address yields a file "fn\nfile:line:col\nmodule\n" -/mem/maps /proc/self/maps served by pread at the requested offset (any size) -/mem/ raw bytes at address+offset via process_vm_readv/writev (never faults); writable -/hex/ hexdump text of 256 bytes at address (+offset), like std.debug.dumpHex -/breakpoints/ directory of tids currently stopped in @breakpoint() -/breakpoints//stack, regs as above -/breakpoints//ctl write "continue" (or "step"? no: continue only) to resume -/panic/message the panic message, empty before any panic -/panic/stack frames of the panicking thread -/panic/ctl write "continue" to let the default panic handler run (abort) -``` - -**Capturing another thread** (`stack`, `regs`): the server thread `tgkill`s -the target with `capture_signal`. The handler (SA_SIGINFO, async-signal-safe: -no allocation, no locks) copies the `cpu_context.Native` obtained through -`std.debug.cpu_context.fromPosixSignalContext` into a slot and futex-waits. -The server thread unwinds with `std.debug.StackIterator.init(&ctx)` while the -target is parked, symbolizes, then releases the slot; the target resumes. The -server's own thread unwinds itself directly. Timeout 250 ms → Rerror -`thread did not respond`. Threads blocked in uninterruptible syscalls simply -time out. A target parked while holding std.debug's `SelfInfo` lock (it was -printing a stack trace itself) cannot be unwound without deadlocking; the -probe detects that with `tryLock`, releases the target and answers -`i/o error` (registers still work). - -**Breakpoints**: `@breakpoint()` raises SIGTRAP on the executing thread only. -The installed handler stores the context in a slot, marks the thread paused, -and futex-waits until `/breakpoints//ctl` receives `continue`. On x86_64 -the saved PC already points past `int3`; on aarch64 the handler advances PC by -4 (`brk`) before returning, but only for kernel-generated traps -(`si_code > 0`); a user-sent `SIGTRAP` (`kill -TRAP`, `tgkill`) parks the -thread exactly where it was, which makes it a usable "pause this thread" -request. Other threads keep running; a slot table (`max_paused`, default 16) -bounds simultaneous pauses and, when it is full, the trapping thread simply -steps over the breakpoint (`traps_skipped` counts these). The probe's own -serving thread is never parked or held: a trap or panic on it goes straight -to the default behaviour, since nobody could write its `ctl` files. - -**Panics**: `pub const panic = introspect.linux.panic;` in the root module -(built with `std.debug.FullPanic`). The first panic records message and a -stack capture (`captureCurrentStackTrace` with `first_address`), publishes -them, and, if `hold_on_panic` and the probe is running, futex-waits until -`/panic/ctl` says `continue`; then `std.debug.defaultPanic` runs (prints the -trace and aborts). A nested or second panic goes straight to the default. - -Signal handlers are installed by `Probe.init` (breakpoints optional) and -restored by `stop`. - -## Demo (`demo/main.zig`, binary `introspect`) - -Keeps every path the existing tests read (`/build/*`, `/comptime/types/Qid/*`, -`/comptime/decls`, `/runtime/fn/now|hostname|…`, `/runtime/ctl` with -`add|echo|fib|sleep-ms`, `/runtime/pid|ppid|uptime|argv|cwd|env|clients`, -`/scratch`), served by the library. Adds: - -* a worker thread running `workerLoop` that increments an exposed - `State { ticks: u64, phase: enum, last_job: Job }` (`/vars/state/...`); -* `/runtime/ctl` commands `trap` (the worker executes `@breakpoint()` on its - next tick) and `panic` (the worker panics with a message); -* `--stdio | --unix PATH | --tcp IP:PORT` as today, `--no-hold` to disable - panic holding. - -`main` passes `init.io` and an explicit allocator to the pieces that need one; -the demo's Storage is a global. - -## Verification - -* Unit tests: core (in-memory client drives every op incl. providers and - snapshots), vars (render/set for every category), scratch, debug (capture - own thread and a helper thread; breakpoint pause/continue on a helper - thread; panic record path without holding). -* `zig build introspect-check-freestanding`: compiles `core.zig` + `vars.zig` - for `riscv32-freestanding-none` with a tiny freestanding root that - instantiates `Server(cfg)` with static Storage. -* `zig build introspect-test` (library and demo unit tests) and - `zig build introspect-debug-test` (linux/debug.zig). -* 9player's `test/integration.sh` unchanged and passing; `test/debug.sh` - (`zig build introspect-debug-itest`) through 9player: read the worker's - stack (contains `workerLoop` and `demo/main.zig:`), - resolve a frame through `/addr`, dump `/hex` of the exposed state, read and - write `/vars/state/f/ticks/value`, trap → `/breakpoints` lists the worker, - its stack shows `workerLoop`, `continue` resumes (ticks keep increasing), - panic → `/panic/message`, `/panic/stack`, `continue` → server exits - non-zero. -* Adversarial pass afterwards (`zig build introspect-adv`: hostile client - against the server and the core, signal races, memory reads of unmapped - addresses, panic while a capture is in flight; `test/adversarial.sh` runs - the same suites by hand). - -## Known upstream issue (Zig 0.16 std.debug) - -`std.debug.SelfInfo` for ELF (`std/debug/SelfInfo/Elf.zig`, `findModule`) -rebuilds its module list whenever it is asked about an address outside every -known module. That frees each module's `Dwarf.Unwind` and CIE list but leaves -`unwind_cache` entries pointing into the freed memory, so later unwinds read -freed data: empty traces, "unwind info invalid", or segfaults once the arena -reuses the block. `linux/debug.zig` records the executable's `PT_LOAD` ranges -at init and refuses to hand std an address outside them (`knownCode`), which -is why `/addr/` of a bogus address renders `?` instead of poisoning the -process. Worth reporting upstream; the guard can go once std clears the cache. diff --git a/introspect/src/core.zig b/introspect/src/core.zig deleted file mode 100644 index 2707af2..0000000 --- a/introspect/src/core.zig +++ /dev/null @@ -1,2656 +0,0 @@ -//! The freestanding 9P2000 introspection engine: a static tree generated at -//! comptime from a `Config` (README, /build, /comptime, /runtime/fn, /ctl, -//! /vars) plus runtime `Provider`s mounted at the top level, served over a -//! `cloud9.Server` connection. No allocator, no OS, no threads: every buffer is -//! caller-owned (`Storage`, `Shared`, `Conn`), every table is sized at comptime. -//! See docs/LIBRARY.md. -const std = @import("std"); -const builtin = @import("builtin"); -const cloud9 = @import("cloud9"); -const vars = @import("vars.zig"); -const Writer = std.Io.Writer; - -/// Longest file name accepted in a create or rename. -pub const max_name: usize = 255; - -/// A dynamic-file generator (`Config.fns`, `Config.runtime`): writes the file's -/// content into `w` at open time (and again at each read from offset 0). -pub const Gen = *const fn (ctx: *anyopaque, w: *Writer) anyerror!void; -/// The /ctl command handler: `cmd` is the written text, `out` receives the -/// result that later reads of /ctl return. -pub const Ctl = *const fn (ctx: *anyopaque, cmd: []const u8, out: *Writer) anyerror!void; - -pub const Config = struct { - /// Appears in /README and as uid/gid/muid of every Stat. - name: []const u8 = "introspect", - /// A type whose pub decls `zig_version`, `target`, `optimize`, `time` - /// and `change` (all `[]const u8`) become the files of /build. `null` - /// omits /build. - build: ?type = null, - /// /comptime/types//{name,size,align,fields}. - types: []const type = &.{}, - /// /comptime/decls lists this type's pub decls (empty when null). - decls_of: ?type = null, - /// /runtime/fn/: every pub decl is a `fn (ctx: *anyopaque, w: *std.Io.Writer) anyerror!void`. - fns: type = struct {}, - /// /runtime/: same signature as `fns`, one level up (pid, uptime, ...). - runtime: type = struct {}, - /// The /ctl handler; `null` omits /ctl. - ctl: ?Ctl = null, - /// Where /ctl lives: the root or /runtime/ctl. - ctl_dir: enum { root, runtime } = .root, - /// Capacity of the ctl result (in `Shared`). - ctl_bytes: u32 = 4096, - /// Largest negotiable msize; sizes `Storage.in/out/data`. - msize: u32 = 8192, - max_fids: u16 = 64, - max_providers: u8 = 8, - max_vars: u8 = 32, - /// Dynamic file contents are generated at open time into per-fid snapshot - /// slots so that reads at arbitrary offsets are consistent. - snapshot_slots: u8 = 8, - snapshot_bytes: u32 = 16 * 1024, -}; - -/// Attributes of a provider node, filled by `VTable.stat` and `VTable.list`. -pub const NodeStat = struct { - /// Permission bits plus `cloud9.dmdir`/`dmappend`/`dmexcl`. - mode: u32, - length: u64 = 0, - atime: u32 = 0, - mtime: u32 = 0, - /// Becomes the qid version. - version: u32 = 0, - /// The entry name (`list`) or the node's own name (`stat`; ignored for the - /// provider root, whose name is the mount name). Must stay valid until the - /// provider's next call. - name: []const u8 = "", - /// Filled by `list`: the entry's handle. Not retained by the core. - handle: Provider.Handle = 0, - /// A stable identity for the qid path (low 56 bits), for providers whose - /// handles are not stable across the node's life (e.g. memory addresses - /// that an allocator may reuse). 0 means "the handle is the path". - path: u64 = 0, - - pub fn isDir(s: NodeStat) bool { - return s.mode & cloud9.dmdir != 0; - } -}; - -/// A runtime subtree mounted at a top-level name. -/// -/// Handle lifetime: every handle returned by `walk` or `create` is released by -/// the core with exactly one `clunk` (after `close` if the fid was open). The -/// root handle 0 is never obtained through `walk`, so providers must treat -/// `clunk(0)` as a no-op. `walk` must accept "." on any node, file or directory -/// (a fresh reference to the same node; the core clones fids with it), and ".." -/// on directories (except at the root, which the core resolves itself). -pub const Provider = struct { - name: []const u8, - ctx: *anyopaque, - vtable: *const VTable, - - /// Provider-defined node id; 0 = provider root. - pub const Handle = u64; - pub const root: Handle = 0; - - pub const Error = error{ NotFound, Exists, Perm, NotDir, IsDir, NotEmpty, BadOffset, NoSpace, Io, Unsupported, Excl }; - - pub const VTable = struct { - walk: *const fn (ctx: *anyopaque, parent: Handle, name: []const u8) Error!Handle, - stat: *const fn (ctx: *anyopaque, h: Handle, out: *NodeStat) Error!void, - /// The `index`-th entry of `dir`; false when done. - list: *const fn (ctx: *anyopaque, dir: Handle, index: usize, out: *NodeStat) Error!bool, - open: *const fn (ctx: *anyopaque, h: Handle, mode: u8) Error!void, - read: *const fn (ctx: *anyopaque, h: Handle, offset: u64, buf: []u8) Error!usize, - write: *const fn (ctx: *anyopaque, h: Handle, offset: u64, data: []const u8) Error!usize, - /// Returns the new node, already open with `mode`. - create: ?*const fn (ctx: *anyopaque, dir: Handle, name: []const u8, perm: u32, mode: u8) Error!Handle = null, - remove: ?*const fn (ctx: *anyopaque, h: Handle) Error!void = null, - /// Only name, length, mode and mtime can differ from the current stat - /// (the core has already checked the immutable fields and the name). - wstat: ?*const fn (ctx: *anyopaque, h: Handle, st: *const cloud9.Stat) Error!void = null, - /// An open fid on `h` was released (before `clunk`). - close: ?*const fn (ctx: *anyopaque, h: Handle) void = null, - /// A fid holding `h` was released (also on hangup and Tversion). - clunk: *const fn (ctx: *anyopaque, h: Handle) void, - }; -}; - -/// The Plan 9 error string for any error the engine or a provider can raise. -pub fn ename(err: anyerror) []const u8 { - return switch (err) { - error.NotFound, error.NoFile => "file does not exist", - error.Perm => "permission denied", - error.Exists => "file already exists", - error.NotEmpty => "directory not empty", - error.NotDir => "not a directory", - error.IsDir => "is a directory", - error.BadOffset => "bad offset", - error.NoSpace => "no space left on device", - error.Io => "i/o error", - error.Unsupported => "not supported", - error.Excl => "exclusive use file already open", - error.FidInUse => "fid in use", - error.UnknownFid => "unknown fid", - error.NotOpen => "file not open", - error.AlreadyOpen => "file already open", - error.AuthNotRequired => "authentication not required", - error.BadCommand => "bad command", - error.BadName => "bad file name", - error.Invalid, error.BadValue => "bad value", - error.TooManyFids => "too many fids", - error.NoSnapshot => "too many open dynamic files", - error.WriteFailed => "no space in buffer", - error.ReplyTooLarge => "reply too large for msize", - error.OutOfMemory => "out of memory", - else => "i/o error", - }; -} - -/// A "don't care" Twstat: every field left as it is. -pub const stat_dontcare: cloud9.Stat = .{ - .type = 0xFFFF, - .dev = 0xFFFF_FFFF, - .qid = .{ .type = 0xFF, .version = 0xFFFF_FFFF, .path = 0xFFFF_FFFF_FFFF_FFFF }, - .mode = 0xFFFF_FFFF, - .atime = 0xFFFF_FFFF, - .mtime = 0xFFFF_FFFF, - .length = 0xFFFF_FFFF_FFFF_FFFF, - .name = "", - .uid = "", - .gid = "", - .muid = "", -}; - -pub fn validName(name: []const u8) error{BadName}!void { - if (name.len == 0 or name.len > max_name) return error.BadName; - if (std.mem.eql(u8, name, ".") or std.mem.eql(u8, name, "..")) return error.BadName; - if (std.mem.indexOfAny(u8, name, "/\x00") != null) return error.BadName; -} - -/// "YYYY-MM-DDTHH:MM:SSZ" as unix seconds, or null. -pub fn parseIso8601(s: []const u8) ?u32 { - if (s.len != 20 or s[4] != '-' or s[7] != '-' or s[10] != 'T' or s[13] != ':' or s[16] != ':' or s[19] != 'Z') return null; - const y = std.fmt.parseInt(i64, s[0..4], 10) catch return null; - const mo = std.fmt.parseInt(i64, s[5..7], 10) catch return null; - const d = std.fmt.parseInt(i64, s[8..10], 10) catch return null; - const h = std.fmt.parseInt(i64, s[11..13], 10) catch return null; - const mi = std.fmt.parseInt(i64, s[14..16], 10) catch return null; - const sec = std.fmt.parseInt(i64, s[17..19], 10) catch return null; - if (mo < 1 or mo > 12 or d < 1 or d > 31 or h > 23 or mi > 59 or sec > 60) return null; - // Howard Hinnant's days_from_civil. - const yy = if (mo <= 2) y - 1 else y; - const era = @divFloor(yy, 400); - const yoe = yy - era * 400; - const mp = if (mo > 2) mo - 3 else mo + 9; - const doy = @divFloor(153 * mp + 2, 5) + d - 1; - const doe = yoe * 365 + @divFloor(yoe, 4) - @divFloor(yoe, 100) + doy; - const days = era * 146097 + doe - 719468; - const total = days * 86400 + h * 3600 + mi * 60 + sec; - if (total < 0 or total > std.math.maxInt(u32)) return null; - return @intCast(total); -} - -/// The last component of @typeName(T): "wire.Qid" -> "Qid". -pub fn shortTypeName(comptime T: type) []const u8 { - const full = @typeName(T); - const dot = std.mem.lastIndexOfScalar(u8, full, '.') orelse return full; - return full[dot + 1 ..]; -} - -/// The /comptime/types//fields text: "name: type @offset" per line. -pub fn fieldsText(comptime T: type) []const u8 { - comptime { - @setEvalBranchQuota(200_000); - var s: []const u8 = ""; - switch (@typeInfo(T)) { - .@"struct" => |info| for (info.fields) |f| { - if (info.layout == .@"packed") { - s = s ++ std.fmt.comptimePrint("{s}: {s} @{d}b\n", .{ f.name, @typeName(f.type), @bitOffsetOf(T, f.name) }); - } else if (f.is_comptime) { - s = s ++ std.fmt.comptimePrint("{s}: {s} (comptime)\n", .{ f.name, @typeName(f.type) }); - } else { - s = s ++ std.fmt.comptimePrint("{s}: {s} @{d}\n", .{ f.name, @typeName(f.type), @offsetOf(T, f.name) }); - } - }, - .@"union" => |info| for (info.fields) |f| { - s = s ++ f.name ++ ": " ++ @typeName(f.type) ++ "\n"; - }, - .@"enum" => |info| for (info.fields) |f| { - s = s ++ std.fmt.comptimePrint("{s} = {d}\n", .{ f.name, f.value }); - }, - else => s = @typeName(T) ++ "\n", - } - return s; - } -} - -fn declsText(comptime T: type) []const u8 { - comptime { - @setEvalBranchQuota(20_000); - const decls = switch (@typeInfo(T)) { - inline .@"struct", .@"union", .@"enum", .@"opaque" => |info| info.decls, - else => &[_]std.builtin.Type.Declaration{}, - }; - var s: []const u8 = ""; - for (decls) |d| s = s ++ d.name ++ "\n"; - return s; - } -} - -/// Wraps `Fns.` in a function of exactly the `Gen` signature. -fn genFor(comptime Fns: type, comptime name: []const u8) Gen { - return &struct { - fn g(ctx: *anyopaque, w: *Writer) anyerror!void { - return @field(Fns, name)(ctx, w); - } - }.g; -} - -/// A node of the static tree, described at comptime. -pub const Node = struct { - name: []const u8, - kind: Kind, - children: []const Node = &.{}, - content: []const u8 = "", - gen: ?Gen = null, - - pub const Kind = enum(u8) { dir, static, dynamic, ctl, vars }; - - fn isDir(n: Node) bool { - return n.kind == .dir or n.kind == .vars; - } -}; - -fn genNodes(comptime Fns: type) [@typeInfo(Fns).@"struct".decls.len]Node { - const decls = @typeInfo(Fns).@"struct".decls; - var arr: [decls.len]Node = undefined; - for (decls, 0..) |d, i| arr[i] = .{ .name = d.name, .kind = .dynamic, .gen = genFor(Fns, d.name) }; - return arr; -} - -pub fn Server(comptime cfg: Config) type { - return struct { - const Self = @This(); - - // -- the static tree ------------------------------------------------ - - pub const readme_text = std.fmt.comptimePrint( - \\{s}: a 9P2000 introspection server (built on cloud9). - \\ - \\/build facts baked in at build time (zig version, target, optimize, time, change id) - \\/comptime facts computed by the Zig compiler: type layouts under types//, pub decls - \\/runtime live facts; fn/ calls a Zig function on every read - \\/ctl write a command, read the result - \\/vars exposed variables: /{{value,type,size,addr,raw,f//...}} - \\ - \\Other top-level directories are providers mounted at runtime. - \\ - , .{cfg.name}); - - fn typeDir(comptime T: type) Node { - return .{ .name = shortTypeName(T), .kind = .dir, .children = &.{ - .{ .name = "name", .kind = .static, .content = @typeName(T) }, - .{ .name = "size", .kind = .static, .content = std.fmt.comptimePrint("{d}", .{@sizeOf(T)}) }, - .{ .name = "align", .kind = .static, .content = std.fmt.comptimePrint("{d}", .{@alignOf(T)}) }, - .{ .name = "fields", .kind = .static, .content = fieldsText(T) }, - } }; - } - - const type_dirs: [cfg.types.len]Node = blk: { - @setEvalBranchQuota(200_000); - var arr: [cfg.types.len]Node = undefined; - for (cfg.types, 0..) |T, i| arr[i] = typeDir(T); - for (arr, 0..) |a, i| for (arr[i + 1 ..]) |b| { - if (std.mem.eql(u8, a.name, b.name)) @compileError("duplicate short type name " ++ a.name); - }; - break :blk arr; - }; - - const decls_text: []const u8 = if (cfg.decls_of) |T| declsText(T) else ""; - const fn_nodes = genNodes(cfg.fns); - const runtime_nodes = genNodes(cfg.runtime); - const ctl_node: Node = .{ .name = "ctl", .kind = .ctl }; - - const build_nodes: []const Node = if (cfg.build) |B| &[_]Node{ - .{ .name = "zig_version", .kind = .static, .content = B.zig_version }, - .{ .name = "target", .kind = .static, .content = B.target }, - .{ .name = "optimize", .kind = .static, .content = B.optimize }, - .{ .name = "time", .kind = .static, .content = B.time }, - .{ .name = "change", .kind = .static, .content = B.change }, - } else &.{}; - - /// Build time as unix seconds (for static atime/mtime), or 0. - pub const build_secs: u32 = if (cfg.build) |B| (parseIso8601(B.time) orelse 0) else 0; - - const runtime_children: []const Node = blk: { - var list: []const Node = &runtime_nodes; - list = list ++ &[_]Node{.{ .name = "fn", .kind = .dir, .children = &fn_nodes }}; - if (cfg.ctl != null and cfg.ctl_dir == .runtime) list = list ++ &[_]Node{ctl_node}; - break :blk list; - }; - - const root_children: []const Node = blk: { - var list: []const Node = &[_]Node{.{ .name = "README", .kind = .static, .content = readme_text }}; - if (cfg.build != null) list = list ++ &[_]Node{.{ .name = "build", .kind = .dir, .children = build_nodes }}; - list = list ++ &[_]Node{ - .{ .name = "comptime", .kind = .dir, .children = &.{ - .{ .name = "types", .kind = .dir, .children = &type_dirs }, - .{ .name = "decls", .kind = .static, .content = decls_text }, - } }, - .{ .name = "runtime", .kind = .dir, .children = runtime_children }, - }; - if (cfg.ctl != null and cfg.ctl_dir == .root) list = list ++ &[_]Node{ctl_node}; - list = list ++ &[_]Node{.{ .name = "vars", .kind = .vars }}; - break :blk list; - }; - - pub const root_node: Node = .{ .name = "/", .kind = .dir, .children = root_children }; - - /// The static tree flattened so nodes can be referenced by index; the - /// children of a node occupy consecutive slots `first..first+count`. - const Flat = struct { node: Node, parent: u32, first: u32, count: u32 }; - - fn countNodes(n: Node) usize { - var c: usize = 1; - for (n.children) |ch| c += countNodes(ch); - return c; - } - - fn fillFlat(arr: []Flat, next: *usize, idx: usize, n: Node, parent: u32) void { - const first = next.*; - next.* += n.children.len; - arr[idx] = .{ .node = n, .parent = parent, .first = @intCast(first), .count = @intCast(n.children.len) }; - for (n.children, 0..) |ch, i| fillFlat(arr, next, first + i, ch, @intCast(idx)); - } - - pub const flat_len = countNodes(root_node); - pub const flat: [flat_len]Flat = blk: { - @setEvalBranchQuota(100_000); - var arr: [flat_len]Flat = undefined; - var next: usize = 1; - fillFlat(&arr, &next, 0, root_node, 0); - break :blk arr; - }; - const vars_idx: u32 = blk: { - for (flat, 0..) |f, i| if (f.node.kind == .vars) break :blk @intCast(i); - @compileError("no vars node"); - }; - - comptime { - for (flat) |f| if (f.node.kind == .dynamic and f.node.gen == null) @compileError("dynamic node without generator"); - std.debug.assert(cfg.msize >= cloud9.Server.msize_min); - std.debug.assert(cfg.snapshot_slots > 0 and cfg.max_fids > 0); - // Provider index 0xFE/0xFF would collide with the var/static qid tags. - std.debug.assert(cfg.max_providers < 0xFE); - } - - // -- qid paths ------------------------------------------------------ - - const static_tag: u64 = 0xFF << 56; - const var_tag: u64 = 0xFE << 56; - const handle_mask: u64 = (1 << 56) - 1; - - // -- storage -------------------------------------------------------- - - /// Per-connection buffers; the caller places one in static memory. - pub const Storage = struct { - in: [cfg.msize]u8, - out: [cfg.msize]u8, - /// Staging area for read replies (directory records, provider and raw reads). - data: [cfg.msize]u8, - snapshots: [cfg.snapshot_slots][cfg.snapshot_bytes]u8, - }; - - const Var = struct { - name: []const u8, - ptr: *anyopaque, - vt: *const vars.VTable, - }; - - /// State common to all connections: providers, exposed variables, the - /// ctl result. Not internally synchronized: one thread serves all - /// connections (or the caller serializes). - pub const Shared = struct { - ctx: *anyopaque, - providers: [cfg.max_providers]Provider = undefined, - nprov: u8 = 0, - vars: [cfg.max_vars]Var = undefined, - nvars: u8 = 0, - /// The ctl result is double-buffered: a command writes into the - /// buffer that is not current and commits it only on success, so - /// a failed command leaves the previous result intact. - ctl_bufs: [2][cfg.ctl_bytes]u8 = undefined, - ctl_cur: u1 = 0, - /// Length of the current ctl result (in `ctl_bufs[ctl_cur]`). - ctl_len: u32 = 0, - ctl_version: u32 = 0, - /// Entropy for the per-connection fid hash. The core mixes in a - /// connection counter and buffer addresses; a platform layer with a - /// random source may set this once after `init` to make the seed - /// unpredictable even where addresses are static. - hash_seed: u32 = 0, - conn_seq: u32 = 0, - - /// `ctx` is passed to every `fns`/`runtime` generator and to `ctl`. - pub fn init(ctx: *anyopaque) Shared { - return .{ .ctx = ctx }; - } - - /// Mounts `p` at `/`. The name must not collide with a - /// static entry or another provider. - pub fn addProvider(s: *Shared, p: Provider) error{Full}!void { - if (s.nprov == cfg.max_providers) return error.Full; - std.debug.assert(validName(p.name) != error.BadName); - std.debug.assert(s.findProvider(p.name) == null); - std.debug.assert(staticChild(0, p.name) == null); - s.providers[s.nprov] = p; - s.nprov += 1; - } - - /// Publishes `ptr.*` as /vars/. `name` and the pointee must - /// outlive the server. - pub fn expose(s: *Shared, name: []const u8, ptr: anytype) error{Full}!void { - const P = @TypeOf(ptr); - const info = @typeInfo(P); - if (info != .pointer or info.pointer.size != .one or info.pointer.is_const) @compileError("expose wants a *T, got " ++ @typeName(P)); - if (s.nvars == cfg.max_vars) return error.Full; - std.debug.assert(validName(name) != error.BadName); - std.debug.assert(s.findVar(name) == null); - s.vars[s.nvars] = .{ .name = name, .ptr = @ptrCast(ptr), .vt = vars.vtableFor(info.pointer.child) }; - s.nvars += 1; - } - - /// The result of the last successful ctl command. - pub fn ctlResult(s: *const Shared) []const u8 { - return s.ctl_bufs[s.ctl_cur][0..s.ctl_len]; - } - - fn findProvider(s: *const Shared, name: []const u8) ?u8 { - for (s.providers[0..s.nprov], 0..) |p, i| if (std.mem.eql(u8, p.name, name)) return @intCast(i); - return null; - } - - fn findVar(s: *const Shared, name: []const u8) ?u8 { - for (s.vars[0..s.nvars], 0..) |v, i| if (std.mem.eql(u8, v.name, name)) return @intCast(i); - return null; - } - }; - - fn staticChild(idx: u32, name: []const u8) ?u32 { - const f = flat[idx]; - for (f.first..f.first + f.count) |ci| { - if (std.mem.eql(u8, flat[ci].node.name, name)) return @intCast(ci); - } - return null; - } - - // -- connection ----------------------------------------------------- - - const NodeRef = union(enum) { - static: u32, - prov: struct { idx: u8, h: Provider.Handle }, - @"var": struct { idx: u8, node: u32 }, - }; - - const Fid = struct { - id: u32 = 0, - used: bool = false, - node: NodeRef = .{ .static = 0 }, - is_dir: bool = true, - open: bool = false, - mode: u8 = 0, - rclose: bool = false, - dir_offset: u64 = 0, - dir_index: usize = 0, - /// Snapshot slot of an open dynamic file. - snap: ?u8 = null, - /// Free-list link, meaningful while `!used`. - next_free: u16 = no_slot, - }; - - const no_slot: u16 = std.math.maxInt(u16); - /// The fid index is an open-addressing (linear probing) table from fid - /// number to a slot of `Conn.fids`, sized to stay at most half full so - /// that lookups are O(1) with any number of fids. - const index_len: usize = std.math.ceilPowerOfTwoAssert(usize, @as(usize, cfg.max_fids) * 2); - const index_mask: usize = index_len - 1; - const index_shift: u5 = @intCast(32 - @as(usize, std.math.log2_int(usize, index_len))); - - /// MurmurHash3's 32-bit finalizer: every input bit affects every output bit. - fn fmix32(x: u32) u32 { - var h = x; - h ^= h >> 16; - h *%= 0x85EB_CA6B; - h ^= h >> 13; - h *%= 0xC2B2_AE35; - h ^= h >> 16; - return h; - } - - /// Everything the engine needs to know about a node for qid/stat. - const Info = struct { - is_dir: bool, - mode: u32, - length: u64, - atime: u32, - mtime: u32, - version: u32, - path: u64, - name: []const u8, - - fn qid(i: Info) cloud9.Qid { - var t: u8 = if (i.is_dir) cloud9.qtdir else cloud9.qtfile; - if (i.mode & cloud9.dmappend != 0) t |= cloud9.qtappend; - if (i.mode & cloud9.dmexcl != 0) t |= cloud9.qtexcl; - return .{ .type = t, .version = i.version, .path = i.path }; - } - - fn stat(i: Info) cloud9.Stat { - return .{ - .type = 0, - .dev = 0, - .qid = i.qid(), - .mode = i.mode, - .atime = i.atime, - .mtime = i.mtime, - .length = i.length, - .name = i.name, - .uid = cfg.name, - .gid = cfg.name, - .muid = cfg.name, - }; - } - }; - - /// One 9P connection: a cloud9.Server plus a fid table and snapshot slots. - pub const Conn = struct { - shared: *Shared, - storage: *Storage, - server: cloud9.Server, - /// Largest msize this connection negotiates. - msize_cap: u32, - fids: [cfg.max_fids]Fid = @splat(.{}), - /// XORed into every fid number before hashing so that a client - /// cannot precompute fid numbers that collide (which would turn the - /// index back into a linear scan). - hash_seed: u32, - /// fid number -> slot of `fids` (`no_slot` = empty bucket). - index: [index_len]u16 = @splat(no_slot), - /// Head of the free list threaded through `Fid.next_free`. - free_head: u16 = no_slot, - /// Slots `high_water..` have never been used (bump allocation). - high_water: u16 = 0, - nfids: u16 = 0, - slot_used: [cfg.snapshot_slots]bool = @splat(false), - slot_len: [cfg.snapshot_slots]u32 = @splat(0), - name_buf: [max_name]u8 = undefined, - - pub fn init(shared: *Shared, storage: *Storage, msize: u32) Conn { - shared.conn_seq +%= 1; - const addr = @intFromPtr(storage) ^ (@intFromPtr(shared) << 7); - const seed = fmix32(shared.hash_seed ^ (shared.conn_seq *% 0x9E37_79B1) ^ @as(u32, @truncate(addr)) ^ @as(u32, @truncate(addr >> 16))); - return .{ - .shared = shared, - .storage = storage, - .server = .init(.{ .in = &storage.in, .out = &storage.out }), - .msize_cap = @max(@min(msize, cfg.msize), cloud9.Server.msize_min), - .hash_seed = seed, - }; - } - - /// Fibonacci hashing of the (seeded) fid number into `index_len` buckets. - fn fidHome(c: *const Conn, id: u32) usize { - return @intCast(((id ^ c.hash_seed) *% 0x9E37_79B1) >> index_shift); - } - - /// Feeds transport bytes; returns how many were taken. - pub fn push(c: *Conn, bytes: []const u8) usize { - return c.server.push(bytes); - } - - /// Bytes to send to the client. - pub fn output(c: *const Conn) []const u8 { - return c.server.output(); - } - - pub fn wrote(c: *Conn, n: usize) void { - c.server.wrote(n); - } - - /// Drops every fid (telling providers) and kills the session. - pub fn hangup(c: *Conn) void { - c.resetFids(); - c.server.hangup(); - } - - /// Handles at most one request. Returns false when more input (or - /// output drainage) is needed. `error.Protocol` is terminal. - pub fn step(c: *Conn) error{Protocol}!bool { - const req = (c.server.receive() catch return error.Protocol) orelse return false; - defer c.server.release(); - switch (req.msg) { - .tversion => |m| { - c.resetFids(); - c.server.negotiate(@min(m.msize, c.msize_cap), m.version) catch return error.Protocol; - }, - else => { - const reply = c.dispatch(req.msg) catch |e| cloud9.Msg{ .rerror = .{ .ename = ename(e) } }; - c.server.reply(req.tag, reply) catch |e| switch (e) { - // The reply does not fit the negotiated msize (Rstat or a long - // Rwalk at a tiny msize). receive() guarantees room for one - // msize-sized message, so this is never backpressure: answer with - // an Rerror (truncated to fit by cloud9). Rwalk, the only - // variable-size reply that follows a state change, is size-checked - // in walk() before anything is mutated. - error.TooLarge => c.server.reply(req.tag, .{ .rerror = .{ .ename = ename(error.ReplyTooLarge) } }) catch return error.Protocol, - else => return error.Protocol, - }; - }, - } - return true; - } - - /// Number of fids currently held. - pub fn fidCount(c: *const Conn) usize { - return c.nfids; - } - - fn dispatch(c: *Conn, msg: cloud9.Msg) anyerror!cloud9.Msg { - return switch (msg) { - .tauth => error.AuthNotRequired, - .tattach => |m| c.attach(m), - .tflush => .rflush, - .twalk => |m| c.walk(m), - .topen => |m| c.open(m), - .tcreate => |m| c.create(m), - .tread => |m| c.read(m), - .twrite => |m| c.write(m), - .tclunk => |m| c.clunk(m), - .tremove => |m| c.remove(m), - .tstat => |m| c.stat(m), - .twstat => |m| c.wstat(m), - else => error.Protocol, - }; - } - - // -- fid table -- - - /// The index bucket holding `id`, if any. - fn findBucket(c: *const Conn, id: u32) ?usize { - var pos = c.fidHome(id); - while (true) : (pos = (pos + 1) & index_mask) { - const slot = c.index[pos]; - if (slot == no_slot) return null; - if (c.fids[slot].id == id) return pos; - } - } - - fn findFid(c: *Conn, id: u32) ?*Fid { - const pos = c.findBucket(id) orelse return null; - return &c.fids[c.index[pos]]; - } - - fn allocFid(c: *Conn, id: u32) !*Fid { - if (c.findBucket(id) != null) return error.FidInUse; - if (c.nfids >= cfg.max_fids) return error.TooManyFids; - const slot: u16 = if (c.free_head != no_slot) blk: { - const slot = c.free_head; - c.free_head = c.fids[slot].next_free; - break :blk slot; - } else blk: { - const slot = c.high_water; - c.high_water += 1; - break :blk slot; - }; - c.fids[slot] = .{ .id = id, .used = true }; - var pos = c.fidHome(id); - while (c.index[pos] != no_slot) pos = (pos + 1) & index_mask; - c.index[pos] = slot; - c.nfids += 1; - return &c.fids[slot]; - } - - /// Removes `id` from the index (backward-shift deletion: no tombstones). - fn unlinkFid(c: *Conn, id: u32) void { - var i = c.findBucket(id).?; - var j = i; - while (true) { - j = (j + 1) & index_mask; - const slot = c.index[j]; - if (slot == no_slot) break; - const k = c.fidHome(c.fids[slot].id); - // The entry at j may move into the hole at i unless its home - // lies in the cyclic interval (i, j]. - const stays = if (i <= j) (k > i and k <= j) else (k > i or k <= j); - if (!stays) { - c.index[i] = slot; - i = j; - } - } - c.index[i] = no_slot; - } - - /// Releases everything a fid holds; the slot stays allocated. - fn dropContents(c: *Conn, f: *Fid) void { - if (f.snap) |s| c.slot_used[s] = false; - f.snap = null; - if (f.node == .prov) { - const p = c.shared.providers[f.node.prov.idx]; - if (f.open) if (p.vtable.close) |close| close(p.ctx, f.node.prov.h); - if (f.rclose and f.open) if (p.vtable.remove) |rm| rm(p.ctx, f.node.prov.h) catch {}; - p.vtable.clunk(p.ctx, f.node.prov.h); - } - f.open = false; - f.rclose = false; - } - - fn freeFid(c: *Conn, f: *Fid) void { - c.dropContents(f); - c.unlinkFid(f.id); - const slot: u16 = @intCast((@intFromPtr(f) - @intFromPtr(&c.fids)) / @sizeOf(Fid)); - f.* = .{ .next_free = c.free_head }; - c.free_head = slot; - c.nfids -= 1; - } - - fn resetFids(c: *Conn) void { - for (c.fids[0..c.high_water]) |*f| { - if (f.used) c.dropContents(f); - f.* = .{}; - } - @memset(&c.index, no_slot); - c.free_head = no_slot; - c.high_water = 0; - c.nfids = 0; - } - - /// Releases a provider handle that is not held by any fid. - fn releaseRef(c: *Conn, ref: NodeRef) void { - if (ref == .prov) { - const p = c.shared.providers[ref.prov.idx]; - p.vtable.clunk(p.ctx, ref.prov.h); - } - } - - // -- node helpers -- - - fn provider(c: *Conn, idx: u8) Provider { - return c.shared.providers[idx]; - } - - fn varBase(c: *Conn, idx: u8, node: u32) [*]u8 { - const v = c.shared.vars[idx]; - return @as([*]u8, @ptrCast(v.ptr)) + v.vt.nodes[node].offset; - } - - fn info(c: *Conn, ref: NodeRef) !Info { - switch (ref) { - .static => |idx| { - const n = flat[idx].node; - return .{ - .is_dir = n.isDir(), - .mode = switch (n.kind) { - .dir, .vars => cloud9.dmdir | 0o555, - .ctl => 0o666, - else => 0o444, - }, - .length = switch (n.kind) { - .static => n.content.len, - .ctl => c.shared.ctl_len, - else => 0, - }, - .atime = build_secs, - .mtime = build_secs, - .version = if (n.kind == .ctl) c.shared.ctl_version else 0, - .path = static_tag | idx, - .name = n.name, - }; - }, - .@"var" => |v| { - const sv = c.shared.vars[v.idx]; - const n = sv.vt.nodes[v.node]; - return .{ - .is_dir = n.isDir(), - .mode = if (n.isDir()) cloud9.dmdir | 0o555 else if (n.writable()) 0o644 else 0o444, - .length = switch (n.kind) { - .type_name, .size => n.content.len, - .raw => n.size, - else => 0, - }, - .atime = 0, - .mtime = 0, - .version = 0, - .path = var_tag | (@as(u64, v.idx) << 32) | v.node, - .name = if (v.node == 0) sv.name else n.name, - }; - }, - .prov => |p| { - const pr = c.provider(p.idx); - var st: NodeStat = .{ .mode = 0 }; - try pr.vtable.stat(pr.ctx, p.h, &st); - return provInfo(pr, p.idx, p.h, st); - }, - } - } - - fn provInfo(pr: Provider, idx: u8, h: Provider.Handle, st: NodeStat) Info { - return .{ - .is_dir = st.isDir(), - .mode = st.mode, - .length = if (st.isDir()) 0 else st.length, - .atime = st.atime, - .mtime = st.mtime, - .version = st.version, - .path = (@as(u64, idx) << 56) | ((if (st.path != 0) st.path else h) & handle_mask), - .name = if (h == Provider.root) pr.name else st.name, - }; - } - - /// Copies `st.name` into the connection so the reply cannot dangle. - fn pinName(c: *Conn, st: cloud9.Stat) cloud9.Stat { - var out = st; - const n = @min(st.name.len, c.name_buf.len); - @memcpy(c.name_buf[0..n], st.name[0..n]); - out.name = c.name_buf[0..n]; - return out; - } - - const Looked = struct { ref: NodeRef, info: Info }; - - /// Resolves `name` in the directory `ref`. A returned provider ref - /// is a fresh handle the caller must release or retain. - fn lookup(c: *Conn, ref: NodeRef, name: []const u8) !Looked { - switch (ref) { - .static => |idx| { - const dot = std.mem.eql(u8, name, "."); - const dotdot = std.mem.eql(u8, name, ".."); - var next: NodeRef = undefined; - if (dot) { - next = ref; - } else if (dotdot) { - next = .{ .static = flat[idx].parent }; - } else if (flat[idx].node.kind == .vars) { - const vi = c.shared.findVar(name) orelse return error.NotFound; - next = .{ .@"var" = .{ .idx = vi, .node = 0 } }; - } else if (staticChild(idx, name)) |ci| { - next = .{ .static = ci }; - } else if (idx == 0) { - const pi = c.shared.findProvider(name) orelse return error.NotFound; - next = .{ .prov = .{ .idx = pi, .h = Provider.root } }; - } else return error.NotFound; - return .{ .ref = next, .info = try c.info(next) }; - }, - .@"var" => |v| { - const vt = c.shared.vars[v.idx].vt; - var next = ref; - if (std.mem.eql(u8, name, ".")) { - // unchanged - } else if (std.mem.eql(u8, name, "..")) { - next = if (v.node == 0) .{ .static = vars_idx } else .{ .@"var" = .{ .idx = v.idx, .node = vt.nodes[v.node].parent } }; - } else { - const ci = vt.child(v.node, name) orelse return error.NotFound; - next = .{ .@"var" = .{ .idx = v.idx, .node = ci } }; - } - return .{ .ref = next, .info = try c.info(next) }; - }, - .prov => |p| { - if (p.h == Provider.root and std.mem.eql(u8, name, "..")) { - const next: NodeRef = .{ .static = 0 }; - return .{ .ref = next, .info = try c.info(next) }; - } - const pr = c.provider(p.idx); - const h = try pr.vtable.walk(pr.ctx, p.h, name); - const next: NodeRef = .{ .prov = .{ .idx = p.idx, .h = h } }; - errdefer c.releaseRef(next); - return .{ .ref = next, .info = try c.info(next) }; - }, - } - } - - /// The i-th entry of directory `ref` as a Stat, or null past the end. - /// The name borrows either static memory or the provider's NodeStat. - fn entryStat(c: *Conn, ref: NodeRef, i: usize) !?cloud9.Stat { - switch (ref) { - .static => |idx| { - const f = flat[idx]; - if (f.node.kind == .vars) { - if (i >= c.shared.nvars) return null; - return (try c.info(.{ .@"var" = .{ .idx = @intCast(i), .node = 0 } })).stat(); - } - if (i < f.count) return (try c.info(.{ .static = f.first + @as(u32, @intCast(i)) })).stat(); - if (idx == 0) { - const pi = i - f.count; - if (pi >= c.shared.nprov) return null; - return (try c.info(.{ .prov = .{ .idx = @intCast(pi), .h = Provider.root } })).stat(); - } - return null; - }, - .@"var" => |v| { - const n = c.shared.vars[v.idx].vt.nodes[v.node]; - if (i >= n.count) return null; - return (try c.info(.{ .@"var" = .{ .idx = v.idx, .node = n.first + @as(u32, @intCast(i)) } })).stat(); - }, - .prov => |p| { - const pr = c.provider(p.idx); - var st: NodeStat = .{ .mode = 0 }; - if (!try pr.vtable.list(pr.ctx, p.h, i, &st)) return null; - return provInfo(pr, p.idx, st.handle, st).stat(); - }, - } - } - - // -- snapshots -- - - fn takeSlot(c: *Conn) !u8 { - for (&c.slot_used, 0..) |*u, i| if (!u.*) { - u.* = true; - return @intCast(i); - }; - return error.NoSnapshot; - } - - /// (Re)generates the content of a dynamic file into its slot. - fn generate(c: *Conn, f: *Fid) !void { - const s = f.snap.?; - var w: Writer = .fixed(&c.storage.snapshots[s]); - c.slot_len[s] = 0; - switch (f.node) { - .static => |idx| try flat[idx].node.gen.?(c.shared.ctx, &w), - .@"var" => |v| { - const n = c.shared.vars[v.idx].vt.nodes[v.node]; - const base = c.varBase(v.idx, v.node); - switch (n.kind) { - .value => try n.render.?(base, &w), - .addr => try w.print("0x{x}", .{@intFromPtr(base)}), - else => unreachable, - } - }, - .prov => unreachable, - } - c.slot_len[s] = @intCast(w.buffered().len); - } - - fn isDynamic(c: *Conn, ref: NodeRef) bool { - return switch (ref) { - .static => |idx| flat[idx].node.kind == .dynamic, - .@"var" => |v| switch (c.shared.vars[v.idx].vt.nodes[v.node].kind) { - .value, .addr => true, - else => false, - }, - .prov => false, - }; - } - - // -- request handlers -- - - fn attach(c: *Conn, m: anytype) !cloud9.Msg { - const f = try c.allocFid(m.fid); - f.node = .{ .static = 0 }; - f.is_dir = true; - return .{ .rattach = .{ .qid = (try c.info(f.node)).qid() } }; - } - - fn walk(c: *Conn, m: anytype) !cloud9.Msg { - const f = c.findFid(m.fid) orelse return error.UnknownFid; - if (m.newfid != m.fid and c.findFid(m.newfid) != null) return error.FidInUse; - if (m.newfid != m.fid and c.nfids >= cfg.max_fids) return error.TooManyFids; - if (m.nwname > 0 and f.open) return error.AlreadyOpen; - // Cloning a fid onto itself changes nothing; in particular it must not - // close an open fid or discard generated content. - if (m.nwname == 0 and m.newfid == m.fid) return .{ .rwalk = .{ .nwqid = 0 } }; - // A full Rwalk must fit the negotiated msize; check before binding anything. - if (cloud9.header_len + 2 + cloud9.qid_len * @as(usize, m.nwname) > c.server.msize) return error.ReplyTooLarge; - var cur = f.node; - var cur_is_dir = f.is_dir; - var held = false; // cur is a provider handle obtained here, not the fid's - var reply: cloud9.Msg = .{ .rwalk = .{ .nwqid = 0 } }; - const names = m.wname[0..m.nwname]; - for (names, 0..) |name, i| { - if (!cur_is_dir) { - if (i == 0) return error.NotDir; - break; - } - const next = c.lookup(cur, name) catch |e| { - if (i == 0) return e; - break; - }; - if (held) c.releaseRef(cur); - cur = next.ref; - cur_is_dir = next.info.is_dir; - held = cur == .prov; - reply.rwalk.wqid[i] = next.info.qid(); - reply.rwalk.nwqid += 1; - } - if (reply.rwalk.nwqid != names.len) { - if (held) c.releaseRef(cur); - return reply; - } - if (names.len == 0 and cur == .prov) { - // A clone of a provider handle needs its own reference. - const dup = try c.lookup(cur, "."); - cur = dup.ref; - cur_is_dir = dup.info.is_dir; - held = true; - } - const target = if (m.newfid == m.fid) f else c.allocFid(m.newfid) catch |e| { - if (held) c.releaseRef(cur); - return e; - }; - if (target == f) c.dropContents(f); - target.node = cur; - target.is_dir = cur_is_dir; - return reply; - } - - fn open(c: *Conn, m: anytype) !cloud9.Msg { - const f = c.findFid(m.fid) orelse return error.UnknownFid; - if (f.open) return error.AlreadyOpen; - const acc = m.mode & 3; - const want_write = acc == cloud9.owrite or acc == cloud9.ordwr; - const trunc = m.mode & cloud9.otrunc != 0; - if (f.is_dir and (want_write or trunc)) return error.IsDir; - switch (f.node) { - .static => |idx| switch (flat[idx].node.kind) { - .dir, .vars, .ctl => {}, - .static, .dynamic => if (want_write or trunc) return error.Perm, - }, - .@"var" => |v| { - const n = c.shared.vars[v.idx].vt.nodes[v.node]; - if ((want_write or trunc) and !n.writable()) return error.Perm; - }, - .prov => |p| { - const pr = c.provider(p.idx); - try pr.vtable.open(pr.ctx, p.h, m.mode); - }, - } - const qid = (c.info(f.node) catch |e| { - // The provider's open succeeded but its stat did not: undo the open. - if (f.node == .prov) { - const pr = c.provider(f.node.prov.idx); - if (pr.vtable.close) |close| close(pr.ctx, f.node.prov.h); - } - return e; - }).qid(); - if (c.isDynamic(f.node)) { - f.snap = try c.takeSlot(); - c.generate(f) catch |e| { - c.slot_used[f.snap.?] = false; - f.snap = null; - if (f.node == .prov) unreachable; - return e; - }; - } - f.open = true; - f.mode = m.mode; - f.rclose = m.mode & cloud9.orclose != 0; - f.dir_offset = 0; - f.dir_index = 0; - return .{ .ropen = .{ .qid = qid, .iounit = 0 } }; - } - - fn create(c: *Conn, m: anytype) !cloud9.Msg { - const f = c.findFid(m.fid) orelse return error.UnknownFid; - if (f.open) return error.AlreadyOpen; - const p = switch (f.node) { - .prov => |p| p, - else => return error.Perm, - }; - if (!f.is_dir) return error.NotDir; - const pr = c.provider(p.idx); - const create_fn = pr.vtable.create orelse return error.Perm; - try validName(m.name); - const is_dir = m.perm & cloud9.dmdir != 0; - const acc = m.mode & 3; - if (is_dir and (acc != cloud9.oread or m.mode & cloud9.otrunc != 0)) return error.IsDir; - const h = try create_fn(pr.ctx, p.h, m.name, m.perm, m.mode); - const node: NodeRef = .{ .prov = .{ .idx = p.idx, .h = h } }; - const qid = (c.info(node) catch |e| { - if (pr.vtable.close) |close| close(pr.ctx, h); - pr.vtable.clunk(pr.ctx, h); - return e; - }).qid(); - c.dropContents(f); - f.node = node; - f.is_dir = is_dir; - f.open = true; - f.mode = m.mode; - f.rclose = m.mode & cloud9.orclose != 0; - f.dir_offset = 0; - f.dir_index = 0; - return .{ .rcreate = .{ .qid = qid, .iounit = 0 } }; - } - - fn read(c: *Conn, m: anytype) !cloud9.Msg { - const f = c.findFid(m.fid) orelse return error.UnknownFid; - if (!f.open or (f.mode & 3) == cloud9.owrite) return error.NotOpen; - const count: usize = @min(m.count, c.server.msize -| cloud9.iohdrsz, c.storage.data.len); - if (f.is_dir) return c.readDir(f, m.offset, count); - const data = &c.storage.data; - const src: []const u8 = switch (f.node) { - .static => |idx| blk: { - const n = flat[idx].node; - switch (n.kind) { - .static => break :blk n.content, - .ctl => break :blk c.shared.ctlResult(), - .dynamic => { - if (m.offset == 0) try c.generate(f); - break :blk c.storage.snapshots[f.snap.?][0..c.slot_len[f.snap.?]]; - }, - .dir, .vars => unreachable, - } - }, - .@"var" => |v| blk: { - const n = c.shared.vars[v.idx].vt.nodes[v.node]; - switch (n.kind) { - .type_name, .size => break :blk n.content, - .value, .addr => { - if (m.offset == 0) try c.generate(f); - break :blk c.storage.snapshots[f.snap.?][0..c.slot_len[f.snap.?]]; - }, - .raw => break :blk c.varBase(v.idx, v.node)[0..n.size], - .dir, .fields => unreachable, - } - }, - .prov => |p| { - const pr = c.provider(p.idx); - const n = try pr.vtable.read(pr.ctx, p.h, m.offset, data[0..count]); - return .{ .rread = .{ .data = data[0..@min(n, count)] } }; - }, - }; - if (m.offset >= src.len) return .{ .rread = .{ .data = "" } }; - const off: usize = @intCast(m.offset); - const n = @min(count, src.len - off); - if (f.node == .@"var" and c.shared.vars[f.node.@"var".idx].vt.nodes[f.node.@"var".node].kind == .raw) { - // Copy out of the variable so the reply does not read live memory twice. - @memcpy(data[0..n], src[off..][0..n]); - return .{ .rread = .{ .data = data[0..n] } }; - } - return .{ .rread = .{ .data = src[off..][0..n] } }; - } - - fn readDir(c: *Conn, f: *Fid, offset: u64, count: usize) !cloud9.Msg { - if (offset == 0) { - f.dir_offset = 0; - f.dir_index = 0; - } else if (offset != f.dir_offset) return error.BadOffset; - const data = &c.storage.data; - var used: usize = 0; - var i = f.dir_index; - while (try c.entryStat(f.node, i)) |st| : (i += 1) { - const rec = st.encode(data[used..count]) catch |e| switch (e) { - error.NoSpace => break, - else => return error.Io, - }; - used += rec.len; - } - f.dir_offset += used; - f.dir_index = i; - return .{ .rread = .{ .data = data[0..used] } }; - } - - fn write(c: *Conn, m: anytype) !cloud9.Msg { - const f = c.findFid(m.fid) orelse return error.UnknownFid; - const acc = f.mode & 3; - if (!f.open or (acc != cloud9.owrite and acc != cloud9.ordwr)) return error.NotOpen; - if (f.is_dir) return error.IsDir; - switch (f.node) { - .static => |idx| switch (flat[idx].node.kind) { - .ctl => try c.ctlCommand(m.data), - else => return error.Perm, - }, - .@"var" => |v| { - const n = c.shared.vars[v.idx].vt.nodes[v.node]; - const set = n.set orelse return error.Perm; - try set(c.varBase(v.idx, v.node), m.data); - }, - .prov => |p| { - const pr = c.provider(p.idx); - const n = try pr.vtable.write(pr.ctx, p.h, m.offset, m.data); - return .{ .rwrite = .{ .count = @intCast(@min(n, m.data.len)) } }; - }, - } - return .{ .rwrite = .{ .count = @intCast(m.data.len) } }; - } - - /// Runs `cfg.ctl`; on success its output becomes the ctl result. - /// On failure the previous result (and its qid version) survive: - /// the handler writes into the staging half of `ctl_bufs`. - fn ctlCommand(c: *Conn, line: []const u8) !void { - const s = c.shared; - const next = s.ctl_cur ^ 1; - var w: Writer = .fixed(&s.ctl_bufs[next]); - try cfg.ctl.?(s.ctx, line, &w); - s.ctl_cur = next; - s.ctl_len = @intCast(w.buffered().len); - s.ctl_version +%= 1; - } - - fn clunk(c: *Conn, m: anytype) !cloud9.Msg { - const f = c.findFid(m.fid) orelse return error.UnknownFid; - c.freeFid(f); - return .rclunk; - } - - fn remove(c: *Conn, m: anytype) !cloud9.Msg { - const f = c.findFid(m.fid) orelse return error.UnknownFid; - defer c.freeFid(f); // Tremove always clunks - f.rclose = false; - switch (f.node) { - .prov => |p| { - const pr = c.provider(p.idx); - const rm = pr.vtable.remove orelse return error.Perm; - try rm(pr.ctx, p.h); - }, - else => return error.Perm, - } - return .rremove; - } - - fn stat(c: *Conn, m: anytype) !cloud9.Msg { - const f = c.findFid(m.fid) orelse return error.UnknownFid; - return .{ .rstat = .{ .stat = c.pinName((try c.info(f.node)).stat()) } }; - } - - fn wstat(c: *Conn, m: anytype) !cloud9.Msg { - const f = c.findFid(m.fid) orelse return error.UnknownFid; - const p = switch (f.node) { - .prov => |p| p, - else => return error.Perm, - }; - const pr = c.provider(p.idx); - const ws = pr.vtable.wstat orelse return error.Perm; - const cur = try c.info(f.node); - const st = m.stat; - const q = cur.qid(); - // Fields we cannot change must be "don't care" or unchanged. - if (st.type != 0xFFFF and st.type != 0) return error.Perm; - if (st.dev != 0xFFFF_FFFF and st.dev != 0) return error.Perm; - if (st.qid.type != 0xFF and st.qid.type != q.type) return error.Perm; - if (st.qid.version != 0xFFFF_FFFF and st.qid.version != q.version) return error.Perm; - if (st.qid.path != 0xFFFF_FFFF_FFFF_FFFF and st.qid.path != q.path) return error.Perm; - if (st.uid.len != 0 and !std.mem.eql(u8, st.uid, cfg.name)) return error.Perm; - if (st.gid.len != 0 and !std.mem.eql(u8, st.gid, cfg.name)) return error.Perm; - if (st.muid.len != 0 and !std.mem.eql(u8, st.muid, cfg.name)) return error.Perm; - if (st.name.len != 0 and !std.mem.eql(u8, st.name, cur.name)) { - if (p.h == Provider.root) return error.Perm; - try validName(st.name); - } - if (st.length != 0xFFFF_FFFF_FFFF_FFFF and st.length != cur.length and cur.is_dir) return error.IsDir; - if (st.mode != 0xFFFF_FFFF and (st.mode & cloud9.dmdir) != (cur.mode & cloud9.dmdir)) return error.Perm; - try ws(pr.ctx, p.h, &st); - return .rwstat; - } - }; - - // -- in-memory test harness ------------------------------------------- - - /// Drives a `Conn` with a `cloud9.Client` in memory. Test-only (uses - /// std.testing.allocator); never referenced by non-test code. - pub const Harness = struct { - shared: *Shared, - storage: *Storage, - conn: Conn, - client: cloud9.Client, - cin: []u8, - cout: []u8, - - pub fn init(h: *Harness, shared: *Shared, storage: *Storage) !void { - h.shared = shared; - h.storage = storage; - h.conn = .init(shared, storage, cfg.msize); - h.cin = try testing.allocator.alloc(u8, cfg.msize); - errdefer testing.allocator.free(h.cin); - h.cout = try testing.allocator.alloc(u8, cfg.msize); - errdefer testing.allocator.free(h.cout); - h.client = .init(.{ .in = h.cin, .out = h.cout }); - try h.version(cfg.msize); - _ = try h.ok(.{ .attach = .{ .fid = 0, .uname = "tester" } }); - } - - pub fn deinit(h: *Harness) void { - h.conn.hangup(); - testing.allocator.free(h.cin); - testing.allocator.free(h.cout); - } - - pub fn version(h: *Harness, msize: u32) !void { - const v = try h.rpc(.{ .version = .{ .msize = msize } }); - try testing.expectEqual(msize, v.version.msize); - try testing.expectEqualStrings("9P2000", v.version.version); - } - - /// One round trip; the result borrows the client input buffer until the next call. - pub fn rpc(h: *Harness, req: cloud9.Client.Request) !cloud9.Client.Result { - _ = try h.client.submit(req); - while (true) { - var moved = false; - while (h.client.output().len > 0) { - const k = h.conn.push(h.client.output()); - h.client.wrote(k); - moved = moved or k > 0; - while (try h.conn.step()) {} - while (h.conn.output().len > 0) { - const n = h.client.push(h.conn.output()); - h.conn.wrote(n); - moved = moved or n > 0; - } - if (k == 0) break; - } - while (try h.conn.step()) {} - while (h.conn.output().len > 0) { - const n = h.client.push(h.conn.output()); - h.conn.wrote(n); - moved = moved or n > 0; - } - if (h.client.take()) |done| return done.result; - if (!moved) return error.Stuck; - } - } - - pub fn ok(h: *Harness, req: cloud9.Client.Request) !cloud9.Client.Result { - const r = try h.rpc(req); - if (r == .fail) { - std.debug.print("unexpected Rerror: {s}\n", .{r.fail}); - return error.Rerror; - } - return r; - } - - pub fn expectFail(h: *Harness, req: cloud9.Client.Request, msg: []const u8) !void { - const r = try h.rpc(req); - if (r != .fail) return error.ExpectedRerror; - try testing.expectEqualStrings(msg, r.fail); - } - - pub fn walkTo(h: *Harness, newfid: u32, names: []const []const u8) !void { - const r = try h.ok(.{ .walk = .{ .fid = 0, .newfid = newfid, .names = names } }); - try testing.expectEqual(@as(u16, @intCast(names.len)), r.walk.nwqid); - } - - /// Opens `fid` for reading and reads it whole (across consecutive offsets); caller frees. - pub fn readAll(h: *Harness, fid: u32) ![]u8 { - _ = try h.ok(.{ .open = .{ .fid = fid, .mode = cloud9.oread } }); - return h.readOpen(fid); - } - - pub fn readOpen(h: *Harness, fid: u32) ![]u8 { - var acc: std.ArrayList(u8) = .empty; - errdefer acc.deinit(testing.allocator); - while (true) { - const r = try h.ok(.{ .read = .{ .fid = fid, .offset = acc.items.len, .count = 1024 } }); - if (r.read.len == 0) break; - try acc.appendSlice(testing.allocator, r.read); - } - return acc.toOwnedSlice(testing.allocator); - } - - pub fn readPath(h: *Harness, names: []const []const u8) ![]u8 { - try h.walkTo(99, names); - defer _ = h.rpc(.{ .clunk = .{ .fid = 99 } }) catch {}; - return h.readAll(99); - } - - pub fn writePath(h: *Harness, names: []const []const u8, data: []const u8) !void { - try h.walkTo(98, names); - defer _ = h.rpc(.{ .clunk = .{ .fid = 98 } }) catch {}; - _ = try h.ok(.{ .open = .{ .fid = 98, .mode = cloud9.owrite } }); - const w = try h.ok(.{ .write = .{ .fid = 98, .offset = 0, .data = data } }); - try testing.expectEqual(@as(u32, @intCast(data.len)), w.write); - } - - /// Reads a whole directory in `count`-byte reads at consecutive offsets; returns owned names. - pub fn listDir(h: *Harness, fid: u32, count: u32) ![][]u8 { - var names: std.ArrayList([]u8) = .empty; - errdefer { - for (names.items) |n| testing.allocator.free(n); - names.deinit(testing.allocator); - } - var offset: u64 = 0; - while (true) { - const r = try h.ok(.{ .read = .{ .fid = fid, .offset = offset, .count = count } }); - if (r.read.len == 0) break; - offset += r.read.len; - var rest = r.read; - while (rest.len > 0) { - const n = std.mem.readInt(u16, rest[0..2], .little) + 2; - const st = try cloud9.Stat.decode(rest[0..n]); - try names.append(testing.allocator, try testing.allocator.dupe(u8, st.name)); - rest = rest[n..]; - } - } - return names.toOwnedSlice(testing.allocator); - } - - pub fn listPath(h: *Harness, names: []const []const u8) ![][]u8 { - try h.walkTo(97, names); - defer _ = h.rpc(.{ .clunk = .{ .fid = 97 } }) catch {}; - _ = try h.ok(.{ .open = .{ .fid = 97, .mode = cloud9.oread } }); - return h.listDir(97, 1024); - } - - pub fn freeNames(names: [][]u8) void { - for (names) |n| testing.allocator.free(n); - testing.allocator.free(names); - } - - pub fn hasName(names: []const []const u8, want: []const u8) bool { - for (names) |n| if (std.mem.eql(u8, n, want)) return true; - return false; - } - }; - }; -} - -// --------------------------------------------------------------------------- -// Tests -// --------------------------------------------------------------------------- - -const testing = std.testing; - -const TestBuild = struct { - pub const zig_version: []const u8 = builtin.zig_version_string; - pub const target: []const u8 = "test-target"; - pub const optimize: []const u8 = "Debug"; - pub const time: []const u8 = "2023-11-14T22:13:20Z"; - pub const change: []const u8 = "abc123"; -}; - -const Layout = struct { a: u8, b: u32, c: u64 }; -const Decls = struct { - pub const one = 1; - pub const two = 2; - pub fn three() void {} -}; - -/// The context every generator and the ctl handler receive in tests. -const TestCtx = struct { - calls: u32 = 0, - ctl_state: i64 = 0, -}; - -const TestFns = struct { - pub fn counter(ctx: *anyopaque, w: *Writer) anyerror!void { - const t: *TestCtx = @ptrCast(@alignCast(ctx)); - t.calls += 1; - try w.print("{d}", .{t.calls}); - } - pub fn fib30(_: *anyopaque, w: *Writer) anyerror!void { - try w.print("{d}", .{fib(30)}); - } - pub fn failing(_: *anyopaque, _: *Writer) anyerror!void { - return error.BadCommand; - } - pub fn huge(_: *anyopaque, w: *Writer) anyerror!void { - try w.splatByteAll('x', 1 << 20); - } -}; - -const TestRuntime = struct { - pub fn pid(_: *anyopaque, w: *Writer) anyerror!void { - try w.writeAll("4242"); - } -}; - -fn fib(n: u32) u64 { - if (n == 0) return 0; - var a: u64 = 0; - var b: u64 = 1; - for (1..n) |_| { - const c = a + b; - a = b; - b = c; - } - return b; -} - -fn testCtl(ctx: *anyopaque, cmd: []const u8, out: *Writer) anyerror!void { - const t: *TestCtx = @ptrCast(@alignCast(ctx)); - const line = std.mem.trim(u8, cmd, " \t\r\n\x00"); - var it = std.mem.tokenizeScalar(u8, line, ' '); - const verb = it.next() orelse return error.BadCommand; - if (std.mem.eql(u8, verb, "echo")) { - try out.writeAll(std.mem.trimStart(u8, line[verb.len..], " \t")); - } else if (std.mem.eql(u8, verb, "add")) { - const a = std.fmt.parseInt(i64, it.next() orelse return error.BadCommand, 10) catch return error.BadCommand; - const b = std.fmt.parseInt(i64, it.next() orelse return error.BadCommand, 10) catch return error.BadCommand; - t.ctl_state = a +% b; - try out.print("{d}", .{t.ctl_state}); - } else if (std.mem.eql(u8, verb, "partial")) { - try out.writeAll("half-written"); - return error.BadCommand; - } else return error.BadCommand; -} - -const test_cfg: Config = .{ - .name = "tester", - .build = TestBuild, - .types = &.{ Layout, cloud9.Qid }, - .decls_of = Decls, - .fns = TestFns, - .runtime = TestRuntime, - .ctl = &testCtl, - .msize = 8192, - .max_fids = 8, - .max_providers = 2, - .max_vars = 4, - .snapshot_slots = 2, - .snapshot_bytes = 512, -}; - -const TS = Server(test_cfg); - -/// A small in-memory provider: /prov/{hello,dir/{inner}} with create/remove/wstat, -/// counting every handle reference so tests can check clunk discipline. -const TestProv = struct { - const max_nodes = 16; - const Entry = struct { - used: bool = false, - name: [max_name]u8 = undefined, - name_len: u8 = 0, - parent: u32 = 0, - is_dir: bool = false, - mode: u32 = 0o644, - data: [64]u8 = undefined, - len: usize = 0, - refs: u32 = 0, - opens: u32 = 0, - mtime: u32 = 0, - - fn nameSlice(e: *const Entry) []const u8 { - return e.name[0..e.name_len]; - } - }; - nodes: [max_nodes]Entry = @splat(.{}), - total_refs: u32 = 0, - clunks: u32 = 0, - fail_io: bool = false, - fail_stat: bool = false, - - fn init() TestProv { - var p: TestProv = .{}; - p.nodes[0] = .{ .used = true, .is_dir = true, .mode = cloud9.dmdir | 0o755 }; - _ = p.add(0, "hello", false, 0o644); - p.nodes[1].len = 5; - @memcpy(p.nodes[1].data[0..5], "hello"); - const d = p.add(0, "dir", true, cloud9.dmdir | 0o755); - _ = p.add(d, "inner", false, 0o600); - _ = p.add(0, "locked", false, 0o000); - return p; - } - - fn add(p: *TestProv, parent: u32, name: []const u8, is_dir: bool, mode: u32) u32 { - for (&p.nodes, 0..) |*e, i| if (!e.used) { - e.* = .{ .used = true, .parent = parent, .is_dir = is_dir, .mode = mode }; - @memcpy(e.name[0..name.len], name); - e.name_len = @intCast(name.len); - return @intCast(i); - }; - unreachable; - } - - fn self(ctx: *anyopaque) *TestProv { - return @ptrCast(@alignCast(ctx)); - } - - fn node(p: *TestProv, h: Provider.Handle) Provider.Error!*Entry { - if (h >= max_nodes or !p.nodes[h].used) return error.NotFound; - return &p.nodes[h]; - } - - fn retain(p: *TestProv, h: Provider.Handle) Provider.Handle { - if (h != 0) { - p.nodes[h].refs += 1; - p.total_refs += 1; - } - return h; - } - - fn walk(ctx: *anyopaque, parent: Provider.Handle, name: []const u8) Provider.Error!Provider.Handle { - const p = self(ctx); - if (p.fail_io) return error.Io; - const d = try p.node(parent); - if (std.mem.eql(u8, name, ".")) return p.retain(parent); - if (!d.is_dir) return error.NotDir; - if (std.mem.eql(u8, name, "..")) return p.retain(d.parent); - for (p.nodes[0..], 0..) |*e, i| { - if (e.used and e.parent == parent and i != 0 and std.mem.eql(u8, e.nameSlice(), name)) return p.retain(@intCast(i)); - } - return error.NotFound; - } - - fn fillStat(e: *const Entry, h: Provider.Handle, out: *NodeStat) void { - out.* = .{ .mode = e.mode, .length = e.len, .mtime = e.mtime, .name = e.nameSlice(), .handle = h }; - } - - fn stat(ctx: *anyopaque, h: Provider.Handle, out: *NodeStat) Provider.Error!void { - const p = self(ctx); - if (p.fail_stat) return error.Io; - fillStat(try p.node(h), h, out); - } - - fn list(ctx: *anyopaque, dir: Provider.Handle, index: usize, out: *NodeStat) Provider.Error!bool { - const p = self(ctx); - const d = try p.node(dir); - if (!d.is_dir) return error.NotDir; - var k: usize = 0; - for (p.nodes[0..], 0..) |*e, i| { - if (!e.used or e.parent != dir or i == 0) continue; - if (k == index) { - fillStat(e, @intCast(i), out); - return true; - } - k += 1; - } - return false; - } - - fn open(ctx: *anyopaque, h: Provider.Handle, mode: u8) Provider.Error!void { - const p = self(ctx); - const e = try p.node(h); - const acc = mode & 3; - if (acc != cloud9.owrite and e.mode & 0o400 == 0) return error.Perm; - if (acc != cloud9.oread and e.mode & 0o200 == 0) return error.Perm; - if (mode & cloud9.otrunc != 0) e.len = 0; - e.opens += 1; - } - - fn close(ctx: *anyopaque, h: Provider.Handle) void { - const p = self(ctx); - p.nodes[h].opens -= 1; - } - - fn read(ctx: *anyopaque, h: Provider.Handle, offset: u64, buf: []u8) Provider.Error!usize { - const p = self(ctx); - const e = try p.node(h); - if (offset >= e.len) return 0; - const n = @min(buf.len, e.len - @as(usize, @intCast(offset))); - @memcpy(buf[0..n], e.data[@intCast(offset)..][0..n]); - return n; - } - - fn write(ctx: *anyopaque, h: Provider.Handle, offset: u64, data: []const u8) Provider.Error!usize { - const p = self(ctx); - const e = try p.node(h); - if (offset + data.len > e.data.len) return error.NoSpace; - const off: usize = @intCast(offset); - @memcpy(e.data[off..][0..data.len], data); - e.len = @max(e.len, off + data.len); - e.mtime += 1; - return data.len; - } - - fn create(ctx: *anyopaque, dir: Provider.Handle, name: []const u8, perm: u32, mode: u8) Provider.Error!Provider.Handle { - const p = self(ctx); - const d = try p.node(dir); - if (!d.is_dir) return error.NotDir; - for (p.nodes[0..]) |*e| if (e.used and e.parent == dir and std.mem.eql(u8, e.nameSlice(), name)) return error.Exists; - var free: ?u32 = null; - for (p.nodes[0..], 0..) |*e, i| if (!e.used) { - free = @intCast(i); - break; - }; - const idx = free orelse return error.NoSpace; - const h = p.add(@intCast(dir), name, perm & cloud9.dmdir != 0, perm); - std.debug.assert(h == idx); - p.nodes[h].opens = 1; - _ = mode; - return p.retain(h); - } - - fn remove(ctx: *anyopaque, h: Provider.Handle) Provider.Error!void { - const p = self(ctx); - const e = try p.node(h); - if (h == 0) return error.Perm; - for (p.nodes[0..]) |*c| if (c.used and c.parent == h) return error.NotEmpty; - e.used = false; // refs still keep the slot "alive" for clunk accounting - e.used = true; - e.parent = std.math.maxInt(u32); // unlinked - } - - fn wstat(ctx: *anyopaque, h: Provider.Handle, st: *const cloud9.Stat) Provider.Error!void { - const p = self(ctx); - const e = try p.node(h); - if (st.name.len != 0) { - @memcpy(e.name[0..st.name.len], st.name); - e.name_len = @intCast(st.name.len); - } - if (st.length != 0xFFFF_FFFF_FFFF_FFFF) { - if (st.length > e.data.len) return error.NoSpace; - e.len = @intCast(st.length); - } - if (st.mode != 0xFFFF_FFFF) e.mode = st.mode; - if (st.mtime != 0xFFFF_FFFF) e.mtime = st.mtime; - } - - fn clunk(ctx: *anyopaque, h: Provider.Handle) void { - const p = self(ctx); - p.clunks += 1; - if (h != 0) { - p.nodes[h].refs -= 1; - p.total_refs -= 1; - } - } - - const vtable: Provider.VTable = .{ - .walk = &walk, - .stat = &stat, - .list = &list, - .open = &open, - .read = &read, - .write = &write, - .create = &create, - .remove = &remove, - .wstat = &wstat, - .close = &close, - .clunk = &clunk, - }; - - fn provider(p: *TestProv) Provider { - return .{ .name = "prov", .ctx = p, .vtable = &vtable }; - } -}; - -const Inner = struct { x: f32 }; -const Exposed = struct { a: u32, b: bool, name: []const u8, inner: Inner }; - -/// Everything a core test needs, in one place; `harness.init` runs version+attach. -const Fixture = struct { - ctx: TestCtx = .{}, - shared: TS.Shared = undefined, - storage: TS.Storage = undefined, - prov: TestProv = undefined, - exposed: Exposed = .{ .a = 1, .b = true, .name = "hello", .inner = .{ .x = 0.5 } }, - counter: u64 = 7, - h: TS.Harness = undefined, - - fn init(x: *Fixture) !void { - x.shared = .init(&x.ctx); - x.prov = TestProv.init(); - try x.shared.addProvider(x.prov.provider()); - try x.shared.expose("state", &x.exposed); - try x.shared.expose("counter", &x.counter); - try x.h.init(&x.shared, &x.storage); - } - - fn deinit(x: *Fixture) void { - x.h.deinit(); - } -}; - -test "README, /build and the static tree read as expected" { - var x: Fixture = .{}; - try x.init(); - defer x.deinit(); - const readme = try x.h.readPath(&.{"README"}); - defer testing.allocator.free(readme); - try testing.expect(std.mem.startsWith(u8, readme, "tester: a 9P2000 introspection server")); - const zv = try x.h.readPath(&.{ "build", "zig_version" }); - defer testing.allocator.free(zv); - try testing.expectEqualStrings(builtin.zig_version_string, zv); - const ch = try x.h.readPath(&.{ "build", "change" }); - defer testing.allocator.free(ch); - try testing.expectEqualStrings("abc123", ch); - try testing.expectEqual(@as(u32, 1_700_000_000), TS.build_secs); - const names = try x.h.listPath(&.{}); - defer TS.Harness.freeNames(names); - for ([_][]const u8{ "README", "build", "comptime", "runtime", "ctl", "vars", "prov" }) |n| try testing.expect(TS.Harness.hasName(names, n)); - try testing.expectEqual(@as(usize, 7), names.len); - // static files are read-only; the static tree admits no creates or removes - try x.h.walkTo(1, &.{ "build", "target" }); - try x.h.expectFail(.{ .open = .{ .fid = 1, .mode = cloud9.owrite } }, "permission denied"); - try x.h.expectFail(.{ .remove = .{ .fid = 1 } }, "permission denied"); - try x.h.walkTo(2, &.{"build"}); - try x.h.expectFail(.{ .create = .{ .fid = 2, .name = "nope", .perm = 0o644, .mode = cloud9.owrite } }, "permission denied"); - try x.h.expectFail(.{ .wstat = .{ .fid = 2, .stat = stat_dontcare } }, "permission denied"); - const st = try x.h.ok(.{ .stat = .{ .fid = 2 } }); - try testing.expectEqualStrings("build", st.stat.name); - try testing.expectEqualStrings("tester", st.stat.uid); - try testing.expect(st.stat.qid.type & cloud9.qtdir != 0); - try testing.expectEqual(TS.build_secs, st.stat.mtime); - try testing.expectEqual(@as(u32, 0), parseIso8601("1970-01-01T00:00:00Z").?); - try testing.expectEqual(@as(?u32, null), parseIso8601("unknown")); -} - -test "comptime/types fields carry @offsetOf and comptime/decls lists pub decls" { - var x: Fixture = .{}; - try x.init(); - defer x.deinit(); - const names = try x.h.listPath(&.{ "comptime", "types" }); - defer TS.Harness.freeNames(names); - try testing.expectEqual(@as(usize, 2), names.len); - try testing.expect(TS.Harness.hasName(names, "Layout")); - try testing.expect(TS.Harness.hasName(names, "Qid")); - const fields = try x.h.readPath(&.{ "comptime", "types", "Layout", "fields" }); - defer testing.allocator.free(fields); - var expect_buf: [128]u8 = undefined; - const expect = try std.fmt.bufPrint(&expect_buf, "a: u8 @{d}\nb: u32 @{d}\nc: u64 @{d}\n", .{ @offsetOf(Layout, "a"), @offsetOf(Layout, "b"), @offsetOf(Layout, "c") }); - try testing.expectEqualStrings(expect, fields); - const size = try x.h.readPath(&.{ "comptime", "types", "Layout", "size" }); - defer testing.allocator.free(size); - try testing.expectEqualStrings(std.fmt.comptimePrint("{d}", .{@sizeOf(Layout)}), size); - const name = try x.h.readPath(&.{ "comptime", "types", "Qid", "name" }); - defer testing.allocator.free(name); - try testing.expectEqualStrings(@typeName(cloud9.Qid), name); - const decls = try x.h.readPath(&.{ "comptime", "decls" }); - defer testing.allocator.free(decls); - try testing.expectEqualStrings("one\ntwo\nthree\n", decls); -} - -test "runtime/fn calls the function at open and at each read from offset 0" { - var x: Fixture = .{}; - try x.init(); - defer x.deinit(); - const names = try x.h.listPath(&.{ "runtime", "fn" }); - defer TS.Harness.freeNames(names); - try testing.expectEqual(@typeInfo(TestFns).@"struct".decls.len, names.len); - const fib_text = try x.h.readPath(&.{ "runtime", "fn", "fib30" }); - defer testing.allocator.free(fib_text); - try testing.expectEqualStrings("832040", fib_text); - const pid = try x.h.readPath(&.{ "runtime", "pid" }); - defer testing.allocator.free(pid); - try testing.expectEqualStrings("4242", pid); - // the generator runs at open, then again at each read from offset 0, not at offset > 0 - try x.h.walkTo(1, &.{ "runtime", "fn", "counter" }); - _ = try x.h.ok(.{ .open = .{ .fid = 1, .mode = cloud9.oread } }); - try testing.expectEqual(@as(u32, 1), x.ctx.calls); - const r1 = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = 100 } }); - try testing.expectEqualStrings("2", r1.read); - const r2 = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 1, .count = 100 } }); - try testing.expectEqualStrings("", r2.read); - try testing.expectEqual(@as(u32, 2), x.ctx.calls); - // stat of a dynamic file reports length 0 - const st = try x.h.ok(.{ .stat = .{ .fid = 1 } }); - try testing.expectEqual(@as(u64, 0), st.stat.length); - try testing.expectEqual(@as(u32, 0o444), st.stat.mode); - // a generator error is the file's Rerror; a generator that overflows the slot too - try x.h.walkTo(2, &.{ "runtime", "fn", "failing" }); - try x.h.expectFail(.{ .open = .{ .fid = 2, .mode = cloud9.oread } }, "bad command"); - try x.h.walkTo(3, &.{ "runtime", "fn", "huge" }); - try x.h.expectFail(.{ .open = .{ .fid = 3, .mode = cloud9.oread } }, "no space in buffer"); - // a failed open frees its slot: two more dynamic opens still succeed - try x.h.walkTo(4, &.{ "runtime", "fn", "fib30" }); - _ = try x.h.ok(.{ .open = .{ .fid = 4, .mode = cloud9.oread } }); - _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); - _ = try x.h.ok(.{ .walk = .{ .fid = 4, .newfid = 5, .names = &.{} } }); - _ = try x.h.ok(.{ .open = .{ .fid = 5, .mode = cloud9.oread } }); -} - -test "snapshot slot exhaustion is an Rerror and clunk frees the slot" { - var x: Fixture = .{}; - try x.init(); - defer x.deinit(); - try x.h.walkTo(1, &.{ "runtime", "fn", "fib30" }); - try x.h.walkTo(2, &.{ "runtime", "fn", "fib30" }); - try x.h.walkTo(3, &.{ "vars", "counter", "value" }); - _ = try x.h.ok(.{ .open = .{ .fid = 1, .mode = cloud9.oread } }); - _ = try x.h.ok(.{ .open = .{ .fid = 2, .mode = cloud9.oread } }); - try x.h.expectFail(.{ .open = .{ .fid = 3, .mode = cloud9.oread } }, "too many open dynamic files"); - // static and provider files need no slot - const t = try x.h.readPath(&.{ "vars", "counter", "type" }); - defer testing.allocator.free(t); - try testing.expectEqualStrings("u64", t); - _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); - _ = try x.h.ok(.{ .open = .{ .fid = 3, .mode = cloud9.oread } }); - const r = try x.h.ok(.{ .read = .{ .fid = 3, .offset = 0, .count = 100 } }); - try testing.expectEqualStrings("7", r.read); - // cloning an open fid onto itself keeps it open and its content - const w = try x.h.ok(.{ .walk = .{ .fid = 3, .newfid = 3, .names = &.{} } }); - try testing.expectEqual(@as(u16, 0), w.walk.nwqid); - const r2 = try x.h.ok(.{ .read = .{ .fid = 3, .offset = 0, .count = 100 } }); - try testing.expectEqualStrings("7", r2.read); - try x.h.expectFail(.{ .walk = .{ .fid = 3, .newfid = 4, .names = &.{".."} } }, "file already open"); - _ = try x.h.ok(.{ .walk = .{ .fid = 3, .newfid = 4, .names = &.{} } }); - try x.h.expectFail(.{ .read = .{ .fid = 4, .offset = 0, .count = 100 } }, "file not open"); -} - -test "ctl round trip" { - var x: Fixture = .{}; - try x.init(); - defer x.deinit(); - try x.h.walkTo(1, &.{"ctl"}); - _ = try x.h.ok(.{ .open = .{ .fid = 1, .mode = cloud9.ordwr } }); - const w = try x.h.ok(.{ .write = .{ .fid = 1, .offset = 0, .data = "add 2 3\n" } }); - try testing.expectEqual(@as(u32, 8), w.write); - const r = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = 100 } }); - try testing.expectEqualStrings("5", r.read); - try testing.expectEqual(@as(i64, 5), x.ctx.ctl_state); - const st = try x.h.ok(.{ .stat = .{ .fid = 1 } }); - try testing.expectEqual(@as(u64, 1), st.stat.length); - try testing.expectEqualStrings("ctl", st.stat.name); - try testing.expectEqual(@as(u32, 0o666), st.stat.mode); - const v1 = st.stat.qid.version; - _ = try x.h.ok(.{ .write = .{ .fid = 1, .offset = 0, .data = "echo hello world" } }); - const r2 = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = 100 } }); - try testing.expectEqualStrings("hello world", r2.read); - const r3 = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 6, .count = 100 } }); - try testing.expectEqualStrings("world", r3.read); - try testing.expect((try x.h.ok(.{ .stat = .{ .fid = 1 } })).stat.qid.version != v1); - const v2 = (try x.h.ok(.{ .stat = .{ .fid = 1 } })).stat.qid.version; - try x.h.expectFail(.{ .write = .{ .fid = 1, .offset = 0, .data = "frobnicate" } }, "bad command"); - // a failed command leaves the previous result, length and version in place - const r4 = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = 100 } }); - try testing.expectEqualStrings("hello world", r4.read); - const st4 = try x.h.ok(.{ .stat = .{ .fid = 1 } }); - try testing.expectEqual(@as(u64, 11), st4.stat.length); - try testing.expectEqual(v2, st4.stat.qid.version); - // even when the handler wrote part of a result before failing - try x.h.expectFail(.{ .write = .{ .fid = 1, .offset = 0, .data = "partial" } }, "bad command"); - const r5 = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = 100 } }); - try testing.expectEqualStrings("hello world", r5.read); - try testing.expectEqualStrings("hello world", x.shared.ctlResult()); - // the empty result is a legitimate result too - _ = try x.h.ok(.{ .write = .{ .fid = 1, .offset = 0, .data = "echo" } }); - try testing.expectEqualStrings("", (try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = 100 } })).read); - try testing.expectEqual(@as(u64, 0), (try x.h.ok(.{ .stat = .{ .fid = 1 } })).stat.length); - // Tversion resets the ctl fid like any other - try x.h.version(4096); - try x.h.expectFail(.{ .clunk = .{ .fid = 1 } }, "unknown fid"); -} - -test "auth is not required and flush is answered" { - var x: Fixture = .{}; - try x.init(); - defer x.deinit(); - try x.h.expectFail(.{ .auth = .{ .afid = 5, .uname = "tester" } }, "authentication not required"); - const f = try x.h.ok(.{ .flush = .{ .oldtag = 1 } }); - try testing.expect(f == .flush); - try x.h.expectFail(.{ .attach = .{ .fid = 0, .uname = "tester" } }, "fid in use"); -} - -test "vars: value/type/size/addr/raw, fields and writes" { - var x: Fixture = .{}; - try x.init(); - defer x.deinit(); - const names = try x.h.listPath(&.{"vars"}); - defer TS.Harness.freeNames(names); - try testing.expectEqual(@as(usize, 2), names.len); - try testing.expectEqualStrings("state", names[0]); - const entries = try x.h.listPath(&.{ "vars", "state" }); - defer TS.Harness.freeNames(entries); - for ([_][]const u8{ "value", "type", "size", "addr", "raw", "f" }) |n| try testing.expect(TS.Harness.hasName(entries, n)); - try testing.expectEqual(@as(usize, 6), entries.len); - const value = try x.h.readPath(&.{ "vars", "state", "value" }); - defer testing.allocator.free(value); - try testing.expectEqualStrings("a: 1\nb: true\nname: \"hello\"\ninner:\n x: 0.5\n", value); - const tn = try x.h.readPath(&.{ "vars", "state", "type" }); - defer testing.allocator.free(tn); - try testing.expectEqualStrings(@typeName(Exposed), tn); - const size = try x.h.readPath(&.{ "vars", "state", "size" }); - defer testing.allocator.free(size); - try testing.expectEqualStrings(std.fmt.comptimePrint("{d}", .{@sizeOf(Exposed)}), size); - const addr = try x.h.readPath(&.{ "vars", "state", "addr" }); - defer testing.allocator.free(addr); - var addr_buf: [32]u8 = undefined; - try testing.expectEqualStrings(try std.fmt.bufPrint(&addr_buf, "0x{x}", .{@intFromPtr(&x.exposed)}), addr); - const raw = try x.h.readPath(&.{ "vars", "state", "raw" }); - defer testing.allocator.free(raw); - try testing.expectEqualSlices(u8, std.mem.asBytes(&x.exposed), raw); - try x.h.walkTo(1, &.{ "vars", "state", "raw" }); - const raw_st = try x.h.ok(.{ .stat = .{ .fid = 1 } }); - try testing.expectEqual(@as(u64, @sizeOf(Exposed)), raw_st.stat.length); - try testing.expectEqual(@as(u32, 0o444), raw_st.stat.mode); - _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); - // fields - const fnames = try x.h.listPath(&.{ "vars", "state", "f" }); - defer TS.Harness.freeNames(fnames); - try testing.expectEqual(@as(usize, 4), fnames.len); - const a_value = try x.h.readPath(&.{ "vars", "state", "f", "a", "value" }); - defer testing.allocator.free(a_value); - try testing.expectEqualStrings("1", a_value); - const a_type = try x.h.readPath(&.{ "vars", "state", "f", "a", "type" }); - defer testing.allocator.free(a_type); - try testing.expectEqualStrings("u32", a_type); - const xv = try x.h.readPath(&.{ "vars", "state", "f", "inner", "f", "x", "value" }); - defer testing.allocator.free(xv); - try testing.expectEqualStrings("0.5", xv); - const b_raw = try x.h.readPath(&.{ "vars", "state", "f", "b", "raw" }); - defer testing.allocator.free(b_raw); - try testing.expectEqualSlices(u8, &.{1}, b_raw); - // writes - try x.h.writePath(&.{ "vars", "state", "f", "a", "value" }, "42"); - try testing.expectEqual(@as(u32, 42), x.exposed.a); - try x.h.writePath(&.{ "vars", "state", "f", "b", "value" }, "false\n"); - try testing.expect(!x.exposed.b); - try x.h.writePath(&.{ "vars", "state", "f", "inner", "f", "x", "value" }, "2.25"); - try testing.expectEqual(@as(f32, 2.25), x.exposed.inner.x); - try x.h.writePath(&.{ "vars", "counter", "value" }, "0x10"); - try testing.expectEqual(@as(u64, 16), x.counter); - try x.h.walkTo(2, &.{ "vars", "state", "f", "a", "value" }); - _ = try x.h.ok(.{ .open = .{ .fid = 2, .mode = cloud9.ordwr } }); - try x.h.expectFail(.{ .write = .{ .fid = 2, .offset = 0, .data = "abc" } }, "bad value"); - const rd = try x.h.ok(.{ .read = .{ .fid = 2, .offset = 0, .count = 100 } }); - try testing.expectEqualStrings("42", rd.read); - const a_st = try x.h.ok(.{ .stat = .{ .fid = 2 } }); - try testing.expectEqual(@as(u32, 0o644), a_st.stat.mode); - try testing.expectEqualStrings("value", a_st.stat.name); - // non-scalar values, type/size/addr/raw and directories are read-only - try x.h.walkTo(3, &.{ "vars", "state", "value" }); - try x.h.expectFail(.{ .open = .{ .fid = 3, .mode = cloud9.owrite } }, "permission denied"); - _ = try x.h.ok(.{ .clunk = .{ .fid = 3 } }); - try x.h.walkTo(4, &.{ "vars", "state", "f", "name", "value" }); - try x.h.expectFail(.{ .open = .{ .fid = 4, .mode = cloud9.owrite } }, "permission denied"); - _ = try x.h.ok(.{ .clunk = .{ .fid = 4 } }); - try x.h.walkTo(5, &.{ "vars", "state" }); - try x.h.expectFail(.{ .open = .{ .fid = 5, .mode = cloud9.owrite } }, "is a directory"); - try x.h.expectFail(.{ .create = .{ .fid = 5, .name = "z", .perm = 0o644, .mode = cloud9.owrite } }, "permission denied"); - // .. climbs back out of the var tree; unknown names fail - const up = try x.h.ok(.{ .walk = .{ .fid = 5, .newfid = 6, .names = &.{ "f", "inner", "..", "..", "..", "..", "README" } } }); - try testing.expectEqual(@as(u16, 7), up.walk.nwqid); - _ = try x.h.ok(.{ .clunk = .{ .fid = 6 } }); - try x.h.walkTo(7, &.{"vars"}); - try x.h.expectFail(.{ .walk = .{ .fid = 7, .newfid = 8, .names = &.{"nope"} } }, "file does not exist"); - try x.h.expectFail(.{ .walk = .{ .fid = 2, .newfid = 8, .names = &.{"x"} } }, "file already open"); - _ = try x.h.ok(.{ .clunk = .{ .fid = 2 } }); - try x.h.walkTo(2, &.{ "vars", "state", "f", "a", "value" }); - try x.h.expectFail(.{ .walk = .{ .fid = 2, .newfid = 8, .names = &.{"x"} } }, "not a directory"); -} - -test "provider: walk/list/stat/open/read/write/create/remove/wstat/clunk and error mapping" { - var x: Fixture = .{}; - try x.init(); - defer x.deinit(); - const names = try x.h.listPath(&.{"prov"}); - defer TS.Harness.freeNames(names); - try testing.expectEqual(@as(usize, 3), names.len); - try testing.expect(TS.Harness.hasName(names, "hello") and TS.Harness.hasName(names, "dir") and TS.Harness.hasName(names, "locked")); - const hello = try x.h.readPath(&.{ "prov", "hello" }); - defer testing.allocator.free(hello); - try testing.expectEqualStrings("hello", hello); - try testing.expectEqual(@as(u32, 0), x.prov.total_refs); // every temp handle was clunked - // stat and qid scheme - try x.h.walkTo(1, &.{ "prov", "dir", "inner" }); - const st = try x.h.ok(.{ .stat = .{ .fid = 1 } }); - try testing.expectEqualStrings("inner", st.stat.name); - try testing.expectEqual(@as(u32, 0o600), st.stat.mode); - try testing.expectEqual(@as(u64, 3), st.stat.qid.path); // provider 0, handle 3 - try testing.expectEqualStrings("tester", st.stat.gid); - try x.h.walkTo(2, &.{"prov"}); - const root_st = try x.h.ok(.{ .stat = .{ .fid = 2 } }); - try testing.expectEqualStrings("prov", root_st.stat.name); - try testing.expect(root_st.stat.qid.type & cloud9.qtdir != 0); - try testing.expectEqual(@as(u64, 0), root_st.stat.qid.path); - try testing.expectEqual(@as(u32, 1), x.prov.total_refs); // fid 1 holds inner; fid 2 holds root (unref'd) - // write then read back; opens are tracked through close - _ = try x.h.ok(.{ .open = .{ .fid = 1, .mode = cloud9.ordwr } }); - try testing.expectEqual(@as(u32, 1), x.prov.nodes[3].opens); - _ = try x.h.ok(.{ .write = .{ .fid = 1, .offset = 0, .data = "abc" } }); - _ = try x.h.ok(.{ .write = .{ .fid = 1, .offset = 3, .data = "def" } }); - const r = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 1, .count = 100 } }); - try testing.expectEqualStrings("bcdef", r.read); - try x.h.expectFail(.{ .write = .{ .fid = 1, .offset = 100, .data = "z" } }, "no space left on device"); - _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); - try testing.expectEqual(@as(u32, 0), x.prov.nodes[3].opens); - try testing.expectEqual(@as(u32, 0), x.prov.nodes[3].refs); - // permission and kind errors come from the provider - try x.h.walkTo(3, &.{ "prov", "locked" }); - try x.h.expectFail(.{ .open = .{ .fid = 3, .mode = cloud9.oread } }, "permission denied"); - try x.h.walkTo(4, &.{ "prov", "hello" }); - try x.h.expectFail(.{ .walk = .{ .fid = 4, .newfid = 5, .names = &.{"x"} } }, "not a directory"); - try x.h.expectFail(.{ .walk = .{ .fid = 2, .newfid = 5, .names = &.{"missing"} } }, "file does not exist"); - try x.h.expectFail(.{ .open = .{ .fid = 2, .mode = cloud9.owrite } }, "is a directory"); - // create in a provider directory: the fid becomes the new open file - const cr = try x.h.ok(.{ .create = .{ .fid = 2, .name = "new", .perm = 0o644, .mode = cloud9.ordwr } }); - try testing.expectEqual(cloud9.qtfile, cr.create.qid.type); - _ = try x.h.ok(.{ .write = .{ .fid = 2, .offset = 0, .data = "fresh" } }); - const rr = try x.h.ok(.{ .read = .{ .fid = 2, .offset = 0, .count = 100 } }); - try testing.expectEqualStrings("fresh", rr.read); - try x.h.walkTo(6, &.{"prov"}); - try x.h.expectFail(.{ .create = .{ .fid = 6, .name = "new", .perm = 0o644, .mode = cloud9.oread } }, "file already exists"); - const long_name = [_]u8{'n'} ** (max_name + 1); - try x.h.expectFail(.{ .create = .{ .fid = 6, .name = &long_name, .perm = 0o644, .mode = cloud9.oread } }, "bad file name"); - try x.h.expectFail(.{ .create = .{ .fid = 6, .name = "d", .perm = cloud9.dmdir | 0o755, .mode = cloud9.owrite } }, "is a directory"); - const dr = try x.h.ok(.{ .create = .{ .fid = 6, .name = "d", .perm = cloud9.dmdir | 0o755, .mode = cloud9.oread } }); - try testing.expectEqual(cloud9.qtdir, dr.create.qid.type); - // wstat: rename, truncate, mode, mtime; immutable fields are refused - var ws = stat_dontcare; - ws.name = "renamed"; - ws.length = 2; - ws.mode = 0o600; - ws.mtime = 99; - _ = try x.h.ok(.{ .wstat = .{ .fid = 2, .stat = ws } }); - const st2 = try x.h.ok(.{ .stat = .{ .fid = 2 } }); - try testing.expectEqualStrings("renamed", st2.stat.name); - try testing.expectEqual(@as(u64, 2), st2.stat.length); - try testing.expectEqual(@as(u32, 0o600), st2.stat.mode); - try testing.expectEqual(@as(u32, 99), st2.stat.mtime); - ws = stat_dontcare; - ws.uid = "someone-else"; - try x.h.expectFail(.{ .wstat = .{ .fid = 2, .stat = ws } }, "permission denied"); - ws = stat_dontcare; - ws.mode = cloud9.dmdir | 0o755; - try x.h.expectFail(.{ .wstat = .{ .fid = 2, .stat = ws } }, "permission denied"); - ws = stat_dontcare; - ws.name = "bad/name"; - try x.h.expectFail(.{ .wstat = .{ .fid = 2, .stat = ws } }, "bad file name"); - ws.name = ".."; - try x.h.expectFail(.{ .wstat = .{ .fid = 2, .stat = ws } }, "bad file name"); - ws = stat_dontcare; - ws.length = 5; - try x.h.walkTo(7, &.{ "prov", "d" }); - try x.h.expectFail(.{ .wstat = .{ .fid = 7, .stat = ws } }, "is a directory"); - ws = stat_dontcare; - ws.name = "root2"; // the provider root cannot be renamed - try x.h.walkTo(14, &.{"prov"}); - try x.h.expectFail(.{ .wstat = .{ .fid = 14, .stat = ws } }, "permission denied"); - _ = try x.h.ok(.{ .clunk = .{ .fid = 14 } }); - // remove always clunks; a non-empty directory refuses - _ = try x.h.ok(.{ .clunk = .{ .fid = 6 } }); - try x.h.walkTo(8, &.{ "prov", "dir" }); - try x.h.expectFail(.{ .remove = .{ .fid = 8 } }, "directory not empty"); - try x.h.expectFail(.{ .clunk = .{ .fid = 8 } }, "unknown fid"); - _ = try x.h.ok(.{ .remove = .{ .fid = 2 } }); - try x.h.walkTo(9, &.{"prov"}); - try x.h.expectFail(.{ .walk = .{ .fid = 9, .newfid = 15, .names = &.{"renamed"} } }, "file does not exist"); - _ = try x.h.ok(.{ .clunk = .{ .fid = 9 } }); - // an i/o error from the provider maps to "i/o error" - try x.h.walkTo(9, &.{"prov"}); - x.prov.fail_io = true; - try x.h.expectFail(.{ .walk = .{ .fid = 9, .newfid = 15, .names = &.{"hello"} } }, "i/o error"); - x.prov.fail_io = false; - _ = try x.h.ok(.{ .clunk = .{ .fid = 9 } }); - // ORCLOSE removes on clunk - try x.h.walkTo(9, &.{"prov"}); - _ = try x.h.ok(.{ .create = .{ .fid = 9, .name = "tmp", .perm = 0o644, .mode = cloud9.owrite | cloud9.orclose } }); - _ = try x.h.ok(.{ .clunk = .{ .fid = 9 } }); - try x.h.walkTo(9, &.{"prov"}); - try x.h.expectFail(.{ .walk = .{ .fid = 9, .newfid = 15, .names = &.{"tmp"} } }, "file does not exist"); - _ = try x.h.ok(.{ .clunk = .{ .fid = 9 } }); - // walking .. out of the provider root and cloning provider fids keeps refs balanced - const up = try x.h.ok(.{ .walk = .{ .fid = 0, .newfid = 10, .names = &.{ "prov", "dir", "..", "..", "build" } } }); - try testing.expectEqual(@as(u16, 5), up.walk.nwqid); - try x.h.walkTo(11, &.{ "prov", "dir", "inner" }); - _ = try x.h.ok(.{ .walk = .{ .fid = 11, .newfid = 12, .names = &.{} } }); - try testing.expectEqual(@as(u32, 2), x.prov.nodes[3].refs); - _ = try x.h.ok(.{ .clunk = .{ .fid = 11 } }); - try testing.expectEqual(@as(u32, 1), x.prov.nodes[3].refs); - // a partial walk releases the handles it obtained - const part = try x.h.ok(.{ .walk = .{ .fid = 0, .newfid = 13, .names = &.{ "prov", "dir", "nope" } } }); - try testing.expectEqual(@as(u16, 2), part.walk.nwqid); - try x.h.expectFail(.{ .clunk = .{ .fid = 13 } }, "unknown fid"); - _ = try x.h.ok(.{ .clunk = .{ .fid = 12 } }); - for (x.h.conn.fids) |f| { - if (f.used) _ = try x.h.ok(.{ .clunk = .{ .fid = f.id } }); - } - try testing.expectEqual(@as(u32, 0), x.prov.total_refs); -} - -test "directory reads across offsets, bad offset, and records never split" { - var x: Fixture = .{}; - try x.init(); - defer x.deinit(); - try x.h.walkTo(1, &.{}); - _ = try x.h.ok(.{ .open = .{ .fid = 1, .mode = cloud9.oread } }); - const first = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = 4096 } }); - try testing.expect(first.read.len > 0); - try x.h.expectFail(.{ .read = .{ .fid = 1, .offset = 5, .count = 4096 } }, "bad offset"); - // offset 0 restarts; the same bytes come back - const again = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = 4096 } }); - try testing.expectEqual(first.read.len, again.read.len); - // small reads at consecutive offsets return every record exactly once - const names = try x.h.listDir(1, 80); - defer TS.Harness.freeNames(names); - try testing.expectEqual(@as(usize, 7), names.len); - // a count too small for even one record returns nothing rather than splitting it - const tiny = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = 10 } }); - try testing.expectEqual(@as(usize, 0), tiny.read.len); - // the same for provider and var directories - const pn = try x.h.listPath(&.{ "prov", "dir" }); - defer TS.Harness.freeNames(pn); - try testing.expectEqual(@as(usize, 1), pn.len); - try x.h.walkTo(2, &.{ "vars", "state", "f" }); - _ = try x.h.ok(.{ .open = .{ .fid = 2, .mode = cloud9.oread } }); - const vn = try x.h.listDir(2, 100); - defer TS.Harness.freeNames(vn); - try testing.expectEqual(@as(usize, 4), vn.len); - try x.h.expectFail(.{ .read = .{ .fid = 2, .offset = 1, .count = 100 } }, "bad offset"); -} - -test "Tversion mid-session resets fids and clunks every provider handle" { - var x: Fixture = .{}; - try x.init(); - defer x.deinit(); - try x.h.walkTo(1, &.{ "prov", "hello" }); - try x.h.walkTo(2, &.{ "prov", "dir", "inner" }); - _ = try x.h.ok(.{ .open = .{ .fid = 2, .mode = cloud9.oread } }); - try x.h.walkTo(3, &.{ "runtime", "fn", "fib30" }); - _ = try x.h.ok(.{ .open = .{ .fid = 3, .mode = cloud9.oread } }); - try testing.expectEqual(@as(u32, 2), x.prov.total_refs); - try testing.expectEqual(@as(u32, 1), x.prov.nodes[3].opens); - try testing.expectEqual(@as(usize, 4), x.h.conn.fidCount()); - const before = x.prov.clunks; - try x.h.version(4096); - try testing.expectEqual(@as(usize, 0), x.h.conn.fidCount()); - try testing.expectEqual(@as(u32, 0), x.prov.total_refs); - try testing.expectEqual(@as(u32, 0), x.prov.nodes[3].opens); - try testing.expectEqual(before + 2, x.prov.clunks); - try testing.expect(!x.h.conn.slot_used[0] and !x.h.conn.slot_used[1]); - try x.h.expectFail(.{ .clunk = .{ .fid = 1 } }, "unknown fid"); - _ = try x.h.ok(.{ .attach = .{ .fid = 0, .uname = "tester" } }); - try x.h.walkTo(1, &.{ "prov", "hello" }); - // hangup does the same - x.h.conn.hangup(); - try testing.expectEqual(@as(u32, 0), x.prov.total_refs); - try testing.expectEqual(@as(usize, 0), x.h.conn.fidCount()); -} - -test "a reply that does not fit msize is an Rerror, not a dead connection" { - var x: Fixture = .{}; - try x.init(); - defer x.deinit(); - try x.h.version(64); - _ = try x.h.ok(.{ .attach = .{ .fid = 0, .uname = "t" } }); - // Rstat of the root is ~70 bytes. - try x.h.expectFail(.{ .stat = .{ .fid = 0 } }, "reply too large for msize"); - // Rwalk with 5 qids is 74 bytes; the walk must not bind newfid. - try x.h.expectFail(.{ .walk = .{ .fid = 0, .newfid = 1, .names = &.{ ".", ".", ".", ".", "." } } }, "reply too large for msize"); - try x.h.expectFail(.{ .clunk = .{ .fid = 1 } }, "unknown fid"); - try x.h.expectFail(.{ .walk = .{ .fid = 0, .newfid = 0, .names = &.{ ".", ".", ".", ".", "." } } }, "reply too large for msize"); - const r = try x.h.ok(.{ .walk = .{ .fid = 0, .newfid = 1, .names = &.{"README"} } }); - try testing.expectEqual(@as(u16, 1), r.walk.nwqid); - _ = try x.h.ok(.{ .open = .{ .fid = 1, .mode = cloud9.oread } }); - const rd = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = 40 } }); - try testing.expect(rd.read.len > 0 and rd.read.len <= 64 - cloud9.iohdrsz); - _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); -} - -test "fid table is bounded per connection" { - var x: Fixture = .{}; - try x.init(); - defer x.deinit(); - var i: u32 = 1; - while (x.h.conn.fidCount() < test_cfg.max_fids) : (i += 1) { - _ = try x.h.ok(.{ .walk = .{ .fid = 0, .newfid = i, .names = &.{} } }); - } - try x.h.expectFail(.{ .walk = .{ .fid = 0, .newfid = i, .names = &.{} } }, "too many fids"); - try x.h.expectFail(.{ .attach = .{ .fid = i, .uname = "tester" } }, "too many fids"); - // self-walks and clunks still work at the limit - _ = try x.h.ok(.{ .walk = .{ .fid = 0, .newfid = 0, .names = &.{"build"} } }); - _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); - _ = try x.h.ok(.{ .walk = .{ .fid = 0, .newfid = i, .names = &.{} } }); - try x.h.expectFail(.{ .walk = .{ .fid = 0, .newfid = 2, .names = &.{"README"} } }, "fid in use"); - try x.h.expectFail(.{ .walk = .{ .fid = 1234, .newfid = 2, .names = &.{} } }, "unknown fid"); - // a walk into a provider at the limit must not leak the handle - _ = try x.h.ok(.{ .clunk = .{ .fid = 2 } }); - _ = try x.h.ok(.{ .walk = .{ .fid = 0, .newfid = 2, .names = &.{ "..", "prov", "hello" } } }); - try x.h.expectFail(.{ .walk = .{ .fid = 2, .newfid = i + 1, .names = &.{} } }, "too many fids"); - try testing.expectEqual(@as(u32, 1), x.prov.total_refs); -} - -test "Shared refuses more providers or vars than configured" { - var ctx: TestCtx = .{}; - var shared: TS.Shared = .init(&ctx); - var p1 = TestProv.init(); - var p2 = TestProv.init(); - var p3 = TestProv.init(); - try shared.addProvider(.{ .name = "a", .ctx = &p1, .vtable = &TestProv.vtable }); - try shared.addProvider(.{ .name = "b", .ctx = &p2, .vtable = &TestProv.vtable }); - try testing.expectError(error.Full, shared.addProvider(.{ .name = "c", .ctx = &p3, .vtable = &TestProv.vtable })); - var v: [5]u32 = @splat(0); - try shared.expose("v0", &v[0]); - try shared.expose("v1", &v[1]); - try shared.expose("v2", &v[2]); - try shared.expose("v3", &v[3]); - try testing.expectError(error.Full, shared.expose("v4", &v[4])); -} - -/// A server with a large fid table for the index tests. -const big_cfg: Config = .{ - .name = "big", - .msize = 8192, - .max_fids = 4096, - .max_providers = 1, - .max_vars = 1, - .snapshot_slots = 1, - .snapshot_bytes = 256, -}; -const BigS = Server(big_cfg); - -/// Fid numbers chosen to stress the index: dense low ids, ids with only high -/// bits set, and ids counting down from 2^32-1 (all distinct for i < 2^20). -fn adversarialId(i: u32) u32 { - return switch (i % 3) { - 0 => i * 8192 + 1, - 1 => 0x8000_0000 | i, - else => 0xFFFF_FFFF - i, - }; -} - -/// Every index bucket points at a used fid that finds itself, and every used -/// fid is found: the invariant the hostile fid tests check after each phase. -fn checkFidIndex(c: *BigS.Conn) !void { - var indexed: usize = 0; - for (c.index) |slot| { - if (slot == BigS.no_slot) continue; - indexed += 1; - try testing.expect(c.fids[slot].used); - try testing.expectEqual(&c.fids[slot], c.findFid(c.fids[slot].id).?); - } - var used: usize = 0; - for (c.fids[0..c.high_water]) |*f| if (f.used) { - used += 1; - try testing.expectEqual(f, c.findFid(f.id).?); - }; - for (c.fids[c.high_water..]) |*f| try testing.expect(!f.used); - try testing.expectEqual(indexed, used); - try testing.expectEqual(used, c.nfids); -} - -test "fid index: thousands of fids, clunk in hostile orders, reuse, Tversion" { - var ctx: TestCtx = .{}; - var shared: BigS.Shared = .init(&ctx); - var prov = TestProv.init(); - try shared.addProvider(prov.provider()); - const storage = try testing.allocator.create(BigS.Storage); - defer testing.allocator.destroy(storage); - var h: BigS.Harness = undefined; - try h.init(&shared, storage); - defer h.deinit(); - const n: u32 = big_cfg.max_fids - 1; // fid 0 is the attach - var i: u32 = 0; - while (i < n) : (i += 1) { - _ = try h.ok(.{ .walk = .{ .fid = 0, .newfid = adversarialId(i), .names = &.{ "prov", "hello" } } }); - } - try testing.expectEqual(@as(usize, n + 1), h.conn.fidCount()); - try testing.expectEqual(n, prov.total_refs); - try h.expectFail(.{ .walk = .{ .fid = 0, .newfid = 0x7FFF_FFFF, .names = &.{} } }, "too many fids"); - try h.expectFail(.{ .walk = .{ .fid = 0, .newfid = adversarialId(5), .names = &.{} } }, "fid in use"); - try h.expectFail(.{ .attach = .{ .fid = adversarialId(7), .uname = "t" } }, "fid in use"); - try testing.expect(h.conn.findFid(0x7FFF_FFFF) == null); - try testing.expect(h.conn.findFid(adversarialId(n)) == null); - try checkFidIndex(&h.conn); - // clunk every third fid, then the rest from the top: backward-shift deletion under churn - i = 0; - while (i < n) : (i += 3) _ = try h.ok(.{ .clunk = .{ .fid = adversarialId(i) } }); - try checkFidIndex(&h.conn); - i = n; - while (i > 0) { - i -= 1; - if (i % 3 == 0) { - try h.expectFail(.{ .clunk = .{ .fid = adversarialId(i) } }, "unknown fid"); - } else { - _ = try h.ok(.{ .clunk = .{ .fid = adversarialId(i) } }); - } - } - try testing.expectEqual(@as(usize, 1), h.conn.fidCount()); - try testing.expectEqual(@as(u32, 0), prov.total_refs); - try checkFidIndex(&h.conn); - // the whole table is reusable after the churn, through the free list - i = 0; - while (i < n) : (i += 1) _ = try h.ok(.{ .walk = .{ .fid = 0, .newfid = n - i, .names = &.{} } }); - try h.expectFail(.{ .walk = .{ .fid = 0, .newfid = n + 1, .names = &.{} } }, "too many fids"); - try checkFidIndex(&h.conn); - // pseudo-random alloc/free storm with verification - var prng = std.Random.DefaultPrng.init(0x9a11); - const rnd = prng.random(); - var live: [n + 1]bool = @splat(true); - live[0] = false; // never touch the attach fid - var round: usize = 0; - while (round < 20_000) : (round += 1) { - const id = 1 + rnd.uintLessThan(u32, n); - if (live[id]) { - _ = try h.ok(.{ .clunk = .{ .fid = id } }); - } else { - _ = try h.ok(.{ .walk = .{ .fid = 0, .newfid = id, .names = &.{"prov"} } }); - } - live[id] = !live[id]; - if (round % 997 == 0) try checkFidIndex(&h.conn); - } - try checkFidIndex(&h.conn); - // Tversion drops everything and the table starts over, provider refs balanced - try h.version(big_cfg.msize); - try testing.expectEqual(@as(usize, 0), h.conn.fidCount()); - try testing.expectEqual(@as(u32, 0), prov.total_refs); - try testing.expectEqual(@as(u16, 0), h.conn.high_water); - try checkFidIndex(&h.conn); - _ = try h.ok(.{ .attach = .{ .fid = 0xFFFF_FFFE, .uname = "t" } }); - _ = try h.ok(.{ .walk = .{ .fid = 0xFFFF_FFFE, .newfid = 0, .names = &.{} } }); - try checkFidIndex(&h.conn); -} - -test "open: a provider stat failure after a successful open closes the file again" { - var x: Fixture = .{}; - try x.init(); - defer x.deinit(); - try x.h.walkTo(1, &.{ "prov", "hello" }); - x.prov.fail_stat = true; - try x.h.expectFail(.{ .open = .{ .fid = 1, .mode = cloud9.oread } }, "i/o error"); - x.prov.fail_stat = false; - try testing.expectEqual(@as(u32, 0), x.prov.nodes[1].opens); - try x.h.expectFail(.{ .read = .{ .fid = 1, .offset = 0, .count = 10 } }, "file not open"); - _ = try x.h.ok(.{ .open = .{ .fid = 1, .mode = cloud9.oread } }); - try testing.expectEqual(@as(u32, 1), x.prov.nodes[1].opens); - // the same for create: a stat failure after the provider created the node releases it - try x.h.walkTo(2, &.{"prov"}); - x.prov.fail_stat = true; - try x.h.expectFail(.{ .create = .{ .fid = 2, .name = "born", .perm = 0o644, .mode = cloud9.owrite } }, "i/o error"); - x.prov.fail_stat = false; - for (x.prov.nodes) |e| if (e.used and std.mem.eql(u8, e.nameSlice(), "born")) { - try testing.expectEqual(@as(u32, 0), e.opens); - try testing.expectEqual(@as(u32, 0), e.refs); - }; - try testing.expect(!x.h.conn.findFid(2).?.open); - try testing.expectEqual(@as(u32, 1), x.prov.total_refs); // fid 1 only -} - -test "fid state machine: open twice, walk from open, remove/clunk of open provider fids" { - var x: Fixture = .{}; - try x.init(); - defer x.deinit(); - try x.h.walkTo(1, &.{ "prov", "dir", "inner" }); - _ = try x.h.ok(.{ .open = .{ .fid = 1, .mode = cloud9.ordwr } }); - try x.h.expectFail(.{ .open = .{ .fid = 1, .mode = cloud9.oread } }, "file already open"); - try x.h.expectFail(.{ .walk = .{ .fid = 1, .newfid = 2, .names = &.{"."} } }, "file already open"); - try x.h.expectFail(.{ .create = .{ .fid = 1, .name = "z", .perm = 0o644, .mode = cloud9.oread } }, "file already open"); - // a clone of an open fid is a fresh, unopened reference - _ = try x.h.ok(.{ .walk = .{ .fid = 1, .newfid = 2, .names = &.{} } }); - try testing.expectEqual(@as(u32, 2), x.prov.nodes[3].refs); - try testing.expectEqual(@as(u32, 1), x.prov.nodes[3].opens); - // walking newfid == fid with names on an unopened provider fid swaps the handle, refs balanced - try x.h.walkTo(7, &.{ "prov", "dir" }); - try testing.expectEqual(@as(u32, 1), x.prov.nodes[2].refs); - _ = try x.h.ok(.{ .walk = .{ .fid = 7, .newfid = 7, .names = &.{ "..", "dir", "inner", "..", "..", "dir" } } }); - try testing.expectEqual(@as(u32, 1), x.prov.nodes[2].refs); - try testing.expectEqual(@as(u32, 2), x.prov.nodes[3].refs); - _ = try x.h.ok(.{ .clunk = .{ .fid = 7 } }); - try testing.expectEqual(@as(u32, 0), x.prov.nodes[2].refs); - // remove of an open fid: close, then remove, then clunk; refs and opens return to zero - _ = try x.h.ok(.{ .remove = .{ .fid = 1 } }); - try testing.expectEqual(@as(u32, 0), x.prov.nodes[3].opens); - try testing.expectEqual(@as(u32, 1), x.prov.nodes[3].refs); - try x.h.expectFail(.{ .open = .{ .fid = 1, .mode = cloud9.oread } }, "unknown fid"); - _ = try x.h.ok(.{ .clunk = .{ .fid = 2 } }); - try testing.expectEqual(@as(u32, 0), x.prov.total_refs); - // walking "." on a file fid is "not a directory" at the protocol level, without a provider walk - try x.h.walkTo(3, &.{ "prov", "hello" }); - const before = x.prov.clunks; - try x.h.expectFail(.{ .walk = .{ .fid = 3, .newfid = 4, .names = &.{"."} } }, "not a directory"); - try testing.expectEqual(before, x.prov.clunks); - try testing.expectEqual(@as(u32, 1), x.prov.total_refs); - // a partial walk through a file releases the handles it took - const part = try x.h.ok(.{ .walk = .{ .fid = 0, .newfid = 5, .names = &.{ "prov", "hello", "x", "y" } } }); - try testing.expectEqual(@as(u16, 2), part.walk.nwqid); - try testing.expectEqual(@as(u32, 1), x.prov.total_refs); - try x.h.expectFail(.{ .clunk = .{ .fid = 5 } }, "unknown fid"); - // Tremove is always a clunk, even of a static node or when the provider refuses - try x.h.walkTo(6, &.{"README"}); - try x.h.expectFail(.{ .remove = .{ .fid = 6 } }, "permission denied"); - try x.h.expectFail(.{ .clunk = .{ .fid = 6 } }, "unknown fid"); - _ = try x.h.ok(.{ .clunk = .{ .fid = 3 } }); - try testing.expectEqual(@as(u32, 0), x.prov.total_refs); -} - -test "snapshot slots: exhaust, hold, Tversion frees; reads past the end and at huge offsets" { - var x: Fixture = .{}; - try x.init(); - defer x.deinit(); - try x.h.walkTo(1, &.{ "runtime", "fn", "fib30" }); - try x.h.walkTo(2, &.{ "vars", "state", "value" }); - try x.h.walkTo(3, &.{ "vars", "state", "addr" }); - _ = try x.h.ok(.{ .open = .{ .fid = 1, .mode = cloud9.oread } }); - _ = try x.h.ok(.{ .open = .{ .fid = 2, .mode = cloud9.oread } }); - try x.h.expectFail(.{ .open = .{ .fid = 3, .mode = cloud9.oread } }, "too many open dynamic files"); - try testing.expect(!x.h.conn.findFid(3).?.open); - // reads at offsets near 2^64 never trap (counts above msize are a raw-9P - // case: the cloud9 client refuses to send them; test/adv_core_hostile.py covers it) - const max_count = test_cfg.msize - cloud9.iohdrsz; - const r = try x.h.ok(.{ .read = .{ .fid = 1, .offset = std.math.maxInt(u64), .count = max_count } }); - try testing.expectEqualStrings("", r.read); - const r2 = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 1 << 63, .count = 0 } }); - try testing.expectEqualStrings("", r2.read); - const r3 = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = max_count } }); - try testing.expectEqualStrings("832040", r3.read); - // raw beyond @sizeOf is empty; a partial raw read at the tail is bounded - try x.h.walkTo(4, &.{ "vars", "state", "raw" }); - _ = try x.h.ok(.{ .open = .{ .fid = 4, .mode = cloud9.oread } }); - const raw_end = try x.h.ok(.{ .read = .{ .fid = 4, .offset = @sizeOf(Exposed), .count = 100 } }); - try testing.expectEqualStrings("", raw_end.read); - const raw_tail = try x.h.ok(.{ .read = .{ .fid = 4, .offset = @sizeOf(Exposed) - 1, .count = 100 } }); - try testing.expectEqual(@as(usize, 1), raw_tail.read.len); - const raw_huge = try x.h.ok(.{ .read = .{ .fid = 4, .offset = std.math.maxInt(u64) - 1, .count = 100 } }); - try testing.expectEqualStrings("", raw_huge.read); - // Tversion releases the held slots - try x.h.version(test_cfg.msize); - try testing.expect(!x.h.conn.slot_used[0] and !x.h.conn.slot_used[1]); - _ = try x.h.ok(.{ .attach = .{ .fid = 0, .uname = "tester" } }); - try x.h.walkTo(3, &.{ "vars", "state", "addr" }); - _ = try x.h.ok(.{ .open = .{ .fid = 3, .mode = cloud9.oread } }); -} - -test "static and var nodes refuse create, remove and wstat; directories refuse writes" { - var x: Fixture = .{}; - try x.init(); - defer x.deinit(); - const dirs = [_][]const []const u8{ &.{}, &.{"build"}, &.{"comptime"}, &.{ "comptime", "types" }, &.{ "comptime", "types", "Layout" }, &.{"runtime"}, &.{ "runtime", "fn" }, &.{"vars"}, &.{ "vars", "state" }, &.{ "vars", "state", "f" }, &.{ "vars", "state", "f", "inner" } }; - for (dirs, 0..) |d, k| { - const fid: u32 = @intCast(10 + k); - try x.h.walkTo(fid, d); - try x.h.expectFail(.{ .create = .{ .fid = fid, .name = "x", .perm = 0o644, .mode = cloud9.owrite } }, "permission denied"); - try x.h.expectFail(.{ .wstat = .{ .fid = fid, .stat = stat_dontcare } }, "permission denied"); - try x.h.expectFail(.{ .open = .{ .fid = fid, .mode = cloud9.owrite } }, "is a directory"); - try x.h.expectFail(.{ .open = .{ .fid = fid, .mode = cloud9.oread | cloud9.otrunc } }, "is a directory"); - try x.h.expectFail(.{ .remove = .{ .fid = fid } }, "permission denied"); - try x.h.expectFail(.{ .clunk = .{ .fid = fid } }, "unknown fid"); - } - const files = [_][]const []const u8{ &.{"README"}, &.{ "build", "time" }, &.{ "comptime", "decls" }, &.{ "runtime", "pid" }, &.{ "runtime", "fn", "fib30" }, &.{"ctl"}, &.{ "vars", "state", "value" }, &.{ "vars", "state", "raw" }, &.{ "vars", "state", "f", "a", "value" }, &.{ "vars", "counter", "type" } }; - for (files, 0..) |f, k| { - const fid: u32 = @intCast(30 + k); - try x.h.walkTo(fid, f); - try x.h.expectFail(.{ .wstat = .{ .fid = fid, .stat = stat_dontcare } }, "permission denied"); - try x.h.expectFail(.{ .walk = .{ .fid = fid, .newfid = 99, .names = &.{".."} } }, "not a directory"); - try x.h.expectFail(.{ .remove = .{ .fid = fid } }, "permission denied"); - } - // writes to a var value at a non-zero offset and with an empty payload - try x.h.walkTo(1, &.{ "vars", "state", "f", "a", "value" }); - _ = try x.h.ok(.{ .open = .{ .fid = 1, .mode = cloud9.owrite | cloud9.otrunc } }); - try x.h.expectFail(.{ .write = .{ .fid = 1, .offset = 0, .data = "" } }, "bad value"); - try x.h.expectFail(.{ .write = .{ .fid = 1, .offset = 0, .data = "-1" } }, "bad value"); - try x.h.expectFail(.{ .write = .{ .fid = 1, .offset = 0, .data = "1e3" } }, "bad value"); - try x.h.expectFail(.{ .write = .{ .fid = 1, .offset = 0, .data = "99999999999999999999" } }, "bad value"); - try testing.expectEqual(@as(u32, 1), x.exposed.a); - _ = try x.h.ok(.{ .write = .{ .fid = 1, .offset = std.math.maxInt(u64), .data = "77\n" } }); - try testing.expectEqual(@as(u32, 77), x.exposed.a); - // reads of a write-only fid are refused; OEXEC reads like OREAD - try x.h.expectFail(.{ .read = .{ .fid = 1, .offset = 0, .count = 10 } }, "file not open"); - try x.h.walkTo(2, &.{"README"}); - _ = try x.h.ok(.{ .open = .{ .fid = 2, .mode = cloud9.oexec } }); - try testing.expect((try x.h.ok(.{ .read = .{ .fid = 2, .offset = 0, .count = 10 } })).read.len == 10); -} - -test "msize 24: every request that fits is answered, every reply that cannot fit is an Rerror" { - var x: Fixture = .{}; - try x.init(); - defer x.deinit(); - try x.h.version(24); - _ = try x.h.ok(.{ .attach = .{ .fid = 0, .uname = "u" } }); // Tattach 20, Rattach 20 - try x.h.expectFail(.{ .stat = .{ .fid = 0 } }, "reply too large"); // Rerror truncated to fit 24 bytes - const w = try x.h.ok(.{ .walk = .{ .fid = 0, .newfid = 1, .names = &.{"ctl"} } }); // Rwalk 22 - try testing.expectEqual(@as(u16, 1), w.walk.nwqid); - try x.h.expectFail(.{ .walk = .{ .fid = 0, .newfid = 2, .names = &.{ ".", "." } } }, "reply too large"); - try x.h.expectFail(.{ .clunk = .{ .fid = 2 } }, "unknown fid"); - _ = try x.h.ok(.{ .open = .{ .fid = 1, .mode = cloud9.ordwr } }); // Ropen 24 - try x.h.expectFail(.{ .write = .{ .fid = 1, .offset = 0, .data = "e" } }, "bad command"); // Twrite 24 - // the largest read the client may ask for is msize - iohdrsz = 0 bytes - const r = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = 0 } }); - try testing.expectEqual(@as(usize, 0), r.read.len); - _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); - try x.h.walkTo(3, &.{"build"}); - _ = try x.h.ok(.{ .open = .{ .fid = 3, .mode = cloud9.oread } }); - const d = try x.h.ok(.{ .read = .{ .fid = 3, .offset = 0, .count = 0 } }); - try testing.expectEqual(@as(usize, 0), d.read.len); // no record fits in 0 bytes, nothing is split - try x.h.expectFail(.{ .read = .{ .fid = 3, .offset = 1, .count = 0 } }, "bad offset"); -} - -test "Conn.init clamps the msize cap to [msize_min, cfg.msize]" { - var ctx: TestCtx = .{}; - var shared: TS.Shared = .init(&ctx); - var storage: TS.Storage = undefined; - const lo: TS.Conn = .init(&shared, &storage, 0); - try testing.expectEqual(cloud9.Server.msize_min, lo.msize_cap); - const hi: TS.Conn = .init(&shared, &storage, std.math.maxInt(u32)); - try testing.expectEqual(test_cfg.msize, hi.msize_cap); - const mid: TS.Conn = .init(&shared, &storage, 4096); - try testing.expectEqual(@as(u32, 4096), mid.msize_cap); -} - -test "parseIso8601 rejects malformed stamps and never traps" { - try testing.expectEqual(@as(?u32, null), parseIso8601("")); - try testing.expectEqual(@as(?u32, null), parseIso8601("2023-11-14T22:13:20")); - try testing.expectEqual(@as(?u32, null), parseIso8601("2023-13-14T22:13:20Z")); - try testing.expectEqual(@as(?u32, null), parseIso8601("2023-11-32T22:13:20Z")); - try testing.expectEqual(@as(?u32, null), parseIso8601("2023-11-14T24:13:20Z")); - try testing.expectEqual(@as(?u32, null), parseIso8601("2023-11-14T22:60:20Z")); - try testing.expectEqual(@as(?u32, null), parseIso8601("1969-12-31T23:59:59Z")); - try testing.expectEqual(@as(?u32, null), parseIso8601("9999-12-31T23:59:59Z")); - try testing.expectEqual(@as(?u32, null), parseIso8601("20x3-11-14T22:13:20Z")); - try testing.expectEqual(@as(?u32, null), parseIso8601("0000-01-01T00:00:00Z")); - try testing.expectEqual(@as(u32, 1_700_000_000), parseIso8601("2023-11-14T22:13:20Z").?); - try testing.expectEqual(@as(u32, 951_782_400), parseIso8601("2000-02-29T00:00:00Z").?); - try testing.expectEqual(@as(u32, 4_102_444_799), parseIso8601("2099-12-31T23:59:59Z").?); - try testing.expectEqual(@as(u32, std.math.maxInt(u32)), parseIso8601("2106-02-07T06:28:15Z").?); - try testing.expectEqual(@as(?u32, null), parseIso8601("2106-02-07T06:28:16Z")); -} - -/// Multiplicative inverse of an odd 32-bit constant (Newton iteration). -fn inverseMod32(a: u32) u32 { - var x: u32 = a; - for (0..5) |_| x *%= 2 -% a *% x; - return x; -} - -test "fid index: fid numbers crafted to collide under the public hash do not cluster a seeded connection" { - var ctx: TestCtx = .{}; - var shared: BigS.Shared = .init(&ctx); - const storage = try testing.allocator.create(BigS.Storage); - defer testing.allocator.destroy(storage); - var h: BigS.Harness = undefined; - try h.init(&shared, storage); - defer h.deinit(); - // two connections on the same Shared never share a seed - const other: BigS.Conn = .init(&shared, storage, big_cfg.msize); - try testing.expect(other.hash_seed != h.conn.hash_seed); - // ids whose products with the golden ratio share their top bits: all one bucket when unseeded - const inv = inverseMod32(0x9E37_79B1); - try testing.expectEqual(@as(u32, 1), inv *% 0x9E37_79B1); - const n: u32 = big_cfg.max_fids - 1; - const base: u32 = 0x4242_0000; - var i: u32 = 0; - while (i < n) : (i += 1) { - const id = (base + i) *% inv; - try testing.expectEqual(@as(usize, base >> BigS.index_shift), @as(usize, @intCast((id *% 0x9E37_79B1) >> BigS.index_shift))); - _ = try h.ok(.{ .walk = .{ .fid = 0, .newfid = id, .names = &.{} } }); - } - try checkFidIndex(&h.conn); - // the longest probe sequence in the seeded table is short; unseeded it would be ~n - var worst: usize = 0; - i = 0; - while (i < n) : (i += 1) { - const id = (base + i) *% inv; - var pos = h.conn.fidHome(id); - var steps: usize = 0; - while (h.conn.fids[h.conn.index[pos]].id != id) : (pos = (pos + 1) & BigS.index_mask) steps += 1; - worst = @max(worst, steps); - } - try testing.expect(worst < 64); -} diff --git a/introspect/src/freestanding_check.zig b/introspect/src/freestanding_check.zig deleted file mode 100644 index ad88d0e..0000000 --- a/introspect/src/freestanding_check.zig +++ /dev/null @@ -1,71 +0,0 @@ -//! A tiny freestanding root proving that `core` and `vars` compile without an -//! OS: `zig build introspect-check-freestanding` builds this for riscv32-freestanding-none. -//! It instantiates `Server(cfg)` with static Storage/Shared, exposes one -//! variable, and runs one push/step over a canned Tversion frame. It must not -//! import scratch.zig (allocator) or anything OS-specific. -const std = @import("std"); -const core = @import("core.zig"); -const Writer = std.Io.Writer; - -const Build = struct { - pub const zig_version: []const u8 = @import("builtin").zig_version_string; - pub const target: []const u8 = "riscv32-freestanding-none"; - pub const optimize: []const u8 = "check"; - pub const time: []const u8 = "1970-01-01T00:00:00Z"; - pub const change: []const u8 = "none"; -}; - -const State = struct { ticks: u32, phase: enum { idle, busy }, inner: struct { x: f32 } }; - -const Fns = struct { - pub fn ticks(ctx: *anyopaque, w: *Writer) anyerror!void { - const s: *State = @ptrCast(@alignCast(ctx)); - try w.print("{d}", .{s.ticks}); - } -}; - -fn ctl(ctx: *anyopaque, cmd: []const u8, out: *Writer) anyerror!void { - const s: *State = @ptrCast(@alignCast(ctx)); - if (std.mem.startsWith(u8, cmd, "reset")) s.ticks = 0; - try out.writeAll("ok"); -} - -const cfg: core.Config = .{ - .name = "fw", - .build = Build, - .types = &.{ State, core.NodeStat }, - .decls_of = Fns, - .fns = Fns, - .ctl = &ctl, - .msize = 2048, - .max_fids = 16, - .snapshot_slots = 2, - .snapshot_bytes = 1024, -}; - -const S = core.Server(cfg); - -var state: State = .{ .ticks = 0, .phase = .idle, .inner = .{ .x = 0 } }; -var storage: S.Storage = undefined; -var shared: S.Shared = undefined; -var conn: S.Conn = undefined; - -/// Tversion msize=2048 version="9P2000". -const tversion = [_]u8{ 19, 0, 0, 0, 100, 0xFF, 0xFF, 0, 8, 0, 0, 6, 0, '9', 'P', '2', '0', '0', '0' }; - -/// Runs one Tversion through the engine; returns the number of reply bytes. -pub export fn introspect_check() u32 { - shared = .init(&state); - shared.expose("state", &state) catch unreachable; - conn = .init(&shared, &storage, cfg.msize); - _ = conn.push(&tversion); - _ = conn.step() catch return 0; - const out = conn.output(); - conn.wrote(out.len); - return @intCast(out.len); -} - -pub export fn _start() noreturn { - _ = introspect_check(); - while (true) {} -} diff --git a/introspect/src/linux/debug.zig b/introspect/src/linux/debug.zig deleted file mode 100644 index 4d43d5b..0000000 --- a/introspect/src/linux/debug.zig +++ /dev/null @@ -1,1458 +0,0 @@ -//! Linux debug facilities for the introspect server: threads, stacks, -//! registers, address → source, memory, breakpoints and panics. -//! -//! This file is a pure API; a later adapter turns it into a core `Provider`. -//! All text is written to a `*std.Io.Writer`. Nothing here allocates after -//! `init` except from the caller-provided `text_buf`, which is used as a fixed -//! arena for symbol text and reset before every query. -//! -//! Only one `Debug` may exist per process: the signal handlers and the panic -//! hook find their state through the global `current` pointer set by `init`. -//! -//! Mechanics -//! -//! * Capturing another thread's stack or registers: the calling (server) -//! thread sends `capture_signal` with `tgkill`. The SA_SIGINFO handler copies -//! the interrupted register state (`cpu_context.fromPosixSignalContext`) into -//! the single capture slot and parks on a futex. The server unwinds the -//! parked thread's stack from that context, releases the target, then -//! symbolizes. The handler is async-signal-safe: no allocation, no -//! `std.debug`, no locks other than the futex. A target that does not run -//! the handler within `capture_timeout_ns` (signal masked, thread in D -//! state, ...) yields `error.Timeout`; a late-arriving handler run cannot -//! corrupt a reused slot because it must match the requested tid and win a -//! compare-and-swap from `armed` on the slot state (that pair plays the role -//! of a generation counter: a stale run finds the slot idle, armed for -//! another tid, or armed for itself, in which case its capture is simply the -//! valid answer to the new request). -//! * Breakpoints: `@breakpoint()` raises SIGTRAP on the executing thread only. -//! The handler claims a pause slot, saves the context and parks on a futex -//! until `resumeThread`. On x86_64 the saved PC is already past `int3`; on -//! aarch64 the handler advances PC by 4 in the ucontext before returning -//! (only for a real `brk`, i.e. a kernel-generated si_code; a SIGTRAP sent -//! with kill/tgkill parks the thread where it was). With no free slot the -//! thread steps over the breakpoint and keeps running (`traps_skipped` -//! counts them): the debug layer never kills the process. The server thread -//! itself (`server_tid`) is never parked, a breakpoint there is stepped -//! over, because nobody could resume it. Only a stale handler run after -//! `deinit` (no `current`) falls back to the default disposition. -//! * Panics: `panicHook` records the message and a stack capture, then, if -//! `hold_on_panic` and a `Debug` exists, parks until `panicContinue`; then -//! `std.debug.defaultPanic` runs. A nested or second panic, or a panic on -//! the server thread itself (which could never be continued), goes -//! straight to the default handler. -//! * std.debug's `SelfInfo` guards its state with an `Io.RwLock`. A target -//! parked while holding it (a thread inside a stack-trace dump, say) would -//! deadlock the unwind, so after parking a thread the lock is probed with -//! `tryLock`; a held lock yields `error.Busy` and the target is released. -//! * Known-module guard: `std.debug.SelfInfo` (Zig 0.16) rebuilds its module -//! list whenever it is asked about an address outside every known module, -//! freeing the CIE lists its unwind cache still points into; later unwinds -//! then read freed memory. `init` records the PT_LOAD ranges of the -//! executable (the same source std uses) and every lookup or unwind is -//! first checked against them; addresses outside (unmapped, vDSO, ...) -//! render as "?" and are never handed to std. - -const std = @import("std"); -const builtin = @import("builtin"); -const linux = std.os.linux; -const cpu_context = std.debug.cpu_context; -const Writer = std.Io.Writer; -const Native = cpu_context.Native; -const arch = builtin.cpu.arch; - -pub const Options = struct { - /// Used for `std.debug` symbolization (reading debug info from disk). - io: std.Io, - /// Fixed arena for symbol text. A `FixedBufferAllocator` is placed over it - /// and reset before every query. 16 KiB is plenty; 4 KiB is a sane floor. - text_buf: []u8, - /// Real-time signal used to snapshot other threads. SIGRTMIN is 32 on - /// Linux without libc; the default is SIGRTMIN+3. - capture_signal: u8 = default_capture_signal, - /// How long to wait for a target thread to run the capture handler. - capture_timeout_ns: u64 = 250 * std.time.ns_per_ms, - /// How many threads may be parked in `@breakpoint()` at once (≤ 32). - max_paused: u8 = 16, -}; - -pub const default_capture_signal: u8 = 32 + 3; - -/// Hard upper bound of `Options.max_paused` (slot storage is static). -pub const max_paused_cap = 32; -/// Maximum number of frames written by any stack function. -pub const max_frames = 64; -/// Maximum number of tids enumerated from /proc/self/task. -pub const max_threads = 512; -/// Upper bound of the recorded panic message. -pub const panic_msg_cap = 1024; -/// Maximum number of PT_LOAD ranges recorded by the known-module guard. -pub const max_ranges = 64; - -/// Consulted by `panicHook`: when true and a `Debug` is initialized, the -/// panicking thread is held until `panicContinue`. -pub var hold_on_panic: bool = true; - -/// The one live instance, set by `init`, cleared by `deinit`. -pub var current: ?*Debug = null; - -/// The tid of the thread serving requests (0 = none). That thread is never -/// parked by a breakpoint or held by a panic, since nobody could release it. -pub var server_tid: std.atomic.Value(u32) = .init(0); - -/// Breakpoints stepped over because no pause slot was free, or because they -/// were hit on the server thread. -pub var traps_skipped: std.atomic.Value(u32) = .init(0); - -pub const Error = error{ - /// The target thread did not run the capture handler in time. - Timeout, - /// No thread with that tid exists in this process. - NoThread, - /// The address is not mapped (EFAULT from process_vm_readv/writev). - Unmapped, - /// The thread is not parked in a breakpoint. - NotPaused, - /// No panic has been recorded / is being held. - NoPanic, - /// Another `Debug` already exists in this process. - AlreadyInitialized, - /// The operation is not available on this architecture / kernel. - Unsupported, - /// The target thread is parked inside std.debug (holding its lock); its - /// stack cannot be unwound without deadlocking. Retry later. - Busy, - /// Invalid option value. - InvalidOptions, - /// A syscall or /proc read failed unexpectedly. - Unexpected, - /// The writer failed. - WriteFailed, -}; - -// Capture slot states. -const cap_idle: u32 = 0; -const cap_armed: u32 = 1; -const cap_capturing: u32 = 2; -const cap_captured: u32 = 3; -const cap_failed: u32 = 4; - -// Pause slot states. -const pause_free: u32 = 0; -const pause_claimed: u32 = 1; -const pause_paused: u32 = 2; -const pause_resuming: u32 = 3; - -const CaptureSlot = struct { - state: std.atomic.Value(u32) = .init(cap_idle), - target_tid: std.atomic.Value(u32) = .init(0), - ctx: Native = undefined, -}; - -const PauseSlot = struct { - state: std.atomic.Value(u32) = .init(pause_free), - tid: std.atomic.Value(u32) = .init(0), - ctx: Native = undefined, -}; - -pub const Debug = struct { - io: std.Io, - text_buf: []u8, - capture_signal: linux.SIG, - capture_timeout_ns: u64, - max_paused: u8, - - capture: CaptureSlot = .{}, - paused: [max_paused_cap]PauseSlot = [_]PauseSlot{.{}} ** max_paused_cap, - - old_capture_action: linux.Sigaction = undefined, - old_trap_action: linux.Sigaction = undefined, - breakpoints_enabled: bool = false, - - tids: [max_threads]u32 = undefined, - tid_count: usize = 0, - - ranges: [max_ranges]Range = undefined, - range_count: usize = 0, - - const Range = struct { start: usize, len: usize }; - - /// Installs the capture handler (not the SIGTRAP handler) and publishes - /// `d` as `current`. - pub fn init(d: *Debug, opts: Options) Error!void { - if (current != null) return error.AlreadyInitialized; - if (opts.capture_signal < 32 or opts.capture_signal >= linux.NSIG) return error.InvalidOptions; - if (opts.max_paused == 0 or opts.max_paused > max_paused_cap) return error.InvalidOptions; - if (Native == noreturn) return error.Unsupported; - d.* = .{ - .io = opts.io, - .text_buf = opts.text_buf, - .capture_signal = @enumFromInt(opts.capture_signal), - .capture_timeout_ns = opts.capture_timeout_ns, - .max_paused = opts.max_paused, - }; - d.scanModules(); - const act: linux.Sigaction = .{ - .handler = .{ .sigaction = captureHandler }, - .mask = linux.sigemptyset(), - .flags = linux.SA.SIGINFO | linux.SA.RESTART, - }; - current = d; - if (linux.errno(linux.sigaction(d.capture_signal, &act, &d.old_capture_action)) != .SUCCESS) { - current = null; - return error.Unexpected; - } - } - - /// Restores the signal dispositions and clears `current`. Threads parked - /// in a breakpoint are resumed first. - pub fn deinit(d: *Debug) void { - d.disableBreakpoints(); - _ = linux.sigaction(d.capture_signal, &d.old_capture_action, null); - if (current == d) current = null; - } - - /// Installs the SIGTRAP handler so that `@breakpoint()` parks the thread. - pub fn enableBreakpoints(d: *Debug) Error!void { - if (d.breakpoints_enabled) return; - if (arch != .x86_64 and !arch.isAARCH64()) return error.Unsupported; - const act: linux.Sigaction = .{ - .handler = .{ .sigaction = trapHandler }, - .mask = linux.sigemptyset(), - .flags = linux.SA.SIGINFO | linux.SA.RESTART, - }; - if (linux.errno(linux.sigaction(.TRAP, &act, &d.old_trap_action)) != .SUCCESS) return error.Unexpected; - d.breakpoints_enabled = true; - } - - /// Restores the previous SIGTRAP disposition and resumes every parked thread. - pub fn disableBreakpoints(d: *Debug) void { - if (!d.breakpoints_enabled) return; - _ = linux.sigaction(.TRAP, &d.old_trap_action, null); - d.breakpoints_enabled = false; - for (&d.paused) |*slot| { - if (slot.state.cmpxchgStrong(pause_paused, pause_resuming, .acq_rel, .acquire) == null) - futexWake(&slot.state); - } - } - - // ---------------------------------------------------------------- threads - - /// The nth tid of this process, numerically sorted; null past the end. - /// Index 0 rescans /proc/self/task; higher indices reuse that scan. - pub fn threadAt(d: *Debug, index: usize) ?u32 { - if (index == 0 or d.tid_count == 0) d.scanThreads(); - if (index >= d.tid_count) return null; - return d.tids[index]; - } - - pub fn threadExists(d: *Debug, tid: u32) bool { - _ = d; - var path_buf: [64]u8 = undefined; - const path = std.fmt.bufPrintZ(&path_buf, "/proc/self/task/{d}/comm", .{tid}) catch return false; - var buf: [32]u8 = undefined; - _ = readFile(path, &buf) catch return false; - return true; - } - - /// The thread's comm (without the trailing newline). - pub fn threadName(d: *Debug, tid: u32, w: *Writer) Error!void { - _ = d; - var path_buf: [64]u8 = undefined; - const path = std.fmt.bufPrintZ(&path_buf, "/proc/self/task/{d}/comm", .{tid}) catch return error.Unexpected; - var buf: [64]u8 = undefined; - const text = readFile(path, &buf) catch |err| switch (err) { - error.NotFound => return error.NoThread, - else => return error.Unexpected, - }; - w.writeAll(std.mem.trimEnd(u8, text, "\n")) catch return error.WriteFailed; - } - - /// A few fields of /proc/self/task//stat, one "name value" per line: - /// state, utime, stime, minflt, majflt, priority, nice, processor. - pub fn threadStat(d: *Debug, tid: u32, w: *Writer) Error!void { - _ = d; - var path_buf: [64]u8 = undefined; - const path = std.fmt.bufPrintZ(&path_buf, "/proc/self/task/{d}/stat", .{tid}) catch return error.Unexpected; - var buf: [1024]u8 = undefined; - const text = readFile(path, &buf) catch |err| switch (err) { - error.NotFound => return error.NoThread, - else => return error.Unexpected, - }; - // " () S "; comm may contain spaces and parens. - const close = std.mem.lastIndexOfScalar(u8, text, ')') orelse return error.Unexpected; - var it = std.mem.tokenizeScalar(u8, text[close + 1 ..], ' '); - // Field numbers below are 0-based from `state`. - const wanted = [_]struct { idx: usize, name: []const u8 }{ - .{ .idx = 0, .name = "state" }, - .{ .idx = 11, .name = "utime" }, - .{ .idx = 12, .name = "stime" }, - .{ .idx = 7, .name = "minflt" }, - .{ .idx = 9, .name = "majflt" }, - .{ .idx = 15, .name = "priority" }, - .{ .idx = 16, .name = "nice" }, - .{ .idx = 36, .name = "processor" }, - }; - var fields: [40][]const u8 = undefined; - var n: usize = 0; - while (it.next()) |f| : (n += 1) { - if (n == fields.len) break; - fields[n] = f; - } - for (wanted) |want| { - const value = if (want.idx < n) fields[want.idx] else "?"; - w.print("{s} {s}\n", .{ want.name, value }) catch return error.WriteFailed; - } - } - - /// "#n 0x in (::)" per frame. The calling - /// thread unwinds itself directly; any other thread is captured with the - /// capture signal. - pub fn threadStack(d: *Debug, tid: u32, w: *Writer) Error!void { - var addrs: [max_frames]usize = undefined; - var trace: std.debug.StackTrace = undefined; - if (tid == selfTid()) { - trace = std.debug.captureCurrentStackTrace(.{}, &addrs); - } else { - try d.captureThread(tid); - if (!d.selfInfoFree()) { - d.releaseCapture(); - return error.Busy; - } - trace = d.unwindContext(&d.capture.ctx, &addrs); - d.releaseCapture(); - } - try d.writeFrames(trace.return_addresses, w); - } - - /// " 0x" per general register, plus pc/sp/fp aliases. - pub fn threadRegs(d: *Debug, tid: u32, w: *Writer) Error!void { - if (tid == selfTid()) { - const ctx = Native.current(); - return writeRegs(&ctx, w); - } - try d.captureThread(tid); - const ctx = d.capture.ctx; - d.releaseCapture(); - return writeRegs(&ctx, w); - } - - // ------------------------------------------------------ addresses & memory - - /// "fn\nfile:line:col\nmodule\n", unknown parts as "?". - pub fn resolveAddr(d: *Debug, addr: usize, w: *Writer) Error!void { - if (!d.knownCode(addr)) return w.writeAll("?\n?\n?\n") catch error.WriteFailed; - var fba = std.heap.FixedBufferAllocator.init(d.text_buf); - const alloc = fba.allocator(); - const di = std.debug.getSelfDebugInfo() catch return error.Unsupported; - var sym = std.debug.Symbol.unknown; - var symbols: std.ArrayList(std.debug.Symbol) = .empty; - if (di.getSymbols(d.io, alloc, alloc, addr, true, &symbols)) { - if (symbols.items.len > 0) sym = symbols.items[0]; - } else |_| {} - w.print("{s}\n", .{sym.name orelse "?"}) catch return error.WriteFailed; - if (sym.source_location) |sl| { - w.print("{s}:{d}:{d}\n", .{ sl.file_name, sl.line, sl.column }) catch return error.WriteFailed; - } else { - w.writeAll("?\n") catch return error.WriteFailed; - } - const module = di.getModuleName(d.io, addr) catch "?"; - w.print("{s}\n", .{module}) catch return error.WriteFailed; - } - - /// Reads `buf.len` bytes at `addr` via process_vm_readv on the own - /// process. Never faults. Returns the number of bytes read (short when the - /// range crosses into an unmapped page); `error.Unmapped` when nothing - /// could be read. - pub fn readMem(d: *Debug, addr: usize, buf: []u8) Error!usize { - _ = d; - if (buf.len == 0) return 0; - // Page 0 is never mapped (mmap_min_addr) and a null `iovec.base` is a - // safety-checked cast; the same answer without the trap. - if (addr == 0) return error.Unmapped; - const local = [_]std.posix.iovec{.{ .base = buf.ptr, .len = buf.len }}; - const remote = [_]std.posix.iovec_const{.{ .base = @ptrFromInt(addr), .len = buf.len }}; - const rc = linux.process_vm_readv(linux.getpid(), &local, &remote, 0); - switch (linux.errno(rc)) { - .SUCCESS => return rc, - .FAULT => return error.Unmapped, - .NOSYS, .PERM => return error.Unsupported, - else => return error.Unexpected, - } - } - - /// Writes `data` at `addr` via process_vm_writev. Read-only mappings also - /// report `error.Unmapped` (the kernel says EFAULT for both). - pub fn writeMem(d: *Debug, addr: usize, data: []const u8) Error!usize { - _ = d; - if (data.len == 0) return 0; - if (addr == 0) return error.Unmapped; - const local = [_]std.posix.iovec_const{.{ .base = data.ptr, .len = data.len }}; - const remote = [_]std.posix.iovec_const{.{ .base = @ptrFromInt(addr), .len = data.len }}; - const rc = linux.process_vm_writev(linux.getpid(), &local, &remote, 0); - switch (linux.errno(rc)) { - .SUCCESS => return rc, - .FAULT => return error.Unmapped, - .NOSYS, .PERM => return error.Unsupported, - else => return error.Unexpected, - } - } - - /// Hexdump of `len` bytes at `addr` in the shape of `std.debug.dumpHex` - /// (16 bytes per line, address column, bytes in two groups, ASCII column). - /// Stops early at the first unmapped byte; `error.Unmapped` only when the - /// very first chunk is unreadable. - pub fn hexdump(d: *Debug, addr: usize, len: usize, w: *Writer) Error!void { - var chunk: [256]u8 = undefined; - var done: usize = 0; - while (done < len) { - const want = @min(chunk.len, len - done); - const got = d.readMem(addr +% done, chunk[0..want]) catch |err| switch (err) { - error.Unmapped => if (done == 0) return error.Unmapped else break, - else => return err, - }; - if (got == 0) break; - try writeHexLines(addr +% done, chunk[0..got], w); - done += got; - if (got < want) break; - } - } - - /// Copies /proc/self/maps to `w`. - pub fn maps(d: *Debug, w: *Writer) Error!void { - _ = d; - return streamFile("/proc/self/maps", w); - } - - /// Reads `buf.len` bytes of /proc/self/maps at `offset` (0 at the end). - /// Not a consistent snapshot across reads; a map appearing between two - /// reads shifts the text, like `cat` on /proc itself. - pub fn readMaps(d: *Debug, offset: u64, buf: []u8) Error!usize { - _ = d; - if (offset > std.math.maxInt(i64)) return 0; - return preadFile("/proc/self/maps", offset, buf); - } - - // ------------------------------------------------------------ breakpoints - - /// The nth tid currently parked in `@breakpoint()`. - pub fn pausedAt(d: *Debug, index: usize) ?u32 { - var n: usize = 0; - for (d.paused[0..d.max_paused]) |*slot| { - if (slot.state.load(.acquire) != pause_paused) continue; - if (n == index) return slot.tid.load(.acquire); - n += 1; - } - return null; - } - - pub fn isPaused(d: *Debug, tid: u32) bool { - return d.pausedSlot(tid) != null; - } - - pub fn pausedStack(d: *Debug, tid: u32, w: *Writer) Error!void { - const slot = d.pausedSlot(tid) orelse return error.NotPaused; - if (!d.selfInfoFree()) return error.Busy; - var addrs: [max_frames]usize = undefined; - const trace = d.unwindContext(&slot.ctx, &addrs); - try d.writeFrames(trace.return_addresses, w); - } - - pub fn pausedRegs(d: *Debug, tid: u32, w: *Writer) Error!void { - const slot = d.pausedSlot(tid) orelse return error.NotPaused; - return writeRegs(&slot.ctx, w); - } - - /// Lets a parked thread continue past its breakpoint. - pub fn resumeThread(d: *Debug, tid: u32) Error!void { - const slot = d.pausedSlot(tid) orelse return error.NotPaused; - if (slot.state.cmpxchgStrong(pause_paused, pause_resuming, .acq_rel, .acquire) != null) return error.NotPaused; - futexWake(&slot.state); - } - - fn pausedSlot(d: *Debug, tid: u32) ?*PauseSlot { - for (d.paused[0..d.max_paused]) |*slot| { - if (slot.state.load(.acquire) == pause_paused and slot.tid.load(.acquire) == tid) return slot; - } - return null; - } - - // ------------------------------------------------------------------ panic - - /// The recorded panic message; nothing before any panic. - pub fn panicMessage(d: *Debug, w: *Writer) Error!void { - _ = d; - if (panic_state.load(.acquire) == panic_none) return; - w.writeAll(panic_msg[0..panic_msg_len]) catch return error.WriteFailed; - } - - /// Frames of the panicking thread, symbolized lazily. - pub fn panicStack(d: *Debug, w: *Writer) Error!void { - if (panic_state.load(.acquire) == panic_none) return; - try d.writeFrames(panic_addrs[0..panic_addr_count], w); - } - - /// True while a panicking thread is parked waiting for `panicContinue`. - pub fn panicHeld(d: *Debug) bool { - _ = d; - return panic_state.load(.acquire) == panic_held; - } - - /// Releases the held panicking thread into `std.debug.defaultPanic`. - pub fn panicContinue(d: *Debug) Error!void { - _ = d; - if (panic_state.cmpxchgStrong(panic_held, panic_continued, .acq_rel, .acquire) != null) return error.NoPanic; - futexWake(&panic_state); - } - - // -------------------------------------------------------------- internals - - fn scanThreads(d: *Debug) void { - d.tid_count = 0; - const fd_rc = linux.open("/proc/self/task", .{ .ACCMODE = .RDONLY, .DIRECTORY = true, .CLOEXEC = true }, 0); - if (linux.errno(fd_rc) != .SUCCESS) return; - const fd: i32 = @intCast(fd_rc); - defer _ = linux.close(fd); - var buf: [4096]u8 align(@alignOf(linux.dirent64)) = undefined; - while (true) { - const rc = linux.getdents64(fd, &buf, buf.len); - if (linux.errno(rc) != .SUCCESS or rc == 0) break; - var off: usize = 0; - while (off < rc) { - const ent: *align(1) const linux.dirent64 = @ptrCast(&buf[off]); - const name_ptr: [*:0]const u8 = @ptrCast(&buf[off + @offsetOf(linux.dirent64, "name")]); - const name = std.mem.span(name_ptr); - if (std.fmt.parseInt(u32, name, 10)) |tid| { - if (d.tid_count < max_threads) { - d.tids[d.tid_count] = tid; - d.tid_count += 1; - } - } else |_| {} - off += ent.reclen; - } - } - std.mem.sort(u32, d.tids[0..d.tid_count], {}, std.sort.asc(u32)); - } - - /// Arms the capture slot for `tid`, signals it and waits until the handler - /// has parked with its context copied. On success the caller owns the - /// slot until `releaseCapture`. - fn captureThread(d: *Debug, tid: u32) Error!void { - const slot = &d.capture; - slot.target_tid.store(tid, .release); - slot.state.store(cap_armed, .release); - const rc = linux.tgkill(linux.getpid(), @intCast(tid), d.capture_signal); - switch (linux.errno(rc)) { - .SUCCESS => {}, - .SRCH => { - slot.state.store(cap_idle, .release); - return error.NoThread; - }, - else => { - slot.state.store(cap_idle, .release); - return error.Unexpected; - }, - } - const deadline = monotonicNs() + d.capture_timeout_ns; - while (true) { - const s = slot.state.load(.acquire); - switch (s) { - cap_captured => return, - cap_failed => { - slot.state.store(cap_idle, .release); - return error.Unsupported; - }, - cap_armed => { - const now = monotonicNs(); - if (now >= deadline) { - // Disarm; if the handler raced us it has moved on to - // `capturing` and we simply keep waiting for it. - if (slot.state.cmpxchgStrong(cap_armed, cap_idle, .acq_rel, .acquire) == null) return error.Timeout; - continue; - } - futexWaitNs(&slot.state, cap_armed, deadline - now); - }, - // The handler is copying registers; it finishes promptly. - cap_capturing => futexWaitNs(&slot.state, cap_capturing, 1 * std.time.ns_per_ms), - else => unreachable, - } - } - } - - /// Records the PT_LOAD ranges of every module `dl_iterate_phdr` reports - /// (for a static executable: the executable itself, not the vDSO). - fn scanModules(d: *Debug) void { - d.range_count = 0; - std.posix.dl_iterate_phdr(d, error{}, struct { - fn cb(info: *std.posix.dl_phdr_info, _: usize, ctx: *Debug) error{}!void { - for (info.phdr[0..info.phnum]) |phdr| { - if (phdr.type != .LOAD) continue; - if (ctx.range_count == max_ranges) return; - ctx.ranges[ctx.range_count] = .{ .start = info.addr +% phdr.vaddr, .len = phdr.memsz }; - ctx.range_count += 1; - } - } - }.cb) catch {}; - } - - /// True when `addr` lies in a module `std.debug` already knows about, so - /// that asking it about `addr` cannot trigger a module rescan. - fn knownCode(d: *const Debug, addr: usize) bool { - for (d.ranges[0..d.range_count]) |r| { - if (addr >= r.start and addr - r.start < r.len) return true; - } - return false; - } - - /// Unwinds from a saved context. A pc outside every known module (e.g. a - /// thread inside the vDSO) is reported as a single frame and not unwound, - /// because std would otherwise rescan its module list (see the header). - fn unwindContext(d: *const Debug, ctx: *const Native, addrs: *[max_frames]usize) std.debug.StackTrace { - if (!d.knownCode(ctx.getPc())) { - addrs[0] = ctx.getPc() +| 1; - return .{ .return_addresses = addrs[0..1], .skipped = .unknown }; - } - return std.debug.captureCurrentStackTrace(.{ .context = ctx }, addrs); - } - - /// True when nobody holds std.debug's `SelfInfo` lock right now. Called - /// with the target parked, so a held lock means the *target* (or another - /// live thread, which will let go) holds it; only the former deadlocks, - /// and the caller cannot tell them apart, so both yield `error.Busy`. - fn selfInfoFree(d: *const Debug) bool { - if (comptime !@hasField(std.debug.SelfInfo, "rwlock")) return true; - const di = std.debug.getSelfDebugInfo() catch return true; - if (!di.rwlock.tryLock(d.io)) return false; - di.rwlock.unlock(d.io); - return true; - } - - fn releaseCapture(d: *Debug) void { - d.capture.state.store(cap_idle, .release); - futexWake(&d.capture.state); - } - - fn writeFrames(d: *Debug, addrs: []const usize, w: *Writer) Error!void { - var fba = std.heap.FixedBufferAllocator.init(d.text_buf); - const alloc = fba.allocator(); - const di = std.debug.getSelfDebugInfo() catch return error.Unsupported; - for (addrs, 0..) |ret_addr, i| { - // Return addresses point after the call; the first frame of a - // context capture is stored as pc+1 by std for the same reason. - const addr = ret_addr -| 1; - fba.reset(); - var symbols: std.ArrayList(std.debug.Symbol) = .empty; - var sym = std.debug.Symbol.unknown; - if (d.knownCode(addr)) { - if (di.getSymbols(d.io, alloc, alloc, addr, true, &symbols)) { - if (symbols.items.len > 0) sym = symbols.items[0]; - } else |_| {} - } - w.print("#{d} 0x{x} in {s} (", .{ i, addr, sym.name orelse "?" }) catch return error.WriteFailed; - if (sym.source_location) |sl| { - w.print("{s}:{d}:{d})\n", .{ sl.file_name, sl.line, sl.column }) catch return error.WriteFailed; - } else { - w.writeAll("?)\n") catch return error.WriteFailed; - } - } - } -}; - -// ------------------------------------------------------------------ handlers - -fn selfTid() u32 { - return @intCast(linux.gettid()); -} - -fn captureHandler(_: linux.SIG, _: *const linux.siginfo_t, ctx_ptr: ?*anyopaque) callconv(.c) void { - const d = current orelse return; - const slot = &d.capture; - const me = selfTid(); - if (slot.target_tid.load(.acquire) != me) return; - if (slot.state.cmpxchgStrong(cap_armed, cap_capturing, .acq_rel, .acquire) != null) return; - // The tid check and the swap are not one atomic step: a stale run (a - // signal that stayed pending while its request timed out) may have read - // the old tid and then won the swap of a request re-armed for another - // thread. `target_tid` is fixed while the slot is armed, so re-checking - // after the swap closes the window; hand the slot back untouched. - if (slot.target_tid.load(.acquire) != me) { - slot.state.store(cap_armed, .release); - futexWake(&slot.state); - return; - } - if (cpu_context.fromPosixSignalContext(ctx_ptr)) |ctx| { - slot.ctx = ctx; - slot.state.store(cap_captured, .release); - futexWake(&slot.state); - while (slot.state.load(.acquire) == cap_captured) futexWaitNs(&slot.state, cap_captured, null); - } else { - slot.state.store(cap_failed, .release); - futexWake(&slot.state); - } -} - -/// aarch64 Linux ucontext_t, only as far as `mcontext.pc` (see -/// std.debug.cpu_context's signal_ucontext_t). -const UcontextAarch64 = extern struct { - flags: usize, - link: ?*UcontextAarch64, - stack: linux.stack_t, - sigmask: linux.sigset_t, - unused: [120]u8, - mcontext: extern struct { - fault_address: u64 align(16), - x: [30]u64, - lr: u64, - sp: u64, - pc: u64, - }, -}; - -fn trapHandler(_: linux.SIG, info: *const linux.siginfo_t, ctx_ptr: ?*anyopaque) callconv(.c) void { - const d = current orelse return trapFallback(); - const ctx = cpu_context.fromPosixSignalContext(ctx_ptr) orelse return trapFallback(); - // si_code > 0 is kernel-generated (TRAP_BRKPT for int3/brk); <= 0 is - // kill/tgkill/sigqueue from user space, where PC points at the - // interrupted instruction and must not be touched. - const from_instruction = info.code > 0; - if (comptime arch.isAARCH64()) { - // `brk #imm` does not advance PC; step over it so returning from the - // handler does not re-trap. - if (from_instruction) { - const uc: *UcontextAarch64 = @ptrCast(@alignCast(ctx_ptr.?)); - uc.mcontext.pc += 4; - } - } else if (comptime arch != .x86_64) { - return trapFallback(); - } - const tid = selfTid(); - if (tid == server_tid.load(.acquire)) { - // Nobody could resume the thread that serves /breakpoints: step over. - _ = traps_skipped.fetchAdd(1, .acq_rel); - return; - } - const slot: *PauseSlot = for (d.paused[0..d.max_paused]) |*slot| { - if (slot.state.cmpxchgStrong(pause_free, pause_claimed, .acq_rel, .acquire) == null) break slot; - } else { - _ = traps_skipped.fetchAdd(1, .acq_rel); - return; - }; - slot.ctx = ctx; - slot.tid.store(tid, .release); - slot.state.store(pause_paused, .release); - while (slot.state.load(.acquire) == pause_paused) futexWaitNs(&slot.state, pause_paused, null); - slot.state.store(pause_free, .release); -} - -/// Restores the default SIGTRAP disposition and re-raises it: the signal is -/// blocked while the handler runs, so it is delivered (fatally) on return. -/// Only for a handler run with no `Debug` (a trap in flight during `deinit`) -/// or on an architecture whose context cannot be read. -fn trapFallback() void { - const act: linux.Sigaction = .{ - .handler = .{ .handler = linux.SIG.DFL }, - .mask = linux.sigemptyset(), - .flags = 0, - }; - _ = linux.sigaction(.TRAP, &act, null); - _ = linux.tkill(linux.gettid(), .TRAP); -} - -// --------------------------------------------------------------------- panic - -const panic_none: u32 = 0; -const panic_recording: u32 = 1; -const panic_recorded: u32 = 2; -const panic_held: u32 = 3; -const panic_continued: u32 = 4; - -var panic_state: std.atomic.Value(u32) = .init(panic_none); -var panic_msg: [panic_msg_cap]u8 = undefined; -var panic_msg_len: usize = 0; -var panic_addrs: [max_frames]usize = undefined; -var panic_addr_count: usize = 0; -/// The tid of the panicking thread (0 before any panic). -pub var panic_tid: u32 = 0; - -/// Records the first panic: message (bounded copy) and stack addresses. -/// Returns false if a panic was already recorded (nested or second panic). -pub fn recordPanic(msg: []const u8, first_trace_addr: ?usize) bool { - if (panic_state.cmpxchgStrong(panic_none, panic_recording, .acq_rel, .acquire) != null) return false; - panic_tid = selfTid(); - panic_msg_len = @min(msg.len, panic_msg.len); - @memcpy(panic_msg[0..panic_msg_len], msg[0..panic_msg_len]); - const trace = std.debug.captureCurrentStackTrace(.{ .first_address = first_trace_addr }, &panic_addrs); - panic_addr_count = trace.return_addresses.len; - panic_state.store(panic_recorded, .release); - return true; -} - -/// Parks the panicking thread until `Debug.panicContinue` when holding is -/// enabled and a `Debug` exists; then hands over to `std.debug.defaultPanic`. -pub fn panicHook(msg: []const u8, first_trace_addr: ?usize) noreturn { - @branchHint(.cold); - if (recordPanic(msg, first_trace_addr)) { - // The server thread cannot be held: it is the one that would have to - // serve /panic/ctl. - if (hold_on_panic and current != null and panic_tid != server_tid.load(.acquire)) { - if (panic_state.cmpxchgStrong(panic_recorded, panic_held, .acq_rel, .acquire) == null) { - while (panic_state.load(.acquire) == panic_held) futexWaitNs(&panic_state, panic_held, null); - } - } - } - std.debug.defaultPanic(msg, first_trace_addr); -} - -/// Clears the recorded panic. Only meaningful in tests of the record path. -pub fn resetPanicRecord() void { - panic_msg_len = 0; - panic_addr_count = 0; - panic_tid = 0; - panic_state.store(panic_none, .release); -} - -// ------------------------------------------------------------------- helpers - -fn futexWake(word: *std.atomic.Value(u32)) void { - _ = linux.futex_3arg(&word.raw, .{ .cmd = .WAKE, .private = true }, std.math.maxInt(u32)); -} - -/// Waits while `*word == expect`, at most `timeout_ns` (forever when null). -/// Returns on wake, timeout, value change or EINTR; callers loop. -fn futexWaitNs(word: *std.atomic.Value(u32), expect: u32, timeout_ns: ?u64) void { - var ts: linux.timespec = undefined; - const ts_ptr: ?*const linux.timespec = if (timeout_ns) |ns| blk: { - ts = .{ .sec = @intCast(ns / std.time.ns_per_s), .nsec = @intCast(ns % std.time.ns_per_s) }; - break :blk &ts; - } else null; - _ = linux.futex_4arg(&word.raw, .{ .cmd = .WAIT, .private = true }, expect, ts_ptr); -} - -fn monotonicNs() u64 { - var ts: linux.timespec = undefined; - _ = linux.clock_gettime(.MONOTONIC, &ts); - return @as(u64, @intCast(ts.sec)) * std.time.ns_per_s + @as(u64, @intCast(ts.nsec)); -} - -const FileError = error{ NotFound, Unexpected, TooBig }; - -/// Reads a whole (small) file with raw syscalls. -fn readFile(path: [*:0]const u8, buf: []u8) FileError![]u8 { - const fd_rc = linux.open(path, .{ .ACCMODE = .RDONLY, .CLOEXEC = true }, 0); - switch (linux.errno(fd_rc)) { - .SUCCESS => {}, - .NOENT, .SRCH => return error.NotFound, - else => return error.Unexpected, - } - const fd: i32 = @intCast(fd_rc); - defer _ = linux.close(fd); - var len: usize = 0; - while (len < buf.len) { - const rc = linux.read(fd, buf[len..].ptr, buf.len - len); - switch (linux.errno(rc)) { - .SUCCESS => {}, - .INTR => continue, - .SRCH, .NOENT => return error.NotFound, - else => return error.Unexpected, - } - if (rc == 0) return buf[0..len]; - len += rc; - } - return error.TooBig; -} - -/// One pread of `buf.len` bytes at `offset`; 0 at the end of the file. -fn preadFile(path: [*:0]const u8, offset: u64, buf: []u8) Error!usize { - const fd_rc = linux.open(path, .{ .ACCMODE = .RDONLY, .CLOEXEC = true }, 0); - if (linux.errno(fd_rc) != .SUCCESS) return error.Unexpected; - const fd: i32 = @intCast(fd_rc); - defer _ = linux.close(fd); - var len: usize = 0; - while (len < buf.len) { - const rc = linux.pread(fd, buf[len..].ptr, buf.len - len, @intCast(offset + len)); - switch (linux.errno(rc)) { - .SUCCESS => {}, - .INTR => continue, - else => return error.Unexpected, - } - if (rc == 0) break; - len += rc; - } - return len; -} - -/// Streams a file of any size to `w`. -fn streamFile(path: [*:0]const u8, w: *Writer) Error!void { - const fd_rc = linux.open(path, .{ .ACCMODE = .RDONLY, .CLOEXEC = true }, 0); - if (linux.errno(fd_rc) != .SUCCESS) return error.Unexpected; - const fd: i32 = @intCast(fd_rc); - defer _ = linux.close(fd); - var buf: [4096]u8 = undefined; - while (true) { - const rc = linux.read(fd, &buf, buf.len); - switch (linux.errno(rc)) { - .SUCCESS => {}, - .INTR => continue, - else => return error.Unexpected, - } - if (rc == 0) return; - w.writeAll(buf[0..rc]) catch return error.WriteFailed; - } -} - -fn writeHexLines(base: usize, bytes: []const u8, w: *Writer) Error!void { - var offset: usize = 0; - while (offset < bytes.len) : (offset += 16) { - const line = bytes[offset..@min(offset + 16, bytes.len)]; - w.print("{x:0>[1]} ", .{ base +% offset, @sizeOf(usize) * 2 }) catch return error.WriteFailed; - for (line, 0..) |byte, i| { - w.print("{X:0>2} ", .{byte}) catch return error.WriteFailed; - if (i == 7) w.writeByte(' ') catch return error.WriteFailed; - } - w.writeByte(' ') catch return error.WriteFailed; - if (line.len < 16) { - var missing = (16 - line.len) * 3; - if (line.len < 8) missing += 1; - w.splatByteAll(' ', missing) catch return error.WriteFailed; - } - for (line) |byte| { - w.writeByte(if (std.ascii.isPrint(byte)) byte else '.') catch return error.WriteFailed; - } - w.writeByte('\n') catch return error.WriteFailed; - } -} - -fn writeRegs(ctx: *const Native, w: *Writer) Error!void { - if (comptime arch == .x86_64) { - inline for (@typeInfo(Native.Gpr).@"enum".fields) |f| { - w.print("{s} 0x{x}\n", .{ f.name, ctx.gprs.get(@field(Native.Gpr, f.name)) }) catch return error.WriteFailed; - } - w.print("pc 0x{x}\nsp 0x{x}\nfp 0x{x}\n", .{ - ctx.gprs.get(.rip), ctx.gprs.get(.rsp), ctx.gprs.get(.rbp), - }) catch return error.WriteFailed; - } else if (comptime arch.isAARCH64()) { - for (ctx.x, 0..) |x, i| w.print("x{d} 0x{x}\n", .{ i, x }) catch return error.WriteFailed; - w.print("sp 0x{x}\npc 0x{x}\nfp 0x{x}\nlr 0x{x}\n", .{ - ctx.sp, ctx.pc, ctx.x[29], ctx.x[30], - }) catch return error.WriteFailed; - } else { - w.print("pc 0x{x}\nfp 0x{x}\n", .{ ctx.getPc(), ctx.getFp() }) catch return error.WriteFailed; - } -} - -// --------------------------------------------------------------------- tests - -const testing = std.testing; - -fn testOptions(text_buf: []u8) Options { - return .{ .io = testing.io, .text_buf = text_buf }; -} - -noinline fn sleepMs(ms: u64) void { - var ts: linux.timespec = .{ .sec = @intCast(ms / 1000), .nsec = @intCast((ms % 1000) * std.time.ns_per_ms) }; - _ = linux.nanosleep(&ts, null); -} - -// The test threads use atomic builtins rather than `std.atomic.Value` methods -// so that, in release modes, their pc is never inside an inlined callee: the -// DWARF symbolizer names the innermost inlined function at an address (see -// the notes on `writeFrames`). -const SpinState = struct { - tid: std.atomic.Value(u32) = .init(0), - stop: bool = false, - counter: u32 = 0, - done: bool = false, -}; - -noinline fn spinHere(st: *SpinState) void { - while (!@atomicLoad(bool, &st.stop, .acquire)) { - _ = @atomicRmw(u32, &st.counter, .Add, 1, .monotonic); - } -} - -fn spinThreadMain(st: *SpinState) void { - st.tid.store(selfTid(), .release); - spinHere(st); - @atomicStore(bool, &st.done, true, .release); // keeps the call above from becoming a tail call -} - -fn waitForTid(st: *SpinState) u32 { - var tries: usize = 0; - while (st.tid.load(.acquire) == 0) : (tries += 1) { - if (tries > 2000) return 0; - sleepMs(1); - } - return st.tid.load(.acquire); -} - -test "capture own stack" { - var text_buf: [16 * 1024]u8 = undefined; - var d: Debug = undefined; - try d.init(testOptions(&text_buf)); - defer d.deinit(); - try testing.expect(current == &d); - - var out: Writer.Allocating = .init(testing.allocator); - defer out.deinit(); - try d.threadStack(selfTid(), &out.writer); - const text = out.written(); - try testing.expect(std.mem.indexOf(u8, text, "#0 0x") != null); - try testing.expect(std.mem.indexOf(u8, text, "debug.zig:") != null); - try testing.expect(std.mem.indexOf(u8, text, "test.capture own stack") != null); - - out.clearRetainingCapacity(); - try d.threadRegs(selfTid(), &out.writer); - try testing.expect(std.mem.indexOf(u8, out.written(), "pc 0x") != null); - try testing.expect(std.mem.indexOf(u8, out.written(), "pc 0x0\n") == null); -} - -test "capture another thread: stack, regs, name, stat" { - var text_buf: [16 * 1024]u8 = undefined; - var d: Debug = undefined; - try d.init(testOptions(&text_buf)); - defer d.deinit(); - - var st: SpinState = .{}; - const th = try std.Thread.spawn(.{}, spinThreadMain, .{&st}); - const tid = waitForTid(&st); - try testing.expect(tid != 0); - - var out: Writer.Allocating = .init(testing.allocator); - defer out.deinit(); - try d.threadStack(tid, &out.writer); - try testing.expect(std.mem.indexOf(u8, out.written(), "spinHere") != null); - try testing.expect(std.mem.indexOf(u8, out.written(), "spinThreadMain") != null); - - out.clearRetainingCapacity(); - try d.threadRegs(tid, &out.writer); - try testing.expect(std.mem.indexOf(u8, out.written(), "pc 0x") != null); - try testing.expect(std.mem.indexOf(u8, out.written(), "pc 0x0\n") == null); - - out.clearRetainingCapacity(); - try d.threadName(tid, &out.writer); - try testing.expect(out.written().len > 0); - try testing.expect(std.mem.indexOfScalar(u8, out.written(), '\n') == null); - - out.clearRetainingCapacity(); - try d.threadStat(tid, &out.writer); - try testing.expect(std.mem.startsWith(u8, out.written(), "state ")); - try testing.expect(std.mem.indexOf(u8, out.written(), "\nutime ") != null); - - // Enumeration lists both threads and nothing bogus. - try testing.expect(d.threadExists(tid)); - try testing.expect(d.threadExists(selfTid())); - var found_self = false; - var found_other = false; - var i: usize = 0; - var prev: u32 = 0; - while (d.threadAt(i)) |t| : (i += 1) { - try testing.expect(t > prev); - prev = t; - if (t == tid) found_other = true; - if (t == selfTid()) found_self = true; - } - try testing.expect(found_self and found_other); - - // Repeated captures of the same thread keep working. - var k: usize = 0; - while (k < 5) : (k += 1) { - out.clearRetainingCapacity(); - try d.threadStack(tid, &out.writer); - try testing.expect(std.mem.indexOf(u8, out.written(), "spinHere") != null); - } - const before = @atomicLoad(u32, &st.counter, .acquire); - sleepMs(2); - try testing.expect(@atomicLoad(u32, &st.counter, .acquire) != before); // the thread is running again - - @atomicStore(bool, &st.stop, true, .release); - th.join(); - try testing.expect(!d.threadExists(tid)); - try testing.expectError(error.NoThread, d.threadStack(tid, &out.writer)); - try testing.expectError(error.NoThread, d.threadName(tid, &out.writer)); -} - -/// The address of the call site in the caller, i.e. inside this file's test. -noinline fn callerAddress() usize { - return @returnAddress() - 1; -} - -test "resolveAddr names this file" { - var text_buf: [16 * 1024]u8 = undefined; - var d: Debug = undefined; - try d.init(testOptions(&text_buf)); - defer d.deinit(); - var out: Writer.Allocating = .init(testing.allocator); - defer out.deinit(); - try d.resolveAddr(callerAddress(), &out.writer); - const text = out.written(); - var lines = std.mem.splitScalar(u8, text, '\n'); - const fn_name = lines.next().?; - const loc = lines.next().?; - const module = lines.next().?; - try testing.expect(fn_name.len > 0 and !std.mem.eql(u8, fn_name, "?")); - try testing.expect(std.mem.indexOf(u8, loc, "debug.zig:") != null); - try testing.expect(module.len > 0); - - out.clearRetainingCapacity(); - try d.resolveAddr(8, &out.writer); - try testing.expectEqualStrings("?\n?\n?\n", out.written()); - - // Regression: an unmapped lookup must not poison std's unwind cache (see - // the header); unwinding afterwards still works. - out.clearRetainingCapacity(); - try d.threadStack(selfTid(), &out.writer); - try testing.expect(std.mem.indexOf(u8, out.written(), "test.resolveAddr names this file") != null); -} - -test "readMem, writeMem, hexdump" { - var text_buf: [16 * 1024]u8 = undefined; - var d: Debug = undefined; - try d.init(testOptions(&text_buf)); - defer d.deinit(); - - var value: [8]u8 = .{ 1, 2, 3, 4, 5, 6, 7, 8 }; - var got: [8]u8 = undefined; - try testing.expectEqual(@as(usize, 8), try d.readMem(@intFromPtr(&value), &got)); - try testing.expectEqualSlices(u8, &value, &got); - try testing.expectError(error.Unmapped, d.readMem(8, &got)); - - const new = [_]u8{ 0xaa, 0xbb, 0xcc }; - try testing.expectEqual(@as(usize, 3), try d.writeMem(@intFromPtr(&value) + 2, &new)); - try testing.expectEqualSlices(u8, &.{ 1, 2, 0xaa, 0xbb, 0xcc, 6, 7, 8 }, &value); - try testing.expectError(error.Unmapped, d.writeMem(8, &new)); - - var bytes: [19]u8 = .{ 0x00, 0x11, 0x22, 0x33, 0x44, 0x55, 0x66, 0x77, 0x88, 0x99, 0xaa, 0xbb, 0xcc, 0xdd, 0xee, 0xff, 0x01, 0x12, 0x13 }; - var out: Writer.Allocating = .init(testing.allocator); - defer out.deinit(); - try d.hexdump(@intFromPtr(&bytes), bytes.len, &out.writer); - const expected = try std.fmt.allocPrint(testing.allocator, - \\{x:0>[2]} 00 11 22 33 44 55 66 77 88 99 AA BB CC DD EE FF .."3DUfw........ - \\{x:0>[2]} 01 12 13 ... - \\ - , .{ @intFromPtr(&bytes), @intFromPtr(&bytes) + 16, @sizeOf(usize) * 2 }); - defer testing.allocator.free(expected); - try testing.expectEqualStrings(expected, out.written()); - try testing.expectError(error.Unmapped, d.hexdump(8, 16, &out.writer)); - - // Address 0 (also reached by an offset that wraps) must be an error, not a - // safety-checked null pointer cast on the server thread. - try testing.expectError(error.Unmapped, d.readMem(0, &got)); - try testing.expectError(error.Unmapped, d.writeMem(0, &new)); - try testing.expectError(error.Unmapped, d.hexdump(0, 16, &out.writer)); - try testing.expectError(error.Unmapped, d.readMem(std.math.maxInt(usize) - 3, &got)); - try testing.expectError(error.Unmapped, d.hexdump(std.math.maxInt(usize) - 3, 16, &out.writer)); - - out.clearRetainingCapacity(); - try d.maps(&out.writer); - try testing.expect(std.mem.indexOf(u8, out.written(), "[stack]") != null); - - // readMaps serves the file piecewise at any offset and ends with 0. - var piece: [4096]u8 = undefined; - var total: usize = 0; - while (true) { - const n = try d.readMaps(total, &piece); - if (n == 0) break; - total += n; - } - try testing.expect(total >= out.written().len / 2); - try testing.expectEqual(@as(usize, 0), try d.readMaps(std.math.maxInt(u64), &piece)); -} - -test "breakpoint on the server thread and past the slot table steps over; tgkill SIGTRAP parks" { - if (arch != .x86_64 and !arch.isAARCH64()) return error.SkipZigTest; - var text_buf: [16 * 1024]u8 = undefined; - var d: Debug = undefined; - var opts = testOptions(&text_buf); - opts.max_paused = 1; - try d.init(opts); - defer d.deinit(); - try d.enableBreakpoints(); - defer d.disableBreakpoints(); - var out: Writer.Allocating = .init(testing.allocator); - defer out.deinit(); - - // The "server" thread (this one, for the test) hits a breakpoint: it keeps running. - const skipped0 = traps_skipped.load(.acquire); - server_tid.store(selfTid(), .release); - defer server_tid.store(0, .release); - @breakpoint(); - try testing.expectEqual(skipped0 + 1, traps_skipped.load(.acquire)); - try testing.expect(!d.isPaused(selfTid())); - - // One slot: the first trapping thread parks, the second steps over. - var a: TrapState = .{}; - const ta = try std.Thread.spawn(.{}, trapThreadMain, .{&a}); - var tries: usize = 0; - while (a.tid.load(.acquire) == 0 or !d.isPaused(a.tid.load(.acquire))) : (tries += 1) { - try testing.expect(tries < 5000); - sleepMs(1); - } - var b: TrapState = .{}; - const tb = try std.Thread.spawn(.{}, trapThreadMain, .{&b}); - tb.join(); - try testing.expectEqual(@as(u32, 1), @atomicLoad(u32, &b.counter, .acquire)); - try testing.expectEqual(skipped0 + 2, traps_skipped.load(.acquire)); - try testing.expectEqual(@as(u32, 0), @atomicLoad(u32, &a.counter, .acquire)); - try d.resumeThread(a.tid.load(.acquire)); - ta.join(); - try testing.expectEqual(@as(u32, 1), @atomicLoad(u32, &a.counter, .acquire)); - - // A SIGTRAP sent with tgkill (not an int3/brk) parks the thread where it - // was; resuming it must not skip an instruction: the spinner keeps counting. - var st: SpinState = .{}; - const th = try std.Thread.spawn(.{}, spinThreadMain, .{&st}); - const tid = waitForTid(&st); - try testing.expect(tid != 0); - try testing.expectEqual(linux.E.SUCCESS, linux.errno(linux.tgkill(linux.getpid(), @intCast(tid), .TRAP))); - tries = 0; - while (!d.isPaused(tid)) : (tries += 1) { - try testing.expect(tries < 5000); - sleepMs(1); - } - const frozen = @atomicLoad(u32, &st.counter, .acquire); - sleepMs(5); - try testing.expectEqual(frozen, @atomicLoad(u32, &st.counter, .acquire)); - out.clearRetainingCapacity(); - try d.pausedStack(tid, &out.writer); - try testing.expect(std.mem.indexOf(u8, out.written(), "spinHere") != null); - try d.resumeThread(tid); - sleepMs(5); - try testing.expect(@atomicLoad(u32, &st.counter, .acquire) != frozen); - @atomicStore(bool, &st.stop, true, .release); - th.join(); -} - -const LockState = struct { - tid: std.atomic.Value(u32) = .init(0), - release: std.atomic.Value(bool) = .init(false), - unlocked: std.atomic.Value(bool) = .init(false), - stop: std.atomic.Value(bool) = .init(false), - io: std.Io, -}; - -fn lockHolderMain(st: *LockState) void { - const di = std.debug.getSelfDebugInfo() catch return; - di.rwlock.lockUncancelable(st.io); - st.tid.store(selfTid(), .release); - while (!st.release.load(.acquire)) sleepMs(1); - di.rwlock.unlock(st.io); - st.unlocked.store(true, .release); - while (!st.stop.load(.acquire)) sleepMs(1); -} - -test "a target parked while holding std.debug's lock is Busy, not a deadlock" { - if (comptime !@hasField(std.debug.SelfInfo, "rwlock")) return error.SkipZigTest; - var text_buf: [16 * 1024]u8 = undefined; - var d: Debug = undefined; - try d.init(testOptions(&text_buf)); - defer d.deinit(); - var st: LockState = .{ .io = testing.io }; - const th = try std.Thread.spawn(.{}, lockHolderMain, .{&st}); - var tries: usize = 0; - while (st.tid.load(.acquire) == 0) : (tries += 1) { - try testing.expect(tries < 2000); - sleepMs(1); - } - const tid = st.tid.load(.acquire); - // No allocation while the holder has the lock: `testing.allocator` - // records a stack trace per allocation, which needs that same lock. - var buf: [16 * 1024]u8 = undefined; - var w: Writer = .fixed(&buf); - try testing.expectError(error.Busy, d.threadStack(tid, &w)); - try testing.expectEqual(cap_idle, d.capture.state.load(.acquire)); - // Registers need no unwind and are still available. - try d.threadRegs(tid, &w); - try testing.expect(std.mem.indexOf(u8, w.buffered(), "pc 0x") != null); - // Handshake, not a sleep: a slow holder would otherwise still hold the - // lock and the next capture would legitimately be Busy again. - st.release.store(true, .release); - tries = 0; - while (!st.unlocked.load(.acquire)) : (tries += 1) { - try testing.expect(tries < 5000); - sleepMs(1); - } - w = .fixed(&buf); - try d.threadStack(tid, &w); - try testing.expect(std.mem.indexOf(u8, w.buffered(), "lockHolderMain") != null); - st.stop.store(true, .release); - th.join(); -} - -const TrapState = struct { - tid: std.atomic.Value(u32) = .init(0), - counter: u32 = 0, -}; - -noinline fn trapThreadMain(st: *TrapState) void { - st.tid.store(selfTid(), .release); - @breakpoint(); - _ = @atomicRmw(u32, &st.counter, .Add, 1, .acq_rel); -} - -test "breakpoint: pause, inspect, resume" { - if (arch != .x86_64 and !arch.isAARCH64()) return error.SkipZigTest; - var text_buf: [16 * 1024]u8 = undefined; - var d: Debug = undefined; - try d.init(testOptions(&text_buf)); - defer d.deinit(); - try d.enableBreakpoints(); - - var st: TrapState = .{}; - const th = try std.Thread.spawn(.{}, trapThreadMain, .{&st}); - var tries: usize = 0; - while (st.tid.load(.acquire) == 0 or !d.isPaused(st.tid.load(.acquire))) : (tries += 1) { - try testing.expect(tries < 5000); - sleepMs(1); - } - const tid = st.tid.load(.acquire); - try testing.expectEqual(@as(?u32, tid), d.pausedAt(0)); - try testing.expectEqual(@as(?u32, null), d.pausedAt(1)); - try testing.expectEqual(@as(u32, 0), @atomicLoad(u32, &st.counter, .acquire)); - - var out: Writer.Allocating = .init(testing.allocator); - defer out.deinit(); - try d.pausedStack(tid, &out.writer); - try testing.expect(std.mem.indexOf(u8, out.written(), "trapThreadMain") != null); - out.clearRetainingCapacity(); - try d.pausedRegs(tid, &out.writer); - try testing.expect(std.mem.indexOf(u8, out.written(), "pc 0x") != null); - - // A paused thread can also be captured through the signal path. - out.clearRetainingCapacity(); - try d.threadStack(tid, &out.writer); - try testing.expect(std.mem.indexOf(u8, out.written(), "#0 0x") != null); - - sleepMs(5); - try testing.expectEqual(@as(u32, 0), @atomicLoad(u32, &st.counter, .acquire)); - try d.resumeThread(tid); - th.join(); - try testing.expectEqual(@as(u32, 1), @atomicLoad(u32, &st.counter, .acquire)); - try testing.expect(!d.isPaused(tid)); - try testing.expectEqual(@as(?u32, null), d.pausedAt(0)); - try testing.expectError(error.NotPaused, d.resumeThread(tid)); - try testing.expectError(error.NotPaused, d.pausedStack(tid, &out.writer)); - d.disableBreakpoints(); -} - -/// Stands in for `FullPanic`'s call: the first trace address is the return -/// address into the panicking function. -noinline fn panicLike(msg: []const u8) bool { - return recordPanic(msg, @returnAddress()); -} - -test "panic record path" { - var text_buf: [16 * 1024]u8 = undefined; - var d: Debug = undefined; - try d.init(testOptions(&text_buf)); - defer d.deinit(); - defer resetPanicRecord(); - - var out: Writer.Allocating = .init(testing.allocator); - defer out.deinit(); - try d.panicMessage(&out.writer); - try testing.expectEqualStrings("", out.written()); - try testing.expect(!d.panicHeld()); - try testing.expectError(error.NoPanic, d.panicContinue()); - - try testing.expect(panicLike("something broke")); - try testing.expect(!recordPanic("nested", null)); - try testing.expectEqual(selfTid(), panic_tid); - - try d.panicMessage(&out.writer); - try testing.expectEqualStrings("something broke", out.written()); - out.clearRetainingCapacity(); - try d.panicStack(&out.writer); - try testing.expect(std.mem.indexOf(u8, out.written(), "#0 0x") != null); - try testing.expect(std.mem.indexOf(u8, out.written(), "test.panic record path") != null); - try testing.expect(!d.panicHeld()); - try testing.expectError(error.NoPanic, d.panicContinue()); - - // A long message is truncated, not overflowed. - resetPanicRecord(); - const long = [_]u8{'x'} ** (panic_msg_cap + 100); - try testing.expect(recordPanic(&long, null)); - out.clearRetainingCapacity(); - try d.panicMessage(&out.writer); - try testing.expectEqual(@as(usize, panic_msg_cap), out.written().len); -} - -const MaskState = struct { - tid: std.atomic.Value(u32) = .init(0), - unblock: std.atomic.Value(bool) = .init(false), - stop: std.atomic.Value(bool) = .init(false), - signal: linux.SIG, -}; - -fn maskedThreadMain(st: *MaskState) void { - var set = linux.sigemptyset(); - linux.sigaddset(&set, st.signal); - _ = linux.sigprocmask(linux.SIG.BLOCK, &set, null); - st.tid.store(selfTid(), .release); - while (!st.unblock.load(.acquire)) sleepMs(1); - _ = linux.sigprocmask(linux.SIG.UNBLOCK, &set, null); - while (!st.stop.load(.acquire)) sleepMs(1); -} - -test "capture timeout on a thread with the signal masked" { - var text_buf: [16 * 1024]u8 = undefined; - var d: Debug = undefined; - var opts = testOptions(&text_buf); - opts.capture_timeout_ns = 50 * std.time.ns_per_ms; - try d.init(opts); - defer d.deinit(); - - var st: MaskState = .{ .signal = d.capture_signal }; - const th = try std.Thread.spawn(.{}, maskedThreadMain, .{&st}); - var tries: usize = 0; - while (st.tid.load(.acquire) == 0) : (tries += 1) { - try testing.expect(tries < 2000); - sleepMs(1); - } - const masked_tid = st.tid.load(.acquire); - - var out: Writer.Allocating = .init(testing.allocator); - defer out.deinit(); - const t0 = monotonicNs(); - try testing.expectError(error.Timeout, d.threadStack(masked_tid, &out.writer)); - try testing.expect(monotonicNs() - t0 >= 50 * std.time.ns_per_ms); - try testing.expectEqual(cap_idle, d.capture.state.load(.acquire)); - - // The process is healthy: another thread can still be captured... - var spin: SpinState = .{}; - const spinner = try std.Thread.spawn(.{}, spinThreadMain, .{&spin}); - const spin_tid = waitForTid(&spin); - try testing.expect(spin_tid != 0); - out.clearRetainingCapacity(); - try d.threadStack(spin_tid, &out.writer); - try testing.expect(std.mem.indexOf(u8, out.written(), "spinHere") != null); - - // ...and the late delivery of the pending signal is harmless. - st.unblock.store(true, .release); - sleepMs(20); - out.clearRetainingCapacity(); - try d.threadStack(spin_tid, &out.writer); - try testing.expect(std.mem.indexOf(u8, out.written(), "spinHere") != null); - out.clearRetainingCapacity(); - try d.threadStack(masked_tid, &out.writer); - try testing.expect(std.mem.indexOf(u8, out.written(), "maskedThreadMain") != null); - - @atomicStore(bool, &spin.stop, true, .release); - spinner.join(); - st.stop.store(true, .release); - th.join(); -} - -test "options validation and single instance" { - var text_buf: [4096]u8 = undefined; - var d: Debug = undefined; - var opts = testOptions(&text_buf); - opts.capture_signal = 5; - try testing.expectError(error.InvalidOptions, d.init(opts)); - opts = testOptions(&text_buf); - opts.max_paused = max_paused_cap + 1; - try testing.expectError(error.InvalidOptions, d.init(opts)); - try d.init(testOptions(&text_buf)); - defer d.deinit(); - var d2: Debug = undefined; - try testing.expectError(error.AlreadyInitialized, d2.init(testOptions(&text_buf))); -} diff --git a/introspect/src/linux/probe.zig b/introspect/src/linux/probe.zig deleted file mode 100644 index f38c491..0000000 --- a/introspect/src/linux/probe.zig +++ /dev/null @@ -1,829 +0,0 @@ -//! The Linux platform layer: one background thread runs a `poll()` loop over -//! a listener and every client connection, feeding each connection's core -//! `Conn` with `push`/`step`/`output`/`wrote`. No per-connection threads, no -//! allocation after `init`; every buffer lives in a caller-placed `Storage`. -//! -//! Also home of the debug facilities (`debug`, `provider`) and the /runtime -//! generators (`runtime`). See docs/LIBRARY.md. -//! -//! Client admission: a new connection takes a free slot. When every slot is -//! taken, the connection that has held a slot without any fid (never -//! attached, or fully clunked) for longer than `evict_idle_ms` is dropped in -//! its favour; if there is none, the new connection is closed ("refused"). -//! Nothing that holds a fid is ever evicted. -//! -//! `sleepServing(ms)` lets a request handler (a ctl command, say) wait -//! without stalling the other clients: called on the probe thread from inside -//! a request it keeps running the poll loop for every client whose request -//! is not in progress until the time is up. Requests served from inside such -//! a wait may wait themselves, up to `max_nested_sleeps` deep (each level is a -//! different client, so the depth is bounded by the client table anyway); the -//! level past that, and any call off the probe thread, is a plain sleep. -const std = @import("std"); -const builtin = @import("builtin"); -const linux = std.os.linux; -const cloud9 = @import("cloud9"); -const core = @import("../core.zig"); - -pub const debug = @import("debug.zig"); -pub const provider = @import("provider.zig"); -pub const runtime = @import("runtime.zig"); -pub const DebugProvider = provider.DebugProvider; - -pub const Listen = union(enum) { - /// A unix socket path (< 108 bytes); a stale socket file is unlinked first. - unix: []const u8, - /// An IPv4 literal "a.b.c.d:port". - tcp: []const u8, - /// An already listening socket, owned by the caller. - fd: i32, - /// One pre-connected client on these descriptors (stdio: 0 and 1). Nothing - /// is accepted and the loop ends when the client hangs up. - client: struct { in: i32, out: i32 }, -}; - -pub const Options = struct { - /// For `std.debug` symbolization. - io: std.Io, - listen: Listen, - /// Largest msize offered to clients (clamped to the server's `cfg.msize`). - msize: u32 = 64 * 1024, - /// Hold a panicking thread until /panic/ctl says "continue". - hold_on_panic: bool = true, - /// Real-time signal used to snapshot other threads. - capture_signal: u8 = debug.default_capture_signal, - /// Install the SIGTRAP handler so `@breakpoint()` parks the thread. - breakpoints: bool = true, - /// Mount /threads, /addr, /mem, /hex, /breakpoints, /panic (six provider slots). - mount_debug: bool = true, -}; - -pub const Error = error{ - /// Another `Debug` (another probe) exists in this process. - AlreadyInitialized, - /// No register capture on this architecture. - Unsupported, - /// `Shared` has fewer than six free provider slots. - TooManyProviders, - PathTooLong, - BadAddress, - /// A syscall failed; `last_errno` says which error. - Syscall, -}; - -/// Idle time without fids after which a slot holder may be evicted. -pub const evict_idle_ms: i64 = 500; -/// Largest number of connections accepted per poll wakeup. -const accept_burst = 64; -/// How deep `sleepServing` may nest (each level keeps a poll round on the stack). -pub const max_nested_sleeps = 8; - -/// Static per-client storage: `max_clients` core `Storage`s and `Conn`s, the -/// poll table, the debug text arena and the debug provider's snapshot pool. -pub fn Storage(comptime max_clients: u8, comptime Srv: type) type { - return Probe(Srv).Storage(max_clients); -} - -pub fn Probe(comptime Srv: type) type { - return struct { - const Self = @This(); - - pub fn Storage(comptime max_clients: u8) type { - comptime std.debug.assert(max_clients > 0); - return struct { - pub const capacity = max_clients; - conns: [max_clients]Srv.Storage, - clients: [max_clients]Client, - /// [0] wake eventfd, [1] listener, [2..] one per client slot. - pollfds: [max_clients + 2]linux.pollfd, - text_buf: [16 * 1024]u8, - dp: DebugProvider, - }; - } - - pub const Client = struct { - conn: Srv.Conn, - in: i32 = -1, - out: i32 = -1, - used: bool = false, - /// Descriptors we opened (accepted) are closed on drop; borrowed ones are not. - owned: bool = false, - /// Send with MSG_NOSIGNAL; falls back to write(2) on ENOTSOCK. - is_socket: bool = true, - /// Monotonic ms of the last byte received. - last_active: i64 = 0, - }; - - shared: *Srv.Shared, - clients: []Client, - conns: []Srv.Storage, - pollfds: []linux.pollfd, - dbg: debug.Debug, - dp: *DebugProvider, - msize: u32, - listen_fd: i32 = -1, - own_listener: bool = false, - is_tcp: bool = false, - single: bool = false, - wake_fd: i32 = -1, - unix_path: [108]u8 = undefined, - unix_len: usize = 0, - thread: ?std.Thread = null, - thread_tid: std.atomic.Value(u32) = .init(0), - nclients: std.atomic.Value(u32) = .init(0), - /// Connections closed because no slot was free. - refused: u64 = 0, - stopping: std.atomic.Value(bool) = .init(false), - /// Slots whose request is being handled (excluded from nested servicing and eviction). - serving: std.StaticBitSet(256) = .initEmpty(), - /// Current `sleepServing` nesting depth. - nested: u8 = 0, - debug_ready: bool = false, - last_errno: linux.E = .SUCCESS, - - // -- lifecycle ------------------------------------------------------- - - /// Installs the debug facilities, mounts the debug providers into - /// `shared` and opens the listener. `storage` is a `*Storage(n)`. On - /// failure `shared` may already hold the debug providers and must be - /// discarded. - pub fn init(p: *Self, shared: *Srv.Shared, storage: anytype, opts: Options) Error!void { - p.* = .{ - .shared = shared, - .clients = &storage.clients, - .conns = &storage.conns, - .pollfds = &storage.pollfds, - .dbg = undefined, - .dp = &storage.dp, - .msize = opts.msize, - }; - for (p.clients) |*c| c.used = false; - shared.hash_seed = randomSeed(); - debug.hold_on_panic = opts.hold_on_panic; - p.dbg.init(.{ .io = opts.io, .text_buf = &storage.text_buf, .capture_signal = opts.capture_signal }) catch |e| return switch (e) { - error.AlreadyInitialized => error.AlreadyInitialized, - error.Unsupported => error.Unsupported, - else => error.Syscall, - }; - p.debug_ready = true; - errdefer { - p.dbg.deinit(); - p.debug_ready = false; - } - if (opts.breakpoints) p.dbg.enableBreakpoints() catch |e| switch (e) { - // No breakpoint support on this architecture: everything else still works. - error.Unsupported => {}, - else => return error.Syscall, - }; - if (opts.mount_debug) { - p.dp.init(&p.dbg); - p.dp.mountAll(shared) catch return error.TooManyProviders; - } - const efd = linux.eventfd(0, linux.EFD.CLOEXEC | linux.EFD.NONBLOCK); - try p.check(efd); - p.wake_fd = @intCast(efd); - errdefer { - _ = linux.close(p.wake_fd); - p.wake_fd = -1; - } - switch (opts.listen) { - .unix => |path| try p.listenUnix(path), - .tcp => |text| try p.listenTcp(text), - .fd => |fd| { - try p.setNonblock(fd); - p.listen_fd = fd; - }, - .client => |c| { - p.single = true; - try p.setNonblock(c.in); - if (c.out != c.in) try p.setNonblock(c.out); - _ = p.addClient(c.in, c.out, false); - }, - } - } - - /// Spawns the poll thread. - pub fn start(p: *Self) std.Thread.SpawnError!void { - std.debug.assert(p.thread == null); - p.stopping.store(false, .release); - p.thread = try std.Thread.spawn(.{}, run, .{p}); - } - - /// Waits for the poll thread to end (only happens by itself in - /// `.client` mode, when the client hangs up). - pub fn wait(p: *Self) void { - if (p.thread) |t| { - t.join(); - p.thread = null; - } - } - - /// Stops the poll thread, drops every client, closes what `init` - /// opened and restores the signal dispositions. - pub fn stop(p: *Self) void { - // Joining the poll thread from itself would hang forever; a - // request handler that wants the server gone uses `requestStop`. - std.debug.assert(p.thread_tid.load(.acquire) != @as(u32, @intCast(linux.gettid()))); - p.stopping.store(true, .release); - p.wakeLoop(); - p.wait(); - for (p.clients, 0..) |*c, i| if (c.used) p.dropClient(i); - if (p.listen_fd >= 0) { - if (p.own_listener) _ = linux.close(p.listen_fd); - p.listen_fd = -1; - } - if (p.unix_len > 0) { - _ = linux.unlink(@ptrCast(&p.unix_path)); - p.unix_len = 0; - } - if (p.wake_fd >= 0) { - _ = linux.close(p.wake_fd); - p.wake_fd = -1; - } - if (p.debug_ready) { - p.dbg.deinit(); - p.debug_ready = false; - } - } - - /// Live client count (for /runtime/clients). - pub fn clientCount(p: *const Self) u32 { - return p.nclients.load(.acquire); - } - - pub fn clientCounter(p: *const Self) *const std.atomic.Value(u32) { - return &p.nclients; - } - - /// Waits `ms` while keeping the other clients served (see the file comment). - pub fn sleepServing(p: *Self, ms: u64) void { - const on_thread = p.thread_tid.load(.acquire) == @as(u32, @intCast(linux.gettid())); - if (!on_thread or p.serving.count() == 0 or p.nested >= max_nested_sleeps) return sleepMs(ms); - p.nested += 1; - defer p.nested -= 1; - const deadline = monotonicMs() + @as(i64, @intCast(@min(ms, std.math.maxInt(i32)))); - while (!p.stopping.load(.acquire)) { - const now = monotonicMs(); - if (now >= deadline) break; - p.pollOnce(@intCast(deadline - now)); - } - } - - /// Asks the poll thread to stop; safe to call from a signal handler - /// (an atomic store and one write to the wake eventfd). `stop` (or - /// `wait`) still has to run afterwards to release everything. - pub fn requestStop(p: *Self) void { - p.stopping.store(true, .release); - p.wakeLoop(); - } - - // -- the loop -------------------------------------------------------- - - fn run(p: *Self) void { - const tid: u32 = @intCast(linux.gettid()); - p.thread_tid.store(tid, .release); - // A breakpoint or panic on this thread must never park it (see debug.zig). - debug.server_tid.store(tid, .release); - setThreadName("introspect"); - while (!p.stopping.load(.acquire)) { - if (p.single and p.clientCount() == 0) break; - p.pollOnce(-1); - } - debug.server_tid.store(0, .release); - p.thread_tid.store(0, .release); - } - - fn wakeLoop(p: *Self) void { - if (p.wake_fd < 0) return; - const one: u64 = 1; - _ = linux.write(p.wake_fd, @ptrCast(&one), 8); - } - - /// One `poll()` round: accept, read, step, write. Slots whose request - /// is in progress (`serving`, only inside `sleepServing`) are left untouched. - fn pollOnce(p: *Self, timeout_ms: i32) void { - p.pollfds[0] = .{ .fd = p.wake_fd, .events = linux.POLL.IN, .revents = 0 }; - p.pollfds[1] = .{ .fd = p.listen_fd, .events = linux.POLL.IN, .revents = 0 }; - for (p.clients, 0..) |*c, i| { - var fd: i32 = -1; - var events: i16 = 0; - if (c.used and !p.serving.isSet(i)) { - fd = c.in; - if (c.conn.output().len > 0) { - fd = c.out; - events = linux.POLL.OUT; - } else if (inputRoom(&c.conn) > 0) { - events = linux.POLL.IN; - } - } - p.pollfds[2 + i] = .{ .fd = fd, .events = events, .revents = 0 }; - } - const rc = linux.poll(p.pollfds.ptr, p.pollfds.len, timeout_ms); - switch (linux.errno(rc)) { - .SUCCESS => {}, - .INTR => return, - else => { - sleepMs(10); - return; - }, - } - if (p.pollfds[0].revents != 0) { - var v: u64 = 0; - _ = linux.read(p.wake_fd, @ptrCast(&v), 8); - } - if (p.stopping.load(.acquire)) return; - if (p.pollfds[1].revents != 0) p.acceptSome(); - for (p.clients, 0..) |*c, i| { - const re = p.pollfds[2 + i].revents; - if (re == 0 or !c.used or p.serving.isSet(i)) continue; - if (re & (linux.POLL.IN | linux.POLL.HUP | linux.POLL.ERR | linux.POLL.NVAL) != 0) { - p.readClient(i, re & linux.POLL.HUP != 0); - } else if (re & linux.POLL.OUT != 0) { - p.service(i); - } - if (p.stopping.load(.acquire)) return; - } - } - - fn acceptSome(p: *Self) void { - var n: usize = 0; - while (n < accept_burst) : (n += 1) { - const rc = linux.accept4(p.listen_fd, null, null, linux.SOCK.NONBLOCK | linux.SOCK.CLOEXEC); - switch (linux.errno(rc)) { - .SUCCESS => {}, - .AGAIN => return, - .INTR, .CONNABORTED => continue, - // Descriptor/memory exhaustion is transient (clients hang up); - // back off instead of spinning on a readable listener. - .MFILE, .NFILE, .NOBUFS, .NOMEM, .PERM => { - sleepMs(100); - return; - }, - else => return, - } - const cfd: i32 = @intCast(rc); - if (p.is_tcp) { - const one: u32 = 1; - _ = linux.setsockopt(cfd, linux.IPPROTO.TCP, linux.TCP.NODELAY, @ptrCast(&one), @sizeOf(u32)); - } - if (p.addClient(cfd, cfd, true) != null) continue; - if (p.evictable()) |victim| { - p.dropClient(victim); - _ = p.addClient(cfd, cfd, true); - continue; - } - p.refused += 1; - // A flood must not flood stderr. - if (p.refused == 1 or p.refused % 1000 == 0) - std.debug.print("introspect: refused connection ({d} clients open, {d} refused so far)\n", .{ p.clients.len, p.refused }); - _ = linux.close(cfd); - } - } - - /// The longest-idle slot holder without fids, if idle long enough. - fn evictable(p: *Self) ?usize { - const now = monotonicMs(); - var best: ?usize = null; - for (p.clients, 0..) |*c, i| { - if (!c.used or !c.owned) continue; - if (p.serving.isSet(i)) continue; - if (c.conn.fidCount() != 0) continue; - if (now - c.last_active < evict_idle_ms) continue; - if (best == null or c.last_active < p.clients[best.?].last_active) best = i; - } - return best; - } - - fn addClient(p: *Self, in: i32, out: i32, owned: bool) ?usize { - for (p.clients, 0..) |*c, i| { - if (c.used) continue; - c.conn = Srv.Conn.init(p.shared, &p.conns[i], p.msize); - c.in = in; - c.out = out; - c.used = true; - c.owned = owned; - c.is_socket = true; - c.last_active = monotonicMs(); - _ = p.nclients.fetchAdd(1, .acq_rel); - return i; - } - return null; - } - - fn dropClient(p: *Self, i: usize) void { - const c = &p.clients[i]; - if (!c.used) return; - c.conn.hangup(); - if (c.owned) { - _ = linux.close(c.in); - if (c.out != c.in) _ = linux.close(c.out); - } - c.used = false; - c.in = -1; - c.out = -1; - _ = p.nclients.fetchSub(1, .acq_rel); - } - - /// Free space in the connection's input buffer (cloud9 keeps one - /// msize-sized frame; `push` copies at most this much). - fn inputRoom(conn: *const Srv.Conn) usize { - return conn.server.in.len - conn.server.in_len; - } - - fn readClient(p: *Self, i: usize, hup: bool) void { - const c = &p.clients[i]; - var buf: [64 * 1024]u8 = undefined; - const room = inputRoom(&c.conn); - if (room == 0) return p.service(i); - const want = @min(room, buf.len); - while (true) { - const rc = linux.read(c.in, &buf, want); - switch (linux.errno(rc)) { - .SUCCESS => { - if (rc == 0) return p.dropClient(i); - const taken = c.conn.push(buf[0..rc]); - std.debug.assert(taken == rc); - c.last_active = monotonicMs(); - return p.service(i); - }, - .INTR => continue, - .AGAIN => { - if (hup) p.dropClient(i); - return; - }, - else => return p.dropClient(i), - } - } - } - - /// Runs requests and drains output until nothing moves. - fn service(p: *Self, i: usize) void { - const c = &p.clients[i]; - std.debug.assert(!p.serving.isSet(i)); - p.serving.set(i); - defer p.serving.unset(i); - while (c.used) { - var moved = false; - while (true) { - const more = c.conn.step() catch return p.dropClient(i); - if (!more) break; - moved = true; - } - const before = c.conn.output().len; - p.flush(c) catch return p.dropClient(i); - if (c.conn.output().len != before) moved = true; - if (!moved) return; - } - } - - fn flush(p: *Self, c: *Client) error{Closed}!void { - _ = p; - while (c.conn.output().len > 0) { - const chunk = c.conn.output(); - const rc = if (c.is_socket) - linux.sendto(c.out, chunk.ptr, chunk.len, linux.MSG.NOSIGNAL, null, 0) - else - linux.write(c.out, chunk.ptr, chunk.len); - switch (linux.errno(rc)) { - .SUCCESS => { - if (rc == 0) return; - c.conn.wrote(rc); - }, - .INTR => continue, - .AGAIN => return, - .NOTSOCK => c.is_socket = false, - else => return error.Closed, - } - } - } - - // -- listeners ------------------------------------------------------- - - fn check(p: *Self, rc: usize) Error!void { - const e = linux.errno(rc); - if (e != .SUCCESS) { - p.last_errno = e; - return error.Syscall; - } - } - - fn setNonblock(p: *Self, fd: i32) Error!void { - const rc = linux.fcntl(fd, linux.F.GETFL, 0); - try p.check(rc); - const nonblock: u32 = @bitCast(linux.O{ .NONBLOCK = true }); - try p.check(linux.fcntl(fd, linux.F.SETFL, rc | nonblock)); - } - - fn listenUnix(p: *Self, path: []const u8) Error!void { - var sa: linux.sockaddr.un = .{ .path = @splat(0) }; - if (path.len == 0 or path.len >= sa.path.len) return error.PathTooLong; - @memcpy(sa.path[0..path.len], path); - const rc = linux.socket(linux.AF.UNIX, linux.SOCK.STREAM | linux.SOCK.CLOEXEC | linux.SOCK.NONBLOCK, 0); - try p.check(rc); - const lfd: i32 = @intCast(rc); - errdefer _ = linux.close(lfd); - // No libc, so no "is it still listening" probe: unlink a stale socket and bind. - _ = linux.unlink(@ptrCast(&sa.path)); - try p.check(linux.bind(lfd, @ptrCast(&sa), @sizeOf(linux.sockaddr.un))); - try p.check(linux.listen(lfd, 128)); - p.listen_fd = lfd; - p.own_listener = true; - p.unix_path = sa.path; - p.unix_len = path.len; - } - - fn listenTcp(p: *Self, text: []const u8) Error!void { - const sa = parseIpv4(text) orelse return error.BadAddress; - const rc = linux.socket(linux.AF.INET, linux.SOCK.STREAM | linux.SOCK.CLOEXEC | linux.SOCK.NONBLOCK, 0); - try p.check(rc); - const lfd: i32 = @intCast(rc); - errdefer _ = linux.close(lfd); - const one: u32 = 1; - _ = linux.setsockopt(lfd, linux.SOL.SOCKET, linux.SO.REUSEADDR, @ptrCast(&one), @sizeOf(u32)); - try p.check(linux.bind(lfd, @ptrCast(&sa), @sizeOf(linux.sockaddr.in))); - try p.check(linux.listen(lfd, 128)); - p.listen_fd = lfd; - p.own_listener = true; - p.is_tcp = true; - } - }; -} - -/// Entropy for the core's fid hash (so fid numbers cannot be chosen to -/// collide); falls back to the clock if getrandom fails. -fn randomSeed() u32 { - var b: [4]u8 = undefined; - if (linux.errno(linux.getrandom(&b, b.len, 0)) == .SUCCESS) return std.mem.readInt(u32, &b, .little); - var ts: linux.timespec = undefined; - _ = linux.clock_gettime(.MONOTONIC, &ts); - return @truncate(@as(u64, @bitCast(ts.nsec)) ^ (@as(u64, @bitCast(ts.sec)) << 20)); -} - -/// "a.b.c.d:port" as a socket address, or null. -pub fn parseIpv4(text: []const u8) ?linux.sockaddr.in { - const colon = std.mem.lastIndexOfScalar(u8, text, ':') orelse return null; - const port = std.fmt.parseInt(u16, text[colon + 1 ..], 10) catch return null; - var octets: [4]u8 = undefined; - var it = std.mem.splitScalar(u8, text[0..colon], '.'); - for (&octets) |*o| o.* = std.fmt.parseInt(u8, it.next() orelse return null, 10) catch return null; - if (it.next() != null) return null; - return .{ .port = std.mem.nativeToBig(u16, port), .addr = @bitCast(octets) }; -} - -/// Names the calling thread (comm, at most 15 bytes) via prctl. -pub fn setThreadName(name: []const u8) void { - var buf: [16]u8 = @splat(0); - const n = @min(name.len, 15); - @memcpy(buf[0..n], name[0..n]); - _ = linux.prctl(@intFromEnum(linux.PR.SET_NAME), @intFromPtr(&buf), 0, 0, 0); -} - -pub fn sleepMs(ms: u64) void { - var req: linux.timespec = .{ .sec = @intCast(ms / 1000), .nsec = @intCast((ms % 1000) * 1_000_000) }; - var rem: linux.timespec = undefined; - while (linux.errno(linux.nanosleep(&req, &rem)) == .INTR) req = rem; -} - -pub fn monotonicMs() i64 { - var ts: linux.timespec = undefined; - _ = linux.clock_gettime(.MONOTONIC, &ts); - return ts.sec * 1000 + @divTrunc(ts.nsec, 1_000_000); -} - -// --------------------------------------------------------------------------- -// Tests: a real unix socket, a cloud9.Client on the other end. -// --------------------------------------------------------------------------- - -const testing = std.testing; - -test { - _ = debug; - _ = provider; - _ = runtime; -} - -const TestBuild = struct { - pub const zig_version: []const u8 = builtin.zig_version_string; - pub const target: []const u8 = "test"; - pub const optimize: []const u8 = "Debug"; - pub const time: []const u8 = "2024-01-01T00:00:00Z"; - pub const change: []const u8 = "none"; -}; - -const test_cfg: core.Config = .{ - .name = "probetest", - .build = TestBuild, - .msize = 8192, - .max_fids = 16, - .max_providers = 6, - .snapshot_slots = 2, - .snapshot_bytes = 1024, -}; -const TS = core.Server(test_cfg); -const TP = Probe(TS); - -/// A blocking client over a connected socket. -const SockClient = struct { - fd: i32, - client: cloud9.Client, - cin: [8192]u8 = undefined, - cout: [8192]u8 = undefined, - - fn connect(sc: *SockClient, path: []const u8) !void { - var sa: linux.sockaddr.un = .{ .path = @splat(0) }; - @memcpy(sa.path[0..path.len], path); - const rc = linux.socket(linux.AF.UNIX, linux.SOCK.STREAM | linux.SOCK.CLOEXEC, 0); - if (linux.errno(rc) != .SUCCESS) return error.Socket; - sc.fd = @intCast(rc); - if (linux.errno(linux.connect(sc.fd, &sa, @sizeOf(linux.sockaddr.un))) != .SUCCESS) return error.Connect; - sc.client = .init(.{ .in = &sc.cin, .out = &sc.cout }); - } - - fn close(sc: *SockClient) void { - _ = linux.close(sc.fd); - } - - /// One round trip; null when the server closed the connection. - fn rpc(sc: *SockClient, req: cloud9.Client.Request) !?cloud9.Client.Result { - _ = try sc.client.submit(req); - while (sc.client.output().len > 0) { - const out = sc.client.output(); - const rc = linux.write(sc.fd, out.ptr, out.len); - switch (linux.errno(rc)) { - .SUCCESS => sc.client.wrote(rc), - .PIPE, .CONNRESET => return null, - else => return error.Write, - } - } - var buf: [8192]u8 = undefined; - while (true) { - if (sc.client.take()) |done| return done.result; - const rc = linux.read(sc.fd, &buf, buf.len); - switch (linux.errno(rc)) { - .SUCCESS => {}, - .CONNRESET => return null, - else => return error.Read, - } - if (rc == 0) return null; - var rest: []const u8 = buf[0..rc]; - while (rest.len > 0) rest = rest[sc.client.push(rest)..]; - } - } - - fn session(sc: *SockClient) !void { - const v = (try sc.rpc(.{ .version = .{ .msize = 8192 } })) orelse return error.Closed; - try testing.expectEqual(@as(u32, 8192), v.version.msize); - const a = (try sc.rpc(.{ .attach = .{ .fid = 0, .uname = "t" } })) orelse return error.Closed; - try testing.expect(a == .attach); - } - - fn readFile(sc: *SockClient, names: []const []const u8, out: []u8) ![]u8 { - const w = (try sc.rpc(.{ .walk = .{ .fid = 0, .newfid = 1, .names = names } })) orelse return error.Closed; - try testing.expectEqual(@as(u16, @intCast(names.len)), w.walk.nwqid); - _ = (try sc.rpc(.{ .open = .{ .fid = 1, .mode = cloud9.oread } })) orelse return error.Closed; - const r = (try sc.rpc(.{ .read = .{ .fid = 1, .offset = 0, .count = @intCast(out.len) } })) orelse return error.Closed; - const n = r.read.len; - @memcpy(out[0..n], r.read); - _ = (try sc.rpc(.{ .clunk = .{ .fid = 1 } })) orelse return error.Closed; - return out[0..n]; - } -}; - -fn testSockPath(buf: []u8, tag: []const u8) ![]const u8 { - return std.fmt.bufPrint(buf, "/tmp/introspect-probe-{d}-{s}.sock", .{ linux.getpid(), tag }); -} - -const TestCtx = struct { info: runtime.Info }; - -test "probe: start, serve a client over a unix socket, stop" { - var ctx: TestCtx = .{ .info = .now() }; - var shared: TS.Shared = .init(&ctx); - const storage = try testing.allocator.create(TP.Storage(2)); - defer testing.allocator.destroy(storage); - var probe: TP = undefined; - var path_buf: [64]u8 = undefined; - const path = try testSockPath(&path_buf, "basic"); - try probe.init(&shared, storage, .{ .io = testing.io, .listen = .{ .unix = path } }); - defer probe.stop(); - try probe.start(); - ctx.info.clients = probe.clientCounter(); - - var sc: SockClient = undefined; - try sc.connect(path); - defer sc.close(); - try sc.session(); - var buf: [1024]u8 = undefined; - const zv = try sc.readFile(&.{ "build", "zig_version" }, &buf); - try testing.expectEqualStrings(builtin.zig_version_string, zv); - try testing.expectEqual(@as(u32, 1), probe.clientCount()); - - // The debug providers are mounted: /threads lists the probe thread by name. - const names = (try sc.rpc(.{ .walk = .{ .fid = 0, .newfid = 2, .names = &.{"threads"} } })) orelse return error.Closed; - try testing.expectEqual(@as(u16, 1), names.walk.nwqid); - _ = (try sc.rpc(.{ .open = .{ .fid = 2, .mode = cloud9.oread } })) orelse return error.Closed; - const dir = (try sc.rpc(.{ .read = .{ .fid = 2, .offset = 0, .count = 4096 } })) orelse return error.Closed; - try testing.expect(dir.read.len > 0); - _ = (try sc.rpc(.{ .clunk = .{ .fid = 2 } })) orelse return error.Closed; - - // A missing file is the Plan 9 error string. - const bad = (try sc.rpc(.{ .walk = .{ .fid = 0, .newfid = 3, .names = &.{"nope"} } })) orelse return error.Closed; - try testing.expect(bad == .fail); - try testing.expectEqualStrings("file does not exist", bad.fail); - - probe.stop(); - // stop() is idempotent and the socket file is gone. - probe.stop(); - var gone: SockClient = undefined; - try testing.expectError(error.Connect, gone.connect(path)); - // The debug facilities can be set up again after stop. - var probe2: TP = undefined; - var shared2: TS.Shared = .init(&ctx); - try probe2.init(&shared2, storage, .{ .io = testing.io, .listen = .{ .unix = path } }); - probe2.stop(); -} - -test "probe: max_clients refusal and idle eviction" { - var ctx: TestCtx = .{ .info = .now() }; - var shared: TS.Shared = .init(&ctx); - const storage = try testing.allocator.create(TP.Storage(2)); - defer testing.allocator.destroy(storage); - var probe: TP = undefined; - var path_buf: [64]u8 = undefined; - const path = try testSockPath(&path_buf, "limit"); - try probe.init(&shared, storage, .{ .io = testing.io, .listen = .{ .unix = path } }); - defer probe.stop(); - try probe.start(); - - // Two attached clients fill the table; a third is accepted then closed. - var a: SockClient = undefined; - try a.connect(path); - defer a.close(); - try a.session(); - var b: SockClient = undefined; - try b.connect(path); - defer b.close(); - try b.session(); - var c: SockClient = undefined; - try c.connect(path); - defer c.close(); - try testing.expectEqual(@as(?cloud9.Client.Result, null), try c.rpc(.{ .version = .{ .msize = 8192 } })); - try testing.expectEqual(@as(u64, 1), probe.refused); - // Attached clients are never evicted, even when idle for long. - sleepMs(evict_idle_ms + 100); - var d: SockClient = undefined; - try d.connect(path); - defer d.close(); - try testing.expectEqual(@as(?cloud9.Client.Result, null), try d.rpc(.{ .version = .{ .msize = 8192 } })); - var buf: [256]u8 = undefined; - _ = try a.readFile(&.{"README"}, &buf); - - // A client without fids that has been idle long enough gives way. - _ = (try b.rpc(.{ .clunk = .{ .fid = 0 } })) orelse return error.Closed; - sleepMs(evict_idle_ms + 100); - var e: SockClient = undefined; - try e.connect(path); - defer e.close(); - try e.session(); - try testing.expectEqual(@as(?cloud9.Client.Result, null), try b.rpc(.{ .version = .{ .msize = 8192 } })); - try testing.expectEqual(@as(u32, 2), probe.clientCount()); -} - -test "probe: single pre-connected client mode ends when the client hangs up" { - var ctx: TestCtx = .{ .info = .now() }; - var shared: TS.Shared = .init(&ctx); - const storage = try testing.allocator.create(TP.Storage(1)); - defer testing.allocator.destroy(storage); - var sv: [2]i32 = undefined; - try testing.expectEqual(linux.E.SUCCESS, linux.errno(linux.socketpair(linux.AF.UNIX, linux.SOCK.STREAM | linux.SOCK.CLOEXEC, 0, &sv))); - var probe: TP = undefined; - try probe.init(&shared, storage, .{ .io = testing.io, .listen = .{ .client = .{ .in = sv[1], .out = sv[1] } } }); - defer probe.stop(); - try probe.start(); - var sc: SockClient = .{ .fd = sv[0], .client = undefined }; - sc.client = .init(.{ .in = &sc.cin, .out = &sc.cout }); - try sc.session(); - var buf: [256]u8 = undefined; - try testing.expect((try sc.readFile(&.{"README"}, &buf)).len > 0); - try testing.expectEqual(@as(u32, 1), probe.clientCount()); - _ = linux.close(sv[0]); - probe.wait(); - try testing.expectEqual(@as(u32, 0), probe.clientCount()); - _ = linux.close(sv[1]); -} - -test "parseIpv4 and sleepServing off the probe thread" { - const sa = parseIpv4("127.0.0.1:564").?; - try testing.expectEqual(std.mem.nativeToBig(u16, 564), sa.port); - try testing.expectEqual(@as(u32, @bitCast([4]u8{ 127, 0, 0, 1 })), sa.addr); - try testing.expect(parseIpv4("localhost:1") == null); - try testing.expect(parseIpv4("1.2.3:1") == null); - try testing.expect(parseIpv4("1.2.3.4") == null); - try testing.expect(parseIpv4("1.2.3.4:70000") == null); - const t0 = monotonicMs(); - var probe: TP = undefined; - probe.thread_tid = .init(0); - probe.serving = .initEmpty(); - probe.nested = 0; - probe.sleepServing(20); - try testing.expect(monotonicMs() - t0 >= 20); -} diff --git a/introspect/src/linux/provider.zig b/introspect/src/linux/provider.zig deleted file mode 100644 index 9952398..0000000 --- a/introspect/src/linux/provider.zig +++ /dev/null @@ -1,604 +0,0 @@ -//! Adapts `debug.Debug` into core `Provider`s. The core mounts providers at -//! top level only, so one `DebugProvider` registers six of them, all sharing -//! the same `Debug` and the same snapshot pool: -//! -//! /threads//{name,stat,stack,regs} (lists only live tids) -//! /addr/ "fn\nfile:line:col\nmodule\n" -//! /mem/maps, /mem/ /proc/self/maps; raw bytes at address+offset (writable) -//! /hex/ hexdump of 256 bytes at the address -//! /breakpoints//{stack,regs,ctl} ctl accepts "continue" (lists only paused tids) -//! /panic/{message,stack,ctl} ctl accepts "continue" -//! -//! /addr, /mem, /hex and /panic carry a README; /threads and /breakpoints -//! list nothing but tids so that a shell glob over them sees only threads. -//! -//! Handles encode `(kind, tid-or-address)` in 56 bits (the core keeps the -//! low 56 bits of a handle for the qid path): kind in bits 48..55, value in -//! bits 0..47. Handles carry no reference count, so `clunk` is a no-op. -//! -//! The core hands providers a buffer-based `read`, not a writer, so every -//! text file is generated at `open` into one of `snapshot_slots` fixed slots -//! (keyed by handle, reference counted across fids) and served from there; a -//! read at offset 0 regenerates, like the core's own dynamic files. `/mem/` -//! is read and written directly at address+offset and never snapshotted, and -//! `/mem/maps` is read straight from /proc/self/maps at the requested offset -//! (a big process has more mappings than a snapshot slot holds). -const std = @import("std"); -const cloud9 = @import("cloud9"); -const core = @import("../core.zig"); -const debug = @import("debug.zig"); -const Writer = std.Io.Writer; -const Provider = core.Provider; -const Handle = Provider.Handle; -const Error = Provider.Error; -const NodeStat = core.NodeStat; - -pub const snapshot_slots = 8; -pub const snapshot_bytes = 32 * 1024; -/// Bytes shown by /hex/. -pub const hex_bytes = 256; - -pub const Tree = enum(u8) { threads, addr, mem, hex, breakpoints, panic }; -pub const tree_names = [_][]const u8{ "threads", "addr", "mem", "hex", "breakpoints", "panic" }; - -const Kind = enum(u8) { - root = 0, - readme, - thread_dir, - thread_name, - thread_stat, - thread_stack, - thread_regs, - addr_file, - maps, - mem_file, - hex_file, - bp_dir, - bp_stack, - bp_regs, - bp_ctl, - panic_message, - panic_stack, - panic_ctl, - - fn isDir(k: Kind) bool { - return k == .root or k == .thread_dir or k == .bp_dir; - } - - /// Text files generated into a snapshot slot at open. - fn isText(k: Kind) bool { - return switch (k) { - .readme, .thread_name, .thread_stat, .thread_stack, .thread_regs, .addr_file, .hex_file, .bp_stack, .bp_regs, .panic_message, .panic_stack => true, - else => false, - }; - } - - fn isCtl(k: Kind) bool { - return k == .bp_ctl or k == .panic_ctl; - } - - fn mode(k: Kind) u32 { - if (k.isDir()) return cloud9.dmdir | 0o555; - if (k.isCtl()) return 0o222; - if (k == .mem_file) return 0o666; - return 0o444; - } - - fn fixedName(k: Kind) ?[]const u8 { - return switch (k) { - .readme => "README", - .thread_name => "name", - .thread_stat => "stat", - .thread_stack, .bp_stack, .panic_stack => "stack", - .thread_regs, .bp_regs => "regs", - .maps => "maps", - .bp_ctl, .panic_ctl => "ctl", - .panic_message => "message", - else => null, - }; - } -}; - -const value_bits = 48; -const value_mask: u64 = (1 << value_bits) - 1; - -fn mk(kind: Kind, value: u64) Handle { - return (@as(u64, @intFromEnum(kind)) << value_bits) | (value & value_mask); -} - -fn kindOf(h: Handle) Kind { - return @enumFromInt(@as(u8, @truncate(h >> value_bits))); -} - -fn valueOf(h: Handle) u64 { - return h & value_mask; -} - -const readme_threads = - \\One directory per thread of this process, named by tid: - \\ name the thread's comm - \\ stat state and a few fields of /proc/self/task//stat - \\ stack "#n 0x in (::)" per frame - \\ regs general registers captured while the thread was stopped - \\ -; -const readme_addr = - \\Walk any hex address: /addr/ reads as "fn\nfile:line:col\nmodule\n". - \\ -; -const readme_mem = - \\maps /proc/self/maps - \\ raw process memory at that address (+ file offset); writable - \\ -; -const readme_hex = - \\Walk any hex address: /hex/ is a hexdump of the 256 bytes there. - \\ -; -const readme_breakpoints = - \\One directory per thread stopped in @breakpoint(), named by tid: - \\ stack, regs as under /threads - \\ ctl write "continue" to resume the thread - \\ -; -const readme_panic = - \\message the first panic's message (empty before any panic) - \\stack frames of the panicking thread - \\ctl write "continue" to let the default panic handler run - \\ -; - -fn readmeFor(tree: Tree) []const u8 { - return switch (tree) { - .threads => readme_threads, - .addr => readme_addr, - .mem => readme_mem, - .hex => readme_hex, - .breakpoints => readme_breakpoints, - .panic => readme_panic, - }; -} - -const Slot = struct { - handle: Handle = 0, - refs: u32 = 0, - len: u32 = 0, - buf: [snapshot_bytes]u8 = undefined, -}; - -pub const DebugProvider = struct { - d: *debug.Debug, - slots: [snapshot_slots]Slot = @splat(.{}), - /// Backs `NodeStat.name` until the next call. - name_buf: [32]u8 = undefined, - - pub fn init(dp: *DebugProvider, d: *debug.Debug) void { - dp.* = .{ .d = d }; - } - - /// The provider for one tree, to pass to `Shared.addProvider`. - pub fn provider(dp: *DebugProvider, comptime tree: Tree) Provider { - return .{ .name = tree_names[@intFromEnum(tree)], .ctx = dp, .vtable = vtableFor(tree) }; - } - - /// Mounts all six trees; `shared` is a `Server(cfg).Shared`. - pub fn mountAll(dp: *DebugProvider, shared: anytype) error{Full}!void { - inline for (comptime std.meta.tags(Tree)) |tree| try shared.addProvider(dp.provider(tree)); - } - - fn self(ctx: *anyopaque) *DebugProvider { - return @ptrCast(@alignCast(ctx)); - } - - fn vtableFor(comptime tree: Tree) *const Provider.VTable { - return &struct { - const vt: Provider.VTable = .{ - .walk = walkFn, - .stat = statFn, - .list = listFn, - .open = openFn, - .read = readFn, - .write = writeFn, - .close = closeFn, - .clunk = clunkFn, - }; - fn walkFn(ctx: *anyopaque, parent: Handle, name: []const u8) Error!Handle { - return self(ctx).walk(tree, parent, name); - } - fn statFn(ctx: *anyopaque, h: Handle, out: *NodeStat) Error!void { - return self(ctx).stat(tree, h, out); - } - fn listFn(ctx: *anyopaque, dir: Handle, index: usize, out: *NodeStat) Error!bool { - return self(ctx).list(tree, dir, index, out); - } - fn openFn(ctx: *anyopaque, h: Handle, mode: u8) Error!void { - return self(ctx).open(tree, h, mode); - } - fn readFn(ctx: *anyopaque, h: Handle, offset: u64, buf: []u8) Error!usize { - return self(ctx).read(tree, h, offset, buf); - } - fn writeFn(ctx: *anyopaque, h: Handle, offset: u64, data: []const u8) Error!usize { - return self(ctx).write(tree, h, offset, data); - } - fn closeFn(ctx: *anyopaque, h: Handle) void { - self(ctx).close(h); - } - fn clunkFn(_: *anyopaque, _: Handle) void {} - }.vt; - } - - // -- naming ------------------------------------------------------------ - - fn parseTid(name: []const u8) ?u32 { - if (name.len == 0 or name.len > 10) return null; - for (name) |ch| if (!std.ascii.isDigit(ch)) return null; - return std.fmt.parseInt(u32, name, 10) catch null; - } - - fn parseHex(name: []const u8) ?u64 { - const digits = if (std.mem.startsWith(u8, name, "0x")) name[2..] else name; - if (digits.len == 0 or digits.len > 12) return null; - for (digits) |ch| if (!std.ascii.isHex(ch)) return null; - const v = std.fmt.parseInt(u64, digits, 16) catch return null; - if (v > value_mask) return null; - return v; - } - - fn nodeName(dp: *DebugProvider, h: Handle) []const u8 { - const k = kindOf(h); - if (k.fixedName()) |n| return n; - return switch (k) { - .root => "", - .thread_dir, .bp_dir => std.fmt.bufPrint(&dp.name_buf, "{d}", .{valueOf(h)}) catch unreachable, - .addr_file, .mem_file, .hex_file => std.fmt.bufPrint(&dp.name_buf, "{x}", .{valueOf(h)}) catch unreachable, - else => unreachable, - }; - } - - // -- vtable ------------------------------------------------------------ - - fn walk(dp: *DebugProvider, tree: Tree, parent: Handle, name: []const u8) Error!Handle { - const k = kindOf(parent); - if (std.mem.eql(u8, name, ".")) return parent; - if (!k.isDir()) return error.NotDir; - if (std.mem.eql(u8, name, "..")) return Provider.root; - switch (k) { - .root => { - if (tree != .threads and tree != .breakpoints and std.mem.eql(u8, name, "README")) return mk(.readme, 0); - switch (tree) { - .threads => { - const tid = parseTid(name) orelse return error.NotFound; - if (!dp.d.threadExists(tid)) return error.NotFound; - return mk(.thread_dir, tid); - }, - .addr => return mk(.addr_file, parseHex(name) orelse return error.NotFound), - .mem => { - if (std.mem.eql(u8, name, "maps")) return mk(.maps, 0); - return mk(.mem_file, parseHex(name) orelse return error.NotFound); - }, - .hex => return mk(.hex_file, parseHex(name) orelse return error.NotFound), - .breakpoints => { - const tid = parseTid(name) orelse return error.NotFound; - if (!dp.d.isPaused(tid)) return error.NotFound; - return mk(.bp_dir, tid); - }, - .panic => { - if (std.mem.eql(u8, name, "message")) return mk(.panic_message, 0); - if (std.mem.eql(u8, name, "stack")) return mk(.panic_stack, 0); - if (std.mem.eql(u8, name, "ctl")) return mk(.panic_ctl, 0); - return error.NotFound; - }, - } - }, - .thread_dir => { - const tid = valueOf(parent); - if (std.mem.eql(u8, name, "name")) return mk(.thread_name, tid); - if (std.mem.eql(u8, name, "stat")) return mk(.thread_stat, tid); - if (std.mem.eql(u8, name, "stack")) return mk(.thread_stack, tid); - if (std.mem.eql(u8, name, "regs")) return mk(.thread_regs, tid); - return error.NotFound; - }, - .bp_dir => { - const tid = valueOf(parent); - if (std.mem.eql(u8, name, "stack")) return mk(.bp_stack, tid); - if (std.mem.eql(u8, name, "regs")) return mk(.bp_regs, tid); - if (std.mem.eql(u8, name, "ctl")) return mk(.bp_ctl, tid); - return error.NotFound; - }, - else => unreachable, - } - } - - fn fill(dp: *DebugProvider, tree: Tree, h: Handle, out: *NodeStat) void { - const k = kindOf(h); - out.* = .{ - .mode = k.mode(), - .length = if (k == .readme) readmeFor(tree).len else 0, - .name = dp.nodeName(h), - .handle = h, - }; - } - - fn stat(dp: *DebugProvider, tree: Tree, h: Handle, out: *NodeStat) Error!void { - dp.fill(tree, h, out); - } - - fn list(dp: *DebugProvider, tree: Tree, dir: Handle, index: usize, out: *NodeStat) Error!bool { - const k = kindOf(dir); - if (!k.isDir()) return error.NotDir; - const h: Handle = switch (k) { - .root => switch (tree) { - .threads => mk(.thread_dir, dp.d.threadAt(index) orelse return false), - .addr, .hex => if (index == 0) mk(.readme, 0) else return false, - .mem => switch (index) { - 0 => mk(.readme, 0), - 1 => mk(.maps, 0), - else => return false, - }, - .breakpoints => mk(.bp_dir, dp.d.pausedAt(index) orelse return false), - .panic => switch (index) { - 0 => mk(.readme, 0), - 1 => mk(.panic_message, 0), - 2 => mk(.panic_stack, 0), - 3 => mk(.panic_ctl, 0), - else => return false, - }, - }, - .thread_dir => switch (index) { - 0 => mk(.thread_name, valueOf(dir)), - 1 => mk(.thread_stat, valueOf(dir)), - 2 => mk(.thread_stack, valueOf(dir)), - 3 => mk(.thread_regs, valueOf(dir)), - else => return false, - }, - .bp_dir => switch (index) { - 0 => mk(.bp_stack, valueOf(dir)), - 1 => mk(.bp_regs, valueOf(dir)), - 2 => mk(.bp_ctl, valueOf(dir)), - else => return false, - }, - else => unreachable, - }; - dp.fill(tree, h, out); - return true; - } - - fn open(dp: *DebugProvider, tree: Tree, h: Handle, mode: u8) Error!void { - const k = kindOf(h); - const acc = mode & 3; - const wants_write = acc == cloud9.owrite or acc == cloud9.ordwr; - const wants_read = acc != cloud9.owrite; - if (k.isDir()) { - if (wants_write or mode & cloud9.otrunc != 0) return error.IsDir; - return; - } - if (k.isCtl()) { - if (wants_read) return error.Perm; - return; - } - if (k == .mem_file) return; - if (wants_write or mode & cloud9.otrunc != 0) return error.Perm; - if (k == .maps) return; - std.debug.assert(k.isText()); - const slot = dp.takeSlot(h) orelse return error.NoSpace; - errdefer dp.releaseSlot(slot); - try dp.generate(tree, h, slot); - } - - fn close(dp: *DebugProvider, h: Handle) void { - if (!kindOf(h).isText()) return; - if (dp.findSlot(h)) |s| dp.releaseSlot(s); - } - - fn read(dp: *DebugProvider, tree: Tree, h: Handle, offset: u64, buf: []u8) Error!usize { - const k = kindOf(h); - if (k.isDir()) return error.IsDir; - if (k.isCtl()) return error.Perm; - if (k == .mem_file) { - const addr = valueOf(h) +% offset; - return dp.d.readMem(addr, buf) catch |e| mapErr(e); - } - if (k == .maps) return dp.d.readMaps(offset, buf) catch |e| mapErr(e); - const slot = dp.findSlot(h) orelse return error.Io; - if (offset == 0) try dp.generate(tree, h, slot); - if (offset >= slot.len) return 0; - const off: usize = @intCast(offset); - const n = @min(buf.len, slot.len - off); - @memcpy(buf[0..n], slot.buf[off..][0..n]); - return n; - } - - fn write(dp: *DebugProvider, tree: Tree, h: Handle, offset: u64, data: []const u8) Error!usize { - _ = tree; - const k = kindOf(h); - if (k.isDir()) return error.IsDir; - switch (k) { - .mem_file => { - const addr = valueOf(h) +% offset; - return dp.d.writeMem(addr, data) catch |e| mapErr(e); - }, - .bp_ctl, .panic_ctl => { - const cmd = std.mem.trim(u8, data, " \t\r\n\x00"); - if (!std.mem.eql(u8, cmd, "continue")) return error.Unsupported; - if (k == .bp_ctl) { - dp.d.resumeThread(@intCast(valueOf(h))) catch |e| return mapErr(e); - } else { - dp.d.panicContinue() catch |e| return mapErr(e); - } - return data.len; - }, - else => return error.Perm, - } - } - - // -- snapshots --------------------------------------------------------- - - fn findSlot(dp: *DebugProvider, h: Handle) ?*Slot { - for (&dp.slots) |*s| if (s.refs > 0 and s.handle == h) return s; - return null; - } - - fn takeSlot(dp: *DebugProvider, h: Handle) ?*Slot { - if (dp.findSlot(h)) |s| { - s.refs += 1; - return s; - } - for (&dp.slots) |*s| if (s.refs == 0) { - s.* = .{ .handle = h, .refs = 1 }; - return s; - }; - return null; - } - - fn releaseSlot(_: *DebugProvider, s: *Slot) void { - s.refs -= 1; - } - - /// (Re)generates the text of `h` into `slot`. A text that does not fit is - /// truncated, not an error. - fn generate(dp: *DebugProvider, tree: Tree, h: Handle, slot: *Slot) Error!void { - var w: Writer = .fixed(&slot.buf); - slot.len = 0; - dp.render(tree, h, &w) catch |e| switch (e) { - error.WriteFailed => {}, - else => return mapErr(e), - }; - slot.len = @intCast(w.buffered().len); - } - - fn render(dp: *DebugProvider, tree: Tree, h: Handle, w: *Writer) debug.Error!void { - const d = dp.d; - const v = valueOf(h); - switch (kindOf(h)) { - .readme => w.writeAll(readmeFor(tree)) catch return error.WriteFailed, - .thread_name => try d.threadName(@intCast(v), w), - .thread_stat => try d.threadStat(@intCast(v), w), - .thread_stack => try d.threadStack(@intCast(v), w), - .thread_regs => try d.threadRegs(@intCast(v), w), - .addr_file => try d.resolveAddr(@intCast(v), w), - .hex_file => try d.hexdump(@intCast(v), hex_bytes, w), - .bp_stack => try d.pausedStack(@intCast(v), w), - .bp_regs => try d.pausedRegs(@intCast(v), w), - .panic_message => try d.panicMessage(w), - .panic_stack => try d.panicStack(w), - else => unreachable, - } - } - - fn mapErr(e: debug.Error) Error { - return switch (e) { - error.NoThread, error.NotPaused, error.NoPanic => error.NotFound, - error.Unsupported => error.Unsupported, - error.WriteFailed => error.NoSpace, - error.Timeout, error.Busy, error.Unmapped, error.Unexpected, error.AlreadyInitialized, error.InvalidOptions => error.Io, - }; - } -}; - -// --------------------------------------------------------------------------- -// Tests (through the core's in-memory harness) -// --------------------------------------------------------------------------- - -const testing = std.testing; - -const TestCfg: core.Config = .{ .name = "dbgtest", .msize = 8192, .max_fids = 16, .max_providers = 6, .snapshot_slots = 2, .snapshot_bytes = 512 }; -const TS = core.Server(TestCfg); - -test "debug provider: threads, addr, mem, hex, breakpoints, panic through the core" { - var text_buf: [16 * 1024]u8 = undefined; - var d: debug.Debug = undefined; - try d.init(.{ .io = testing.io, .text_buf = &text_buf }); - defer d.deinit(); - var dp: DebugProvider = undefined; - dp.init(&d); - - var dummy: u8 = 0; - var shared: TS.Shared = .init(&dummy); - try dp.mountAll(&shared); - var storage: TS.Storage = undefined; - var h: TS.Harness = undefined; - try h.init(&shared, &storage); - defer h.deinit(); - - // /threads lists tids only, among them this thread. - const names = try h.listPath(&.{"threads"}); - defer TS.Harness.freeNames(names); - try testing.expect(names.len >= 1); - try testing.expect(!TS.Harness.hasName(names, "README")); - const no_readme = try h.ok(.{ .walk = .{ .fid = 0, .newfid = 7, .names = &.{ "threads", "README" } } }); - try testing.expectEqual(@as(u16, 1), no_readme.walk.nwqid); - var tid_buf: [16]u8 = undefined; - const tid = try std.fmt.bufPrint(&tid_buf, "{d}", .{std.os.linux.gettid()}); - try testing.expect(TS.Harness.hasName(names, tid)); - - // Own stack names this test function's file. - const stack = try h.readPath(&.{ "threads", tid, "stack" }); - defer testing.allocator.free(stack); - try testing.expect(std.mem.indexOf(u8, stack, "#0 0x") != null); - - // /addr/ of a function here resolves to this file. - var addr_buf: [32]u8 = undefined; - const addr_name = try std.fmt.bufPrint(&addr_buf, "{x}", .{@intFromPtr(&DebugProvider.parseTid)}); - const resolved = try h.readPath(&.{ "addr", addr_name }); - defer testing.allocator.free(resolved); - try testing.expect(std.mem.indexOf(u8, resolved, "provider.zig") != null); - const partial = try h.ok(.{ .walk = .{ .fid = 0, .newfid = 5, .names = &.{ "addr", "zzz" } } }); - try testing.expectEqual(@as(u16, 1), partial.walk.nwqid); - try h.walkTo(5, &.{"addr"}); - try h.expectFail(.{ .walk = .{ .fid = 5, .newfid = 6, .names = &.{"zzz"} } }, "file does not exist"); - _ = try h.ok(.{ .clunk = .{ .fid = 5 } }); - - // /mem/ reads and writes live memory; unmapped is an error. - var cell: [8]u8 = "abcdefgh".*; - var mem_buf: [32]u8 = undefined; - const mem_name = try std.fmt.bufPrint(&mem_buf, "{x}", .{@intFromPtr(&cell)}); - try h.walkTo(1, &.{ "mem", mem_name }); - _ = try h.ok(.{ .open = .{ .fid = 1, .mode = cloud9.ordwr } }); - const r = try h.ok(.{ .read = .{ .fid = 1, .offset = 2, .count = 4 } }); - try testing.expectEqualStrings("cdef", r.read); - _ = try h.ok(.{ .write = .{ .fid = 1, .offset = 0, .data = "XY" } }); - try testing.expectEqualStrings("XYcdefgh", &cell); - _ = try h.ok(.{ .clunk = .{ .fid = 1 } }); - try h.walkTo(2, &.{ "mem", "8" }); - _ = try h.ok(.{ .open = .{ .fid = 2, .mode = cloud9.oread } }); - try h.expectFail(.{ .read = .{ .fid = 2, .offset = 0, .count = 4 } }, "i/o error"); - _ = try h.ok(.{ .clunk = .{ .fid = 2 } }); - const maps = try h.readPath(&.{ "mem", "maps" }); - defer testing.allocator.free(maps); - try testing.expect(std.mem.indexOf(u8, maps, "r-xp") != null or std.mem.indexOf(u8, maps, "r--p") != null); - - // /hex/ is a hexdump. - const hex = try h.readPath(&.{ "hex", mem_name }); - defer testing.allocator.free(hex); - try testing.expect(std.mem.indexOf(u8, hex, "XYcdefgh") != null); - - // Nothing paused, no panic. - const bps = try h.listPath(&.{"breakpoints"}); - defer TS.Harness.freeNames(bps); - try testing.expectEqual(@as(usize, 0), bps.len); - const msg = try h.readPath(&.{ "panic", "message" }); - defer testing.allocator.free(msg); - try testing.expectEqualStrings("", msg); - try h.walkTo(3, &.{ "panic", "ctl" }); - _ = try h.ok(.{ .open = .{ .fid = 3, .mode = cloud9.owrite } }); - try h.expectFail(.{ .write = .{ .fid = 3, .offset = 0, .data = "continue" } }, "file does not exist"); - try h.expectFail(.{ .write = .{ .fid = 3, .offset = 0, .data = "bogus" } }, "not supported"); - _ = try h.ok(.{ .clunk = .{ .fid = 3 } }); - - // Snapshot slots are released on clunk: open more files than slots, sequentially. - for (0..4) |_| { - const t = try h.readPath(&.{ "threads", tid, "name" }); - testing.allocator.free(t); - } - for (&dp.slots) |s| try testing.expectEqual(@as(u32, 0), s.refs); -} - -test "handle encoding round-trips" { - const h = mk(.mem_file, 0x7fff_dead_beef); - try testing.expectEqual(Kind.mem_file, kindOf(h)); - try testing.expectEqual(@as(u64, 0x7fff_dead_beef), valueOf(h)); - try testing.expect(h < (1 << 56)); - try testing.expectEqual(@as(?u64, null), DebugProvider.parseHex("1_0")); - try testing.expectEqual(@as(?u64, 0x10), DebugProvider.parseHex("0x10")); - try testing.expectEqual(@as(?u32, null), DebugProvider.parseTid("+5")); -} diff --git a/introspect/src/linux/runtime.zig b/introspect/src/linux/runtime.zig deleted file mode 100644 index 503d6c2..0000000 --- a/introspect/src/linux/runtime.zig +++ /dev/null @@ -1,90 +0,0 @@ -//! Generators for `Config.runtime`: /runtime/{pid,ppid,uptime,argv,cwd,env,clients}. -//! The core passes every generator `Shared.ctx`; `Fns(Ctx, field)` casts it -//! to `*Ctx` and reads the `Info` stored in `@field(ctx, field)`. -const std = @import("std"); -const linux = std.os.linux; -const Writer = std.Io.Writer; - -/// What the generators report. Texts are borrowed for the server's lifetime. -pub const Info = struct { - /// argv, one argument per line. - argv: []const u8 = "", - /// Environment, one KEY=VALUE per line. - env: []const u8 = "", - cwd: []const u8 = "", - /// Monotonic seconds at startup; /runtime/uptime is the difference. - start_mono: i64 = 0, - /// Live client count, published by the probe. - clients: ?*const std.atomic.Value(u32) = null, - - pub fn now() Info { - return .{ .start_mono = monotonicSecs() }; - } -}; - -pub fn monotonicSecs() i64 { - var ts: linux.timespec = undefined; - _ = linux.clock_gettime(.MONOTONIC, &ts); - return ts.sec; -} - -/// Seconds since the epoch, clamped to u32 (for atime/mtime and `fn/now`). -pub fn realtimeSecs() u32 { - var ts: linux.timespec = undefined; - _ = linux.clock_gettime(.REALTIME, &ts); - return @intCast(std.math.clamp(ts.sec, 0, std.math.maxInt(u32))); -} - -/// The `Config.runtime` type: `Ctx` is the type behind `Shared.ctx`, `field` -/// the name of its `Info` field. -pub fn Fns(comptime Ctx: type, comptime field: []const u8) type { - return struct { - fn info(ctx: *anyopaque) *const Info { - const c: *Ctx = @ptrCast(@alignCast(ctx)); - return &@field(c, field); - } - pub fn pid(_: *anyopaque, w: *Writer) anyerror!void { - try w.print("{d}", .{linux.getpid()}); - } - pub fn ppid(_: *anyopaque, w: *Writer) anyerror!void { - try w.print("{d}", .{linux.getppid()}); - } - pub fn uptime(ctx: *anyopaque, w: *Writer) anyerror!void { - try w.print("{d}", .{monotonicSecs() - info(ctx).start_mono}); - } - pub fn argv(ctx: *anyopaque, w: *Writer) anyerror!void { - try w.writeAll(info(ctx).argv); - } - pub fn cwd(ctx: *anyopaque, w: *Writer) anyerror!void { - try w.writeAll(info(ctx).cwd); - } - pub fn env(ctx: *anyopaque, w: *Writer) anyerror!void { - try w.writeAll(info(ctx).env); - } - pub fn clients(ctx: *anyopaque, w: *Writer) anyerror!void { - const n: u32 = if (info(ctx).clients) |c| c.load(.acquire) else 0; - try w.print("{d}", .{n}); - } - }; -} - -test "runtime generators read Info through the context" { - const Ctx = struct { x: u32, info: Info }; - var count: std.atomic.Value(u32) = .init(3); - var ctx: Ctx = .{ .x = 0, .info = .{ .argv = "a\nb\n", .cwd = "/tmp", .env = "K=V\n", .start_mono = monotonicSecs(), .clients = &count } }; - const F = Fns(Ctx, "info"); - var buf: [64]u8 = undefined; - var w: Writer = .fixed(&buf); - try F.clients(&ctx, &w); - try std.testing.expectEqualStrings("3", w.buffered()); - w = .fixed(&buf); - try F.argv(&ctx, &w); - try std.testing.expectEqualStrings("a\nb\n", w.buffered()); - w = .fixed(&buf); - try F.uptime(&ctx, &w); - try std.testing.expect(w.buffered().len >= 1); - w = .fixed(&buf); - try F.pid(&ctx, &w); - try std.testing.expectEqual(linux.getpid(), try std.fmt.parseInt(i32, w.buffered(), 10)); - try std.testing.expectEqual(@as(usize, 7), @typeInfo(F).@"struct".decls.len); -} diff --git a/introspect/src/root.zig b/introspect/src/root.zig deleted file mode 100644 index 1dcc892..0000000 --- a/introspect/src/root.zig +++ /dev/null @@ -1,24 +0,0 @@ -//! introspect: a 9P2000 debug/introspection server as a library. See -//! docs/LIBRARY.md. `core` and `vars` are freestanding; `scratch` takes an -//! allocator; `linux` is the platform layer (only on Linux). -const std = @import("std"); -const builtin = @import("builtin"); - -pub const core = @import("core.zig"); -pub const vars = @import("vars.zig"); -pub const scratch = @import("scratch.zig"); -pub const linux = if (builtin.os.tag == .linux) @import("linux/probe.zig") else struct {}; - -pub const Config = core.Config; -pub const Server = core.Server; -pub const Provider = core.Provider; -pub const NodeStat = core.NodeStat; -pub const Scratch = scratch.Scratch; - -test { - std.testing.refAllDecls(@This()); - _ = core; - _ = vars; - _ = scratch; - if (builtin.os.tag == .linux) _ = linux; -} diff --git a/introspect/src/scratch.zig b/introspect/src/scratch.zig deleted file mode 100644 index 2163a28..0000000 --- a/introspect/src/scratch.zig +++ /dev/null @@ -1,689 +0,0 @@ -//! An in-memory read/write tree as a `Provider`: create, write, truncate, -//! rename, remove, mkdir, DMAPPEND, DMEXCL. The one core-level component that -//! takes an `Allocator` (nodes and file contents live on it); it is optional. -//! -//! Nodes are kept alive by `refs` (fids holding a handle) after removal, so a -//! handle stays valid until the core clunks it. Handles are node addresses; -//! the root is handle 0. Not internally synchronized (like `Shared`). -const std = @import("std"); -const cloud9 = @import("cloud9"); -const core = @import("core.zig"); -const Allocator = std.mem.Allocator; -const Provider = core.Provider; -const Handle = Provider.Handle; -const Error = Provider.Error; -const NodeStat = core.NodeStat; - -/// A node of the tree. -pub const Node = struct { - name: []u8, - path: u64, - version: u32 = 0, - mode: u32, - atime: u32, - mtime: u32, - data: std.ArrayList(u8) = .empty, - children: std.ArrayList(*Node) = .empty, - parent: ?*Node, - /// Handles held by the core. - refs: u32 = 0, - /// Fids currently open on this node (DMEXCL admits at most one). - opens: u32 = 0, - removed: bool = false, - - pub fn isDir(n: *const Node) bool { - return n.mode & cloud9.dmdir != 0; - } - - fn find(n: *const Node, name: []const u8) ?*Node { - for (n.children.items) |ch| if (std.mem.eql(u8, ch.name, name)) return ch; - return null; - } -}; - -/// Seconds since the epoch, for atime/mtime; the default clock reports 0. -pub const Clock = *const fn () u32; - -fn zeroClock() u32 { - return 0; -} - -pub const Scratch = struct { - gpa: Allocator, - root: *Node, - /// Qid paths are a counter, never reused: the root is 0 (the provider - /// root handle), so a removed-and-recreated file gets a fresh identity - /// even when the allocator hands back the same address. - next_path: u64 = 0, - /// Sum of all file lengths, bounded by `budget`. - bytes: usize = 0, - /// Largest total of file contents across all files. - budget: usize, - /// Largest single file; defaults to the budget. - max_file: usize, - /// The time source for atime/mtime (a platform layer sets it). - now: Clock = &zeroClock, - - /// The tree's only allocation policy: every node and every file's - /// contents come from `gpa`, and no file content ever exceeds `budget_bytes` - /// in total. - pub fn init(gpa: Allocator, budget_bytes: usize) Allocator.Error!Scratch { - var s: Scratch = .{ .gpa = gpa, .root = undefined, .budget = budget_bytes, .max_file = budget_bytes }; - s.root = try s.newNode("", cloud9.dmdir | 0o777, null); - return s; - } - - pub fn deinit(s: *Scratch) void { - s.destroyTree(s.root); - s.* = undefined; - } - - /// The provider to mount, at `/`. - pub fn provider(s: *Scratch, name: []const u8) Provider { - return .{ .name = name, .ctx = s, .vtable = &vtable }; - } - - pub const vtable: Provider.VTable = .{ - .walk = &walk, - .stat = &stat, - .list = &list, - .open = &open, - .read = &read, - .write = &write, - .create = &create, - .remove = &remove, - .wstat = &wstat, - .close = &close, - .clunk = &clunk, - }; - - // -- node management -- - - fn destroyTree(s: *Scratch, n: *Node) void { - for (n.children.items) |ch| s.destroyTree(ch); - n.children.clearRetainingCapacity(); - n.removed = true; - if (n.refs == 0 or n == s.root) s.destroyNode(n); - } - - fn destroyNode(s: *Scratch, n: *Node) void { - s.bytes -= n.data.items.len; - s.gpa.free(n.name); - n.data.deinit(s.gpa); - n.children.deinit(s.gpa); - s.gpa.destroy(n); - } - - fn newNode(s: *Scratch, name: []const u8, mode: u32, parent: ?*Node) Allocator.Error!*Node { - const n = try s.gpa.create(Node); - errdefer s.gpa.destroy(n); - const t = s.now(); - n.* = .{ - .name = try s.gpa.dupe(u8, name), - .path = s.next_path, - .mode = mode, - .atime = t, - .mtime = t, - .parent = parent, - }; - errdefer s.gpa.free(n.name); - if (parent) |p| try p.children.append(s.gpa, n); - s.next_path += 1; - return n; - } - - /// Sets a file's length, zero-filling growth and charging the budget. - /// Shrinking releases the memory so a truncated file costs nothing. - fn resizeData(s: *Scratch, n: *Node, new_len: usize) Error!void { - const old = n.data.items.len; - if (new_len > old) { - if (new_len > s.max_file) return error.NoSpace; - if (s.bytes + (new_len - old) > s.budget) return error.NoSpace; - n.data.resize(s.gpa, new_len) catch return error.NoSpace; - @memset(n.data.items[old..new_len], 0); - s.bytes += new_len - old; - } else if (new_len < old) { - n.data.shrinkAndFree(s.gpa, new_len); - s.bytes -= old - new_len; - } - } - - fn touch(s: *Scratch, n: *Node) void { - n.version +%= 1; - n.mtime = s.now(); - } - - fn self(ctx: *anyopaque) *Scratch { - return @ptrCast(@alignCast(ctx)); - } - - fn handle(s: *Scratch, n: *Node) Handle { - return if (n == s.root) Provider.root else @intFromPtr(n); - } - - fn node(s: *Scratch, h: Handle) *Node { - return if (h == Provider.root) s.root else @ptrFromInt(@as(usize, @intCast(h))); - } - - /// A handle the core will clunk exactly once. - fn retain(s: *Scratch, n: *Node) Handle { - if (n != s.root) n.refs += 1; - return s.handle(n); - } - - fn release(s: *Scratch, n: *Node) void { - if (n == s.root) return; - n.refs -= 1; - if (n.refs == 0 and n.removed) s.destroyNode(n); - } - - fn fillStat(n: *const Node, h: Handle, out: *NodeStat) void { - out.* = .{ - .mode = n.mode, - .length = if (n.isDir()) 0 else n.data.items.len, - .atime = n.atime, - .mtime = n.mtime, - .version = n.version, - .name = n.name, - .handle = h, - .path = n.path, - }; - } - - // -- the vtable -- - - fn walk(ctx: *anyopaque, parent: Handle, name: []const u8) Error!Handle { - const s = self(ctx); - const p = s.node(parent); - if (std.mem.eql(u8, name, ".")) return s.retain(p); - if (p.removed) return error.NotFound; - if (!p.isDir()) return error.NotDir; - if (std.mem.eql(u8, name, "..")) return s.retain(p.parent orelse s.root); - return s.retain(p.find(name) orelse return error.NotFound); - } - - fn stat(ctx: *anyopaque, h: Handle, out: *NodeStat) Error!void { - const s = self(ctx); - fillStat(s.node(h), h, out); - } - - fn list(ctx: *anyopaque, dir: Handle, index: usize, out: *NodeStat) Error!bool { - const s = self(ctx); - const d = s.node(dir); - if (!d.isDir()) return error.NotDir; - if (index >= d.children.items.len) return false; - const ch = d.children.items[index]; - fillStat(ch, s.handle(ch), out); - return true; - } - - fn open(ctx: *anyopaque, h: Handle, mode: u8) Error!void { - const s = self(ctx); - const n = s.node(h); - if (n.removed) return error.NotFound; - const acc = mode & 3; - const want_write = acc == cloud9.owrite or acc == cloud9.ordwr; - const want_read = !want_write or acc == cloud9.ordwr; - const trunc = mode & cloud9.otrunc != 0; - if (n.isDir()) { - if (want_write or trunc) return error.IsDir; - if (n.mode & 0o400 == 0) return error.Perm; - } else { - if (want_read and n.mode & 0o400 == 0) return error.Perm; - if ((want_write or trunc) and n.mode & 0o200 == 0) return error.Perm; - if (n.mode & cloud9.dmexcl != 0 and n.opens != 0) return error.Excl; - if (trunc and n.mode & cloud9.dmappend == 0) { - s.resizeData(n, 0) catch unreachable; // shrinking cannot fail - s.touch(n); - } - } - n.opens += 1; - } - - fn close(ctx: *anyopaque, h: Handle) void { - const s = self(ctx); - s.node(h).opens -= 1; - } - - fn read(ctx: *anyopaque, h: Handle, offset: u64, buf: []u8) Error!usize { - const s = self(ctx); - const n = s.node(h); - if (n.isDir()) return error.IsDir; - const src = n.data.items; - if (offset >= src.len) return 0; - const off: usize = @intCast(offset); - const len = @min(buf.len, src.len - off); - @memcpy(buf[0..len], src[off..][0..len]); - return len; - } - - fn write(ctx: *anyopaque, h: Handle, offset: u64, data: []const u8) Error!usize { - const s = self(ctx); - const n = s.node(h); - if (n.isDir()) return error.IsDir; - // A zero-length write changes nothing (and must not extend the file). - if (data.len == 0) return 0; - const off: usize = if (n.mode & cloud9.dmappend != 0) n.data.items.len else @intCast(@min(offset, s.max_file)); - const end = off + data.len; - if (end > s.max_file) return error.NoSpace; - if (end > n.data.items.len) try s.resizeData(n, end); - @memcpy(n.data.items[off..end], data); - s.touch(n); - return data.len; - } - - fn create(ctx: *anyopaque, dir: Handle, name: []const u8, perm: u32, mode: u8) Error!Handle { - const s = self(ctx); - const d = s.node(dir); - if (d.removed) return error.NotFound; - if (!d.isDir()) return error.NotDir; - if (d.mode & 0o200 == 0) return error.Perm; - if (d.find(name) != null) return error.Exists; - const is_dir = perm & cloud9.dmdir != 0; - const inherit: u32 = if (is_dir) d.mode & 0o777 else d.mode & 0o666; - const n = s.newNode(name, perm & (~@as(u32, 0o777) | inherit), d) catch return error.NoSpace; - s.touch(d); - n.opens += 1; - _ = mode; - return s.retain(n); - } - - fn remove(ctx: *anyopaque, h: Handle) Error!void { - const s = self(ctx); - const n = s.node(h); - if (n.removed) return error.NotFound; - const parent = n.parent orelse return error.Perm; - if (parent.mode & 0o200 == 0) return error.Perm; - if (n.isDir() and n.children.items.len != 0) return error.NotEmpty; - const i = std.mem.indexOfScalar(*Node, parent.children.items, n) orelse return error.NotFound; - _ = parent.children.orderedRemove(i); - s.touch(parent); - n.removed = true; - if (n.refs == 0) s.destroyNode(n); - } - - fn wstat(ctx: *anyopaque, h: Handle, st: *const cloud9.Stat) Error!void { - const s = self(ctx); - const n = s.node(h); - if (n.removed) return error.NotFound; - // Validate everything before changing anything. - const rename = st.name.len != 0 and !std.mem.eql(u8, st.name, n.name); - if (rename) { - const parent = n.parent orelse return error.Perm; - if (parent.find(st.name) != null) return error.Exists; - } - const cur_len: u64 = if (n.isDir()) 0 else n.data.items.len; - const set_len = st.length != 0xFFFF_FFFF_FFFF_FFFF and st.length != cur_len; - if (set_len) { - if (n.isDir()) return error.IsDir; - if (st.length > s.max_file) return error.NoSpace; - } - const set_mode = st.mode != 0xFFFF_FFFF and st.mode != n.mode; - if (set_mode and (st.mode & cloud9.dmdir) != (n.mode & cloud9.dmdir)) return error.Perm; - const set_mtime = st.mtime != 0xFFFF_FFFF and st.mtime != n.mtime; - if (!(rename or set_len or set_mode or set_mtime)) return; - const new_name: ?[]u8 = if (rename) s.gpa.dupe(u8, st.name) catch return error.NoSpace else null; - errdefer if (new_name) |nn| s.gpa.free(nn); - if (set_len) try s.resizeData(n, @intCast(st.length)); - // Nothing below can fail. - if (new_name) |nn| { - s.gpa.free(n.name); - n.name = nn; - s.touch(n.parent.?); - } - if (set_mode) n.mode = (n.mode & cloud9.dmdir) | (st.mode & ~cloud9.dmdir); - s.touch(n); - if (set_mtime) n.mtime = st.mtime; - } - - fn clunk(ctx: *anyopaque, h: Handle) void { - const s = self(ctx); - s.release(s.node(h)); - } -}; - -// --------------------------------------------------------------------------- -// Tests: the scratch tree mounted at /scratch of a core server. -// --------------------------------------------------------------------------- - -const testing = std.testing; - -const test_cfg: core.Config = .{ .name = "tester", .msize = 8192, .max_fids = 32 }; -const TS = core.Server(test_cfg); - -const budget: usize = 1 << 20; - -const Fixture = struct { - ctx: u8 = 0, - shared: TS.Shared = undefined, - storage: TS.Storage = undefined, - scratch: Scratch = undefined, - h: TS.Harness = undefined, - - fn init(x: *Fixture) !void { - x.shared = .init(&x.ctx); - x.scratch = try Scratch.init(testing.allocator, budget); - errdefer x.scratch.deinit(); - try x.shared.addProvider(x.scratch.provider("scratch")); - try x.h.init(&x.shared, &x.storage); - } - - fn deinit(x: *Fixture) void { - x.h.deinit(); - x.scratch.deinit(); - } - - fn nodeOf(x: *Fixture, fid: u32) *Node { - for (x.h.conn.fids) |f| if (f.used and f.id == fid) return x.scratch.node(f.node.prov.h); - unreachable; - } -}; - -const dontcare = core.stat_dontcare; - -test "scratch create/write/read/rename/truncate/remove" { - var x: Fixture = .{}; - try x.init(); - defer x.deinit(); - try x.h.walkTo(1, &.{"scratch"}); - // create + write - const cr = try x.h.ok(.{ .create = .{ .fid = 1, .name = "x", .perm = 0o644, .mode = cloud9.ordwr } }); - try testing.expectEqual(cloud9.qtfile, cr.create.qid.type); - _ = try x.h.ok(.{ .write = .{ .fid = 1, .offset = 0, .data = "hello" } }); - _ = try x.h.ok(.{ .write = .{ .fid = 1, .offset = 5, .data = " world" } }); - const r = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = 100 } }); - try testing.expectEqualStrings("hello world", r.read); - _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); - // rename x -> y - try x.h.walkTo(2, &.{ "scratch", "x" }); - var st = dontcare; - st.name = "y"; - _ = try x.h.ok(.{ .wstat = .{ .fid = 2, .stat = st } }); - try x.h.walkTo(3, &.{"scratch"}); - try x.h.expectFail(.{ .walk = .{ .fid = 3, .newfid = 30, .names = &.{"x"} } }, "file does not exist"); - _ = try x.h.ok(.{ .clunk = .{ .fid = 3 } }); - try x.h.walkTo(3, &.{ "scratch", "y" }); - // truncate then extend with zero fill - st = dontcare; - st.length = 2; - _ = try x.h.ok(.{ .wstat = .{ .fid = 3, .stat = st } }); - st.length = 4; - _ = try x.h.ok(.{ .wstat = .{ .fid = 3, .stat = st } }); - const text = try x.h.readAll(3); - defer testing.allocator.free(text); - try testing.expectEqualStrings("he\x00\x00", text); - const s3 = try x.h.ok(.{ .stat = .{ .fid = 3 } }); - try testing.expectEqualStrings("y", s3.stat.name); - try testing.expectEqual(@as(u64, 4), s3.stat.length); - try testing.expectEqualStrings("tester", s3.stat.uid); - _ = try x.h.ok(.{ .clunk = .{ .fid = 3 } }); - _ = try x.h.ok(.{ .clunk = .{ .fid = 2 } }); - // mkdir, nested create, remove rules - try x.h.walkTo(4, &.{"scratch"}); - const dr = try x.h.ok(.{ .create = .{ .fid = 4, .name = "d", .perm = cloud9.dmdir | 0o755, .mode = cloud9.oread } }); - try testing.expectEqual(cloud9.qtdir, dr.create.qid.type); - _ = try x.h.ok(.{ .clunk = .{ .fid = 4 } }); - try x.h.walkTo(5, &.{ "scratch", "d" }); - _ = try x.h.ok(.{ .create = .{ .fid = 5, .name = "inner", .perm = 0o600, .mode = cloud9.owrite } }); - _ = try x.h.ok(.{ .write = .{ .fid = 5, .offset = 0, .data = "z" } }); - _ = try x.h.ok(.{ .clunk = .{ .fid = 5 } }); - try x.h.walkTo(6, &.{ "scratch", "d" }); - try x.h.expectFail(.{ .remove = .{ .fid = 6 } }, "directory not empty"); - try x.h.expectFail(.{ .clunk = .{ .fid = 6 } }, "unknown fid"); // remove always clunks - try x.h.walkTo(7, &.{ "scratch", "d", "inner" }); - _ = try x.h.ok(.{ .remove = .{ .fid = 7 } }); - try x.h.walkTo(8, &.{ "scratch", "d" }); - _ = try x.h.ok(.{ .remove = .{ .fid = 8 } }); - try x.h.walkTo(9, &.{ "scratch", "y" }); - _ = try x.h.ok(.{ .remove = .{ .fid = 9 } }); - try x.h.walkTo(10, &.{"scratch"}); - _ = try x.h.ok(.{ .open = .{ .fid = 10, .mode = cloud9.oread } }); - const names = try x.h.listDir(10, 1024); - defer testing.allocator.free(names); - try testing.expectEqual(@as(usize, 0), names.len); - // append-only files ignore the offset - try x.h.walkTo(11, &.{"scratch"}); - _ = try x.h.ok(.{ .create = .{ .fid = 11, .name = "log", .perm = cloud9.dmappend | 0o644, .mode = cloud9.ordwr } }); - _ = try x.h.ok(.{ .write = .{ .fid = 11, .offset = 100, .data = "a" } }); - _ = try x.h.ok(.{ .write = .{ .fid = 11, .offset = 0, .data = "b" } }); - const lr = try x.h.ok(.{ .read = .{ .fid = 11, .offset = 0, .count = 10 } }); - try testing.expectEqualStrings("ab", lr.read); - try testing.expect(lr.read.len == 2); - const ls = try x.h.ok(.{ .stat = .{ .fid = 11 } }); - try testing.expect(ls.stat.qid.type & cloud9.qtappend != 0); - // the scratch root cannot be removed - try x.h.walkTo(12, &.{"scratch"}); - try x.h.expectFail(.{ .remove = .{ .fid = 12 } }, "permission denied"); -} - -test "walk of a missing name and walking a file" { - var x: Fixture = .{}; - try x.init(); - defer x.deinit(); - try x.h.walkTo(1, &.{"scratch"}); - try x.h.expectFail(.{ .walk = .{ .fid = 1, .newfid = 2, .names = &.{"nope"} } }, "file does not exist"); - _ = try x.h.ok(.{ .create = .{ .fid = 1, .name = "f", .perm = 0o644, .mode = cloud9.oread } }); - _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); - try x.h.walkTo(3, &.{ "scratch", "f" }); - try x.h.expectFail(.{ .walk = .{ .fid = 3, .newfid = 4, .names = &.{"x"} } }, "not a directory"); - // a walk that fails past the first element is a partial Rwalk that leaves newfid unused - const r = try x.h.ok(.{ .walk = .{ .fid = 0, .newfid = 4, .names = &.{ "scratch", "nope", "x" } } }); - try testing.expectEqual(@as(u16, 1), r.walk.nwqid); - try x.h.expectFail(.{ .clunk = .{ .fid = 4 } }, "unknown fid"); - // .. from a file is not a directory; .. from the scratch root reaches the server root - try x.h.expectFail(.{ .walk = .{ .fid = 3, .newfid = 5, .names = &.{".."} } }, "not a directory"); - try x.h.walkTo(5, &.{"scratch"}); - const up = try x.h.ok(.{ .walk = .{ .fid = 5, .newfid = 6, .names = &.{ "..", "scratch", "..", "README" } } }); - try testing.expectEqual(@as(u16, 4), up.walk.nwqid); - try testing.expectEqual(@as(u64, 0), up.walk.wqid[1].path); // provider 0, root - try testing.expect(up.walk.wqid[0].type & cloud9.qtdir != 0); -} - -test "directory read across consecutive offsets returns every record exactly once" { - var x: Fixture = .{}; - try x.init(); - defer x.deinit(); - const n = 40; - for (0..n) |i| { - try x.h.walkTo(1, &.{"scratch"}); - var name_buf: [64]u8 = undefined; - const name = try std.fmt.bufPrint(&name_buf, "file-with-a-long-name-{d:0>3}", .{i}); - _ = try x.h.ok(.{ .create = .{ .fid = 1, .name = name, .perm = 0o644, .mode = cloud9.oread } }); - _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); - } - try x.h.walkTo(2, &.{"scratch"}); - _ = try x.h.ok(.{ .open = .{ .fid = 2, .mode = cloud9.oread } }); - // 200 bytes fits two records, so this takes many reads. - const names = try x.h.listDir(2, 200); - defer TS.Harness.freeNames(names); - try testing.expectEqual(@as(usize, n), names.len); - var seen: [n]bool = @splat(false); - for (names) |nm| { - const idx = try std.fmt.parseInt(usize, nm[nm.len - 3 ..], 10); - try testing.expect(!seen[idx]); - seen[idx] = true; - } - for (seen) |s| try testing.expect(s); - try x.h.expectFail(.{ .read = .{ .fid = 2, .offset = 7, .count = 200 } }, "bad offset"); - // a read that cannot fit even one record returns nothing rather than splitting it - const tiny = try x.h.ok(.{ .read = .{ .fid = 2, .offset = 0, .count = 30 } }); - try testing.expectEqual(@as(usize, 0), tiny.read.len); - _ = try x.h.ok(.{ .clunk = .{ .fid = 2 } }); - // Tversion resets every fid and every reference - try x.h.version(8192); - try testing.expectEqual(@as(usize, 0), x.h.conn.fidCount()); - for (x.scratch.root.children.items) |ch| try testing.expectEqual(@as(u32, 0), ch.refs); - _ = try x.h.ok(.{ .attach = .{ .fid = 0, .uname = "tester" } }); -} - -test "DMEXCL admits one open fid at a time" { - var x: Fixture = .{}; - try x.init(); - defer x.deinit(); - try x.h.walkTo(1, &.{"scratch"}); - const cr = try x.h.ok(.{ .create = .{ .fid = 1, .name = "lock", .perm = cloud9.dmexcl | 0o644, .mode = cloud9.owrite } }); - try testing.expect(cr.create.qid.type & cloud9.qtexcl != 0); - try x.h.walkTo(2, &.{ "scratch", "lock" }); - try x.h.expectFail(.{ .open = .{ .fid = 2, .mode = cloud9.oread } }, "exclusive use file already open"); - _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); - _ = try x.h.ok(.{ .open = .{ .fid = 2, .mode = cloud9.oread } }); - try x.h.walkTo(3, &.{ "scratch", "lock" }); - try x.h.expectFail(.{ .open = .{ .fid = 3, .mode = cloud9.oread } }, "exclusive use file already open"); - // a Tversion reset drops the open and frees the file for the next session - try x.h.version(8192); - _ = try x.h.ok(.{ .attach = .{ .fid = 0, .uname = "tester" } }); - try x.h.walkTo(4, &.{ "scratch", "lock" }); - _ = try x.h.ok(.{ .open = .{ .fid = 4, .mode = cloud9.oread } }); - try testing.expectEqual(@as(u32, 1), x.nodeOf(4).opens); - _ = try x.h.ok(.{ .remove = .{ .fid = 4 } }); -} - -test "scratch memory: zero-length writes, truncation frees, global budget" { - var x: Fixture = .{}; - try x.init(); - defer x.deinit(); - x.scratch.max_file = 4096; - try x.h.walkTo(1, &.{"scratch"}); - _ = try x.h.ok(.{ .create = .{ .fid = 1, .name = "f", .perm = 0o644, .mode = cloud9.ordwr } }); - // a zero-length write at a huge offset must not extend the file - const w0 = try x.h.ok(.{ .write = .{ .fid = 1, .offset = std.math.maxInt(u64), .data = "" } }); - try testing.expectEqual(@as(u32, 0), w0.write); - var st = try x.h.ok(.{ .stat = .{ .fid = 1 } }); - try testing.expectEqual(@as(u64, 0), st.stat.length); - // growth is charged to the budget; truncation releases it (memory too) - _ = try x.h.ok(.{ .write = .{ .fid = 1, .offset = 1000, .data = "x" } }); - try testing.expectEqual(@as(usize, 1001), x.scratch.bytes); - var ws = dontcare; - ws.length = 10; - _ = try x.h.ok(.{ .wstat = .{ .fid = 1, .stat = ws } }); - try testing.expectEqual(@as(usize, 10), x.scratch.bytes); - try testing.expectEqual(@as(usize, 10), x.nodeOf(1).data.capacity); - // per-file cap and the global budget both answer "no space" - try x.h.expectFail(.{ .write = .{ .fid = 1, .offset = 4096, .data = "x" } }, "no space left on device"); - try x.h.expectFail(.{ .write = .{ .fid = 1, .offset = std.math.maxInt(u64), .data = "x" } }, "no space left on device"); - x.scratch.bytes = budget - 10; // pretend other files hold the rest - try x.h.expectFail(.{ .write = .{ .fid = 1, .offset = 10, .data = "0123456789A" } }, "no space left on device"); - _ = try x.h.ok(.{ .write = .{ .fid = 1, .offset = 10, .data = "0123456789" } }); - try testing.expectEqual(budget, x.scratch.bytes); - ws.length = 4096; - try x.h.expectFail(.{ .wstat = .{ .fid = 1, .stat = ws } }, "no space left on device"); - x.scratch.bytes -= budget - 20; - // OTRUNC releases too - try x.h.walkTo(2, &.{ "scratch", "f" }); - _ = try x.h.ok(.{ .open = .{ .fid = 2, .mode = cloud9.owrite | cloud9.otrunc } }); - try testing.expectEqual(@as(usize, 0), x.scratch.bytes); - st = try x.h.ok(.{ .stat = .{ .fid = 2 } }); - try testing.expectEqual(@as(u64, 0), st.stat.length); - // removing a file with content returns its bytes once the last fid lets go - _ = try x.h.ok(.{ .write = .{ .fid = 2, .offset = 0, .data = "abc" } }); - try testing.expectEqual(@as(usize, 3), x.scratch.bytes); - _ = try x.h.ok(.{ .remove = .{ .fid = 2 } }); - try testing.expectEqual(@as(usize, 3), x.scratch.bytes); // fid 1 still holds it - const r = try x.h.ok(.{ .read = .{ .fid = 1, .offset = 0, .count = 10 } }); - try testing.expectEqualStrings("abc", r.read); - _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); - try testing.expectEqual(@as(usize, 0), x.scratch.bytes); -} - -test "wstat with every field equal to the current stat changes nothing" { - var x: Fixture = .{}; - try x.init(); - defer x.deinit(); - try x.h.walkTo(1, &.{"scratch"}); - _ = try x.h.ok(.{ .create = .{ .fid = 1, .name = "same", .perm = 0o640, .mode = cloud9.oread } }); - const before = (try x.h.ok(.{ .stat = .{ .fid = 1 } })).stat; - var copy = before; - var name_buf: [core.max_name]u8 = undefined; - @memcpy(name_buf[0..before.name.len], before.name); - copy.name = name_buf[0..before.name.len]; - copy.uid = "tester"; - copy.gid = "tester"; - copy.muid = "tester"; - _ = try x.h.ok(.{ .wstat = .{ .fid = 1, .stat = copy } }); - const after = (try x.h.ok(.{ .stat = .{ .fid = 1 } })).stat; - try testing.expectEqual(before.qid, after.qid); - try testing.expectEqual(before.mtime, after.mtime); - try testing.expectEqual(before.mode, after.mode); - try testing.expectEqualStrings("same", after.name); - // and a rename to the very same name is also a no-op - var st = dontcare; - st.name = "same"; - _ = try x.h.ok(.{ .wstat = .{ .fid = 1, .stat = st } }); - try testing.expectEqual(before.qid, (try x.h.ok(.{ .stat = .{ .fid = 1 } })).stat.qid); - // renaming onto an existing sibling is refused - _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); - try x.h.walkTo(2, &.{"scratch"}); - _ = try x.h.ok(.{ .create = .{ .fid = 2, .name = "other", .perm = 0o640, .mode = cloud9.oread } }); - st.name = "same"; - try x.h.expectFail(.{ .wstat = .{ .fid = 2, .stat = st } }, "file already exists"); - // the mode's directory bit is immutable, mtime is settable - st = dontcare; - st.mode = cloud9.dmdir | 0o640; - try x.h.expectFail(.{ .wstat = .{ .fid = 2, .stat = st } }, "permission denied"); - st = dontcare; - st.mtime = 12345; - _ = try x.h.ok(.{ .wstat = .{ .fid = 2, .stat = st } }); - try testing.expectEqual(@as(u32, 12345), (try x.h.ok(.{ .stat = .{ .fid = 2 } })).stat.mtime); - _ = try x.h.ok(.{ .remove = .{ .fid = 2 } }); -} - -test "ORCLOSE removes on clunk and removed files stay readable through open fids" { - var x: Fixture = .{}; - try x.init(); - defer x.deinit(); - try x.h.walkTo(1, &.{"scratch"}); - _ = try x.h.ok(.{ .create = .{ .fid = 1, .name = "tmp", .perm = 0o644, .mode = cloud9.ordwr | cloud9.orclose } }); - _ = try x.h.ok(.{ .write = .{ .fid = 1, .offset = 0, .data = "gone" } }); - try x.h.walkTo(2, &.{ "scratch", "tmp" }); - _ = try x.h.ok(.{ .open = .{ .fid = 2, .mode = cloud9.oread } }); - _ = try x.h.ok(.{ .clunk = .{ .fid = 1 } }); - try x.h.walkTo(3, &.{"scratch"}); - try x.h.expectFail(.{ .walk = .{ .fid = 3, .newfid = 4, .names = &.{"tmp"} } }, "file does not exist"); - const r = try x.h.ok(.{ .read = .{ .fid = 2, .offset = 0, .count = 10 } }); - try testing.expectEqualStrings("gone", r.read); - try testing.expectEqual(@as(usize, 4), x.scratch.bytes); - _ = try x.h.ok(.{ .clunk = .{ .fid = 2 } }); - try testing.expectEqual(@as(usize, 0), x.scratch.bytes); - // create inside a removed directory fails - _ = try x.h.ok(.{ .create = .{ .fid = 3, .name = "d", .perm = cloud9.dmdir | 0o755, .mode = cloud9.oread } }); - try x.h.walkTo(5, &.{ "scratch", "d" }); - _ = try x.h.ok(.{ .remove = .{ .fid = 5 } }); - _ = try x.h.ok(.{ .clunk = .{ .fid = 3 } }); - try x.h.walkTo(6, &.{"scratch"}); - try x.h.expectFail(.{ .walk = .{ .fid = 6, .newfid = 7, .names = &.{"d"} } }, "file does not exist"); - try x.h.expectFail(.{ .create = .{ .fid = 5, .name = "x", .perm = 0o644, .mode = cloud9.oread } }, "unknown fid"); // remove clunked it -} - -test "qid paths are stable identities, not addresses: remove + recreate differ" { - var x: Fixture = .{}; - try x.init(); - defer x.deinit(); - try x.h.walkTo(1, &.{"scratch"}); - const a = try x.h.ok(.{ .create = .{ .fid = 1, .name = "f", .perm = 0o644, .mode = cloud9.oread } }); - const path_a = a.create.qid.path; - try testing.expectEqual(x.nodeOf(1).path, path_a & ((1 << 56) - 1)); - try testing.expect(path_a != 0); - // a rename keeps the identity - var st = dontcare; - st.name = "g"; - _ = try x.h.ok(.{ .wstat = .{ .fid = 1, .stat = st } }); - try testing.expectEqual(path_a, (try x.h.ok(.{ .stat = .{ .fid = 1 } })).stat.qid.path); - _ = try x.h.ok(.{ .remove = .{ .fid = 1 } }); - // the allocator very likely reuses the freed node's address here - try x.h.walkTo(2, &.{"scratch"}); - const b = try x.h.ok(.{ .create = .{ .fid = 2, .name = "f", .perm = 0o644, .mode = cloud9.oread } }); - try testing.expect(b.create.qid.path != path_a); - _ = try x.h.ok(.{ .remove = .{ .fid = 2 } }); - // the scratch root keeps path 0 (its handle), like every provider root - try x.h.walkTo(3, &.{"scratch"}); - try testing.expectEqual(@as(u64, 0), (try x.h.ok(.{ .stat = .{ .fid = 3 } })).stat.qid.path); - // directory listing reports the same identities as walking - _ = try x.h.ok(.{ .create = .{ .fid = 3, .name = "listed", .perm = 0o644, .mode = cloud9.oread } }); - const via_create = (try x.h.ok(.{ .stat = .{ .fid = 3 } })).stat.qid; - try x.h.walkTo(4, &.{"scratch"}); - _ = try x.h.ok(.{ .open = .{ .fid = 4, .mode = cloud9.oread } }); - const r = try x.h.ok(.{ .read = .{ .fid = 4, .offset = 0, .count = 1024 } }); - const listed = try cloud9.Stat.decode(r.read[0 .. std.mem.readInt(u16, r.read[0..2], .little) + 2]); - try testing.expectEqual(via_create, listed.qid); - _ = try x.h.ok(.{ .remove = .{ .fid = 3 } }); -} diff --git a/introspect/src/vars.zig b/introspect/src/vars.zig deleted file mode 100644 index 20cadfc..0000000 --- a/introspect/src/vars.zig +++ /dev/null @@ -1,824 +0,0 @@ -//! Comptime value renderers for /vars. For a type T, `vtableFor(T)` builds (at -//! comptime) a flat table of the files and directories that describe a value of -//! that type: -//! -//! /value rendered text /type @typeName -//! /size @sizeOf /addr 0x… -//! /raw the bytes /f//... recursively (depth <= max_depth) -//! -//! The core serves a variable by walking this table; a node index is the whole -//! state it needs. Rendering writes into a `*std.Io.Writer` and never allocates. -//! Writes to scalar `value` files are plain stores (not atomic). -const std = @import("std"); -const Writer = std.Io.Writer; - -/// Deepest `f/` nesting: /vars/x/f/a/f/b/f/c/f/d/value is depth 4. -pub const max_depth: u8 = 4; -/// Longest string rendered from a `[]const u8` / `[*:0]const u8` before "…". -pub const max_string: usize = 256; -/// Most array/slice elements rendered before "…". -pub const max_elems: usize = 64; - -pub const Kind = enum(u8) { - /// The directory of a value: value, type, size, addr, raw, [f]. - dir, - /// Rendered text; writable when `set` is non-null. - value, - /// @typeName, static content. - type_name, - /// @sizeOf, static content. - size, - /// "0x…" of the value's address. - addr, - /// The bytes of the value, length = size. - raw, - /// The `f` directory: one `dir` per struct field. - fields, -}; - -pub const RenderFn = *const fn (base: [*]const u8, w: *Writer) Writer.Error!void; -pub const SetFn = *const fn (base: [*]u8, text: []const u8) SetError!void; -pub const SetError = error{ Invalid, Unsupported }; - -/// One node of a type's tree. Children occupy `first..first+count`. -pub const Node = struct { - name: []const u8, - kind: Kind, - parent: u32, - first: u32 = 0, - count: u32 = 0, - /// Byte offset of the described value from the variable's base address. - offset: usize, - /// @sizeOf the described value. - size: usize, - /// Static text for `type_name` and `size` leaves. - content: []const u8 = "", - render: ?RenderFn = null, - set: ?SetFn = null, - - pub fn isDir(n: Node) bool { - return n.kind == .dir or n.kind == .fields; - } - - pub fn writable(n: Node) bool { - return n.kind == .value and n.set != null; - } -}; - -pub const VTable = struct { - nodes: []const Node, - type_name: []const u8, - size: usize, - - /// The child of `dir` named `name`, if any. - pub fn child(vt: *const VTable, dir: u32, name: []const u8) ?u32 { - const d = vt.nodes[dir]; - for (d.first..d.first + d.count) |i| { - if (std.mem.eql(u8, vt.nodes[i].name, name)) return @intCast(i); - } - return null; - } -}; - -/// The comptime-generated table for `T`; the same pointer for the same `T`. -pub fn vtableFor(comptime T: type) *const VTable { - const S = struct { - const nodes = buildTable(T); - const vt: VTable = .{ .nodes = &nodes, .type_name = @typeName(T), .size = @sizeOf(T) }; - }; - return &S.vt; -} - -/// Whether a struct's fields get an `f/` directory at this depth. -fn hasFields(comptime T: type, depth: u8) bool { - if (depth >= max_depth) return false; - return switch (@typeInfo(T)) { - .@"struct" => |s| s.layout != .@"packed" and fieldCount(T) > 0, - else => false, - }; -} - -fn fieldCount(comptime T: type) usize { - var n: usize = 0; - for (@typeInfo(T).@"struct".fields) |f| { - if (!f.is_comptime and @sizeOf(f.type) != 0) n += 1; - } - return n; -} - -fn countNodes(comptime T: type, depth: u8) usize { - var n: usize = 6; // dir + value, type, size, addr, raw - if (hasFields(T, depth)) { - n += 1; // f - for (@typeInfo(T).@"struct".fields) |f| { - if (f.is_comptime or @sizeOf(f.type) == 0) continue; - n += countNodes(f.type, depth + 1); - } - } - return n; -} - -fn fill(nodes: []Node, next: *usize, idx: usize, comptime T: type, name: []const u8, offset: usize, depth: u8, parent: u32) void { - const with_fields = hasFields(T, depth); - const count: u32 = if (with_fields) 6 else 5; - const first = next.*; - next.* += count; - nodes[idx] = .{ .name = name, .kind = .dir, .parent = parent, .first = @intCast(first), .count = count, .offset = offset, .size = @sizeOf(T) }; - const me: u32 = @intCast(idx); - nodes[first + 0] = .{ .name = "value", .kind = .value, .parent = me, .offset = offset, .size = @sizeOf(T), .render = renderFor(T), .set = setFor(T) }; - nodes[first + 1] = .{ .name = "type", .kind = .type_name, .parent = me, .offset = offset, .size = @sizeOf(T), .content = @typeName(T) }; - nodes[first + 2] = .{ .name = "size", .kind = .size, .parent = me, .offset = offset, .size = @sizeOf(T), .content = std.fmt.comptimePrint("{d}", .{@sizeOf(T)}) }; - nodes[first + 3] = .{ .name = "addr", .kind = .addr, .parent = me, .offset = offset, .size = @sizeOf(T) }; - nodes[first + 4] = .{ .name = "raw", .kind = .raw, .parent = me, .offset = offset, .size = @sizeOf(T) }; - if (with_fields) { - const fdir: u32 = @intCast(first + 5); - const nf = fieldCount(T); - const ffirst = next.*; - next.* += nf; - nodes[fdir] = .{ .name = "f", .kind = .fields, .parent = me, .first = @intCast(ffirst), .count = @intCast(nf), .offset = offset, .size = @sizeOf(T) }; - var i: usize = 0; - for (@typeInfo(T).@"struct".fields) |f| { - if (f.is_comptime or @sizeOf(f.type) == 0) continue; - fill(nodes, next, ffirst + i, f.type, f.name, offset + @offsetOf(T, f.name), depth + 1, fdir); - i += 1; - } - } -} - -fn buildTable(comptime T: type) [countNodes(T, 0)]Node { - @setEvalBranchQuota(1_000_000); - var nodes: [countNodes(T, 0)]Node = undefined; - var next: usize = 1; - fill(&nodes, &next, 0, T, "", 0, 0, 0); - std.debug.assert(next == nodes.len); - return nodes; -} - -fn renderFor(comptime T: type) RenderFn { - return &struct { - fn f(base: [*]const u8, w: *Writer) Writer.Error!void { - const p: *const T = @ptrCast(@alignCast(base)); - try render(p, w, max_depth); - } - }.f; -} - -fn setFor(comptime T: type) ?SetFn { - if (!settable(T)) return null; - return &struct { - fn f(base: [*]u8, text: []const u8) SetError!void { - const p: *T = @ptrCast(@alignCast(base)); - try set(p, text); - } - }.f; -} - -fn settable(comptime T: type) bool { - return switch (@typeInfo(T)) { - .int, .float, .bool, .@"enum" => true, - else => false, - }; -} - -// --------------------------------------------------------------------------- -// Rendering -// --------------------------------------------------------------------------- - -/// Renders `ptr.*`. Structs become "field: value" lines (nested structs -/// indented); everything else is a single line without a trailing newline. -/// `depth` bounds struct/union/optional nesting; deeper values print as "…". -pub fn render(ptr: anytype, w: *Writer, depth: usize) Writer.Error!void { - const T = @TypeOf(ptr.*); - if (comptime isPlainStruct(T)) { - try renderStruct(T, ptr, w, depth, 0); - } else { - try renderValue(T, ptr, w, depth); - } -} - -fn isPlainStruct(comptime T: type) bool { - return switch (@typeInfo(T)) { - .@"struct" => |s| !s.is_tuple and s.fields.len > 0, - else => false, - }; -} - -/// The multi-line form: each field on its own line, nested structs indented. -fn renderStruct(comptime T: type, ptr: *const T, w: *Writer, depth: usize, indent: usize) Writer.Error!void { - if (depth == 0) { - try w.splatByteAll(' ', indent); - try w.writeAll("…\n"); - return; - } - const packed_layout = @typeInfo(T).@"struct".layout == .@"packed"; - inline for (@typeInfo(T).@"struct".fields) |f| { - try w.splatByteAll(' ', indent); - try w.writeAll(f.name); - try w.writeByte(':'); - if (comptime f.is_comptime) { - try w.writeAll(" (comptime)\n"); - } else if (comptime packed_layout) { - // Fields of a packed struct have no byte address: render a copy. - const v = @field(ptr.*, f.name); - try w.writeByte(' '); - try renderValue(f.type, &v, w, depth - 1); - try w.writeByte('\n'); - } else if (comptime isPlainStruct(f.type)) { - try w.writeByte('\n'); - try renderStruct(f.type, &@field(ptr.*, f.name), w, depth - 1, indent + 2); - } else { - try w.writeByte(' '); - try renderValue(f.type, &@field(ptr.*, f.name), w, depth - 1); - try w.writeByte('\n'); - } - } -} - -/// The single-line form of any value. -fn renderValue(comptime T: type, ptr: *const T, w: *Writer, depth: usize) Writer.Error!void { - switch (@typeInfo(T)) { - .int, .comptime_int => try w.print("{d}", .{ptr.*}), - .float, .comptime_float => try w.print("{d}", .{ptr.*}), - .bool => { - // The variable is live memory that anything (a debugger's /mem write, - // a torn update) may have corrupted: judge the byte, not the bool. - const b = @as(*const u8, @ptrCast(ptr)).*; - switch (b) { - 0 => try w.writeAll("false"), - 1 => try w.writeAll("true"), - else => try w.print("{d}", .{b}), - } - }, - .void => try w.writeAll("{}"), - .@"enum" => |e| { - // Read the storage bytes as one integer: @tagName/switch on a corrupt - // value is a safety panic, and the value is caller memory we do not - // control. A load through the tag type would truncate to its bit - // width (a u2 tag in a byte), so the full storage width is read. - if (@sizeOf(T) == 0) return w.writeAll(e.fields[0].name); - const Raw = std.meta.Int(.unsigned, @sizeOf(T) * 8); - const raw = @as(*align(@alignOf(T)) const Raw, @ptrCast(ptr)).*; - const TagU = std.meta.Int(.unsigned, @bitSizeOf(e.tag_type)); - const padding: Raw = if (@bitSizeOf(TagU) == @bitSizeOf(Raw)) 0 else ~@as(Raw, std.math.maxInt(TagU)); - if (raw & padding == 0) { - const low: TagU = @truncate(raw); - inline for (e.fields) |f| { - if (low == @as(TagU, @bitCast(@as(e.tag_type, f.value)))) return w.writeAll(f.name); - } - } - try w.print("{d}", .{raw}); - }, - .error_set => try w.print("error.{s}", .{@errorName(ptr.*)}), - .error_union => |eu| if (ptr.*) |v| { - try renderValue(eu.payload, &v, w, depth); - } else |e| { - try w.print("error.{s}", .{@errorName(e)}); - }, - .optional => |o| if (ptr.*) |v| { - try renderValue(o.child, &v, w, depth); - } else { - try w.writeAll("null"); - }, - .pointer => |p| switch (p.size) { - .slice => if (p.child == u8) { - try renderString(ptr.*, w); - } else { - try renderElems(p.child, ptr.*, w, depth); - }, - .many => if (p.child == u8 and p.sentinel() == 0) { - try renderCString(ptr.*, w); - } else { - try w.print("0x{x}", .{@intFromPtr(ptr.*)}); - }, - .one, .c => try w.print("0x{x}", .{@intFromPtr(ptr.*)}), - }, - .array => |a| if (a.child == u8) { - try renderString(ptr.*[0..], w); - } else { - try renderElems(a.child, ptr.*[0..], w, depth); - }, - .vector => |v| { - const arr: [v.len]v.child = ptr.*; - try renderElems(v.child, &arr, w, depth); - }, - .@"struct" => |s| { - if (depth == 0) { - try w.writeAll("…"); - return; - } - if (s.fields.len == 0) { - try w.writeAll("{}"); - return; - } - try w.writeAll("{ "); - inline for (s.fields, 0..) |f, i| { - if (i != 0) try w.writeAll(", "); - if (!s.is_tuple) { - try w.writeAll(f.name); - try w.writeAll(": "); - } - if (comptime f.is_comptime) { - try w.writeAll("(comptime)"); - } else if (comptime s.layout == .@"packed") { - const v = @field(ptr.*, f.name); - try renderValue(f.type, &v, w, depth - 1); - } else { - try renderValue(f.type, &@field(ptr.*, f.name), w, depth - 1); - } - } - try w.writeAll(" }"); - }, - .@"union" => |u| { - const Tag = u.tag_type orelse { - try w.print("(untagged union, {d} bytes)", .{@sizeOf(T)}); - return; - }; - if (depth == 0) { - try w.writeAll("…"); - return; - } - // A switch on a corrupt tag is a safety panic: match the integer first. - const raw = @intFromEnum(@as(Tag, ptr.*)); - inline for (u.fields) |f| { - if (raw == @intFromEnum(@field(Tag, f.name))) { - try w.writeAll(f.name); - if (f.type != void) { - try w.writeAll(": "); - try renderValue(f.type, &@field(ptr.*, f.name), w, depth - 1); - } - return; - } - } - try w.print("(invalid tag {d})", .{raw}); - }, - .@"fn" => try w.print("0x{x}", .{@intFromPtr(ptr)}), - else => try w.print("<{s}>", .{@typeName(T)}), - } -} - -fn renderElems(comptime E: type, items: []const E, w: *Writer, depth: usize) Writer.Error!void { - try w.writeByte('['); - for (items, 0..) |*item, i| { - if (i == max_elems) { - try w.writeAll(", …"); - break; - } - if (i != 0) try w.writeAll(", "); - try renderValue(E, item, w, depth); - } - try w.writeByte(']'); -} - -/// A NUL-terminated string, scanning at most `max_string` + 1 bytes for the -/// terminator so that a missing one cannot walk off the end of the mapping. -fn renderCString(s: [*:0]const u8, w: *Writer) Writer.Error!void { - var n: usize = 0; - while (n <= max_string and s[n] != 0) n += 1; - try renderString(s[0..n], w); -} - -/// A double-quoted string with C-style escapes, truncated to `max_string` bytes. -fn renderString(s: []const u8, w: *Writer) Writer.Error!void { - try w.writeByte('"'); - for (s[0..@min(s.len, max_string)]) |b| switch (b) { - '\n' => try w.writeAll("\\n"), - '\r' => try w.writeAll("\\r"), - '\t' => try w.writeAll("\\t"), - '\\' => try w.writeAll("\\\\"), - '"' => try w.writeAll("\\\""), - ' '...'!', '#'...'[', ']'...'~' => try w.writeByte(b), - else => { - const hex = "0123456789abcdef"; - try w.writeAll("\\x"); - try w.writeByte(hex[b >> 4]); - try w.writeByte(hex[b & 15]); - }, - }; - try w.writeByte('"'); - if (s.len > max_string) try w.writeAll("…"); -} - -// --------------------------------------------------------------------------- -// Setting -// --------------------------------------------------------------------------- - -/// Parses `text` and stores it into `ptr.*`: ints in decimal or 0x/0o/0b, -/// floats, bools (true/false/1/0), enums by tag name (or by integer value for -/// non-exhaustive enums). Other types are `error.Unsupported`. -pub fn set(ptr: anytype, text: []const u8) SetError!void { - const T = @TypeOf(ptr.*); - const s = std.mem.trim(u8, text, " \t\r\n\x00"); - switch (@typeInfo(T)) { - .int => ptr.* = std.fmt.parseInt(T, s, 0) catch return error.Invalid, - .float => ptr.* = std.fmt.parseFloat(T, s) catch return error.Invalid, - .bool => { - if (std.mem.eql(u8, s, "true") or std.mem.eql(u8, s, "1")) { - ptr.* = true; - } else if (std.mem.eql(u8, s, "false") or std.mem.eql(u8, s, "0")) { - ptr.* = false; - } else return error.Invalid; - }, - .@"enum" => |e| { - if (std.meta.stringToEnum(T, s)) |v| { - ptr.* = v; - } else if (!e.is_exhaustive) { - const raw = std.fmt.parseInt(e.tag_type, s, 0) catch return error.Invalid; - ptr.* = @enumFromInt(raw); - } else return error.Invalid; - }, - else => return error.Unsupported, - } -} - -// --------------------------------------------------------------------------- -// Tests -// --------------------------------------------------------------------------- - -const testing = std.testing; - -fn renderToBuf(buf: []u8, ptr: anytype) ![]const u8 { - var w: Writer = .fixed(buf); - try render(ptr, &w, max_depth); - return w.buffered(); -} - -test "render scalars, strings, pointers, optionals, enums, arrays" { - var buf: [512]u8 = undefined; - const i: i32 = -42; - try testing.expectEqualStrings("-42", try renderToBuf(&buf, &i)); - const f: f32 = 1.5; - try testing.expectEqualStrings("1.5", try renderToBuf(&buf, &f)); - const b: bool = true; - try testing.expectEqualStrings("true", try renderToBuf(&buf, &b)); - const s: []const u8 = "hi \"there\"\n"; - try testing.expectEqualStrings("\"hi \\\"there\\\"\\n\"", try renderToBuf(&buf, &s)); - const z: [*:0]const u8 = "zed"; - try testing.expectEqualStrings("\"zed\"", try renderToBuf(&buf, &z)); - const p: *const i32 = &i; - var expect_buf: [32]u8 = undefined; - const expect = try std.fmt.bufPrint(&expect_buf, "0x{x}", .{@intFromPtr(&i)}); - try testing.expectEqualStrings(expect, try renderToBuf(&buf, &p)); - const o: ?u8 = null; - try testing.expectEqualStrings("null", try renderToBuf(&buf, &o)); - const o2: ?u8 = 7; - try testing.expectEqualStrings("7", try renderToBuf(&buf, &o2)); - const E = enum { red, green }; - const e: E = .green; - try testing.expectEqualStrings("green", try renderToBuf(&buf, &e)); - const NE = enum(u8) { a, _ }; - const ne: NE = @enumFromInt(9); - try testing.expectEqualStrings("9", try renderToBuf(&buf, &ne)); - const arr = [_]u16{ 1, 2, 3 }; - try testing.expectEqualStrings("[1, 2, 3]", try renderToBuf(&buf, &arr)); - const bytes = [_]u8{ 0, 'a', 0xff }; - try testing.expectEqualStrings("\"\\x00a\\xff\"", try renderToBuf(&buf, &bytes)); - const U = union(enum) { none, some: u32 }; - const u: U = .{ .some = 5 }; - try testing.expectEqualStrings("some: 5", try renderToBuf(&buf, &u)); - const un: U = .none; - try testing.expectEqualStrings("none", try renderToBuf(&buf, &un)); -} - -test "render structs multi-line with nested indentation and depth limit" { - const Inner = struct { x: f32, flags: [2]bool }; - const Outer = struct { a: u32, b: bool, name: []const u8, inner: Inner, items: []const Inner }; - const v: Outer = .{ .a = 1, .b = false, .name = "n", .inner = .{ .x = 2.5, .flags = .{ true, false } }, .items = &.{.{ .x = 0, .flags = .{ false, false } }} }; - var buf: [512]u8 = undefined; - try testing.expectEqualStrings( - \\a: 1 - \\b: false - \\name: "n" - \\inner: - \\ x: 2.5 - \\ flags: [true, false] - \\items: [{ x: 0, flags: [false, false] }] - \\ - , try renderToBuf(&buf, &v)); - var w: Writer = .fixed(&buf); - try render(&v, &w, 1); - try testing.expectEqualStrings( - \\a: 1 - \\b: false - \\name: "n" - \\inner: - \\ … - \\items: […] - \\ - , w.buffered()); -} - -test "long strings and arrays are truncated" { - const long = [_]u8{'x'} ** 300; - var buf: [1024]u8 = undefined; - const s: []const u8 = &long; - const out = try renderToBuf(&buf, &s); - try testing.expectEqual(@as(usize, 1 + max_string + 1 + "…".len), out.len); - try testing.expect(std.mem.endsWith(u8, out, "\"…")); - const nums: [100]u32 = @splat(1); - const out2 = try renderToBuf(&buf, &nums); - try testing.expect(std.mem.endsWith(u8, out2, ", …]")); - try testing.expectEqual(@as(usize, max_elems), std.mem.count(u8, out2, "1")); -} - -test "set parses ints, floats, bools and enums" { - var i: u32 = 0; - try set(&i, "42\n"); - try testing.expectEqual(@as(u32, 42), i); - try set(&i, "0x10"); - try testing.expectEqual(@as(u32, 16), i); - try testing.expectError(error.Invalid, set(&i, "-1")); - try testing.expectError(error.Invalid, set(&i, "abc")); - var si: i8 = 0; - try set(&si, " -7 "); - try testing.expectEqual(@as(i8, -7), si); - try testing.expectError(error.Invalid, set(&si, "200")); - var f: f64 = 0; - try set(&f, "2.25"); - try testing.expectEqual(@as(f64, 2.25), f); - var b: bool = false; - try set(&b, "true"); - try testing.expect(b); - try set(&b, "0"); - try testing.expect(!b); - try testing.expectError(error.Invalid, set(&b, "maybe")); - const E = enum { off, on }; - var e: E = .off; - try set(&e, "on"); - try testing.expectEqual(E.on, e); - try testing.expectError(error.Invalid, set(&e, "blue")); - var s: []const u8 = "x"; - try testing.expectError(error.Unsupported, set(&s, "y")); -} - -test "vtable table layout for a nested struct" { - const Inner = struct { x: f32 }; - const T = struct { a: u32, b: bool, name: []const u8, inner: Inner }; - const vt = vtableFor(T); - try testing.expectEqual(vt, vtableFor(T)); - try testing.expectEqualStrings(@typeName(T), vt.type_name); - const root = vt.nodes[0]; - try testing.expect(root.isDir()); - try testing.expectEqual(@as(u32, 6), root.count); - const value = vt.child(0, "value").?; - try testing.expect(!vt.nodes[value].writable()); // a struct is not settable - try testing.expectEqualStrings(std.fmt.comptimePrint("{d}", .{@sizeOf(T)}), vt.nodes[vt.child(0, "size").?].content); - const f = vt.child(0, "f").?; - try testing.expectEqual(Kind.fields, vt.nodes[f].kind); - try testing.expectEqual(@as(u32, 4), vt.nodes[f].count); - const a = vt.child(f, "a").?; - try testing.expectEqual(@offsetOf(T, "a"), vt.nodes[a].offset); - const a_value = vt.child(a, "value").?; - try testing.expect(vt.nodes[a_value].writable()); - try testing.expectEqual(@as(usize, 4), vt.nodes[a_value].size); - try testing.expectEqual(a, vt.nodes[a_value].parent); - const inner = vt.child(f, "inner").?; - const inner_f = vt.child(inner, "f").?; - const x = vt.child(inner_f, "x").?; - try testing.expectEqual(@offsetOf(T, "inner") + @offsetOf(Inner, "x"), vt.nodes[x].offset); - try testing.expectEqualStrings("f32", vt.nodes[vt.child(x, "type").?].content); - try testing.expect(vt.child(f, "nope") == null); - // rendering and setting through the table - var v: T = .{ .a = 1, .b = true, .name = "n", .inner = .{ .x = 0.5 } }; - const base: [*]u8 = @ptrCast(&v); - var buf: [256]u8 = undefined; - var w: Writer = .fixed(&buf); - const x_value = vt.child(x, "value").?; - try vt.nodes[x_value].render.?(base + vt.nodes[x_value].offset, &w); - try testing.expectEqualStrings("0.5", w.buffered()); - try vt.nodes[a_value].set.?(base + vt.nodes[a_value].offset, "42"); - try testing.expectEqual(@as(u32, 42), v.a); - w = .fixed(&buf); - try vt.nodes[value].render.?(base, &w); - try testing.expect(std.mem.startsWith(u8, w.buffered(), "a: 42\nb: true\n")); -} - -test "depth limit stops the f/ tree at max_depth" { - const L4 = struct { v: u8 }; - const L3 = struct { l4: L4 }; - const L2 = struct { l3: L3 }; - const L1 = struct { l2: L2 }; - const L0 = struct { l1: L1 }; - const vt = vtableFor(L0); - var node: u32 = 0; - var depth: usize = 0; - while (vt.child(node, "f")) |f| : (depth += 1) { - node = vt.nodes[f].first; // the single field - } - try testing.expectEqual(@as(usize, max_depth), depth); - try testing.expect(vt.child(node, "value") != null); -} - -test "every @typeInfo category renders without dereferencing anything unbounded" { - var buf: [2048]u8 = undefined; - // packed and extern structs (packed fields have no address: rendered by copy) - const Packed = packed struct { a: u3, b: bool, c: u12, e: enum(u2) { p, q, r } }; - const pk: Packed = .{ .a = 5, .b = true, .c = 300, .e = .r }; - try testing.expectEqualStrings("a: 5\nb: true\nc: 300\ne: r\n", try renderToBuf(&buf, &pk)); - const Ext = extern struct { x: u16, y: f32, inner: extern struct { z: u8 } }; - const ex: Ext = .{ .x = 1, .y = 0.5, .inner = .{ .z = 9 } }; - try testing.expectEqualStrings("x: 1\ny: 0.5\ninner:\n z: 9\n", try renderToBuf(&buf, &ex)); - const Holder = struct { p: Packed, list: [2]Packed }; - const ho: Holder = .{ .p = pk, .list = .{ pk, pk } }; - try testing.expect(std.mem.startsWith(u8, try renderToBuf(&buf, &ho), "p:\n a: 5\n")); - // the f/ tree has no entries for a packed struct and works through a table - const vt = vtableFor(Packed); - try testing.expect(vt.child(0, "f") == null); - var w: Writer = .fixed(&buf); - try vt.nodes[vt.child(0, "value").?].render.?(@ptrCast(&pk), &w); - try testing.expect(std.mem.startsWith(u8, w.buffered(), "a: 5\n")); - // optionals of pointers are printed, never followed - var target: u32 = 7; - const op: ?*u32 = ⌖ - var expect_buf: [32]u8 = undefined; - try testing.expectEqualStrings(try std.fmt.bufPrint(&expect_buf, "0x{x}", .{@intFromPtr(&target)}), try renderToBuf(&buf, &op)); - const np: ?*u32 = null; - try testing.expectEqualStrings("null", try renderToBuf(&buf, &np)); - const dangling: *const u32 = @ptrFromInt(0x1000); - try testing.expectEqualStrings("0x1000", try renderToBuf(&buf, &dangling)); - const cptr: [*c]const u8 = @ptrFromInt(0x2000); - try testing.expectEqualStrings("0x2000", try renderToBuf(&buf, &cptr)); - const manyp: [*]const u32 = @ptrFromInt(0x3000); - try testing.expectEqualStrings("0x3000", try renderToBuf(&buf, &manyp)); - // untagged and tagged unions, error unions, error sets - const Untagged = union { a: u32, b: f32 }; - const un: Untagged = .{ .a = 1 }; - try testing.expectEqualStrings(std.fmt.comptimePrint("(untagged union, {d} bytes)", .{@sizeOf(Untagged)}), try renderToBuf(&buf, &un)); - const Tagged = union(enum(u8)) { none, some: u32, pair: struct { l: u8, r: u8 } }; - const tg: Tagged = .{ .pair = .{ .l = 1, .r = 2 } }; - try testing.expectEqualStrings("pair: { l: 1, r: 2 }", try renderToBuf(&buf, &tg)); - const eu: anyerror!u8 = error.Boom; - try testing.expectEqualStrings("error.Boom", try renderToBuf(&buf, &eu)); - const eu2: error{X}!u8 = 4; - try testing.expectEqualStrings("4", try renderToBuf(&buf, &eu2)); - const es: anyerror = error.Zap; - try testing.expectEqualStrings("error.Zap", try renderToBuf(&buf, &es)); - // wide ints and floats, vectors, sentinel arrays, slices of slices, void, comptime fields - const big: u128 = std.math.maxInt(u128); - try testing.expectEqualStrings("340282366920938463463374607431768211455", try renderToBuf(&buf, &big)); - const neg: i128 = std.math.minInt(i128); - try testing.expectEqualStrings("-170141183460469231731687303715884105728", try renderToBuf(&buf, &neg)); - const h: f16 = 1.5; - try testing.expectEqualStrings("1.5", try renderToBuf(&buf, &h)); - const ld: f80 = 2.25; - try testing.expectEqualStrings("2.25", try renderToBuf(&buf, &ld)); - const quad: f128 = 3.125; - try testing.expectEqualStrings("3.125", try renderToBuf(&buf, &quad)); - const vec: @Vector(4, i16) = .{ 1, -2, 3, -4 }; - try testing.expectEqualStrings("[1, -2, 3, -4]", try renderToBuf(&buf, &vec)); - const sarr: [3:0]u8 = .{ 'a', 'b', 'c' }; - try testing.expectEqualStrings("\"abc\"", try renderToBuf(&buf, &sarr)); - const rows: []const []const u8 = &.{ "ab", "cd" }; - try testing.expectEqualStrings("[\"ab\", \"cd\"]", try renderToBuf(&buf, &rows)); - const Odd = struct { v: void, comptime k: u8 = 3, n: u8 }; - const odd: Odd = .{ .v = {}, .n = 1 }; - try testing.expectEqualStrings("v: {}\nk: (comptime)\nn: 1\n", try renderToBuf(&buf, &odd)); - try testing.expectEqual(@as(u32, 1), vtableFor(Odd).nodes[vtableFor(Odd).child(0, "f").?].count); - // self-referential through a pointer: rendered as an address, table stays finite - const Link = struct { next: ?*const @This(), v: u8 }; - var a: Link = .{ .next = null, .v = 1 }; - const b: Link = .{ .next = &a, .v = 2 }; - a.next = &b; - try testing.expectEqualStrings(try std.fmt.bufPrint(&expect_buf, "next: 0x{x}\nv: 2\n", .{@intFromPtr(&a)}), try renderToBuf(&buf, &b)); - try testing.expect(vtableFor(Link).nodes.len < 32); - // tuples - const tup: struct { u8, []const u8 } = .{ 1, "x" }; - try testing.expectEqualStrings("{ 1, \"x\" }", try renderToBuf(&buf, &tup)); -} - -test "corrupt live memory renders instead of trapping: enums, unions, bools" { - var buf: [128]u8 = undefined; - const E = enum(u8) { a, b }; - var raw_e: u8 = 7; - try testing.expectEqualStrings("7", try renderToBuf(&buf, @as(*const E, @ptrCast(&raw_e)))); - raw_e = 1; - try testing.expectEqualStrings("b", try renderToBuf(&buf, @as(*const E, @ptrCast(&raw_e)))); - // a u2 tag in a byte: the whole byte is judged, not the truncated tag (ReleaseSafe would say "c") - const E3 = enum { a, b, c }; - var raw3: u8 = 0xEE; - try testing.expectEqualStrings("238", try renderToBuf(&buf, @as(*const E3, @ptrCast(&raw3)))); - raw3 = 3; - try testing.expectEqualStrings("3", try renderToBuf(&buf, @as(*const E3, @ptrCast(&raw3)))); - raw3 = 2; - try testing.expectEqualStrings("c", try renderToBuf(&buf, @as(*const E3, @ptrCast(&raw3)))); - const E12 = enum(u12) { p = 5, q = 4095 }; - var raw12: u16 = 0xF005; - try testing.expectEqualStrings("61445", try renderToBuf(&buf, @as(*const E12, @ptrCast(&raw12)))); - raw12 = 4095; - try testing.expectEqualStrings("q", try renderToBuf(&buf, @as(*const E12, @ptrCast(&raw12)))); - const ES = enum(i8) { neg = -3, pos = 7 }; - var raws: u8 = 0xFD; - try testing.expectEqualStrings("neg", try renderToBuf(&buf, @as(*const ES, @ptrCast(&raws)))); - raws = 0x80; - try testing.expectEqualStrings("128", try renderToBuf(&buf, @as(*const ES, @ptrCast(&raws)))); - const E1 = enum { only }; - const e1: E1 = .only; - try testing.expectEqualStrings("only", try renderToBuf(&buf, &e1)); - const NE = enum(u16) { x = 5, _ }; - var raw_ne: u16 = 5; - try testing.expectEqualStrings("x", try renderToBuf(&buf, @as(*const NE, @ptrCast(&raw_ne)))); - raw_ne = 6; - try testing.expectEqualStrings("6", try renderToBuf(&buf, @as(*const NE, @ptrCast(&raw_ne)))); - var raw_b: u8 = 2; - try testing.expectEqualStrings("2", try renderToBuf(&buf, @as(*const bool, @ptrCast(&raw_b)))); - const U = union(enum(u8)) { x: u32, y: bool }; - var raw_u: [@sizeOf(U)]u8 align(@alignOf(U)) = @splat(0x55); - const out = try renderToBuf(&buf, @as(*const U, @ptrCast(&raw_u))); - try testing.expectEqualStrings("(invalid tag 85)", out); - const S = struct { e: E, u: U, b: bool }; - var raw_s: [@sizeOf(S)]u8 align(@alignOf(S)) = @splat(0xEE); - const ps: *const S = @ptrCast(&raw_s); - _ = try renderToBuf(&buf, ps); // no trap - try testing.expect(std.mem.indexOf(u8, try renderToBuf(&buf, ps), "238") != null); -} - -test "a [*:0]const u8 without a terminator is read at most max_string + 1 bytes" { - // Only the first max_string + 1 bytes exist; anything beyond is the - // testing allocator's guard, which a wider scan would touch. - const mem = try testing.allocator.alloc(u8, max_string + 1); - defer testing.allocator.free(mem); - @memset(mem, 'x'); - const z: [*:0]const u8 = @ptrCast(mem.ptr); - var buf: [1024]u8 = undefined; - const out = try renderToBuf(&buf, &z); - try testing.expectEqual(@as(usize, 1 + max_string + 1 + "…".len), out.len); - try testing.expect(std.mem.endsWith(u8, out, "\"…")); - // exactly max_string bytes then NUL: no ellipsis - const mem2 = try testing.allocator.alloc(u8, max_string + 1); - defer testing.allocator.free(mem2); - @memset(mem2, 'y'); - mem2[max_string] = 0; - const z2: [*:0]const u8 = @ptrCast(mem2.ptr); - const out2 = try renderToBuf(&buf, &z2); - try testing.expectEqual(@as(usize, 1 + max_string + 1), out2.len); - // a garbage-length []const u8 still reads at most max_string bytes - const garbage: []const u8 = mem[0..max_string]; - _ = try renderToBuf(&buf, &garbage); -} - -test "set rejects hostile input without partial writes" { - var u: u8 = 200; - for ([_][]const u8{ "-1", "256", "1e3", "0x", "", " ", "1.5", "+", "0b2", "١", "12abc", "0x100", "\x00", "1 2" }) |bad| { - try testing.expectError(error.Invalid, set(&u, bad)); - try testing.expectEqual(@as(u8, 200), u); - } - try set(&u, "0b1111_1111"); - try testing.expectEqual(@as(u8, 255), u); - try set(&u, "+0o17"); - try testing.expectEqual(@as(u8, 15), u); - var i: i64 = 1; - try set(&i, "-9223372036854775808"); - try testing.expectEqual(std.math.minInt(i64), i); - try testing.expectError(error.Invalid, set(&i, "9223372036854775808")); - var w: u128 = 0; - try set(&w, "340282366920938463463374607431768211455"); - try testing.expectEqual(std.math.maxInt(u128), w); - try testing.expectError(error.Invalid, set(&w, "340282366920938463463374607431768211456")); - // floats: exponents, hex floats, inf/nan spellings, and junk - var f: f32 = 1; - try set(&f, "1.5e3"); - try testing.expectEqual(@as(f32, 1500), f); - try set(&f, "-0x1p-2"); - try testing.expectEqual(@as(f32, -0.25), f); - try set(&f, "1e999"); - try testing.expect(std.math.isInf(f)); - try testing.expectError(error.Invalid, set(&f, "1.5.5")); - try testing.expectError(error.Invalid, set(&f, "e5")); - try testing.expectError(error.Invalid, set(&f, "")); - var h: f16 = 0; - try set(&h, "65504"); - try testing.expectEqual(@as(f16, 65504), h); - var q: f128 = 0; - try set(&q, "2.5"); - try testing.expectEqual(@as(f128, 2.5), q); - // enums: NULs inside the tag, case, trailing junk; non-exhaustive by integer only when out of names - const E = enum(u8) { off, on }; - var e: E = .off; - for ([_][]const u8{ "on\x00x", "On", "on x", "1", "0x1", "" }) |bad| { - try testing.expectError(error.Invalid, set(&e, bad)); - try testing.expectEqual(E.off, e); - } - try set(&e, "\x00on\n"); - try testing.expectEqual(E.on, e); - const NE = enum(u8) { a, _ }; - var ne: NE = .a; - try set(&ne, "200"); - try testing.expectEqual(@as(u8, 200), @intFromEnum(ne)); - try testing.expectError(error.Invalid, set(&ne, "256")); - try testing.expectError(error.Invalid, set(&ne, "-1")); - try set(&ne, "a"); - try testing.expectEqual(NE.a, ne); - // bools - var b: bool = true; - for ([_][]const u8{ "yes", "TRUE", "2", "", "01" }) |bad| { - try testing.expectError(error.Invalid, set(&b, bad)); - try testing.expect(b); - } - // unsupported types are refused without touching memory - var opt: ?u8 = 3; - try testing.expectError(error.Unsupported, set(&opt, "4")); - try testing.expectEqual(@as(?u8, 3), opt); - var arr: [2]u8 = .{ 1, 2 }; - try testing.expectError(error.Unsupported, set(&arr, "x")); - var un: union(enum) { a: u8 } = .{ .a = 1 }; - try testing.expectError(error.Unsupported, set(&un, "a")); -} diff --git a/introspect/test/adv_core_hostile.py b/introspect/test/adv_core_hostile.py deleted file mode 100755 index 56ef5a3..0000000 --- a/introspect/test/adv_core_hostile.py +++ /dev/null @@ -1,1018 +0,0 @@ -#!/usr/bin/env python3 -"""Hostile raw-9P2000 client aimed at the introspect *core* (stdlib only). - -Complements adv_introspect_hostile.py in this directory (framing, tags, scratch, floods) -with attacks on the freestanding engine's own paths: the /vars tree and its -comptime renderers, snapshot slots, the static tree, the fid table at its -configured maximum, directory-read offsets, msize 24, the ctl staging rule, -and the demo's debug providers driven as black boxes. - -Usage: - adv_core_hostile.py --server zig-out/bin/introspect # spawns it on a temp unix socket - adv_core_hostile.py --socket PATH # attacks a running server - -Exit status is non-zero if any check fails or the server dies. -""" -import argparse -import os -import signal -import struct -import subprocess -import sys -import tempfile -import threading -import time - -sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) -import adv_introspect_hostile as base # noqa: E402 -from adv_introspect_hostile import ( # noqa: E402 - NOTAG, Tversion, Tflush, Rflush, Twalk, Rwalk, Topen, Ropen, Rcreate, - Tread, Rread, Twrite, Rwrite, Tclunk, Rclunk, Tremove, Rremove, Tstat, Rstat, Twstat, Rwstat, - Rerror, OREAD, OWRITE, ORDWR, OEXEC, OTRUNC, ORCLOSE, DMDIR, - Nine, frame, s16, mkstat, parse_stat, ok, healthy, expect_dead, -) - -MAX_FIDS = 32768 # demo/main.zig cfg.max_fids -SNAPSHOT_SLOTS = 8 # demo/main.zig cfg.snapshot_slots (per connection) -SCRATCH_BUDGET = 512 << 20 -SCRATCH_MAX_FILE = 64 << 20 - - -def records(d): - """Splits a directory read into (name, raw-record) pairs.""" - out = [] - while d: - n, = struct.unpack_from("/name.""" - c.walk_ok(0, 40, [b"threads"]) - c.open(40, OREAD) - d = c.read_all(40) - c.clunk(40) - for name, _ in records(d): - if c.path_read([b"threads", name, b"name"], fid=41) == b"worker": - return name - return None - - -# --------------------------------------------------------------------------- /vars - - -def attack_vars(path): - print("# /vars: deep walks, hostile names, renderer edge cases, hostile writes") - c = Nine(path) - c.session(1 << 20) - deep = [b"vars", b"state", b"f", b"last_job", b"f", b"id", b".", b"..", b"id", b".", b"..", b"id", b".", b"..", b"id", b"value"] - assert len(deep) == 16 - ok("16-element walk deep into /vars/state/f/... succeeds", c.walk_ok(0, 1, deep) == 16) - rt, _, _ = c.open(1, OREAD) - ok("deep walk lands on a readable value file", rt == Ropen, rt) - c.clunk(1) - up = [b"vars", b"state", b"f", b"inner"] if False else [b"vars", b"state", b"f", b"last_job"] + [b".."] * 12 - n = c.walk_ok(0, 1, up) - ok("12 x '..' from inside /vars climbs to the root and stays there", n == 16, n) - rt, st = c.stat(1) - ok("fid after the climb is the root directory", rt == Rstat and st["qid"][2] == 0xFF << 56, st) - c.clunk(1) - # names that are hex/decimal edge cases or otherwise hostile: never anything but Rerror/partial walk - for nm in (b"0", b"-1", b"0x", b"0x0", b"state\x00", b"State", b" state", b"state ", b"a" * 255, b"a" * 65535, b"\xff\xfe", b"..\x00", b"f", b"value"): - n = c.walk_ok(0, 1, [b"vars", nm]) - ok(f"walk /vars/{nm[:12]!r}{'...' if len(nm) > 12 else ''} is a partial walk (1)", n == 1, n) - ok(" and newfid stays unbound", c.err(Tclunk, struct.pack(" state -> /vars -> /", len(q) == 4 and q[3][2] == 0xFF << 56 and (q[1][0] & 0x80), q) - c.clunk(1) - c.clunk(2) - # every file under /vars/state reads; raw reads beyond @sizeOf are empty - size = int(c.path_read([b"vars", b"state", b"size"])) - ok("/vars/state/size is a number", size > 0, size) - raw = c.path_read([b"vars", b"state", b"raw"]) - ok("/vars/state/raw has exactly @sizeOf bytes", raw is not None and len(raw) == size, (len(raw) if raw else raw, size)) - c.walk_ok(0, 1, [b"vars", b"state", b"raw"]) - c.open(1, OREAD) - rt, d = c.read(1, size, 100) - ok("raw read at offset @sizeOf is empty", rt == Rread and d == b"", (rt, d)) - rt, d = c.read(1, size - 1, 100) - ok("raw read at @sizeOf-1 returns one byte", rt == Rread and len(d) == 1, (rt, d)) - rt, d = c.read(1, (1 << 64) - 1, 100) - ok("raw read at 2^64-1 is empty", rt == Rread and d == b"") - rt, d = c.read(1, 0, 0xFFFFFFFF) - ok("raw read with count 2^32-1 is clamped", rt == Rread and len(d) == size, (rt, len(d) if d else d)) - rt, st = c.stat(1) - ok("raw stat length is @sizeOf and mode 0444", rt == Rstat and st["length"] == size and st["mode"] == 0o444, st) - ok("raw is read-only", c.err(Twrite, struct.pack(" the refused one now opens; clunk via Tremove (denied) also frees the slot - c.clunk(100) - rt, _, _ = c.open(100 + opened, OREAD) - ok("after one clunk the refused open succeeds", rt == Ropen, rt) - ok("remove of an open dynamic file is denied", c.err(Tremove, struct.pack("/stack/x is 'not a directory'", c.walk_ok(0, 1, [b"threads", tid, b"stack"]) == 3 and c.err(Twalk, struct.pack("/stack is 'not a directory'", c.err(Twalk, struct.pack("/../..//name walks", c.walk_ok(0, 1, [b"threads", tid, b"..", b"..", b"threads", tid, b"name"]) == 7) - c.clunk(1) - c.walk_ok(0, 1, [b"threads", tid]) - ok("create under /threads/ is denied", c.err(base.Tcreate, struct.pack(" is denied", c.err(Twstat, struct.pack(" is denied", c.err(Tremove, struct.pack(" reads @sizeOf bytes", rt == Rread and len(d) == size, (rt, len(d) if d else d)) - rt, d = c.read(1, 0, 0xFFFFFFFF) - ok("/mem read with count 2^32-1 is clamped and answered", rt in (Rread, Rerror), rt) - c.clunk(1) - hexd = c.path_read([b"hex", hx]) - ok("/hex/ is a hexdump", hexd is not None and len(hexd) > 64, hexd[:40] if hexd else hexd) - # a value written through /mem must render, not trap: corrupt the phase enum and read /vars/state/value - phase_addr = int(c.path_read([b"vars", b"state", b"f", b"phase", b"addr"]), 16) - c.walk_ok(0, 1, [b"mem", b"%x" % phase_addr]) - c.open(1, OWRITE) - rt, _, _ = c.write(1, 0, b"\xee") - ok("write a corrupt enum byte through /mem", rt == Rwrite, rt) - c.clunk(1) - v = c.path_read([b"vars", b"state", b"value"]) - ok("/vars/state/value renders the corrupt enum as a number instead of trapping", v is not None and b"phase: 238" in v, v) - pv = c.path_read([b"vars", b"state", b"f", b"phase", b"value"]) - ok("/vars/state/f/phase/value renders 238", pv == b"238", pv) - c.walk_ok(0, 1, [b"vars", b"state", b"f", b"phase", b"value"]) - c.open(1, OWRITE) - rt, _, _ = c.write(1, 0, b"idle") - ok("the enum can be repaired through /vars", rt == Rwrite, rt) - c.clunk(1) - # /panic: ctl refuses reads and garbage; message/stack read - ok("/panic/message reads (empty, no panic)", c.path_read([b"panic", b"message"]) == b"") - ok("/panic/stack reads", c.path_read([b"panic", b"stack"]) is not None) - c.walk_ok(0, 1, [b"panic", b"ctl"]) - ok("/panic/ctl refuses OREAD", c.err(Topen, struct.pack("= 1, (len(held), err)) - for f in held: - c.clunk(f) - ok("after clunking, /addr opens again", c.path_read([b"addr", b"1000"]) is not None) - # Tversion with debug files open (hexdumps of the exposed state: mapped memory) - base_addr = int(addr, 16) if addr else 0 - opened = 0 - for i in range(4): - c.walk_ok(0, 300 + i, [b"hex", b"%x" % (base_addr + i)]) - opened += c.open(300 + i, OREAD)[0] == Ropen - ok("four /hex snapshots open", opened == 4, opened) - rt, _, _ = c.version(65536) - ok("Tversion with debug snapshots open", rt == base.Rversion) - c.attach() - ok("/hex still opens after the reset", c.path_read([b"hex", b"%x" % base_addr]) is not None) - ok("/hex of unmapped memory is an Rerror at open, not a crash", c.walk_ok(0, 1, [b"hex", b"3000"]) == 2 and c.open(1, OREAD)[0] == Rerror) - c.clunk(1) - c.close() - ok("server healthy after debug provider attacks", healthy(path)) - - -# --------------------------------------------------------------------------- fids at the maximum - - -def flood(c, ids, names): - """Pipelines one Twalk per id and counts the Rwalk replies; returns (ok_count, error_count, seconds).""" - got = [0, 0] - dead = [False] - - def reader(): - try: - for _ in ids: - rt, _, _ = c.recv_frame() - if rt == Rwalk: - got[0] += 1 - else: - got[1] += 1 - except (EOFError, OSError): - dead[0] = True - - t = threading.Thread(target=reader) - t.start() - t0 = time.time() - body = b"".join(frame(Twalk, i & 0xFFFE, struct.pack(" the handler's writer fails - rt2, d = c.read(1, 0, 100) - rt3, st5 = c.stat(1) - ok("an over-long result is an Rerror and the previous result survives", rt == Rerror and d == b"y" * 100 and st5["length"] == 60000, (rt, d[:10] if d else d, st5["length"])) - c.clunk(1) - c.close() - ok("server healthy after ctl attacks", healthy(path)) - - -# --------------------------------------------------------------------------- fid state machine on scratch - - -def attack_fid_states(path): - print("# fid state machine on /scratch") - c = Nine(path) - c.session() - tag = os.urandom(3).hex().encode() - root = b"fs-" + tag - c.walk_ok(0, 1, [b"scratch"]) - c.create(1, root, DMDIR | 0o755, OREAD) - c.clunk(1) - S = [b"scratch", root] - c.walk_ok(0, 1, S) - rt, _, _ = c.create(1, b"f", 0o644, ORDWR) - ok("create f", rt == Rcreate) - ok("open of an open fid is 'file already open'", c.err(Topen, struct.pack("> 20} MiB fill the {SCRATCH_BUDGET >> 20} MiB budget exactly", made == count, made) - print(f" filled the budget in {time.time() - t0:.1f}s") - c.walk_ok(0, 1, S) - c.create(1, b"one-more", 0o644, OWRITE) - ok("one more byte is 'no space left on device'", c.err(Twrite, struct.pack(" [--fast] (part of zig build introspect-adv) -# Spawns the server on a temporary unix socket and attacks it with -# introspect/test/adv_core_hostile.py (Python 3 stdlib). Exit 1 on any failure. -set -u -INTROSPECT=$(realpath "${1:?path to introspect}") -shift -command -v python3 >/dev/null || { echo "SKIP: python3 missing"; exit 0; } -exec python3 "$(dirname "$0")/adv_core_hostile.py" --server "$INTROSPECT" "$@" diff --git a/introspect/test/adv_introspect_hostile.py b/introspect/test/adv_introspect_hostile.py deleted file mode 100755 index 7dbb5c0..0000000 --- a/introspect/test/adv_introspect_hostile.py +++ /dev/null @@ -1,999 +0,0 @@ -#!/usr/bin/env python3 -"""Hostile raw-9P2000 client for the introspect server (stdlib only). - -Usage: - adv_introspect_hostile.py --server zig-out/bin/introspect # spawns it on a temp unix socket - adv_introspect_hostile.py --socket PATH # attacks a running server - -Every attack is followed by a "server still healthy" probe on a fresh connection. -Exit status is non-zero if any check fails, the server dies, or a probe hangs. -""" -import argparse -import os -import signal -import socket -import struct -import subprocess -import sys -import tempfile -import threading -import time - -NOTAG = 0xFFFF -NOFID = 0xFFFFFFFF -Tversion, Rversion, Tauth, Rauth, Tattach, Rattach, Rerror = 100, 101, 102, 103, 104, 105, 107 -Tflush, Rflush, Twalk, Rwalk, Topen, Ropen, Tcreate, Rcreate = 108, 109, 110, 111, 112, 113, 114, 115 -Tread, Rread, Twrite, Rwrite, Tclunk, Rclunk, Tremove, Rremove = 116, 117, 118, 119, 120, 121, 122, 123 -Tstat, Rstat, Twstat, Rwstat = 124, 125, 126, 127 -OREAD, OWRITE, ORDWR, OEXEC, OTRUNC, ORCLOSE = 0, 1, 2, 3, 0x10, 0x40 -DMDIR, DMAPPEND, DMEXCL = 0x80000000, 0x40000000, 0x20000000 -NAMES = {v: k for k, v in globals().items() if k[:1] in "TR" and isinstance(v, int) and 100 <= v <= 127} - -FAILS = [] -PASSES = 0 - - -def ok(name, cond, detail=""): - global PASSES - if cond: - PASSES += 1 - print(f"ok - {name}") - else: - FAILS.append(name) - print(f"FAIL - {name} {detail}") - - -def s16(b): - return struct.pack(" connection closed - c.raw(frame(Twalk, 1, struct.pack("0 - ok("walk to file", c.walk_ok(0, 1, [b"build", b"target"]) == 2) - ok("walk from file fails 'not a directory'", c.err(Twalk, struct.pack(" 0) - ok("read dir at bad offset", c.err(Tread, struct.pack(" 6: - return - if c.walk_ok(0, 9, names) != len(names): - ok("walk " + b"/".join(names).decode(), False) - return - rt, st = c.stat(9) - if st["mode"] & DMDIR: - c.open(9, OREAD) - d = c.read_all(9, 512) - c.clunk(9) - while d: - n, = struct.unpack_from(" 0: - ok("server process still running", proc.poll() is None, proc.poll()) - finally: - if proc is not None: - proc.send_signal(signal.SIGTERM) - try: - _, err = proc.communicate(timeout=5) - except subprocess.TimeoutExpired: - proc.kill() - _, err = proc.communicate() - lines = [ln for ln in err.decode("utf-8", "replace").splitlines() if "connection ended" not in ln and "read: " not in ln] - if lines: - print("# server stderr (filtered):") - for ln in lines[:40]: - print(" " + ln) - if tmp: - try: - os.unlink(path) - os.rmdir(tmp) - except OSError: - pass - print(f"# {PASSES} passed, {len(FAILS)} failed") - for f in FAILS: - print("# FAIL " + f) - sys.exit(1 if FAILS else 0) - - -if __name__ == "__main__": - main() diff --git a/introspect/test/adv_introspect_hostile.sh b/introspect/test/adv_introspect_hostile.sh deleted file mode 100755 index 3ee2ecb..0000000 --- a/introspect/test/adv_introspect_hostile.sh +++ /dev/null @@ -1,10 +0,0 @@ -#!/usr/bin/env bash -# Adversarial raw-9P2000 client tests for the introspect server. -# Usage: bash introspect/test/adv_introspect_hostile.sh [--fast] (part of zig build introspect-adv) -# Spawns the server on a temporary unix socket and attacks it with -# introspect/test/adv_introspect_hostile.py (Python 3 stdlib). Exit 1 on any failure. -set -u -INTROSPECT=$(realpath "${1:?path to introspect}") -shift -command -v python3 >/dev/null || { echo "SKIP: python3 missing"; exit 0; } -exec python3 "$(dirname "$0")/adv_introspect_hostile.py" --server "$INTROSPECT" "$@" diff --git a/introspect/test/adv_linux_probe.py b/introspect/test/adv_linux_probe.py deleted file mode 100755 index 83de771..0000000 --- a/introspect/test/adv_linux_probe.py +++ /dev/null @@ -1,421 +0,0 @@ -#!/usr/bin/env python3 -"""Adversarial tests of the introspect Linux layer: probe loop, debug provider, -signal machinery. Raw 9P2000 over a unix socket, plus one 9player mount. -Usage: adv_linux_probe.py --player <9player> --server -Reuses the client of adv_introspect_hostile.py. Exit 1 on any failure. -""" -import argparse -import ctypes -import os -import signal -import socket -import struct -import subprocess -import sys -import tempfile -import threading -import time - -sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) -from adv_introspect_hostile import ( # noqa: E402 - NOFID, NOTAG, OREAD, OWRITE, Nine, Rerror, Ropen, Rread, Rversion, Rwalk, Rwrite, - Tattach, Tread, Tversion, Twrite, frame, healthy, ok, parse_stat, s16) -import adv_introspect_hostile as hostile # noqa: E402 - -libc = ctypes.CDLL(None, use_errno=True) -SYS_tgkill = 234 if os.uname().machine == "x86_64" else 131 # aarch64: 131 - - -def tgkill(pid, tid, sig): - return libc.syscall(SYS_tgkill, pid, tid, sig) - - -class Srv: - def __init__(self, server, extra=()): - self.tmp = tempfile.mkdtemp(prefix="advlin.") - self.path = os.path.join(self.tmp, "sock") - self.proc = subprocess.Popen([server, "--unix", self.path, *extra], stderr=subprocess.PIPE) - for _ in range(200): - if os.path.exists(self.path): - break - time.sleep(0.02) - self.pid = self.proc.pid - - def alive(self): - return self.proc.poll() is None - - def stop(self): - if self.proc.poll() is None: - self.proc.send_signal(signal.SIGTERM) - try: - self.proc.wait(timeout=5) - except subprocess.TimeoutExpired: - self.proc.kill() - self.proc.wait() - err = self.proc.stderr.read().decode("utf-8", "replace") - try: - os.unlink(self.path) - except OSError: - pass - try: - os.rmdir(self.tmp) - except OSError: - pass - return err - - -def client(path, timeout=10): - c = Nine(path, timeout=timeout) - c.session() - return c - - -def rd(c, names, fid=50, offset=0, count=8192): - """walk+open+one read; returns (rtype-or-tag, data-or-error-string).""" - if c.walk_ok(0, fid, names) != len(names): - c.clunk(fid) - return "walkfail", None - rt, _, rb = c.open(fid, OREAD) - if rt != Ropen: - c.clunk(fid) - return "openfail", rb - rt, d = c.read(fid, offset, count) - c.clunk(fid) - if rt == Rerror: - n, = struct.unpack_from("/maps (read in 4 KiB pieces)", via == real, f"{len(via)} vs {len(real)}") - maps = real.decode() - for tag in ("[stack]", "[vdso]", "[heap]"): - m = [ln for ln in maps.splitlines() if ln.endswith(tag)] - if not m: - continue - lo = int(m[0].split("-")[0], 16) + 0x100 - ok(f"/addr of {tag} renders ? (never handed to std)", rd(c, [b"addr", b"%x" % lo])[1] == b"?\n?\n?\n") - ok(f"/hex of {tag} dumps", rd(c, [b"hex", b"%x" % lo])[0] == Rread) - m = [ln for ln in maps.splitlines() if ln.endswith("[stack]")][0] - hi = int(m.split()[0].split("-")[1], 16) - rt, d = rd(c, [b"mem", b"%x" % (hi - 16)], count=4096) - ok("read across the end of a mapping is a short read of 16 bytes", rt == Rread and len(d) == 16, (rt, d and len(d))) - rt, d = rd(c, [b"hex", b"%x" % (hi - 16)]) - ok("hexdump across the end of a mapping stops at the boundary", rt == Rread and d.count(b"\n") == 1, (rt, d)) - code = [ln for ln in maps.splitlines() if "r-xp" in ln and "introspect" in ln][0] - clo = int(code.split("-")[0], 16) - ok("write into read-only code is an error, not a fault", wr(c, [b"mem", b"%x" % (clo + 0x100)], b"\xcc") == ("err", "i/o error")) - ok("server alive after memory attacks", s.alive() and healthy(s.path)) - # a big write over the demo's own globals (state onwards) must not fault the server path - addr = rd(c, [b"vars", b"state", b"addr"])[1].decode().strip()[2:] - # (it zeroes the demo's own globals, this connection's state included, so - # the reply may never come; the server as a whole must keep working) - wr(c, [b"mem", addr.encode()], b"\x00" * 65536) - time.sleep(0.3) - ok("server alive and serving after a 64 KiB overwrite of its own globals", s.alive() and healthy(s.path)) - s.stop() - - -def attack_signals(server): - print("# signal machinery") - s = Srv(server) - c = client(s.path, timeout=8) - w = thread_by_name(c, s.pid, b"worker") - probe = thread_by_name(c, s.pid, b"introspect") - ok("worker and probe threads found by name", bool(w and probe), (w, probe)) - names = sorted(open(f"/proc/{s.pid}/task/{t}/comm").read().strip() for t in os.listdir(f"/proc/{s.pid}/task")) - ok("thread names are introspect, introspect, worker", names == ["introspect", "introspect", "worker"], names) - - # 4 clients hammer stacks of every thread while trap/continue interleave - errs, count = [], [0] - stop = threading.Event() - - def hammer(k): - try: - cc = client(s.path, timeout=8) - while not stop.is_set(): - for t in (w, probe, str(s.pid).encode()): - rt, d = rd(cc, [b"threads", t, b"stack"], fid=10 + k) - count[0] += 1 - if rt in ("walkfail", "openfail") or (rt == "err" and "i/o" not in d): - errs.append((t, rt, d)) - except Exception as e: # noqa: BLE001 - errs.append(repr(e)) - - ths = [threading.Thread(target=hammer, args=(k,)) for k in range(4)] - for t in ths: - t.start() - rounds_ok = True - for _ in range(5): - wr(c, [b"runtime", b"ctl"], b"trap") - time.sleep(0.15) - rounds_ok &= ls(c, [b"breakpoints"]) == [w] - rounds_ok &= b"workerLoop" in (rd(c, [b"breakpoints", w, b"stack"])[1] or b"") - rounds_ok &= rd(c, [b"threads", w, b"stack"])[0] == Rread # capture of a paused thread - rounds_ok &= wr(c, [b"breakpoints", w, b"ctl"], b"continue")[0] == Rwrite - rounds_ok &= wr(c, [b"breakpoints", w, b"ctl"], b"continue")[0] == "walkfail" # twice: gone - stop.set() - for t in ths: - t.join() - ok("trap/inspect/continue rounds while 4 clients capture stacks", rounds_ok) - ok(f"{count[0]} concurrent captures without a wrong answer", count[0] > 50 and not errs, errs[:3]) - - # SIGTRAP from outside (tgkill, not int3): parks without corrupting the thread - ok("tgkill SIGTRAP to the worker", tgkill(s.pid, int(w), signal.SIGTRAP) == 0) - time.sleep(0.3) - ok("worker listed under /breakpoints after tgkill", ls(c, [b"breakpoints"]) == [w]) - ok("its stack names workerLoop", b"workerLoop" in (rd(c, [b"breakpoints", w, b"stack"])[1] or b"")) - t1 = ticks(c) - time.sleep(0.3) - ok("ticks frozen while parked", ticks(c) == t1) - ok("continue after tgkill", wr(c, [b"breakpoints", w, b"ctl"], b"continue")[0] == Rwrite) - time.sleep(0.4) - ok("ticks advance after continue (no instruction skipped)", ticks(c) > t1) - - # SIGTRAP on the probe thread itself: stepped over, the server keeps serving - ok("tgkill SIGTRAP to the probe thread", tgkill(s.pid, int(probe), signal.SIGTRAP) == 0) - time.sleep(0.2) - ok("server serves after a SIGTRAP on its own thread", healthy(s.path)) - ok("probe thread not parked", ls(c, [b"breakpoints"]) == []) - # process-directed SIGTRAP lands on some thread; whichever it is, it is resumable - os.kill(s.pid, signal.SIGTRAP) - time.sleep(0.3) - ok("alive after kill -TRAP ", s.alive() and healthy(s.path)) - for t in ls(c, [b"breakpoints"]): - ok(f"thread {t.decode()} parked by kill -TRAP resumes", wr(c, [b"breakpoints", t, b"ctl"], b"continue")[0] == Rwrite) - ok("continue on a never-paused tid does not walk", wr(c, [b"breakpoints", w, b"ctl"], b"continue")[0] == "walkfail") - ok("a bogus tid does not walk", c.walk_ok(0, 31, [b"threads", b"999999"]) == 1) - - # panic: held, inspectable, capture of the held thread works, trap meanwhile harmless, continue aborts - wr(c, [b"runtime", b"ctl"], b"panic") - time.sleep(0.3) - ok("panic message published", rd(c, [b"panic", b"message"])[1] == b"demo panic requested over 9p") - ok("panic stack names workerLoop", b"workerLoop" in rd(c, [b"panic", b"stack"])[1]) - ok("capture of the held panicking thread answers", rd(c, [b"threads", w, b"stack"])[0] == Rread) - ok("trap request while a panic is held is harmless", wr(c, [b"runtime", b"ctl"], b"trap")[0] == Rwrite and s.alive()) - ok("panic continue", wr(c, [b"panic", b"ctl"], b"continue")[0] == Rwrite) - ok("second panic continue is an error", wr(c, [b"panic", b"ctl"], b"continue") == ("err", "file does not exist")) - time.sleep(1.0) - ok("process aborted after continue", not s.alive() and s.proc.poll() not in (0, None), s.proc.poll()) - s.stop() - - s = Srv(server, ["--no-hold"]) - c = client(s.path, timeout=5) - wr(c, [b"runtime", b"ctl"], b"panic") - time.sleep(1.0) - ok("--no-hold: panic aborts at once", not s.alive() and s.proc.poll() not in (0, None), s.proc.poll()) - s.stop() - - s = Srv(server) - c = client(s.path, timeout=5) - wr(c, [b"runtime", b"ctl"], b"trap") - time.sleep(0.3) - ok("worker parked", ls(c, [b"breakpoints"]) != []) - path = s.path - s.proc.send_signal(signal.SIGTERM) - try: - rc = s.proc.wait(timeout=5) - except subprocess.TimeoutExpired: - rc = None - ok("SIGTERM with a parked thread exits promptly", rc is not None, rc) - ok("SIGTERM unlinks the unix socket (clean stop path)", not os.path.exists(path)) - s.stop() - - -def attack_probe(player, server): - print("# probe loop and admission") - s = Srv(server) - c0 = cpu_ticks(s.pid) - time.sleep(5.0) - ok("0 CPU ticks over 5 s idle", cpu_ticks(s.pid) - c0 == 0, cpu_ticks(s.pid) - c0) - f0 = fds(s.pid) - for i in range(1000): - so = socket.socket(socket.AF_UNIX, socket.SOCK_STREAM) - so.connect(s.path) - if i % 3 == 0: - so.sendall(frame(Tversion, NOTAG, struct.pack(" (part of zig build introspect-adv) -# Exit 1 on any failure; SKIP (exit 0) without python3. -set -u -PLAYER=$(realpath "${1:?path to 9player}") -INTROSPECT=$(realpath "${2:?path to introspect}") -command -v python3 >/dev/null || { echo "SKIP: python3 missing"; exit 0; } -exec python3 "$(dirname "$0")/adv_linux_probe.py" --player "$PLAYER" --server "$INTROSPECT" diff --git a/introspect/test/adversarial.sh b/introspect/test/adversarial.sh deleted file mode 100755 index 918bb9a..0000000 --- a/introspect/test/adversarial.sh +++ /dev/null @@ -1,19 +0,0 @@ -#!/usr/bin/env bash -# Runs every introspect adversarial suite in sequence: hostile raw-9P clients -# against the demo (framing, tags, floods; the core's /vars, snapshots, fids) -# and the Linux layer through one 9player mount (memory, signals, poll loop). -# Usage: bash introspect/test/adversarial.sh <9player> [--fast] -# `zig build introspect-adv` runs the same suites as separate steps. -set -u -PLAYER=${1:?path to 9player} -INTROSPECT=${2:?path to introspect} -shift 2 -HERE=$(cd "$(dirname "$0")" && pwd) -status=0 -for suite in adv_introspect_hostile adv_core_hostile; do - echo "### $suite" - if bash "$HERE/$suite.sh" "$INTROSPECT" "$@"; then echo "### $suite: ok"; else echo "### $suite: FAILED"; status=1; fi -done -echo "### adv_linux_probe" -if bash "$HERE/adv_linux_probe.sh" "$PLAYER" "$INTROSPECT"; then echo "### adv_linux_probe: ok"; else echo "### adv_linux_probe: FAILED"; status=1; fi -exit $status diff --git a/introspect/test/debug.sh b/introspect/test/debug.sh deleted file mode 100755 index c3be406..0000000 --- a/introspect/test/debug.sh +++ /dev/null @@ -1,84 +0,0 @@ -#!/usr/bin/env bash -# End-to-end test of the introspect debug facilities through 9player: -# threads and stacks, address resolution, memory, exposed values, breakpoints, panics. -# Usage: bash introspect/test/debug.sh <9player> (zig build introspect-debug-itest) -set -u -PLAYER=$(realpath "${1:?path to 9player}") -INTROSPECT=$(realpath "${2:?path to introspect}") -TMP=$(mktemp -d /tmp/9pdbg.XXXXXX) -M=/mnt/9p -FAILED=0; PASSED=0 -SRV= - -cleanup() { [ -n "$SRV" ] && kill "$SRV" 2>/dev/null; rm -rf "$TMP"; } -trap cleanup EXIT -if ! unshare -Urm true 2>/dev/null || [ ! -c /dev/fuse ]; then echo "SKIP: namespaces or /dev/fuse unavailable"; exit 0; fi - -pass() { PASSED=$((PASSED + 1)); echo "ok - $1"; } -fail() { FAILED=$((FAILED + 1)); echo "FAIL - $1"; shift; [ $# -gt 0 ] && printf ' %s\n' "$@"; } -expect_eq() { if [ "$2" = "$3" ]; then pass "$1"; else fail "$1" "expected: $(printf %q "$2")" "actual: $(printf %q "$3")"; fi; } -expect_contains() { case "$3" in *"$2"*) pass "$1" ;; *) fail "$1" "missing: $(printf %q "$2")" "in: $(printf %q "$3")" ;; esac; } -run_in() { timeout 60 "$PLAYER" --unix "$SOCK" -- sh -c "$1" 2>"$TMP/stderr"; } - -SOCK=$TMP/dbg.sock -"$INTROSPECT" --unix "$SOCK" >"$TMP/server.log" 2>&1 & -SRV=$! -for _ in $(seq 1 100); do [ -S "$SOCK" ] && break; sleep 0.05; done -[ -S "$SOCK" ] || { echo "server did not start"; cat "$TMP/server.log"; exit 1; } - -echo "# threads" -expect_eq "threads listed" "yes" "$(run_in "[ \$(ls $M/threads | wc -l) -ge 2 ] && echo yes")" -WORKER=$(run_in "for t in $M/threads/*; do if grep -q '^worker' \$t/name 2>/dev/null; then basename \$t; fi; done | head -1") -expect_eq "worker thread found by name" "yes" "$([ -n "$WORKER" ] && echo yes)" -STACK=$(run_in "cat $M/threads/$WORKER/stack") -expect_contains "worker stack names workerLoop" "workerLoop" "$STACK" -expect_contains "worker stack has source locations" "demo/main.zig:" "$STACK" -expect_contains "worker regs" "0x" "$(run_in "cat $M/threads/$WORKER/regs | head -3")" -expect_eq "own (server) thread stack works" "yes" "$(run_in "for t in $M/threads/*; do cat \$t/stack >/dev/null 2>&1 || echo bad; done; echo yes")" - -echo "# addresses and memory" -FRAME=$(printf '%s\n' "$STACK" | grep -oE '0x[0-9a-f]+' | head -1) -expect_contains "addr resolves a stack frame to the demo source" "demo/main.zig" "$(run_in "cat $M/addr/${FRAME#0x}")" -expect_contains "addr of garbage is an error, not a crash" "No such file" "$(run_in "cat $M/addr/zzz 2>&1")" -expect_contains "mem/maps readable" "r-xp" "$(run_in "head -c 4000 $M/mem/maps")" -STATE_ADDR=$(run_in "cat $M/vars/state/addr") -expect_contains "hexdump of the exposed state" " " "$(run_in "head -2 $M/hex/${STATE_ADDR#0x}")" -expect_eq "raw bytes of the state match its size" "$(run_in "cat $M/vars/state/size")" "$(run_in "cat $M/mem/${STATE_ADDR#0x} | head -c \$(cat $M/vars/state/size) | wc -c")" -expect_eq "reading unmapped memory is an error, not a crash" "no" "$(run_in "cat $M/mem/8 >/dev/null 2>&1 && echo yes || echo no")" - -echo "# exposed values" -T1=$(run_in "cat $M/vars/state/f/ticks/value"); sleep 0.4; T2=$(run_in "cat $M/vars/state/f/ticks/value") -expect_eq "ticks is numeric" "num" "$(printf '%s' "$T1" | grep -Eq '^[0-9]+$' && echo num)" -expect_eq "ticks advance" "yes" "$([ "$T2" -gt "$T1" ] 2>/dev/null && echo yes)" -expect_contains "rendered struct value" "ticks" "$(run_in "cat $M/vars/state/value")" -expect_contains "type name" "State" "$(run_in "cat $M/vars/state/type")" -T3=$(run_in "echo 5 > $M/vars/state/f/ticks/value && cat $M/vars/state/f/ticks/value") -expect_eq "writing a scalar changes the live variable" "yes" "$([ "$T3" -lt "$T2" ] 2>/dev/null && echo yes)" - -echo "# breakpoints" -expect_eq "no breakpoints initially" "" "$(run_in "ls $M/breakpoints")" -run_in "echo trap > $M/runtime/ctl" >/dev/null; sleep 0.6 -PAUSED=$(run_in "ls $M/breakpoints | head -1") -expect_eq "worker paused at @breakpoint()" "$WORKER" "$PAUSED" -expect_contains "paused stack names workerLoop" "workerLoop" "$(run_in "cat $M/breakpoints/$WORKER/stack 2>&1")" -P1=$(run_in "cat $M/vars/state/f/ticks/value"); sleep 0.4; P2=$(run_in "cat $M/vars/state/f/ticks/value") -expect_eq "ticks frozen while paused" "$P1" "$P2" -run_in "echo continue > $M/breakpoints/$WORKER/ctl" >/dev/null; sleep 0.4 -expect_eq "breakpoint list empty after continue" "" "$(run_in "ls $M/breakpoints")" -P3=$(run_in "cat $M/vars/state/f/ticks/value") -expect_eq "ticks advance after continue" "yes" "$([ "$P3" -gt "$P2" ] 2>/dev/null && echo yes)" - -echo "# panic" -expect_eq "no panic recorded" "" "$(run_in "cat $M/panic/message")" -run_in "echo panic > $M/runtime/ctl" >/dev/null; sleep 0.6 -expect_contains "panic message published" "demo panic" "$(run_in "cat $M/panic/message")" -expect_contains "panic stack names the worker" "workerLoop" "$(run_in "cat $M/panic/stack")" -expect_eq "server still alive while holding the panic" "yes" "$(kill -0 $SRV 2>/dev/null && echo yes)" -run_in "echo continue > $M/panic/ctl" >/dev/null 2>&1 -for _ in $(seq 1 50); do kill -0 $SRV 2>/dev/null || break; sleep 0.1; done -if kill -0 $SRV 2>/dev/null; then fail "server exits after panic continue"; else wait $SRV; RC=$?; SRV=; expect_eq "server exit status is non-zero after the panic" "yes" "$([ $RC -ne 0 ] && echo yes)"; fi -expect_contains "default panic output reached stderr" "demo panic" "$(cat "$TMP/server.log")" - -echo -echo "passed=$PASSED failed=$FAILED" -[ "$FAILED" -eq 0 ] diff --git a/test/web/e2e.mjs b/test/web/e2e.mjs index 2ff83ae..ad10295 100644 --- a/test/web/e2e.mjs +++ b/test/web/e2e.mjs @@ -41,7 +41,7 @@ async function launchBridge(upstream, origin, extra = []) { const child = start(bridgeBinary, ['--listen', '127.0.0.1:0', '--upstream', upstream, ...(origin ? ['--origin', origin] : []), ...extra]); // With a public origin the log does not include the bound address. Tests // requiring a proxy instead reserve a port and pass it explicitly below. - const match = await wait(() => child.errors.match(/cloud9-http (http:\/\/127\.0\.0\.1:\d+)/), 'bridge ready'); + const match = await wait(() => child.errors.match(/9web (http:\/\/127\.0\.0\.1:\d+)/), 'bridge ready'); return { child, url: match[1] }; } async function port() { const s = net.createServer(); s.listen(0, '127.0.0.1'); await once(s, 'listening'); const p = s.address().port; await new Promise(r => s.close(r)); return p; } @@ -213,7 +213,7 @@ try { const httpsPort = await port(), backendPort = await port(); const origin = `https://localhost:${httpsPort}`; const secureBridge = start(bridgeBinary, ['--listen', `127.0.0.1:${backendPort}`, '--upstream', `tcp:${upstream}`, '--origin', origin]); - await wait(() => secureBridge.errors.includes('cloud9-http'), 'TLS backend ready'); + await wait(() => secureBridge.errors.includes('9web'), 'TLS backend ready'); tlsServer = tls.createServer({ key: await fs.readFile(key), cert: await fs.readFile(cert), ALPNProtocols: ['http/1.1'] }, socket => { const upstreamSocket = net.connect(backendPort, '127.0.0.1'); sockets.add(socket); sockets.add(upstreamSocket); diff --git a/web/client.zig b/web/client.zig new file mode 100644 index 0000000..2ec9163 --- /dev/null +++ b/web/client.zig @@ -0,0 +1,176 @@ +//! Browser ABI. Protocol state, encoding, reply validation, and directory decoding +//! all use the same cloud9 sources as native clients. No allocator or host imports. +const std = @import("std"); +const c9 = @import("cloud9"); +const capacity = 65536; +var input: [capacity]u8 = undefined; +var output: [capacity]u8 = undefined; +var staging: [capacity]u8 = undefined; +var json: [capacity * 2]u8 = undefined; +var client: c9.Client = undefined; +var data: []const u8 = ""; +var number: u32 = 0; +var directory: bool = false; + +export fn init() void { + client = .init(.{ .in = &input, .out = &output }); + data = ""; + number = 0; + directory = false; +} +export fn input_ptr() [*]u8 { + return &staging; +} +export fn input_capacity() u32 { + return capacity; +} +export fn output_ptr() [*]const u8 { + return client.output().ptr; +} +export fn output_len() u32 { + return @intCast(client.output().len); +} +export fn sent() void { + client.wrote(client.output().len); +} +export fn data_ptr() [*]const u8 { + return data.ptr; +} +export fn data_len() u32 { + return @intCast(data.len); +} +export fn result_number() u32 { + return number; +} +export fn result_directory() bool { + return directory; +} +export fn is_dead() bool { + return client.dead; +} +export fn max_read() u32 { + return client.maxRead(); +} +export fn max_write() u32 { + return client.maxWrite(); +} + +fn fail(message: []const u8) i32 { + data = message; + return -1; +} +fn submit(request: c9.Client.Request) i32 { + _ = client.submit(request) catch |err| return fail(@errorName(err)); + return 0; +} +export fn version() i32 { + return submit(.{ .version = .{} }); +} +export fn attach(user_len: u32, tree_len: u32) i32 { + if (user_len > capacity or tree_len > capacity - user_len) return fail("InputTooLarge"); + return submit(.{ .attach = .{ .fid = 0, .uname = staging[0..user_len], .aname = staging[user_len..][0..tree_len] } }); +} +export fn walk(fid: u32, newfid: u32, len: u32) i32 { + if (len > capacity) return fail("InputTooLarge"); + var names: [c9.max_welem][]const u8 = undefined; + var count: usize = 0; + var it = std.mem.splitScalar(u8, staging[0..len], '/'); + while (it.next()) |name| { + if (name.len == 0) continue; + if (count == names.len) return fail("TooManyNames"); + names[count] = name; + count += 1; + } + return submit(.{ .walk = .{ .fid = fid, .newfid = newfid, .names = names[0..count] } }); +} +export fn open(fid: u32, mode: u8) i32 { + return submit(.{ .open = .{ .fid = fid, .mode = mode } }); +} +export fn read(fid: u32, offset: u64, count: u32) i32 { + return submit(.{ .read = .{ .fid = fid, .offset = offset, .count = count } }); +} +export fn write(fid: u32, offset: u64, len: u32) i32 { + if (len > capacity) return fail("InputTooLarge"); + return submit(.{ .write = .{ .fid = fid, .offset = offset, .data = staging[0..len] } }); +} +export fn clunk(fid: u32) i32 { + return submit(.{ .clunk = .{ .fid = fid } }); +} +export fn stat(fid: u32) i32 { + return submit(.{ .stat = .{ .fid = fid } }); +} + +fn writeStat(s: c9.Stat, writer: *std.Io.Writer) !void { + try std.json.Stringify.value(.{ .name = s.name, .directory = s.qid.type & c9.qtdir != 0, .length = s.length, .mode = s.mode }, .{}, writer); +} + +/// Result codes: 0 incomplete, -1 error, 1 version, 2 attach, 3 walk, +/// 4 open, 5 read, 6 write, 7 clunk, 8 stat. Data is borrowed until next call. +export fn receive(len: u32) i32 { + if (len > capacity) return fail("InputTooLarge"); + if (client.push(staging[0..len]) != len) return fail("InputFull"); + const done = client.take() orelse return if (client.dead) fail("ProtocolError") else 0; + data = ""; + number = 0; + directory = false; + return switch (done.result) { + .fail => |message| fail(message), + .version => |v| blk: { + data = v.version; + number = v.msize; + break :blk 1; + }, + .attach => |qid| blk: { + directory = qid.type & c9.qtdir != 0; + break :blk 2; + }, + .walk => |w| blk: { + number = w.nwqid; + if (w.nwqid != 0) directory = w.wqid[w.nwqid - 1].type & c9.qtdir != 0; + break :blk 3; + }, + .open => |o| blk: { + number = o.iounit; + directory = o.qid.type & c9.qtdir != 0; + break :blk 4; + }, + .read => |bytes| blk: { + data = bytes; + break :blk 5; + }, + .write => |count| blk: { + number = count; + break :blk 6; + }, + .clunk => 7, + .stat => |s| blk: { + var writer: std.Io.Writer = .fixed(&json); + writeStat(s, &writer) catch return fail("OutputTooLarge"); + data = writer.buffered(); + break :blk 8; + }, + else => fail("UnexpectedReply"), + }; +} + +/// Parse complete directory stat records from one Rread, using cloud9.Stat. +export fn decode_directory(len: u32) i32 { + if (len > capacity) return fail("InputTooLarge"); + var remaining: []const u8 = staging[0..len]; + var writer: std.Io.Writer = .fixed(&json); + writer.writeByte('[') catch unreachable; + var first = true; + while (remaining.len != 0) { + if (remaining.len < 2) return fail("TruncatedDirectory"); + const size: usize = @as(usize, std.mem.readInt(u16, remaining[0..2], .little)) + 2; + if (size > remaining.len) return fail("TruncatedDirectory"); + const s = c9.Stat.decode(remaining[0..size]) catch return fail("InvalidDirectory"); + if (!first) writer.writeByte(',') catch return fail("OutputTooLarge"); + writeStat(s, &writer) catch return fail("OutputTooLarge"); + first = false; + remaining = remaining[size..]; + } + writer.writeByte(']') catch return fail("OutputTooLarge"); + data = writer.buffered(); + return 0; +} diff --git a/web/main.zig b/web/main.zig new file mode 100644 index 0000000..e6aa421 --- /dev/null +++ b/web/main.zig @@ -0,0 +1,204 @@ +//! HTTP application; no web or namespace policy is added to the cloud9 library. +const std = @import("std"); +const c9 = @import("cloud9"); +const Io = std.Io; +const max_frame = 65536; +const max_connections = 32; +var serial_busy: std.atomic.Value(bool) = .init(false); +const Upstream = union(enum) { network: c9.transport.Address, file: []const u8 }; +var connections: std.atomic.Value(u32) = .init(0); + +const Config = struct { + upstream: Upstream, + host: []const u8, + origin: []const u8, + public_host: []const u8, + timeout_ms: u32, + browser_config: []const u8, +}; + +pub fn main(init: std.process.Init) !void { + const allocator = init.arena.allocator(); + const args = try init.minimal.args.toSlice(allocator); + var listen_text: []const u8 = "127.0.0.1:8080"; + var upstream_text: []const u8 = "tcp:127.0.0.1:564"; + var timeout_ms: u32 = 300000; + var user: []const u8 = "user"; + var tree: []const u8 = ""; + var public_origin: ?[]const u8 = null; + var i: usize = 1; + while (i < args.len) : (i += 1) { + if (std.mem.eql(u8, args[i], "--help")) { + std.debug.print("Usage: 9web [--listen IP:PORT] [--upstream tcp:IP:PORT|unix:PATH|file:PATH] [--origin https://HOST:PORT] [--timeout-ms 300000] [--user NAME] [--tree NAME]\nDefault: http://127.0.0.1:8080 -> tcp:127.0.0.1:564\n", .{}); + return; + } + if (i + 1 == args.len) return error.MissingArgument; + if (std.mem.eql(u8, args[i], "--listen")) { + i += 1; + listen_text = args[i]; + } else if (std.mem.eql(u8, args[i], "--upstream")) { + i += 1; + upstream_text = args[i]; + } else if (std.mem.eql(u8, args[i], "--origin")) { + i += 1; + public_origin = args[i]; + } else if (std.mem.eql(u8, args[i], "--timeout-ms")) { + i += 1; + timeout_ms = try std.fmt.parseInt(u32, args[i], 10); + } else if (std.mem.eql(u8, args[i], "--user")) { + i += 1; + user = args[i]; + } else if (std.mem.eql(u8, args[i], "--tree")) { + i += 1; + tree = args[i]; + } else return error.UnknownArgument; + } + const address = try Io.net.IpAddress.parseLiteral(listen_text); + const upstream: Upstream = if (std.mem.startsWith(u8, upstream_text, "tcp:")) + .{ .network = .{ .tcp = try Io.net.IpAddress.parseLiteral(upstream_text[4..]) } } + else if (std.mem.startsWith(u8, upstream_text, "unix:")) + .{ .network = .{ .unix = try allocator.dupeZ(u8, upstream_text[5..]) } } + else if (std.mem.startsWith(u8, upstream_text, "file:")) + .{ .file = upstream_text[5..] } + else + return error.InvalidUpstream; + const io = init.io; + var listener = c9.transport.listen(io, .{ .tcp = address }, max_connections) catch |err| { + if (err == error.AddressInUse) { + std.debug.print("9web: cannot listen on {s}: address already in use.\nUse --listen 127.0.0.1:0 to select a free port; the URL is printed at startup.\n", .{listen_text}); + } else { + std.debug.print("9web: cannot listen on {s}: {s}\n", .{ listen_text, @errorName(err) }); + } + return err; + }; + defer listener.deinit(io); + const host = try std.fmt.allocPrint(allocator, "{f}", .{listener.socket.address}); + const origin_text = public_origin orelse try std.fmt.allocPrint(allocator, "http://{s}", .{host}); + const origin_uri = try std.Uri.parse(origin_text); + if ((!std.mem.eql(u8, origin_uri.scheme, "https") and !std.mem.eql(u8, origin_uri.scheme, "http")) or + origin_uri.host == null or origin_uri.user != null or origin_uri.password != null or + origin_uri.path.percent_encoded.len != 0 or origin_uri.query != null or origin_uri.fragment != null) return error.InvalidOrigin; + if (user.len > 256 or tree.len > 256) return error.AttachNameTooLong; + const browser_config = try std.json.Stringify.valueAlloc(allocator, .{ .user = user, .tree = tree }, .{}); + const config: Config = .{ .upstream = upstream, .host = host, .origin = origin_text, .public_host = origin_text[origin_uri.scheme.len + 3 ..], .timeout_ms = timeout_ms, .browser_config = browser_config }; + std.debug.print("9web {s} -> {s}\n", .{ config.origin, upstream_text }); + var group: Io.Group = .init; + defer group.cancel(io); + while (true) { + const stream = try listener.accept(io); + if (connections.fetchAdd(1, .monotonic) >= max_connections) { + _ = connections.fetchSub(1, .monotonic); + stream.close(io); + continue; + } + group.concurrent(io, handle, .{ io, stream, config }) catch |err| { + _ = connections.fetchSub(1, .monotonic); + stream.close(io); + return err; + }; + } +} + +fn handle(io: Io, stream: Io.net.Stream, config: Config) void { + defer _ = connections.fetchSub(1, .monotonic); + defer stream.close(io); + var work: Connection = .{ .io = io, .stream = stream, .config = config }; + var group: Io.Group = .init; + defer group.cancel(io); + group.concurrent(io, Connection.run, .{&work}) catch return; + const timeout: Io.Timeout = if (config.timeout_ms == 0) .none else .{ .duration = .{ .raw = .fromMilliseconds(config.timeout_ms), .clock = .awake } }; + work.done.waitTimeout(io, timeout) catch {}; +} + +const Connection = struct { + io: Io, + stream: Io.net.Stream, + config: Config, + done: Io.Event = .unset, + fn run(connection: *Connection) void { + defer connection.done.set(connection.io); + serve(connection.io, connection.stream, connection.config) catch {}; + } +}; + +fn respond(request: *std.http.Server.Request, content: []const u8, mime: []const u8, status: std.http.Status) !void { + try request.respond(content, .{ .status = status, .keep_alive = false, .extra_headers = &.{ + .{ .name = "content-type", .value = mime }, + .{ .name = "cache-control", .value = "no-store" }, + .{ .name = "x-content-type-options", .value = "nosniff" }, + .{ .name = "content-security-policy", .value = "default-src 'self'; script-src 'self' 'wasm-unsafe-eval'; style-src 'self'; connect-src 'self'; frame-ancestors 'none'; base-uri 'none'" }, + } }); +} + +fn serve(io: Io, stream: Io.net.Stream, config: Config) !void { + var input: [max_frame + 14]u8 = undefined; + var output: [8192]u8 = undefined; + var reader = stream.reader(io, &input); + var writer = stream.writer(io, &output); + var http: std.http.Server = .init(&reader.interface, &writer.interface); + http.reader.max_head_len = 8192; + var request = try http.receiveHead(); + var host: ?[]const u8 = null; + var origin: ?[]const u8 = null; + var version: ?[]const u8 = null; + var headers = request.iterateHeaders(); + while (headers.next()) |header| { + if (std.ascii.eqlIgnoreCase(header.name, "host")) host = header.value; + if (std.ascii.eqlIgnoreCase(header.name, "origin")) origin = header.value; + if (std.ascii.eqlIgnoreCase(header.name, "sec-websocket-version")) version = header.value; + } + if (!std.mem.eql(u8, host orelse "", config.host) and !std.mem.eql(u8, host orelse "", config.public_host)) return respond(&request, "Unexpected Host\n", "text/plain", .forbidden); + if (request.head.method != .GET) return respond(&request, "Use GET\n", "text/plain", .method_not_allowed); + const path = std.mem.sliceTo(request.head.target, '?'); + if (std.mem.eql(u8, path, "/_cloud9/config.json")) return respond(&request, config.browser_config, "application/json", .ok); + if (std.mem.eql(u8, path, "/_cloud9/app.mjs")) return respond(&request, @embedFile("static/app.mjs"), "text/javascript; charset=utf-8", .ok); + if (std.mem.eql(u8, path, "/_cloud9/client.mjs")) return respond(&request, @embedFile("static/client.mjs"), "text/javascript; charset=utf-8", .ok); + if (std.mem.eql(u8, path, "/_cloud9/style.css")) return respond(&request, @embedFile("static/style.css"), "text/css; charset=utf-8", .ok); + if (std.mem.eql(u8, path, "/_cloud9/cloud9.wasm")) return respond(&request, @embedFile("client.wasm"), "application/wasm", .ok); + if (!std.mem.eql(u8, path, "/_cloud9/9p")) { + if (std.mem.eql(u8, path, "/_cloud9") or std.mem.startsWith(u8, path, "/_cloud9/")) + return respond(&request, "Not found\n", "text/plain", .not_found); + // File URLs boot the same browser client. The client resolves the path + // in the upstream 9P namespace, never in this machine's filesystem. + return respond(&request, @embedFile("static/index.html"), "text/html; charset=utf-8", .ok); + } + if (!std.mem.eql(u8, origin orelse "", config.origin)) return respond(&request, "Unexpected Origin\n", "text/plain", .forbidden); + if (!std.mem.eql(u8, version orelse "", "13")) return respond(&request, "WebSocket version 13 required\n", "text/plain", .bad_request); + switch (config.upstream) { + .network => |address| { + const upstream = c9.transport.connect(io, address) catch return respond(&request, "9P upstream unavailable\n", "text/plain", .bad_gateway); + defer upstream.close(io); + var in_buffer: [8192]u8 = undefined; + var out_buffer: [8192]u8 = undefined; + var upstream_reader = upstream.reader(io, &in_buffer); + var upstream_writer = upstream.writer(io, &out_buffer); + try bridge(io, &request, &upstream_reader.interface, &upstream_writer.interface); + }, + .file => |path_name| { + if (serial_busy.swap(true, .acquire)) return respond(&request, "Device already in use\n", "text/plain", .service_unavailable); + defer serial_busy.store(false, .release); + const file = Io.Dir.cwd().openFile(io, path_name, .{ .mode = .read_write }) catch return respond(&request, "9P device unavailable\n", "text/plain", .bad_gateway); + defer file.close(io); + var in_buffer: [8192]u8 = undefined; + var out_buffer: [8192]u8 = undefined; + var upstream_reader = file.readerStreaming(io, &in_buffer); + var upstream_writer = file.writerStreaming(io, &out_buffer); + try bridge(io, &request, &upstream_reader.interface, &upstream_writer.interface); + }, + } +} + +fn bridge(io: Io, request: *std.http.Server.Request, reader: *Io.Reader, writer: *Io.Writer) !void { + var ws = try c9.http.accept(request); + var requests: [max_frame]u8 = undefined; + var replies: [max_frame]u8 = undefined; + var relay: c9.http.Bridge = .{ + .socket = &ws, + .upstream_reader = reader, + .upstream_writer = writer, + .request_buffer = &requests, + .reply_buffer = &replies, + .frame_limit = max_frame, + }; + try relay.run(io, .none); +} diff --git a/web/static/app.mjs b/web/static/app.mjs new file mode 100644 index 0000000..a6985f0 --- /dev/null +++ b/web/static/app.mjs @@ -0,0 +1,183 @@ +import { Client } from './client.mjs'; +const $ = id => document.getElementById(id); +const text = new TextDecoder('utf-8', { fatal: true }); +let client = null, current = null, currentPath = '/', busy = false, dirty = false; +let configuration = null; + +function status(message, error = false) { + $('status').textContent = message; + $('status').classList.toggle('error', error); +} +function controls() { + for (const id of ['go', 'reload']) $(id).disabled = busy; + $('save').disabled = busy || !current || current.stat.directory || $('editor').hidden || !dirty; + $('download').disabled = busy || !current; + $('editor').readOnly = busy; + $('edit-status').textContent = dirty ? 'Unsaved changes.' : 'No unsaved changes.'; + document.querySelector('main').setAttribute('aria-busy', String(busy)); +} +async function action(operation) { + if (busy) return; + busy = true; + controls(); + try { await operation(); } + catch (error) { status(error.message, true); } + finally { busy = false; controls(); } +} +async function connectedClient() { + if (client && !client.closed) return client; + if (!configuration) { + const response = await fetch('/_cloud9/config.json'); + if (!response.ok) throw new Error('Unable to load server settings. Reload to try again.'); + configuration = await response.json(); + } + try { client = await Client.connect('/_cloud9/9p', configuration.user, configuration.tree); } + catch { throw new Error('Unable to reach the file server. Reload to try again.'); } + return client; +} +async function readPath(path) { + const active = await connectedClient(); + try { return await active.readPath(path); } + catch (error) { + // A timed-out/closed transport has lost its fids. Retry reads once on a + // fresh session. Writes are never replayed after an uncertain result. + if (!active.closed) throw error; + return (await connectedClient()).readPath(path); + } +} +function normalizePath(path) { + const parts = []; + for (const name of path.split('/')) { + if (!name || name === '.') continue; + if (name === '..') parts.pop(); else parts.push(name); + } + return '/' + parts.join('/'); +} +function pathURL(path) { + const parts = path.split('/').map(encodeURIComponent); + // Preserve access to an upstream directory named _cloud9 without colliding + // with the literal gateway prefix. Decode each component exactly once. + if (parts[1] === '_cloud9') parts[1] = '%5Fcloud9'; + return parts.join('/'); +} +function locationPath() { + const parts = location.pathname.split('/').map(part => { + let name; + try { name = decodeURIComponent(part); } + catch { throw new Error('The URL contains invalid path encoding.'); } + if (name.includes('/') || name.includes('\0')) throw new Error('The URL contains an invalid path component.'); + return name; + }); + return normalizePath(parts.join('/')); +} +function child(path, name) { return `${path.replace(/\/$/, '')}/${name}`; } +function mode(stat) { + let result = stat.directory ? 'd' : '-'; + for (let bit = 8; bit >= 0; bit--) result += stat.mode & (1 << bit) ? 'rwx'[(8 - bit) % 3] : '-'; + return result; +} +function pathLink(label, path, className = '') { + const link = document.createElement('a'); + link.textContent = label; + link.href = pathURL(path); + link.dataset.path = path; + link.className = className; + return link; +} +function breadcrumbs(path) { + const node = $('breadcrumbs'); + node.replaceChildren(pathLink('root', '/')); + let walked = ''; + for (const name of path.split('/').filter(Boolean)) { + walked += '/' + name; + const separator = document.createElement('span'); + separator.textContent = '/'; separator.className = 'separator'; + node.append(separator, pathLink(name, walked)); + } + node.lastElementChild.setAttribute('aria-current', 'page'); + const parent = path.replace(/\/[^/]+$/, '') || '/'; + $('up').hidden = path === '/'; + $('up').href = pathURL(parent); + $('up').dataset.path = parent; +} +async function navigate(path, historyMode = 'push') { + path = normalizePath(path); + status('Loading…'); + const result = await readPath(path); + current = result; currentPath = path; dirty = false; + $('path').value = path; + const url = pathURL(path); + if (historyMode === 'push' && location.pathname + location.search + location.hash !== url) history.pushState(null, '', url); + document.title = `${path} — cloud9`; + breadcrumbs(path); + $('directory').hidden = !result.stat.directory; + $('file').hidden = result.stat.directory; + if (result.stat.directory) { + const rows = document.createDocumentFragment(); + const sorted = result.entries.sort((a, b) => Number(b.directory) - Number(a.directory) || a.name.localeCompare(b.name)); + for (const entry of sorted) { + const row = document.createElement('tr'); + const permissions = document.createElement('td'), name = document.createElement('td'), size = document.createElement('td'); + permissions.className = 'mode'; permissions.textContent = mode(entry); + name.append(pathLink(entry.name + (entry.directory ? '/' : ''), child(path, entry.name), 'entry' + (entry.directory ? ' directory' : ''))); + size.className = 'size'; size.textContent = entry.directory ? '—' : entry.length.toLocaleString(); + row.append(permissions, name, size); rows.append(row); + } + $('entries').replaceChildren(rows); + $('empty').hidden = sorted.length !== 0; + status(`${sorted.length} ${sorted.length === 1 ? 'entry' : 'entries'}`); + } else { + $('filename').textContent = result.stat.name; + $('file-mode').textContent = mode(result.stat); + $('file-size').textContent = `${result.bytes.length.toLocaleString()} bytes`; + let editable = true; + try { + if (result.bytes.includes(0)) throw new Error('Binary'); + $('editor').value = text.decode(result.bytes); + } catch { editable = false; $('editor').value = ''; } + $('editor').hidden = !editable; + $('binary').hidden = editable; + $('edit-actions').hidden = !editable; + status('File loaded.'); + } +} +function mayLeave() { return !dirty || window.confirm('Discard unsaved changes?'); } +document.addEventListener('click', event => { + const link = event.target.closest('a[data-path]'); + if (!link || event.button !== 0 || event.ctrlKey || event.metaKey || event.shiftKey || event.altKey) return; + event.preventDefault(); + if (!busy && mayLeave()) action(() => navigate(link.dataset.path)); +}); +$('path-form').onsubmit = event => { + event.preventDefault(); + if (mayLeave()) action(() => navigate($('path').value)); +}; +$('reload').onclick = () => { if (mayLeave()) action(() => navigate(current ? currentPath : locationPath(), 'none')); }; +$('editor').oninput = () => { dirty = true; controls(); }; +$('save').onclick = () => action(async () => { + const bytes = new TextEncoder().encode($('editor').value); + const active = await connectedClient(); + let count; + try { count = await active.writePath(currentPath, bytes); } + catch (error) { + if (active.closed) throw new Error('Save interrupted. The file may be partially written; your edits are still here.'); + throw error; + } + // Keep the editor intact if refreshing after a successful write fails. + dirty = false; + await navigate(currentPath, 'none'); + status(`Saved ${count.toLocaleString()} bytes.`); +}); +$('download').onclick = () => { + const url = URL.createObjectURL(new Blob([current.bytes])); + const link = document.createElement('a'); link.href = url; link.download = current.stat.name; link.click(); + setTimeout(() => URL.revokeObjectURL(url), 1000); +}; +window.addEventListener('popstate', () => { + if (busy || !mayLeave()) { history.pushState(null, '', pathURL(currentPath)); return; } + action(() => navigate(locationPath(), 'none')); +}); +window.addEventListener('beforeunload', event => { + if (dirty) { event.preventDefault(); event.returnValue = ''; } +}); +action(() => navigate(locationPath(), 'none')); diff --git a/web/static/client.mjs b/web/static/client.mjs new file mode 100644 index 0000000..69f44d2 --- /dev/null +++ b/web/static/client.mjs @@ -0,0 +1,138 @@ +// Browser transport and file operations. All 9P serialization lives in WASM. +const encoder = new TextEncoder(); +const decoder = new TextDecoder(); +export class Client { + static async connect(url = '/_cloud9/9p', user = 'user', tree = '') { + const { instance } = await WebAssembly.instantiateStreaming(fetch(new URL('./cloud9.wasm', import.meta.url)), {}); + const wsURL = new URL(url, location.href); + wsURL.protocol = location.protocol === 'https:' ? 'wss:' : 'ws:'; + const socket = new WebSocket(wsURL); + socket.binaryType = 'arraybuffer'; + const client = new Client(instance.exports, socket); + try { + await new Promise((resolve, reject) => { + const timer = setTimeout(() => reject(new Error('Connection timed out')), 10000); + socket.onopen = () => { clearTimeout(timer); resolve(); }; + socket.onerror = () => { clearTimeout(timer); reject(new Error('Cannot connect to 9P bridge')); }; + }); + await client.ask('version'); + const u = encoder.encode(user), a = encoder.encode(tree); + client.stage(new Uint8Array([...u, ...a])); + await client.ask('attach', u.length, a.length); + return client; + } catch (error) { client.close(); throw error; } + } + constructor(wasm, socket) { + this.wasm = wasm; + this.socket = socket; + this.pending = null; + this.queue = Promise.resolve(); + this.closed = false; + wasm.init(); + socket.onmessage = ({ data }) => { + if (!this.pending) { this.close(); return; } + try { + this.stage(new Uint8Array(data)); + const kind = wasm.receive(data.byteLength); + if (!kind) return; + const result = { kind, data: this.data(), number: wasm.result_number(), directory: Boolean(wasm.result_directory()) }; + const pending = this.pending; + this.pending = null; + clearTimeout(pending.timer); + if (kind < 0) { + const error = new Error(decoder.decode(result.data)); + pending.reject(error); + if (wasm.is_dead()) this.abort(error); + } + else pending.resolve(result); + } catch (error) { this.abort(error); } + }; + socket.onclose = () => this.abort(new Error('Connection closed')); + } + data() { return new Uint8Array(this.wasm.memory.buffer, this.wasm.data_ptr(), this.wasm.data_len()).slice(); } + stage(bytes) { + if (bytes.length > this.wasm.input_capacity()) throw new Error('Input exceeds 64 KiB'); + new Uint8Array(this.wasm.memory.buffer, this.wasm.input_ptr(), bytes.length).set(bytes); + return bytes.length; + } + ask(operation, ...args) { + if (this.closed) return Promise.reject(new Error('Connection closed')); + if (this.pending) return Promise.reject(new Error('Request already in flight')); + if (this.wasm[operation](...args) < 0) return Promise.reject(new Error(decoder.decode(this.data()))); + return new Promise((resolve, reject) => { + const timer = setTimeout(() => this.abort(new Error('9P request timed out')), 15000); + this.pending = { resolve, reject, timer }; + try { + this.socket.send(new Uint8Array(this.wasm.memory.buffer, this.wasm.output_ptr(), this.wasm.output_len())); + this.wasm.sent(); + } catch (error) { this.abort(error); } + }); + } + abort(error) { + this.closed = true; + if (this.pending) { clearTimeout(this.pending.timer); this.pending.reject(error); this.pending = null; } + this.socket.close(); + } + close() { this.abort(new Error('Disconnected')); } + serial(operation) { + const result = this.queue.then(operation); + this.queue = result.catch(() => {}); + return result; + } + async withFile(path, operation) { + const names = path.split('/').filter(Boolean); + let live = false; + try { + // Walk in spec-sized batches, retaining an unopened root fid. + for (let index = 0; index < Math.max(1, names.length); index += 16) { + const batch = names.slice(index, index + 16); + const len = this.stage(encoder.encode(batch.join('/'))); + const result = await this.ask('walk', index === 0 ? 0 : 1, 1, len); + live = live || result.number > 0 || batch.length === 0; + if (result.number !== batch.length) throw new Error('Path does not exist'); + } + return await operation(); + } finally { + if (live && !this.closed) await this.ask('clunk', 1); + } + } + readPath(path) { + return this.serial(() => this.withFile(path, async () => { + const stat = JSON.parse(decoder.decode((await this.ask('stat', 1)).data)); + const opened = await this.ask('open', 1, 0); + const count = Math.min(this.wasm.max_read(), opened.number || Infinity); + let offset = 0; + const chunks = [], entries = []; + while (true) { + const { data } = await this.ask('read', 1, BigInt(offset), count); + if (!data.length) break; + offset += data.length; + if (offset > 8 * 1024 * 1024) throw new Error('Browser view is limited to 8 MiB'); + if (stat.directory) { + this.stage(data); + if (this.wasm.decode_directory(data.length) < 0) throw new Error(decoder.decode(this.data())); + entries.push(...JSON.parse(decoder.decode(this.data()))); + } else chunks.push(data); + } + const bytes = new Uint8Array(stat.directory ? 0 : offset); + let position = 0; + for (const chunk of chunks) { bytes.set(chunk, position); position += chunk.length; } + return { stat, entries, bytes }; + })); + } + writePath(path, bytes) { + if (!(bytes instanceof Uint8Array) || bytes.length > 8 * 1024 * 1024) return Promise.reject(new Error('Write is limited to 8 MiB')); + return this.serial(() => this.withFile(path, async () => { + const opened = await this.ask('open', 1, 1 | 16); // OWRITE | OTRUNC + const count = Math.min(this.wasm.max_write(), opened.number || Infinity); + let offset = 0; + while (offset < bytes.length) { + const len = this.stage(bytes.subarray(offset, offset + count)); + const result = await this.ask('write', 1, BigInt(offset), len); + if (!result.number) throw new Error('Server made no write progress'); + offset += result.number; + } + return offset; + })); + } +} diff --git a/web/static/index.html b/web/static/index.html new file mode 100644 index 0000000..aeca5d3 --- /dev/null +++ b/web/static/index.html @@ -0,0 +1,53 @@ + + + + + + cloud9 — files + + + + +
+ cloud9 +
A web interface to your file server.
+
+ +
+ +
+

Loading files…

+ +
+
+ + + +
ModeNameSize
+ +
+ +
+ + + diff --git a/web/static/style.css b/web/static/style.css new file mode 100644 index 0000000..a8605f5 --- /dev/null +++ b/web/static/style.css @@ -0,0 +1,87 @@ +:root { + font: 14px/1.45 sans-serif; + color: #333; + background: white; + font-synthesis: none; +} +* { box-sizing: border-box; } +body { margin: 0; padding: 22px 3%; } +a { color: #07539b; text-decoration: none; } +a:hover { text-decoration: underline; } +header { padding: 0 8px 17px; } +.brand { font-size: 30px; font-weight: bold; color: #222; letter-spacing: -1px; } +.description { color: #777; margin-top: 1px; } +.toolbar { + display: flex; + align-items: center; + gap: 2px; + border-bottom: 3px solid #ccc; + padding: 0 7px; +} +.toolbar > a, .toolbar > button { + padding: 5px 14px; + color: #555; + background: none; + border: 0; + font-size: 15px; +} +.toolbar .selected { background: #ccc; color: #111; } +button, input, textarea { font: inherit; color: inherit; } +button { + cursor: pointer; + border: 1px solid #aaa; + padding: 2px 9px; + background: #f5f5f5; + border-radius: 0; +} +button:hover:enabled { background: #e8e8e8; } +button:disabled { color: #999; cursor: default; } +#path-form { margin-left: auto; display: flex; align-items: center; gap: 6px; padding-bottom: 5px; } +#path-form label { color: #777; font-size: 12px; } +#path { width: 250px; border: 1px solid #bbb; padding: 2px 5px; font: 13px/1.5 monospace; } +main { padding: 0 8px; } +#breadcrumbs { display: flex; flex-wrap: wrap; gap: 7px; padding: 13px 0 5px; font-family: monospace; overflow-wrap: anywhere; } +#breadcrumbs .separator { color: #999; } +#breadcrumbs [aria-current] { color: #333; font-weight: bold; } +.summary { display: flex; align-items: baseline; justify-content: space-between; gap: 12px; margin: 2px 0 13px; font-size: 12px; } +#status { color: #777; margin: 0; min-height: 18px; } +#status.error { color: #a22; } +#up { white-space: nowrap; } +table { width: 100%; border-collapse: collapse; text-align: left; } +th { background: #eee; padding: 4px 8px; font-weight: bold; border-bottom: 1px solid #ccc; } +td { padding: 3px 8px; vertical-align: top; } +tbody tr:nth-child(even) { background: #f7f7f7; } +tbody tr:hover { background: #edf3f8; } +.mode { width: 120px; color: #777; font: 12px/1.7 monospace; white-space: nowrap; } +.entry { font-family: monospace; overflow-wrap: anywhere; } +.entry.directory { font-weight: bold; } +.size { width: 110px; text-align: right; font-variant-numeric: tabular-nums; white-space: nowrap; } +td.size { color: #666; font-size: 12px; } +#empty { color: #777; padding: 4px 8px; } +.file-heading { display: flex; flex-wrap: wrap; align-items: baseline; gap: 16px; padding: 6px 8px; background: #eee; border-bottom: 1px solid #ccc; } +h1 { font: bold 14px monospace; margin: 0; overflow-wrap: anywhere; } +#file-mode, #file-size { color: #666; font: 12px monospace; } +#download { margin-left: auto; font-size: 12px; } +textarea { display: block; width: 100%; min-height: 420px; resize: vertical; border: 1px solid #ddd; border-top: 0; padding: 10px; font: 13px/1.6 monospace; tab-size: 4; white-space: pre; overflow: auto; } +#binary { color: #777; padding: 16px 8px; } +.actions { display: flex; align-items: center; gap: 12px; margin-top: 10px; font-size: 12px; } +#edit-status { color: #777; } +footer { margin: 30px 8px 0; padding-top: 7px; border-top: 1px solid #ddd; text-align: right; color: #999; font-size: 11px; } +a:focus-visible, button:focus-visible { outline: 2px solid #07539b; outline-offset: 2px; } +[hidden] { display: none !important; } +@media (max-width: 600px) { + body { padding: 14px 10px; } + header { padding-left: 3px; } + .brand { font-size: 26px; } + .toolbar { flex-wrap: wrap; padding: 0; } + .toolbar > a, .toolbar > button { padding: 5px 10px; } + #path-form { order: -1; width: 100%; margin: 0 0 6px; } + #path { width: auto; min-width: 0; flex: 1; } + main { padding: 0; } + th, td { padding-left: 5px; padding-right: 5px; } + .mode { width: 86px; font-size: 10px; } + .size { width: 65px; } + .summary { align-items: start; } + .file-heading { gap: 7px 12px; } + textarea { min-height: 350px; } +} -- cgit v1.3